parse-bench 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (227) hide show
  1. parse_bench/__init__.py +3 -0
  2. parse_bench/analysis/__init__.py +6 -0
  3. parse_bench/analysis/aggregation_report.py +582 -0
  4. parse_bench/analysis/cli.py +472 -0
  5. parse_bench/analysis/comparison.py +382 -0
  6. parse_bench/analysis/comparison_core.py +357 -0
  7. parse_bench/analysis/comparison_report.py +2066 -0
  8. parse_bench/analysis/detailed_report.py +2254 -0
  9. parse_bench/analysis/leaderboard_report.py +852 -0
  10. parse_bench/analysis/metric_definitions.py +771 -0
  11. parse_bench/cli.py +267 -0
  12. parse_bench/data/__init__.py +1 -0
  13. parse_bench/data/cli.py +118 -0
  14. parse_bench/data/download.py +127 -0
  15. parse_bench/evaluation/__init__.py +11 -0
  16. parse_bench/evaluation/cli.py +435 -0
  17. parse_bench/evaluation/evaluators/__init__.py +17 -0
  18. parse_bench/evaluation/evaluators/base.py +34 -0
  19. parse_bench/evaluation/evaluators/extract.py +429 -0
  20. parse_bench/evaluation/evaluators/layoutdet.py +1682 -0
  21. parse_bench/evaluation/evaluators/parse.py +1353 -0
  22. parse_bench/evaluation/evaluators/qa.py +199 -0
  23. parse_bench/evaluation/layout_adapters/__init__.py +21 -0
  24. parse_bench/evaluation/layout_adapters/adapters.py +3180 -0
  25. parse_bench/evaluation/layout_adapters/base.py +105 -0
  26. parse_bench/evaluation/layout_adapters/registry.py +109 -0
  27. parse_bench/evaluation/layout_label_mappers/__init__.py +22 -0
  28. parse_bench/evaluation/layout_label_mappers/base.py +66 -0
  29. parse_bench/evaluation/layout_label_mappers/mappers.py +332 -0
  30. parse_bench/evaluation/layout_label_mappers/projection.py +74 -0
  31. parse_bench/evaluation/layout_label_mappers/registry.py +119 -0
  32. parse_bench/evaluation/metric_aggregation.py +56 -0
  33. parse_bench/evaluation/metrics/__init__.py +5 -0
  34. parse_bench/evaluation/metrics/attribution/__init__.py +35 -0
  35. parse_bench/evaluation/metrics/attribution/constants.py +12 -0
  36. parse_bench/evaluation/metrics/attribution/core.py +1108 -0
  37. parse_bench/evaluation/metrics/attribution/evaluate.py +446 -0
  38. parse_bench/evaluation/metrics/attribution/geometry.py +161 -0
  39. parse_bench/evaluation/metrics/attribution/text_utils.py +233 -0
  40. parse_bench/evaluation/metrics/base.py +33 -0
  41. parse_bench/evaluation/metrics/downstream/__init__.py +0 -0
  42. parse_bench/evaluation/metrics/extract/__init__.py +29 -0
  43. parse_bench/evaluation/metrics/extract/json_subset_match.py +473 -0
  44. parse_bench/evaluation/metrics/extract/json_subset_match_metric.py +81 -0
  45. parse_bench/evaluation/metrics/extract/list_unwrap.py +340 -0
  46. parse_bench/evaluation/metrics/extract/rule_based_metric.py +90 -0
  47. parse_bench/evaluation/metrics/extract/test_rules.py +409 -0
  48. parse_bench/evaluation/metrics/extract/test_types.py +11 -0
  49. parse_bench/evaluation/metrics/field_grounding/__init__.py +21 -0
  50. parse_bench/evaluation/metrics/field_grounding/core.py +437 -0
  51. parse_bench/evaluation/metrics/field_grounding/extract_adapter.py +1224 -0
  52. parse_bench/evaluation/metrics/field_grounding/parse_adapter.py +697 -0
  53. parse_bench/evaluation/metrics/field_grounding/rule_filters.py +19 -0
  54. parse_bench/evaluation/metrics/field_grounding/value_compare.py +190 -0
  55. parse_bench/evaluation/metrics/layoutdet/__init__.py +17 -0
  56. parse_bench/evaluation/metrics/layoutdet/classification_utils.py +300 -0
  57. parse_bench/evaluation/metrics/layoutdet/iou.py +76 -0
  58. parse_bench/evaluation/metrics/parse/__init__.py +5 -0
  59. parse_bench/evaluation/metrics/parse/_vendor_grits_reference.py +531 -0
  60. parse_bench/evaluation/metrics/parse/cross_page_table_consistency.py +165 -0
  61. parse_bench/evaluation/metrics/parse/emphasis_spans.py +242 -0
  62. parse_bench/evaluation/metrics/parse/fast_tree_edit.py +282 -0
  63. parse_bench/evaluation/metrics/parse/grits_metric.py +1125 -0
  64. parse_bench/evaluation/metrics/parse/grits_reference_metric.py +142 -0
  65. parse_bench/evaluation/metrics/parse/header_accuracy_metric.py +1662 -0
  66. parse_bench/evaluation/metrics/parse/llm_normalization/__init__.py +51 -0
  67. parse_bench/evaluation/metrics/parse/llm_normalization/base.py +125 -0
  68. parse_bench/evaluation/metrics/parse/llm_normalization/config.py +44 -0
  69. parse_bench/evaluation/metrics/parse/llm_normalization/postprocess.py +322 -0
  70. parse_bench/evaluation/metrics/parse/llm_normalization/strategy_judge.py +541 -0
  71. parse_bench/evaluation/metrics/parse/mermaid_graph.py +682 -0
  72. parse_bench/evaluation/metrics/parse/rule_based_judge_metric.py +56 -0
  73. parse_bench/evaluation/metrics/parse/rule_based_metric.py +434 -0
  74. parse_bench/evaluation/metrics/parse/rules_bag.py +1161 -0
  75. parse_bench/evaluation/metrics/parse/rules_base.py +751 -0
  76. parse_bench/evaluation/metrics/parse/rules_chart.py +1556 -0
  77. parse_bench/evaluation/metrics/parse/rules_diagram.py +591 -0
  78. parse_bench/evaluation/metrics/parse/rules_form.py +2274 -0
  79. parse_bench/evaluation/metrics/parse/rules_formatting.py +1500 -0
  80. parse_bench/evaluation/metrics/parse/rules_heading.py +228 -0
  81. parse_bench/evaluation/metrics/parse/rules_list.py +226 -0
  82. parse_bench/evaluation/metrics/parse/rules_page_decoration.py +276 -0
  83. parse_bench/evaluation/metrics/parse/rules_table.py +1666 -0
  84. parse_bench/evaluation/metrics/parse/rules_text.py +340 -0
  85. parse_bench/evaluation/metrics/parse/rules_watermark.py +105 -0
  86. parse_bench/evaluation/metrics/parse/structural_consistency_metric.py +251 -0
  87. parse_bench/evaluation/metrics/parse/table_extraction.py +152 -0
  88. parse_bench/evaluation/metrics/parse/table_merging.py +195 -0
  89. parse_bench/evaluation/metrics/parse/table_pairing.py +87 -0
  90. parse_bench/evaluation/metrics/parse/table_parsing.py +955 -0
  91. parse_bench/evaluation/metrics/parse/table_record_match_metric.py +1453 -0
  92. parse_bench/evaluation/metrics/parse/table_splitting.py +301 -0
  93. parse_bench/evaluation/metrics/parse/table_title_stripping.py +530 -0
  94. parse_bench/evaluation/metrics/parse/teds_metric.py +600 -0
  95. parse_bench/evaluation/metrics/parse/test_rules.py +120 -0
  96. parse_bench/evaluation/metrics/parse/test_types.py +103 -0
  97. parse_bench/evaluation/metrics/parse/text_content_projection.py +175 -0
  98. parse_bench/evaluation/metrics/parse/text_similarity_metric.py +61 -0
  99. parse_bench/evaluation/metrics/parse/utils.py +885 -0
  100. parse_bench/evaluation/metrics/qa/__init__.py +5 -0
  101. parse_bench/evaluation/metrics/qa/answer_comparison.py +380 -0
  102. parse_bench/evaluation/qa/__init__.py +5 -0
  103. parse_bench/evaluation/qa/llm_service.py +335 -0
  104. parse_bench/evaluation/reports/__init__.py +8 -0
  105. parse_bench/evaluation/reports/csv.py +64 -0
  106. parse_bench/evaluation/reports/html.py +338 -0
  107. parse_bench/evaluation/reports/markdown.py +98 -0
  108. parse_bench/evaluation/reports/rule_csv.py +22 -0
  109. parse_bench/evaluation/runner.py +1864 -0
  110. parse_bench/evaluation/stats.py +104 -0
  111. parse_bench/extensions.py +72 -0
  112. parse_bench/inference/__init__.py +33 -0
  113. parse_bench/inference/chunkr_layout_extraction.py +160 -0
  114. parse_bench/inference/cli.py +484 -0
  115. parse_bench/inference/layout_extraction.py +422 -0
  116. parse_bench/inference/pipelines/__init__.py +59 -0
  117. parse_bench/inference/pipelines/extract.py +39 -0
  118. parse_bench/inference/pipelines/layout.py +142 -0
  119. parse_bench/inference/pipelines/parse.py +2603 -0
  120. parse_bench/inference/pipelines.py +0 -0
  121. parse_bench/inference/providers/__init__.py +28 -0
  122. parse_bench/inference/providers/base.py +196 -0
  123. parse_bench/inference/providers/cancellation.py +137 -0
  124. parse_bench/inference/providers/extract/__init__.py +22 -0
  125. parse_bench/inference/providers/extract/citations.py +549 -0
  126. parse_bench/inference/providers/extract/extend.py +851 -0
  127. parse_bench/inference/providers/extract/llamaextract_v2_api.py +583 -0
  128. parse_bench/inference/providers/layoutdet/__init__.py +25 -0
  129. parse_bench/inference/providers/layoutdet/adapters.py +946 -0
  130. parse_bench/inference/providers/layoutdet/base.py +203 -0
  131. parse_bench/inference/providers/layoutdet/chandra.py +449 -0
  132. parse_bench/inference/providers/layoutdet/docling.py +125 -0
  133. parse_bench/inference/providers/layoutdet/dots_ocr.py +606 -0
  134. parse_bench/inference/providers/layoutdet/layout_v3.py +137 -0
  135. parse_bench/inference/providers/layoutdet/layout_v3_byoc.py +204 -0
  136. parse_bench/inference/providers/layoutdet/paddle.py +117 -0
  137. parse_bench/inference/providers/layoutdet/qwen3vl.py +360 -0
  138. parse_bench/inference/providers/layoutdet/surya.py +250 -0
  139. parse_bench/inference/providers/layoutdet/yolo.py +109 -0
  140. parse_bench/inference/providers/parse/__init__.py +64 -0
  141. parse_bench/inference/providers/parse/_docling_common.py +233 -0
  142. parse_bench/inference/providers/parse/_layout_utils.py +611 -0
  143. parse_bench/inference/providers/parse/amazon_nova.py +515 -0
  144. parse_bench/inference/providers/parse/anthropic.py +882 -0
  145. parse_bench/inference/providers/parse/azure_document_intelligence.py +700 -0
  146. parse_bench/inference/providers/parse/chandra2.py +633 -0
  147. parse_bench/inference/providers/parse/chunkr.py +268 -0
  148. parse_bench/inference/providers/parse/databricks_ai_parse.py +724 -0
  149. parse_bench/inference/providers/parse/datalab.py +370 -0
  150. parse_bench/inference/providers/parse/deepseekocr2.py +382 -0
  151. parse_bench/inference/providers/parse/docling.py +281 -0
  152. parse_bench/inference/providers/parse/docling_serve.py +289 -0
  153. parse_bench/inference/providers/parse/dots_ocr.py +574 -0
  154. parse_bench/inference/providers/parse/extend_parse.py +710 -0
  155. parse_bench/inference/providers/parse/falconocr.py +436 -0
  156. parse_bench/inference/providers/parse/florin_parser_nano.py +559 -0
  157. parse_bench/inference/providers/parse/gemma4.py +472 -0
  158. parse_bench/inference/providers/parse/glm_zai.py +229 -0
  159. parse_bench/inference/providers/parse/google.py +1125 -0
  160. parse_bench/inference/providers/parse/google_agentic_vision.py +819 -0
  161. parse_bench/inference/providers/parse/google_docai.py +776 -0
  162. parse_bench/inference/providers/parse/google_docai_layout_normalization.py +573 -0
  163. parse_bench/inference/providers/parse/granite_vision.py +515 -0
  164. parse_bench/inference/providers/parse/infinity_parser2.py +704 -0
  165. parse_bench/inference/providers/parse/kdl_frontier_nano.py +3327 -0
  166. parse_bench/inference/providers/parse/landingai.py +452 -0
  167. parse_bench/inference/providers/parse/liteparse.py +350 -0
  168. parse_bench/inference/providers/parse/llamaparse.py +677 -0
  169. parse_bench/inference/providers/parse/llamaparse_v2_normalization.py +1013 -0
  170. parse_bench/inference/providers/parse/markitdown.py +138 -0
  171. parse_bench/inference/providers/parse/mineru25.py +405 -0
  172. parse_bench/inference/providers/parse/mineru2605pro.py +432 -0
  173. parse_bench/inference/providers/parse/mineru_diffusion.py +371 -0
  174. parse_bench/inference/providers/parse/mistral_ocr.py +546 -0
  175. parse_bench/inference/providers/parse/nemotron_omni.py +473 -0
  176. parse_bench/inference/providers/parse/oi_parser.py +222 -0
  177. parse_bench/inference/providers/parse/openai.py +740 -0
  178. parse_bench/inference/providers/parse/opendataloader.py +152 -0
  179. parse_bench/inference/providers/parse/paddleocr.py +624 -0
  180. parse_bench/inference/providers/parse/pdf_inspector.py +142 -0
  181. parse_bench/inference/providers/parse/pulse.py +785 -0
  182. parse_bench/inference/providers/parse/pymupdf.py +207 -0
  183. parse_bench/inference/providers/parse/pymupdf4llm.py +356 -0
  184. parse_bench/inference/providers/parse/pypdf.py +179 -0
  185. parse_bench/inference/providers/parse/qwen.py +678 -0
  186. parse_bench/inference/providers/parse/rakedoc_nano.py +70 -0
  187. parse_bench/inference/providers/parse/reducto.py +546 -0
  188. parse_bench/inference/providers/parse/surya2.py +372 -0
  189. parse_bench/inference/providers/parse/tesseract.py +301 -0
  190. parse_bench/inference/providers/parse/textract.py +694 -0
  191. parse_bench/inference/providers/parse/unlimitedocr.py +346 -0
  192. parse_bench/inference/providers/parse/unstructured.py +485 -0
  193. parse_bench/inference/providers/parse/warp_ingest.py +199 -0
  194. parse_bench/inference/providers/registry.py +49 -0
  195. parse_bench/inference/renormalize.py +170 -0
  196. parse_bench/inference/runner.py +2023 -0
  197. parse_bench/layout_label_mapping.py +424 -0
  198. parse_bench/layout_projection.py +179 -0
  199. parse_bench/pipeline/__init__.py +1 -0
  200. parse_bench/pipeline/cli.py +549 -0
  201. parse_bench/schemas/__init__.py +33 -0
  202. parse_bench/schemas/evaluation.py +93 -0
  203. parse_bench/schemas/extract_output.py +36 -0
  204. parse_bench/schemas/layout_detection_output.py +545 -0
  205. parse_bench/schemas/layout_ontology.py +315 -0
  206. parse_bench/schemas/metrics.py +69 -0
  207. parse_bench/schemas/parse_output.py +152 -0
  208. parse_bench/schemas/pipeline.py +22 -0
  209. parse_bench/schemas/pipeline_io.py +106 -0
  210. parse_bench/schemas/product.py +97 -0
  211. parse_bench/test_cases/__init__.py +25 -0
  212. parse_bench/test_cases/bbox_value_strict_comparator.py +880 -0
  213. parse_bench/test_cases/extract_field_paths.py +164 -0
  214. parse_bench/test_cases/layout_attribution_generation.py +287 -0
  215. parse_bench/test_cases/loader.py +652 -0
  216. parse_bench/test_cases/parse_rule_schemas.py +1071 -0
  217. parse_bench/test_cases/rule_filters.py +32 -0
  218. parse_bench/test_cases/rule_ids.py +107 -0
  219. parse_bench/test_cases/schema.py +427 -0
  220. parse_bench/utils/__init__.py +15 -0
  221. parse_bench/utils/gemini_layout_utils.py +670 -0
  222. parse_bench/utils/text_aggregation.py +100 -0
  223. parse_bench-1.0.0.dist-info/METADATA +476 -0
  224. parse_bench-1.0.0.dist-info/RECORD +227 -0
  225. parse_bench-1.0.0.dist-info/WHEEL +4 -0
  226. parse_bench-1.0.0.dist-info/entry_points.txt +2 -0
  227. parse_bench-1.0.0.dist-info/licenses/LICENSE +201 -0
@@ -0,0 +1,340 @@
1
+ """Text presence, order, and baseline test rules."""
2
+
3
+ import re
4
+ from typing import cast
5
+
6
+ from fuzzysearch import find_near_matches
7
+ from rapidfuzz import fuzz
8
+
9
+ from parse_bench.evaluation.metrics.parse.rules_base import (
10
+ ParseTestRule,
11
+ _strip_and_replace_latex,
12
+ )
13
+ from parse_bench.evaluation.metrics.parse.test_types import TestType
14
+ from parse_bench.evaluation.metrics.parse.utils import normalize_text, normalize_text_light
15
+ from parse_bench.test_cases.parse_rule_schemas import (
16
+ ParseBaselineRule,
17
+ ParseOrderRule,
18
+ ParsePresenceRule,
19
+ )
20
+
21
+
22
+ class TextPresenceRule(ParseTestRule):
23
+ """Test rule for text presence/absence."""
24
+
25
+ def __init__(self, rule_data: ParsePresenceRule | dict):
26
+ super().__init__(rule_data)
27
+ rule_data = cast(ParsePresenceRule, self._rule_data)
28
+
29
+ if self.type not in {TestType.PRESENT.value, TestType.ABSENT.value}:
30
+ raise ValueError(f"Invalid type for TextPresenceRule: {self.type}")
31
+
32
+ # Check if we should use light normalization that preserves formatting tags
33
+ self.keep_formatting = rule_data.keep_formatting_text_normalisation
34
+ normalize_fn = normalize_text_light if self.keep_formatting else normalize_text
35
+
36
+ self.text = normalize_fn(rule_data.text)
37
+ if not self.text.strip():
38
+ raise ValueError("Text field cannot be empty")
39
+
40
+ self.case_sensitive = rule_data.case_sensitive
41
+ self.first_n = rule_data.first_n
42
+ self.last_n = rule_data.last_n
43
+ self.count = rule_data.count
44
+
45
+ if self.count is not None:
46
+ if not isinstance(self.count, int):
47
+ raise ValueError("Count field must be an integer when provided")
48
+ if self.count < 0:
49
+ raise ValueError("Count field cannot be negative")
50
+
51
+ def _count_non_overlapping_fuzzy_matches(self, query: str, content: str) -> int:
52
+ """Count non-overlapping fuzzy matches for query in content.
53
+
54
+ We keep matches non-overlapping to avoid over-counting near-duplicate
55
+ windows for the same textual occurrence.
56
+ """
57
+ max_distance = min(self.max_diffs, 15)
58
+ matches = sorted(
59
+ find_near_matches(query, content, max_l_dist=max_distance),
60
+ key=lambda match: (match.start, match.end),
61
+ )
62
+
63
+ non_overlapping_count = 0
64
+ last_end = -1
65
+ for match in matches:
66
+ if match.start >= last_end:
67
+ non_overlapping_count += 1
68
+ last_end = match.end
69
+ return non_overlapping_count
70
+
71
+ def run(self, md_content: str, normalized_content: str | None = None) -> tuple[bool, str]:
72
+ """Check if text is present or absent in markdown."""
73
+ reference_query = self.text
74
+
75
+ # When keep_formatting is enabled, we must re-normalize content with light normalization
76
+ # since the pre-normalized content from the metric uses standard normalization
77
+ if self.keep_formatting:
78
+ # Always use light normalization when testing formatting
79
+ normalized_content = normalize_text_light(md_content)
80
+ # Backward compatibility: standard normalize_text lowercases by default,
81
+ # so keep-formatting mode mirrors that behavior unless case sensitivity
82
+ # is explicitly disabled below.
83
+ reference_query = reference_query.lower()
84
+ normalized_content = normalized_content.lower()
85
+ elif normalized_content is None:
86
+ # Use pre-normalized content if provided, otherwise normalize
87
+ normalized_content = normalize_text(md_content)
88
+
89
+ if not self.case_sensitive:
90
+ reference_query = reference_query.lower()
91
+ normalized_content = normalized_content.lower()
92
+
93
+ # Apply first_n/last_n if specified
94
+ if self.first_n and self.last_n:
95
+ normalized_content = normalized_content[: self.first_n] + normalized_content[-self.last_n :]
96
+ elif self.first_n:
97
+ normalized_content = normalized_content[: self.first_n]
98
+ elif self.last_n:
99
+ normalized_content = normalized_content[-self.last_n :]
100
+
101
+ # Threshold for fuzzy matching derived from max_diffs.
102
+ # Floor at 0.7 so short queries (e.g. 2-3 chars) don't produce
103
+ # wildly permissive thresholds that match almost anything.
104
+ raw_threshold = 1.0 - (self.max_diffs / (len(reference_query) if len(reference_query) > 0 else 1))
105
+ threshold = max(0.7, min(1.0, raw_threshold))
106
+ best_ratio = fuzz.partial_ratio(reference_query, normalized_content) / 100.0
107
+
108
+ if self.type == TestType.PRESENT.value:
109
+ # Backward compatibility: count=None or count=0 keeps legacy behavior
110
+ # (presence check only, regardless of number of occurrences).
111
+ if self.count not in {None, 0}:
112
+ if self.max_diffs == 0:
113
+ actual_count = normalized_content.count(reference_query)
114
+ else:
115
+ actual_count = self._count_non_overlapping_fuzzy_matches(reference_query, normalized_content)
116
+
117
+ if actual_count == self.count:
118
+ return True, ""
119
+ msg = f"Expected '{reference_query[:40]}...' exactly {self.count} time(s), but found {actual_count}"
120
+ return False, msg
121
+
122
+ if best_ratio >= threshold:
123
+ return True, ""
124
+ else:
125
+ msg = (
126
+ f"Expected '{reference_query[:40]}...' with threshold {threshold} "
127
+ f"but best match ratio was {best_ratio:.3f}"
128
+ )
129
+ return False, msg
130
+ else: # ABSENT
131
+ if best_ratio < threshold:
132
+ return True, ""
133
+ else:
134
+ msg = (
135
+ f"Expected absence of '{reference_query[:40]}...' with threshold {threshold} "
136
+ f"but best match ratio was {best_ratio:.3f}"
137
+ )
138
+ return False, msg
139
+
140
+
141
+ class BaselineRule(ParseTestRule):
142
+ """Test rule for baseline quality checks (blank pages, repeats, character sets)."""
143
+
144
+ def __init__(self, rule_data: ParseBaselineRule | dict):
145
+ super().__init__(rule_data)
146
+ rule_data = cast(ParseBaselineRule, self._rule_data)
147
+
148
+ self.max_length = rule_data.max_length
149
+ self.max_length_skips_image_alt_tags = rule_data.max_length_skips_image_alt_tags
150
+ self.max_repeats = rule_data.max_repeats
151
+ self.check_disallowed_characters = rule_data.check_disallowed_characters
152
+
153
+ def run(self, content: str, normalized_content: str | None = None) -> tuple[bool, str]:
154
+ """Run baseline quality checks."""
155
+ base_content_len = len("".join(c for c in content if c.isalnum()).strip())
156
+
157
+ # Blank page check
158
+ if self.max_length is not None:
159
+ if self.max_length_skips_image_alt_tags:
160
+ # Remove markdown image tags
161
+ content_for_length_check = re.sub(r"!\[.*?\]\(.*?\)", "", content)
162
+ base_content_len = len("".join(c for c in content_for_length_check if c.isalnum()).strip())
163
+
164
+ if base_content_len > self.max_length:
165
+ return (
166
+ False,
167
+ f"{base_content_len} characters were output for a page we expected to be blank",
168
+ )
169
+ else:
170
+ return True, ""
171
+
172
+ # Check for empty content
173
+ if base_content_len == 0:
174
+ return False, "The text contains no alpha numeric characters"
175
+
176
+ # Check for excessive repetition using sampled windows across the
177
+ # document. Sampling keeps this O(1) even on very large documents.
178
+ if len(content) > 50:
179
+ # Sample up to 5 windows of 200 chars evenly spaced through the doc
180
+ sample_size = min(200, len(content))
181
+ num_samples = min(5, max(1, len(content) // sample_size))
182
+ step = max(1, (len(content) - sample_size) // max(1, num_samples - 1)) if num_samples > 1 else 0
183
+ repetitive_samples = 0
184
+ for i in range(num_samples):
185
+ start = i * step
186
+ window = content[start : start + sample_size]
187
+ if len(set(window)) < 3:
188
+ repetitive_samples += 1
189
+ # Fail if majority of samples are repetitive
190
+ if repetitive_samples > num_samples // 2:
191
+ return False, "Text appears to be excessively repetitive"
192
+
193
+ # Check for disallowed characters (CJK, emoji, etc.)
194
+ if self.check_disallowed_characters:
195
+ pattern = re.compile(
196
+ r"["
197
+ r"\u4e00-\u9FFF" # CJK Unified Ideographs
198
+ r"\u3040-\u309F" # Hiragana
199
+ r"\u30A0-\u30FF" # Katakana
200
+ r"\U0001F600-\U0001F64F" # Emoticons
201
+ r"\U0001F300-\U0001F5FF" # Miscellaneous Symbols
202
+ r"\U0001F680-\U0001F6FF" # Transport and Map Symbols
203
+ r"\U0001F1E0-\U0001F1FF" # Regional Indicator Symbols
204
+ r"]",
205
+ flags=re.UNICODE,
206
+ )
207
+ matches = pattern.findall(content)
208
+ if matches:
209
+ return False, f"Text contains disallowed characters: {matches[:5]}"
210
+
211
+ return True, ""
212
+
213
+
214
+ class TextOrderRule(ParseTestRule):
215
+ """Test rule to verify that one text appears before another."""
216
+
217
+ # Residual HTML markup left behind by ``normalize_text`` (structural tags such
218
+ # as <table>/<tr>/<td>, which it does not touch). Order anchors are authored
219
+ # against plain markdown and routinely span a label and its value, so any page
220
+ # the parser renders as a table puts a `</td><td>` inside the anchor and the
221
+ # rule could never match. Substituting a SPACE — never the empty string —
222
+ # keeps cell contents from welding into one unmatchable token, and matches how
223
+ # the word/sentence bag rules already treat markup.
224
+ _HTML_TAG_PATTERN = re.compile(r"</?[^>]+>")
225
+
226
+ @classmethod
227
+ def _strip_html(cls, text: str) -> str:
228
+ """Replace residual HTML tags with a space and collapse the runs."""
229
+ return re.sub(r" +", " ", cls._HTML_TAG_PATTERN.sub(" ", text)).strip()
230
+
231
+ def __init__(self, rule_data: ParseOrderRule | dict):
232
+ super().__init__(rule_data)
233
+ rule_data = cast(ParseOrderRule, self._rule_data)
234
+
235
+ if self.type != TestType.ORDER.value:
236
+ raise ValueError(f"Invalid type for TextOrderRule: {self.type}")
237
+
238
+ # Check if we should use light normalization that preserves formatting tags
239
+ self.keep_formatting = rule_data.keep_formatting_text_normalisation
240
+ normalize_fn = normalize_text_light if self.keep_formatting else normalize_text
241
+
242
+ # Canonicalize LaTeX in query strings so "$...$" and "LATEX" authored
243
+ # expectations are matched consistently against content.
244
+ before_for_match = _strip_and_replace_latex(rule_data.before)
245
+ after_for_match = _strip_and_replace_latex(rule_data.after)
246
+
247
+ # Import SentenceBagRule lazily to access _MULTI_DOT_PATTERN
248
+ from parse_bench.evaluation.metrics.parse.rules_bag import SentenceBagRule
249
+
250
+ self.before = re.sub(r" +", " ", SentenceBagRule._MULTI_DOT_PATTERN.sub(" ", normalize_fn(before_for_match)))
251
+ self.after = re.sub(r" +", " ", SentenceBagRule._MULTI_DOT_PATTERN.sub(" ", normalize_fn(after_for_match)))
252
+ # Strip markup on the rule side too, so anchor and content are compared in
253
+ # the same alphabet. Skipped for keep_formatting rules, whose whole point is
254
+ # to assert on the styling tags themselves.
255
+ if not self.keep_formatting:
256
+ self.before = self._strip_html(self.before)
257
+ self.after = self._strip_html(self.after)
258
+ if not self.before.strip():
259
+ raise ValueError("Before field cannot be empty")
260
+ if not self.after.strip():
261
+ raise ValueError("After field cannot be empty")
262
+ if self.max_diffs > len(self.before) // 2 or self.max_diffs > len(self.after) // 2:
263
+ raise ValueError("Max diffs is too large for this test, greater than 50% of the search string")
264
+
265
+ def run(self, md_content: str, normalized_content: str | None = None) -> tuple[bool, str]:
266
+ """Check if 'before' text appears before 'after' text.
267
+
268
+ When multiple instances exist, we check that the FIRST occurrence of 'before'
269
+ appears before the LAST occurrence of 'after'. This handles cases where the same
270
+ text may appear multiple times in the document.
271
+ """
272
+ from parse_bench.evaluation.metrics.parse.rules_bag import SentenceBagRule
273
+
274
+ # Order matching must canonicalize LaTeX consistently between rule text and content.
275
+ # Re-normalize from raw markdown so inline math ($...$) becomes a stable LATEX token.
276
+ content_for_match = _strip_and_replace_latex(md_content)
277
+ if self.keep_formatting:
278
+ normalized_content = normalize_text_light(content_for_match)
279
+ else:
280
+ normalized_content = normalize_text(content_for_match)
281
+ normalized_content = re.sub(r" +", " ", SentenceBagRule._MULTI_DOT_PATTERN.sub(" ", normalized_content))
282
+ if not self.keep_formatting:
283
+ normalized_content = self._strip_html(normalized_content)
284
+
285
+ # OPTIMIZATION: Try exact match first (O(n) vs O(n*m) for fuzzy)
286
+ # This provides ~10-100x speedup for rules that match exactly
287
+ # Use find() for FIRST occurrence of before, rfind() for LAST occurrence of after
288
+ before_pos = normalized_content.find(self.before)
289
+ after_pos = normalized_content.rfind(self.after)
290
+
291
+ if before_pos != -1 and after_pos != -1 and before_pos < after_pos:
292
+ # Fast path: both found exactly and in correct order
293
+ return True, ""
294
+
295
+ # Slow path: fall back to fuzzy matching
296
+ # instead of max_diffs = 2, use a relative distance
297
+ # here as the annotation fail quite a lot on typo / unicode chars
298
+ # Cap at 30 to avoid combinatorial explosion in fuzzysearch for long strings
299
+ # (e.g. 7k pattern on 30k text with max_l_dist=350 is ~73B operations)
300
+ before_max_dist = min(max(len(self.before) // 20, self.max_diffs), 15)
301
+ after_max_dist = min(max(len(self.after) // 20, self.max_diffs), 15)
302
+
303
+ # Only do expensive fuzzy search if exact match failed
304
+ if before_pos == -1:
305
+ before_matches = find_near_matches(self.before, normalized_content, max_l_dist=before_max_dist)
306
+ else:
307
+ # Create a fake match object for the exact match (first occurrence)
308
+ before_matches = [type("Match", (), {"start": before_pos, "end": before_pos + len(self.before)})()]
309
+
310
+ if after_pos == -1:
311
+ after_matches = find_near_matches(self.after, normalized_content, max_l_dist=after_max_dist)
312
+ else:
313
+ # Create a fake match object for the exact match (last occurrence)
314
+ after_matches = [type("Match", (), {"start": after_pos, "end": after_pos + len(self.after)})()]
315
+
316
+ if not before_matches:
317
+ return (
318
+ False,
319
+ f"'before' text '{self.before[:40]}...' not found with max_l_dist {before_max_dist}",
320
+ )
321
+ if not after_matches:
322
+ return (
323
+ False,
324
+ f"'after' text '{self.after[:40]}...' not found with max_l_dist {after_max_dist}",
325
+ )
326
+
327
+ # Get FIRST occurrence of before (earliest start position)
328
+ first_before_match = min(before_matches, key=lambda m: m.start)
329
+ # Get LAST occurrence of after (latest start position)
330
+ last_after_match = max(after_matches, key=lambda m: m.start)
331
+
332
+ if first_before_match.start < last_after_match.start:
333
+ return True, ""
334
+
335
+ return (
336
+ False,
337
+ f"Could not find a location where '{self.before}...' appears before '{self.after}...'. "
338
+ f"First 'before' at position {first_before_match.start}, "
339
+ f"last 'after' at position {last_after_match.start}.",
340
+ )
@@ -0,0 +1,105 @@
1
+ """Watermark-removal rule with an explicit content-preservation guard."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from typing import Any, cast
7
+
8
+ from rapidfuzz import fuzz
9
+
10
+ from parse_bench.evaluation.metrics.parse.rules_base import ParseTestRule
11
+ from parse_bench.evaluation.metrics.parse.utils import normalize_text
12
+ from parse_bench.test_cases.parse_rule_schemas import ParseWatermarkRemovalRule
13
+
14
+
15
+ def _partial_similarity(needle: str, content: str) -> float:
16
+ normalized_needle = normalize_text(needle).casefold().strip()
17
+ normalized_content = normalize_text(content).casefold().strip()
18
+ if not normalized_needle or not normalized_content:
19
+ return 0.0
20
+ return float(fuzz.partial_ratio(normalized_needle, normalized_content) / 100.0)
21
+
22
+
23
+ def _occurrence_count(needle: str, content: str) -> int:
24
+ """Count normalized, non-overlapping literal phrase occurrences."""
25
+ normalized_needle = normalize_text(needle).casefold().strip()
26
+ normalized_content = normalize_text(content).casefold().strip()
27
+ if not normalized_needle or not normalized_content:
28
+ return 0
29
+ pattern = re.escape(normalized_needle).replace(r"\ ", r"\s+")
30
+ return sum(1 for _ in re.finditer(pattern, normalized_content))
31
+
32
+
33
+ class WatermarkRemovalRule(ParseTestRule):
34
+ """Require watermark text to disappear while sampled body anchors survive."""
35
+
36
+ def __init__(self, rule_data: ParseWatermarkRemovalRule | dict[str, Any]):
37
+ super().__init__(rule_data)
38
+ if not isinstance(self._rule_data, ParseWatermarkRemovalRule):
39
+ raise TypeError("watermark_removal requires ParseWatermarkRemovalRule")
40
+
41
+ def run(
42
+ self,
43
+ md_content: str,
44
+ normalized_content: str | None = None,
45
+ ) -> tuple[bool, str, float]:
46
+ rule = cast(ParseWatermarkRemovalRule, self._rule_data)
47
+ actual_markdown = md_content
48
+ if self.parse_output is not None:
49
+ for markdown_page in self.parse_output.pages:
50
+ if markdown_page.page_index + 1 == rule.page:
51
+ actual_markdown = markdown_page.markdown
52
+ break
53
+
54
+ allowed_occurrences = rule.allowed_occurrences or [0] * len(rule.watermark_texts)
55
+ watermark_matches: list[dict[str, Any]] = [
56
+ {
57
+ "text": text,
58
+ "similarity": _partial_similarity(text, actual_markdown),
59
+ "occurrences": _occurrence_count(text, actual_markdown),
60
+ "allowed_occurrences": allowed,
61
+ }
62
+ for text, allowed in zip(rule.watermark_texts, allowed_occurrences, strict=True)
63
+ ]
64
+ for match in watermark_matches:
65
+ # Legacy rules have no allowed body occurrences and retain fuzzy
66
+ # leak detection. Occurrence-aware rules distinguish a removed
67
+ # overlay from legitimate copies of the same phrase in body text.
68
+ if rule.allowed_occurrences is None:
69
+ match["removed"] = match["similarity"] < rule.watermark_match_threshold
70
+ else:
71
+ match["removed"] = match["occurrences"] <= match["allowed_occurrences"]
72
+
73
+ preservation_matches: list[dict[str, Any]] = [
74
+ {
75
+ "text": text,
76
+ "similarity": _partial_similarity(text, actual_markdown),
77
+ }
78
+ for text in rule.preserve_texts
79
+ ]
80
+ for match in preservation_matches:
81
+ match["preserved"] = match["similarity"] >= rule.preserve_match_threshold
82
+
83
+ removal_score = sum(bool(match["removed"]) for match in watermark_matches) / len(watermark_matches)
84
+ preservation_score = sum(bool(match["preserved"]) for match in preservation_matches) / len(preservation_matches)
85
+ combined_score = min(removal_score, preservation_score)
86
+ passed = removal_score >= rule.removal_pass_threshold and preservation_score >= rule.preservation_pass_threshold
87
+
88
+ self.result_details = {
89
+ "removal_score": removal_score,
90
+ "preservation_score": preservation_score,
91
+ "watermark_match_threshold": rule.watermark_match_threshold,
92
+ "preserve_match_threshold": rule.preserve_match_threshold,
93
+ "watermark_matches": watermark_matches,
94
+ "preservation_matches": preservation_matches,
95
+ }
96
+ explanation = (
97
+ f"watermark_removal={removal_score:.4f} "
98
+ f"body_preservation={preservation_score:.4f} "
99
+ f"removed={sum(bool(match['removed']) for match in watermark_matches)}/{len(watermark_matches)} "
100
+ f"preserved={sum(bool(match['preserved']) for match in preservation_matches)}/{len(preservation_matches)}"
101
+ )
102
+ return passed, explanation, combined_score
103
+
104
+
105
+ __all__ = ["WatermarkRemovalRule"]