ref-verify 1.1.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,876 @@
1
+ from __future__ import annotations
2
+
3
+ import re
4
+
5
+ from ref_verify.models import ClaimSupportResult, PaperRecord
6
+ from ref_verify.numeric_claim import check_numeric_claim_support
7
+
8
+ _PERCENTAGE_VALUE_PATTERN = r"\d+(?:,\d{3})*(?:\.\d+)?"
9
+ _PERCENTAGE_UNIT_PATTERN = r"(?:%|\bpercent\b|\bper\s+cent\b)"
10
+ _PERCENTAGE_PATTERN = re.compile(
11
+ rf"(?<![\d,])({_PERCENTAGE_VALUE_PATTERN})\s*{_PERCENTAGE_UNIT_PATTERN}",
12
+ re.IGNORECASE,
13
+ )
14
+
15
+ _STOPWORDS = {
16
+ "a",
17
+ "an",
18
+ "and",
19
+ "above",
20
+ "as",
21
+ "at",
22
+ "can",
23
+ "for",
24
+ "in",
25
+ "of",
26
+ "over",
27
+ "that",
28
+ "the",
29
+ "to",
30
+ "up",
31
+ "with",
32
+ }
33
+
34
+ _STRAIN_QUALIFIER_STEMS = {
35
+ "bend",
36
+ "bending",
37
+ "compressive",
38
+ "compression",
39
+ "elongation",
40
+ "shear",
41
+ "tensile",
42
+ "torsion",
43
+ "torsional",
44
+ }
45
+
46
+ _NON_OUTPUT_STRAIN_FOLLOWER_STEMS = {
47
+ "energy",
48
+ "localisation",
49
+ "localization",
50
+ "rate",
51
+ }
52
+
53
+ _TEXT_CLAIM_COMPARATIVE_SUFFIXES = {
54
+ "additional",
55
+ "decreased",
56
+ "extra",
57
+ "fewer",
58
+ "greater",
59
+ "higher",
60
+ "increased",
61
+ "less",
62
+ "longer",
63
+ "lower",
64
+ "max",
65
+ "maximum",
66
+ "min",
67
+ "minimum",
68
+ "more",
69
+ "shorter",
70
+ "than",
71
+ }
72
+
73
+ _TEXT_CLAIM_SCOPE_SUFFIXES = {
74
+ "across",
75
+ "after",
76
+ "among",
77
+ "at",
78
+ "average",
79
+ "averaged",
80
+ "averages",
81
+ "before",
82
+ "during",
83
+ "except",
84
+ "following",
85
+ "for",
86
+ "from",
87
+ "in",
88
+ "mean",
89
+ "median",
90
+ "only",
91
+ "then",
92
+ "typical",
93
+ "typically",
94
+ "under",
95
+ "unless",
96
+ "until",
97
+ "when",
98
+ "while",
99
+ "within",
100
+ }
101
+
102
+ _PERCENTAGE_APPROXIMATION_MODIFIERS = {
103
+ "about",
104
+ "approx",
105
+ "approximately",
106
+ "around",
107
+ "ca",
108
+ "circa",
109
+ "nearly",
110
+ "roughly",
111
+ }
112
+
113
+ _PERCENTAGE_APPROXIMATION_SYMBOLS = ("~", "∼", "≈")
114
+
115
+ _UNRELATED_PERCENTAGE_SUBJECT_STEMS = {
116
+ "breakdown",
117
+ "conductivity",
118
+ "cycle",
119
+ "efficiency",
120
+ "energy",
121
+ "field",
122
+ "force",
123
+ "frequency",
124
+ "lifetime",
125
+ "modulus",
126
+ "power",
127
+ "pressure",
128
+ "speed",
129
+ "stress",
130
+ "temperature",
131
+ "voltage",
132
+ }
133
+
134
+ _GENERIC_PERCENTAGE_QUALIFIER_STEMS = {
135
+ "actuator",
136
+ "device",
137
+ "elastomer",
138
+ "film",
139
+ "material",
140
+ "polymer",
141
+ "sample",
142
+ "specimen",
143
+ }
144
+
145
+ _GENERIC_PERCENTAGE_CLAIM_NON_SUBJECT_STEMS = {
146
+ "at",
147
+ "above",
148
+ "below",
149
+ "equal",
150
+ "exceed",
151
+ "exceeded",
152
+ "exceeding",
153
+ "exceeds",
154
+ "few",
155
+ "fewer",
156
+ "greater",
157
+ "high",
158
+ "higher",
159
+ "least",
160
+ "less",
161
+ "low",
162
+ "lower",
163
+ "more",
164
+ "most",
165
+ "no",
166
+ "not",
167
+ "percent",
168
+ "per",
169
+ "cent",
170
+ "than",
171
+ "under",
172
+ }
173
+
174
+ _UNSUPPORTED_CLAIM_FRAME_PATTERNS = (
175
+ r"\baccording to\b",
176
+ r"\bwhether\b",
177
+ r"\bif\b",
178
+ r"\bnot true that\b",
179
+ r"\bfalse that\b",
180
+ r"\bcannot\b",
181
+ r"\bcan not\b",
182
+ r"\bcould not\b",
183
+ r"\b(?:did|do|does|is|are|was|were|has|have|had) not\b",
184
+ r"\b(?:didn t|don t|doesn t|isn t|aren t|wasn t|weren t)\b",
185
+ r"\b(?:fail|fails|failed) to\b",
186
+ r"\bunable to\b",
187
+ r"\bwithout\b",
188
+ r"\bnever\b",
189
+ r"\bno\b.*\b(?:observed|found|shown|showed|reported|measured|demonstrated)\b",
190
+ r"\bno "
191
+ r"(?:sample|samples|specimen|specimens|device|devices|case|cases|paper|papers|study|studies)\b",
192
+ r"\bnone of (?:the )?"
193
+ r"(?:sample|samples|specimen|specimens|device|devices|case|cases|paper|papers|study|studies)\b",
194
+ r"\bnone "
195
+ r"(?:show|shows|showed|had|has|have|observed|found|reported|reached|exceeded|met|demonstrated)\b",
196
+ r"\b(?:previous|prior|earlier) (?:work|study|studies|research)\b",
197
+ r"\b(?:may|might|would|should)\b",
198
+ r"\b(?:appear|appears|appeared|appearing) to\b",
199
+ r"\b(?:seem|seems|seemed|seeming) to\b",
200
+ r"\b(?:the )?(?:paper|article|study|work) "
201
+ r"(?:report|reports|reported|reporting)\b",
202
+ r"\b(?:the )?authors? "
203
+ r"(?:report|reports|reported|found|finds|observed|observes|noted|notes)\b",
204
+ r"\breportedly\b",
205
+ r"\b(?:claim|claims|claimed|claiming)\b",
206
+ r"\b(?:suggest|suggests|suggested|suggesting)\b",
207
+ r"\b(?:indicate|indicates|indicated|indicating)\b",
208
+ r"\b(?:imply|implies|implied|implying)\b",
209
+ r"\bsaid to\b",
210
+ r"\b(?:expect|expects|expected|expecting) to\b",
211
+ r"\b(?:project|projects|projected|projecting) to\b",
212
+ r"\b(?:estimate|estimates|estimated|estimating) to\b",
213
+ r"\bachievable\b",
214
+ r"\bpossible\b",
215
+ )
216
+
217
+
218
+ def check_claim_support(record: PaperRecord, claim: str) -> ClaimSupportResult:
219
+ if not record.abstract:
220
+ return ClaimSupportResult(
221
+ status="UNVERIFIABLE",
222
+ verdict="WARN",
223
+ reason="No abstract was available from the fetched record.",
224
+ evidence="",
225
+ paper=record,
226
+ claim=claim,
227
+ )
228
+
229
+ threshold = _claim_percentage_threshold(claim)
230
+ comparator = _claim_percentage_comparator(claim)
231
+ evidence_sentences = _ranked_evidence_sentences(record.abstract, claim)
232
+ evidence_sentence = evidence_sentences[0] if evidence_sentences else record.abstract.strip()
233
+
234
+ if threshold is None or not _is_strain_percentage_claim(claim):
235
+ numeric_result = check_numeric_claim_support(record.abstract, claim)
236
+ if numeric_result.status == "SUPPORTED":
237
+ return ClaimSupportResult(
238
+ status="SUPPORTED",
239
+ verdict="ACCEPT",
240
+ reason=numeric_result.reason,
241
+ evidence=numeric_result.evidence,
242
+ paper=record,
243
+ claim=claim,
244
+ )
245
+
246
+ if threshold is not None:
247
+ for sentence_index, sentence in enumerate(evidence_sentences):
248
+ supported = _sentence_supports_percentage_claim(
249
+ sentence,
250
+ threshold,
251
+ comparator,
252
+ claim,
253
+ )
254
+ if supported:
255
+ if _has_cross_sentence_contradictory_percentage_context(
256
+ evidence_sentences,
257
+ sentence_index,
258
+ threshold,
259
+ comparator,
260
+ claim,
261
+ ):
262
+ continue
263
+ return ClaimSupportResult(
264
+ status="SUPPORTED",
265
+ verdict="ACCEPT",
266
+ reason="Fetched abstract explicitly reports a matching quantitative claim.",
267
+ evidence=sentence,
268
+ paper=record,
269
+ claim=claim,
270
+ )
271
+
272
+ for sentence in evidence_sentences:
273
+ if _all_percentage_evidence_is_prestrain(sentence):
274
+ return ClaimSupportResult(
275
+ status="PARTIAL",
276
+ verdict="WARN",
277
+ reason=(
278
+ "The abstract percentage appears in a pre-strain context, "
279
+ "not an actuation output."
280
+ ),
281
+ evidence=sentence,
282
+ paper=record,
283
+ claim=claim,
284
+ )
285
+
286
+ if threshold is None:
287
+ for sentence in evidence_sentences:
288
+ if _sentence_supports_text_claim(sentence, claim):
289
+ return ClaimSupportResult(
290
+ status="SUPPORTED",
291
+ verdict="ACCEPT",
292
+ reason="Fetched abstract explicitly states the claim.",
293
+ evidence=sentence,
294
+ paper=record,
295
+ claim=claim,
296
+ )
297
+
298
+ if _term_overlap(claim, record.abstract) > 0:
299
+ return ClaimSupportResult(
300
+ status="PARTIAL",
301
+ verdict="WARN",
302
+ reason="The abstract is related, but does not explicitly support the specific claim.",
303
+ evidence=evidence_sentence,
304
+ paper=record,
305
+ claim=claim,
306
+ )
307
+
308
+ return ClaimSupportResult(
309
+ status="PARTIAL",
310
+ verdict="WARN",
311
+ reason="The abstract does not explicitly support the specific claim.",
312
+ evidence=evidence_sentence,
313
+ paper=record,
314
+ claim=claim,
315
+ )
316
+
317
+
318
+ def _claim_percentage_threshold(claim: str) -> float | None:
319
+ match = _PERCENTAGE_PATTERN.search(claim)
320
+ return _parse_percentage_value(match.group(1)) if match else None
321
+
322
+
323
+ def _claim_percentage_comparator(claim: str) -> str:
324
+ normalized = claim.lower()
325
+ if ">=" in normalized or "≥" in normalized:
326
+ return "gte"
327
+ if "<=" in normalized or "≤" in normalized:
328
+ return "lte"
329
+ if ">" in normalized:
330
+ return "gt"
331
+ if "<" in normalized:
332
+ return "lt"
333
+ if re.search(r"\b(?:more|greater|higher) than or equal to\b", normalized):
334
+ return "gte"
335
+ if re.search(r"\b(?:less|lower|fewer) than or equal to\b", normalized):
336
+ return "lte"
337
+ if re.search(r"\b(at least|not less than)\b", normalized):
338
+ return "gte"
339
+ if re.search(r"\b(at most|no more than)\b", normalized):
340
+ return "lte"
341
+ if re.search(r"\b(below|under|less than)\b", normalized):
342
+ return "lt"
343
+ if re.search(
344
+ r"\b(above|over|greater than|more than|exceeded|exceeds|exceeding)\b",
345
+ normalized,
346
+ ):
347
+ return "gt"
348
+ return "exact"
349
+
350
+
351
+ def _is_strain_percentage_claim(claim: str) -> bool:
352
+ claim_terms = {_stem(token) for token in _tokens(claim)}
353
+ return "strain" in claim_terms
354
+
355
+
356
+ def _ranked_evidence_sentences(abstract: str, claim: str) -> list[str]:
357
+ sentences = re.split(r"(?<=[.!?])\s+", abstract.strip())
358
+ if not sentences:
359
+ return [abstract.strip()]
360
+ stripped = [sentence.strip() for sentence in sentences if sentence.strip()]
361
+ return sorted(stripped, key=lambda sentence: _term_overlap(claim, sentence), reverse=True)
362
+
363
+
364
+ def _sentence_supports_percentage_claim(
365
+ sentence: str,
366
+ threshold: float,
367
+ comparator: str,
368
+ claim: str,
369
+ ) -> bool:
370
+ if _has_unsupported_claim_frame(sentence):
371
+ return False
372
+ if _has_sentence_scope_prefix(sentence):
373
+ return False
374
+ contexts = _percentage_contexts(sentence)
375
+ for index, (value, context, percentage_start, percentage_end) in enumerate(contexts):
376
+ if _mentions_prestrain_context(context):
377
+ continue
378
+ if _has_unsupported_claim_frame(context):
379
+ continue
380
+ if _has_percentage_scope_prefix(sentence, percentage_start):
381
+ continue
382
+ if _has_percentage_scope_suffix(sentence, percentage_end):
383
+ continue
384
+ if _has_approximate_percentage_context(
385
+ sentence,
386
+ percentage_start,
387
+ percentage_end,
388
+ ):
389
+ continue
390
+ if not _has_percentage_subject_context(context, sentence, claim):
391
+ continue
392
+ evidence_comparator = _evidence_percentage_comparator(
393
+ context,
394
+ sentence[percentage_end:],
395
+ )
396
+ if _evidence_entails_claim(value, evidence_comparator, threshold, comparator):
397
+ if _has_contradictory_percentage_context(
398
+ contexts,
399
+ index,
400
+ threshold,
401
+ comparator,
402
+ claim,
403
+ sentence,
404
+ ):
405
+ continue
406
+ return True
407
+ return False
408
+
409
+
410
+ def _evidence_entails_claim(
411
+ value: float,
412
+ evidence_comparator: str,
413
+ threshold: float,
414
+ claim_comparator: str,
415
+ ) -> bool:
416
+ if evidence_comparator == "up_to":
417
+ if claim_comparator == "exact":
418
+ return False
419
+ return _compare_percentage(value, threshold, claim_comparator)
420
+ if evidence_comparator == "lt":
421
+ return claim_comparator in {"lt", "lte"} and value <= threshold
422
+ if evidence_comparator == "lte":
423
+ if claim_comparator == "lt":
424
+ return value < threshold
425
+ return claim_comparator == "lte" and value <= threshold
426
+ if evidence_comparator == "gt":
427
+ return claim_comparator in {"gt", "gte"} and value >= threshold
428
+ if evidence_comparator == "gte":
429
+ if claim_comparator == "gt":
430
+ return value > threshold
431
+ return claim_comparator == "gte" and value >= threshold
432
+ return _compare_percentage(value, threshold, claim_comparator)
433
+
434
+
435
+ def _compare_percentage(value: float, threshold: float, comparator: str) -> bool:
436
+ if comparator == "lt":
437
+ return value < threshold
438
+ if comparator == "lte":
439
+ return value <= threshold
440
+ if comparator == "gte":
441
+ return value >= threshold
442
+ if comparator == "gt":
443
+ return value > threshold
444
+ return value == threshold
445
+
446
+
447
+ def _has_contradictory_percentage_context(
448
+ contexts: list[tuple[float, str, int, int]],
449
+ supporting_index: int,
450
+ threshold: float,
451
+ claim_comparator: str,
452
+ claim: str,
453
+ sentence: str,
454
+ ) -> bool:
455
+ supporting_context = contexts[supporting_index][1]
456
+ for index, (value, context, _, percentage_end) in enumerate(contexts):
457
+ if index == supporting_index:
458
+ continue
459
+ if _mentions_prestrain_context(context):
460
+ continue
461
+ if _has_unsupported_claim_frame(context):
462
+ continue
463
+ if not _has_percentage_subject_context(context, sentence, claim):
464
+ continue
465
+ if _has_distinct_percentage_qualifiers(
466
+ supporting_context,
467
+ context,
468
+ ):
469
+ continue
470
+ evidence_comparator = _evidence_percentage_comparator(
471
+ context,
472
+ sentence[percentage_end:],
473
+ )
474
+ if _evidence_contradicts_claim(
475
+ value,
476
+ evidence_comparator,
477
+ threshold,
478
+ claim_comparator,
479
+ ):
480
+ return True
481
+ return False
482
+
483
+
484
+ def _has_distinct_percentage_qualifiers(left: str, right: str) -> bool:
485
+ left_terms = _percentage_qualifier_terms(left)
486
+ right_terms = _percentage_qualifier_terms(right)
487
+ return bool(left_terms and right_terms and left_terms.isdisjoint(right_terms))
488
+
489
+
490
+ def _percentage_qualifier_terms(context: str) -> set[str]:
491
+ percentage = _PERCENTAGE_PATTERN.search(context)
492
+ if not percentage:
493
+ return set()
494
+ return {
495
+ _stem(token)
496
+ for token in _tokens(context[percentage.end() :])
497
+ if _stem(token) not in _GENERIC_PERCENTAGE_QUALIFIER_STEMS
498
+ }
499
+
500
+
501
+ def _has_cross_sentence_contradictory_percentage_context(
502
+ sentences: list[str],
503
+ supporting_sentence_index: int,
504
+ threshold: float,
505
+ claim_comparator: str,
506
+ claim: str,
507
+ ) -> bool:
508
+ for sentence_index, sentence in enumerate(sentences):
509
+ if sentence_index == supporting_sentence_index:
510
+ continue
511
+ if _has_unsupported_claim_frame(sentence):
512
+ continue
513
+ contexts = _percentage_contexts(sentence)
514
+ for value, context, percentage_start, percentage_end in contexts:
515
+ if _mentions_prestrain_context(context):
516
+ continue
517
+ if _has_unsupported_claim_frame(context):
518
+ continue
519
+ if _has_percentage_scope_prefix(sentence, percentage_start):
520
+ continue
521
+ if _has_percentage_scope_suffix(sentence, percentage_end):
522
+ continue
523
+ if _has_approximate_percentage_context(
524
+ sentence,
525
+ percentage_start,
526
+ percentage_end,
527
+ ):
528
+ continue
529
+ if not _has_percentage_subject_context(context, sentence, claim):
530
+ continue
531
+ evidence_comparator = _evidence_percentage_comparator(
532
+ context,
533
+ sentence[percentage_end:],
534
+ )
535
+ if _evidence_contradicts_claim(
536
+ value,
537
+ evidence_comparator,
538
+ threshold,
539
+ claim_comparator,
540
+ ):
541
+ return True
542
+ return False
543
+
544
+
545
+ def _has_percentage_subject_context(context: str, sentence: str, claim: str) -> bool:
546
+ if _has_actuation_strain_context(context, claim):
547
+ return True
548
+ if _inherits_actuation_strain_subject(context, sentence, claim):
549
+ return True
550
+ return _has_generic_percentage_subject_context(context, claim)
551
+
552
+
553
+ def _has_generic_percentage_subject_context(context: str, claim: str) -> bool:
554
+ claim_terms = _generic_percentage_subject_terms(claim)
555
+ if not claim_terms:
556
+ return False
557
+ context_terms = {_stem(token) for token in _tokens(context)}
558
+ return claim_terms <= context_terms
559
+
560
+
561
+ def _generic_percentage_subject_terms(claim: str) -> set[str]:
562
+ terms = {_stem(token) for token in _tokens(claim)}
563
+ if "actuat" in terms or "strain" in terms:
564
+ return set()
565
+ return terms - _GENERIC_PERCENTAGE_CLAIM_NON_SUBJECT_STEMS
566
+
567
+
568
+ def _inherits_actuation_strain_subject(context: str, sentence: str, claim: str) -> bool:
569
+ claim_terms = {_stem(token) for token in _tokens(claim)}
570
+ if "actuat" not in claim_terms:
571
+ return False
572
+ if _has_non_output_strain_compound(context):
573
+ return False
574
+
575
+ sentence_terms = {_stem(token) for token in _tokens(sentence)}
576
+ if not {"actuat", "strain"} <= sentence_terms:
577
+ return False
578
+
579
+ context_terms = {_stem(token) for token in _tokens(context)}
580
+ if context_terms & _STRAIN_QUALIFIER_STEMS:
581
+ return False
582
+ return not bool(context_terms & _UNRELATED_PERCENTAGE_SUBJECT_STEMS)
583
+
584
+
585
+ def _evidence_contradicts_claim(
586
+ value: float,
587
+ evidence_comparator: str,
588
+ threshold: float,
589
+ claim_comparator: str,
590
+ ) -> bool:
591
+ if claim_comparator == "lt":
592
+ return (
593
+ (evidence_comparator == "exact" and value >= threshold)
594
+ or (evidence_comparator == "gt" and value >= threshold)
595
+ or (evidence_comparator == "gte" and value >= threshold)
596
+ or (evidence_comparator == "up_to" and value >= threshold)
597
+ )
598
+ if claim_comparator == "lte":
599
+ return (
600
+ (evidence_comparator == "exact" and value > threshold)
601
+ or (evidence_comparator == "gt" and value >= threshold)
602
+ or (evidence_comparator == "gte" and value > threshold)
603
+ or (evidence_comparator == "up_to" and value > threshold)
604
+ )
605
+ if claim_comparator == "gt":
606
+ return (
607
+ (evidence_comparator == "exact" and value <= threshold)
608
+ or (evidence_comparator in {"lt", "lte", "up_to"} and value <= threshold)
609
+ )
610
+ if claim_comparator == "gte":
611
+ return (
612
+ (evidence_comparator == "exact" and value < threshold)
613
+ or (evidence_comparator == "lt" and value <= threshold)
614
+ or (evidence_comparator in {"lte", "up_to"} and value < threshold)
615
+ )
616
+ return (
617
+ (evidence_comparator == "exact" and value != threshold)
618
+ or (evidence_comparator == "lt" and value <= threshold)
619
+ or (evidence_comparator in {"lte", "up_to"} and value < threshold)
620
+ or (evidence_comparator == "gt" and value >= threshold)
621
+ or (evidence_comparator == "gte" and value > threshold)
622
+ )
623
+
624
+
625
+ def _sentence_supports_text_claim(sentence: str, claim: str) -> bool:
626
+ claim_numbers = set(_numbers(claim))
627
+ if claim_numbers and not claim_numbers <= set(_numbers(sentence)):
628
+ return False
629
+
630
+ claim_tokens = _phrase_tokens(claim)
631
+ if not claim_tokens:
632
+ return False
633
+ sentence_tokens = _phrase_tokens(sentence)
634
+
635
+ for start in _token_sequence_offsets(sentence_tokens, claim_tokens):
636
+ end = start + len(claim_tokens)
637
+ if _has_unsupported_claim_frame(sentence):
638
+ continue
639
+ if _has_scope_qualifier_prefix(sentence_tokens, start):
640
+ continue
641
+ if _has_comparative_suffix(sentence_tokens, end):
642
+ continue
643
+ return True
644
+ return False
645
+
646
+
647
+ def _all_percentage_evidence_is_prestrain(value: str) -> bool:
648
+ contexts = [context for _, context, _, _ in _percentage_contexts(value)]
649
+ return bool(contexts) and all(_mentions_prestrain_context(context) for context in contexts)
650
+
651
+
652
+ def _percentage_contexts(value: str) -> list[tuple[float, str, int, int]]:
653
+ contexts: list[tuple[float, str, int, int]] = []
654
+ for match in _PERCENTAGE_PATTERN.finditer(value):
655
+ start, end = _clause_bounds(value, match.start(), match.end())
656
+ contexts.append(
657
+ (
658
+ _parse_percentage_value(match.group(1)),
659
+ value[start:end],
660
+ match.start(),
661
+ match.end(),
662
+ )
663
+ )
664
+ return contexts
665
+
666
+
667
+ def _evidence_percentage_comparator(context: str, trailing_text: str = "") -> str:
668
+ percentage = _PERCENTAGE_PATTERN.search(context)
669
+ if not percentage:
670
+ return "exact"
671
+
672
+ prefix = context[: percentage.start()].lower()
673
+ suffix = f"{context[percentage.end() :]} {trailing_text}".lower()
674
+ stripped_prefix = prefix.rstrip()
675
+ stripped_suffix = suffix.lstrip()
676
+ if stripped_prefix.endswith((">=", "≥")):
677
+ return "gte"
678
+ if stripped_prefix.endswith(("<=", "≤")):
679
+ return "lte"
680
+ if stripped_prefix.endswith(">"):
681
+ return "gt"
682
+ if stripped_prefix.endswith("<"):
683
+ return "lt"
684
+ if re.search(
685
+ r"\b(?:no (?:more|greater|higher) than|"
686
+ r"(?:less|lower|fewer) than or equal to)\s*$",
687
+ prefix,
688
+ ):
689
+ return "lte"
690
+ if re.search(
691
+ r"\b(?:no (?:less|lower|fewer) than|"
692
+ r"(?:more|greater|higher) than or equal to)\s*$",
693
+ prefix,
694
+ ):
695
+ return "gte"
696
+ if re.search(r"\b(at least|not less than)\s*$", prefix):
697
+ return "gte"
698
+ if re.search(r"\b(at most|no more than)\s*$", prefix):
699
+ return "lte"
700
+ if stripped_suffix.startswith("+"):
701
+ return "gte"
702
+ if re.search(
703
+ r"^\s*[,;:]?\s*(?:or\s+)?(?:more|greater|higher|min|minimum)\b",
704
+ suffix,
705
+ ):
706
+ return "gte"
707
+ if re.search(
708
+ r"^\s*[,;:]?\s*(?:or\s+)?(?:less|fewer|lower|max|maximum)\b",
709
+ suffix,
710
+ ):
711
+ return "lte"
712
+ if re.search(r"\bup to\s*$", prefix):
713
+ return "up_to"
714
+ if re.search(r"\b(below|under|less than)\s*$", prefix):
715
+ return "lt"
716
+ if re.search(
717
+ r"\b(above|over|greater than|more than|exceeded|exceeds|exceeding)\s*$",
718
+ prefix,
719
+ ):
720
+ return "gt"
721
+ return "exact"
722
+
723
+
724
+ def _has_sentence_scope_prefix(value: str) -> bool:
725
+ tokens = _phrase_tokens(value)
726
+ return _has_scope_qualifier_tokens(tokens[:2])
727
+
728
+
729
+ def _has_percentage_scope_suffix(sentence: str, percentage_end: int) -> bool:
730
+ suffix_tokens = _phrase_tokens(sentence[percentage_end:])
731
+ return _has_scope_qualifier_tokens(suffix_tokens)
732
+
733
+
734
+ def _has_percentage_scope_prefix(sentence: str, percentage_start: int) -> bool:
735
+ prefix_tokens = _phrase_tokens(sentence[:percentage_start])
736
+ return _has_scope_qualifier_tokens(prefix_tokens)
737
+
738
+
739
+ def _has_approximate_percentage_context(
740
+ sentence: str,
741
+ percentage_start: int,
742
+ percentage_end: int,
743
+ ) -> bool:
744
+ prefix = sentence[:percentage_start]
745
+ if prefix.rstrip().endswith(_PERCENTAGE_APPROXIMATION_SYMBOLS):
746
+ return True
747
+ prefix_tokens = _phrase_tokens(prefix)
748
+ suffix_tokens = _phrase_tokens(sentence[percentage_end:])
749
+ nearby_tokens = prefix_tokens[-3:] + suffix_tokens[:3]
750
+ return any(token in _PERCENTAGE_APPROXIMATION_MODIFIERS for token in nearby_tokens)
751
+
752
+
753
+ def _parse_percentage_value(value: str) -> float:
754
+ return float(value.replace(",", ""))
755
+
756
+
757
+ def _clause_bounds(value: str, start: int, end: int) -> tuple[int, int]:
758
+ boundary = r"(?:[.;:]\s+|,\s+|,?\s+\b(?:and|but|while|whereas|although)\b\s+)"
759
+ context_start = 0
760
+ for match in re.finditer(boundary, value[:start]):
761
+ context_start = match.end()
762
+
763
+ next_boundary = re.search(boundary, value[end:])
764
+ context_end = end + next_boundary.start() if next_boundary else len(value)
765
+ return context_start, context_end
766
+
767
+
768
+ def _mentions_prestrain_context(value: str) -> bool:
769
+ normalized = value.lower()
770
+ return "pre-strain" in normalized or "prestrain" in normalized
771
+
772
+
773
+ def _has_actuation_strain_context(sentence: str, claim: str) -> bool:
774
+ terms = {_stem(token) for token in _tokens(sentence)}
775
+ claim_terms = {_stem(token) for token in _tokens(claim)}
776
+ if "strain" not in terms:
777
+ return False
778
+ if "actuat" in claim_terms:
779
+ if "actuat" not in terms:
780
+ return False
781
+ if _has_non_output_strain_compound(sentence):
782
+ return False
783
+ return not bool(terms & _STRAIN_QUALIFIER_STEMS)
784
+ claim_qualifiers = claim_terms & _STRAIN_QUALIFIER_STEMS
785
+ if claim_qualifiers and not claim_qualifiers <= terms:
786
+ return False
787
+ return True
788
+
789
+
790
+ def _has_non_output_strain_compound(value: str) -> bool:
791
+ tokens = [_stem(token) for token in _tokens(value)]
792
+ return any(
793
+ token == "strain"
794
+ and index + 1 < len(tokens)
795
+ and tokens[index + 1] in _NON_OUTPUT_STRAIN_FOLLOWER_STEMS
796
+ for index, token in enumerate(tokens)
797
+ )
798
+
799
+
800
+ def _term_overlap(left: str, right: str) -> int:
801
+ left_terms = {_stem(token) for token in _tokens(left)}
802
+ right_terms = {_stem(token) for token in _tokens(right)}
803
+ return len(left_terms & right_terms)
804
+
805
+
806
+ def _tokens(value: str) -> list[str]:
807
+ return [
808
+ token
809
+ for token in re.findall(r"[a-zA-Z]+", value.lower())
810
+ if token not in _STOPWORDS
811
+ ]
812
+
813
+
814
+ def _numbers(value: str) -> list[str]:
815
+ return [
816
+ number.replace(",", "")
817
+ for number in re.findall(r"\d+(?:,\d{3})*(?:\.\d+)?", value)
818
+ ]
819
+
820
+
821
+ def _phrase_tokens(value: str) -> list[str]:
822
+ return re.findall(r"[a-zA-Z0-9]+", value.lower())
823
+
824
+
825
+ def _token_sequence_offsets(tokens: list[str], target: list[str]) -> list[int]:
826
+ width = len(target)
827
+ return [
828
+ index
829
+ for index in range(0, len(tokens) - width + 1)
830
+ if tokens[index : index + width] == target
831
+ ]
832
+
833
+
834
+ def _has_unsupported_claim_frame(value: str) -> bool:
835
+ normalized = " ".join(_phrase_tokens(value))
836
+ return any(
837
+ re.search(pattern, normalized)
838
+ for pattern in _UNSUPPORTED_CLAIM_FRAME_PATTERNS
839
+ )
840
+
841
+
842
+ def _has_comparative_suffix(tokens: list[str], claim_end: int) -> bool:
843
+ if claim_end >= len(tokens):
844
+ return False
845
+
846
+ suffix = tokens[claim_end:]
847
+ if any(token in _TEXT_CLAIM_COMPARATIVE_SUFFIXES for token in suffix):
848
+ return True
849
+ if _has_scope_qualifier_tokens(suffix):
850
+ return True
851
+ return False
852
+
853
+
854
+ def _has_scope_qualifier_prefix(tokens: list[str], claim_start: int) -> bool:
855
+ prefix = tokens[:claim_start]
856
+ return _has_scope_qualifier_tokens(prefix)
857
+
858
+
859
+ def _has_scope_qualifier_tokens(tokens: list[str]) -> bool:
860
+ for index, token in enumerate(tokens):
861
+ next_token = tokens[index + 1] if index + 1 < len(tokens) else ""
862
+ if token == "at" and next_token in {"least", "most"}:
863
+ continue
864
+ if token in _TEXT_CLAIM_SCOPE_SUFFIXES:
865
+ return True
866
+ return False
867
+
868
+
869
+ def _stem(token: str) -> str:
870
+ if token.startswith("actuat"):
871
+ return "actuat"
872
+ if token.startswith("strain"):
873
+ return "strain"
874
+ if token.endswith("s") and len(token) > 3:
875
+ return token[:-1]
876
+ return token