ref-verify 1.1.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ref_verify/__init__.py +5 -0
- ref_verify/abstract_lookup.py +226 -0
- ref_verify/claim_check.py +876 -0
- ref_verify/cli.py +162 -0
- ref_verify/crossref.py +89 -0
- ref_verify/doi_check.py +191 -0
- ref_verify/models.py +89 -0
- ref_verify/numeric_claim.py +425 -0
- ref_verify/pubmed.py +156 -0
- ref_verify/semantic_scholar.py +80 -0
- ref_verify-1.1.2.dist-info/METADATA +250 -0
- ref_verify-1.1.2.dist-info/RECORD +16 -0
- ref_verify-1.1.2.dist-info/WHEEL +5 -0
- ref_verify-1.1.2.dist-info/entry_points.txt +2 -0
- ref_verify-1.1.2.dist-info/licenses/LICENSE +21 -0
- ref_verify-1.1.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,876 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
|
|
5
|
+
from ref_verify.models import ClaimSupportResult, PaperRecord
|
|
6
|
+
from ref_verify.numeric_claim import check_numeric_claim_support
|
|
7
|
+
|
|
8
|
+
_PERCENTAGE_VALUE_PATTERN = r"\d+(?:,\d{3})*(?:\.\d+)?"
|
|
9
|
+
_PERCENTAGE_UNIT_PATTERN = r"(?:%|\bpercent\b|\bper\s+cent\b)"
|
|
10
|
+
_PERCENTAGE_PATTERN = re.compile(
|
|
11
|
+
rf"(?<![\d,])({_PERCENTAGE_VALUE_PATTERN})\s*{_PERCENTAGE_UNIT_PATTERN}",
|
|
12
|
+
re.IGNORECASE,
|
|
13
|
+
)
|
|
14
|
+
|
|
15
|
+
_STOPWORDS = {
|
|
16
|
+
"a",
|
|
17
|
+
"an",
|
|
18
|
+
"and",
|
|
19
|
+
"above",
|
|
20
|
+
"as",
|
|
21
|
+
"at",
|
|
22
|
+
"can",
|
|
23
|
+
"for",
|
|
24
|
+
"in",
|
|
25
|
+
"of",
|
|
26
|
+
"over",
|
|
27
|
+
"that",
|
|
28
|
+
"the",
|
|
29
|
+
"to",
|
|
30
|
+
"up",
|
|
31
|
+
"with",
|
|
32
|
+
}
|
|
33
|
+
|
|
34
|
+
_STRAIN_QUALIFIER_STEMS = {
|
|
35
|
+
"bend",
|
|
36
|
+
"bending",
|
|
37
|
+
"compressive",
|
|
38
|
+
"compression",
|
|
39
|
+
"elongation",
|
|
40
|
+
"shear",
|
|
41
|
+
"tensile",
|
|
42
|
+
"torsion",
|
|
43
|
+
"torsional",
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
_NON_OUTPUT_STRAIN_FOLLOWER_STEMS = {
|
|
47
|
+
"energy",
|
|
48
|
+
"localisation",
|
|
49
|
+
"localization",
|
|
50
|
+
"rate",
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
_TEXT_CLAIM_COMPARATIVE_SUFFIXES = {
|
|
54
|
+
"additional",
|
|
55
|
+
"decreased",
|
|
56
|
+
"extra",
|
|
57
|
+
"fewer",
|
|
58
|
+
"greater",
|
|
59
|
+
"higher",
|
|
60
|
+
"increased",
|
|
61
|
+
"less",
|
|
62
|
+
"longer",
|
|
63
|
+
"lower",
|
|
64
|
+
"max",
|
|
65
|
+
"maximum",
|
|
66
|
+
"min",
|
|
67
|
+
"minimum",
|
|
68
|
+
"more",
|
|
69
|
+
"shorter",
|
|
70
|
+
"than",
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
_TEXT_CLAIM_SCOPE_SUFFIXES = {
|
|
74
|
+
"across",
|
|
75
|
+
"after",
|
|
76
|
+
"among",
|
|
77
|
+
"at",
|
|
78
|
+
"average",
|
|
79
|
+
"averaged",
|
|
80
|
+
"averages",
|
|
81
|
+
"before",
|
|
82
|
+
"during",
|
|
83
|
+
"except",
|
|
84
|
+
"following",
|
|
85
|
+
"for",
|
|
86
|
+
"from",
|
|
87
|
+
"in",
|
|
88
|
+
"mean",
|
|
89
|
+
"median",
|
|
90
|
+
"only",
|
|
91
|
+
"then",
|
|
92
|
+
"typical",
|
|
93
|
+
"typically",
|
|
94
|
+
"under",
|
|
95
|
+
"unless",
|
|
96
|
+
"until",
|
|
97
|
+
"when",
|
|
98
|
+
"while",
|
|
99
|
+
"within",
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
_PERCENTAGE_APPROXIMATION_MODIFIERS = {
|
|
103
|
+
"about",
|
|
104
|
+
"approx",
|
|
105
|
+
"approximately",
|
|
106
|
+
"around",
|
|
107
|
+
"ca",
|
|
108
|
+
"circa",
|
|
109
|
+
"nearly",
|
|
110
|
+
"roughly",
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
_PERCENTAGE_APPROXIMATION_SYMBOLS = ("~", "∼", "≈")
|
|
114
|
+
|
|
115
|
+
_UNRELATED_PERCENTAGE_SUBJECT_STEMS = {
|
|
116
|
+
"breakdown",
|
|
117
|
+
"conductivity",
|
|
118
|
+
"cycle",
|
|
119
|
+
"efficiency",
|
|
120
|
+
"energy",
|
|
121
|
+
"field",
|
|
122
|
+
"force",
|
|
123
|
+
"frequency",
|
|
124
|
+
"lifetime",
|
|
125
|
+
"modulus",
|
|
126
|
+
"power",
|
|
127
|
+
"pressure",
|
|
128
|
+
"speed",
|
|
129
|
+
"stress",
|
|
130
|
+
"temperature",
|
|
131
|
+
"voltage",
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
_GENERIC_PERCENTAGE_QUALIFIER_STEMS = {
|
|
135
|
+
"actuator",
|
|
136
|
+
"device",
|
|
137
|
+
"elastomer",
|
|
138
|
+
"film",
|
|
139
|
+
"material",
|
|
140
|
+
"polymer",
|
|
141
|
+
"sample",
|
|
142
|
+
"specimen",
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
_GENERIC_PERCENTAGE_CLAIM_NON_SUBJECT_STEMS = {
|
|
146
|
+
"at",
|
|
147
|
+
"above",
|
|
148
|
+
"below",
|
|
149
|
+
"equal",
|
|
150
|
+
"exceed",
|
|
151
|
+
"exceeded",
|
|
152
|
+
"exceeding",
|
|
153
|
+
"exceeds",
|
|
154
|
+
"few",
|
|
155
|
+
"fewer",
|
|
156
|
+
"greater",
|
|
157
|
+
"high",
|
|
158
|
+
"higher",
|
|
159
|
+
"least",
|
|
160
|
+
"less",
|
|
161
|
+
"low",
|
|
162
|
+
"lower",
|
|
163
|
+
"more",
|
|
164
|
+
"most",
|
|
165
|
+
"no",
|
|
166
|
+
"not",
|
|
167
|
+
"percent",
|
|
168
|
+
"per",
|
|
169
|
+
"cent",
|
|
170
|
+
"than",
|
|
171
|
+
"under",
|
|
172
|
+
}
|
|
173
|
+
|
|
174
|
+
_UNSUPPORTED_CLAIM_FRAME_PATTERNS = (
|
|
175
|
+
r"\baccording to\b",
|
|
176
|
+
r"\bwhether\b",
|
|
177
|
+
r"\bif\b",
|
|
178
|
+
r"\bnot true that\b",
|
|
179
|
+
r"\bfalse that\b",
|
|
180
|
+
r"\bcannot\b",
|
|
181
|
+
r"\bcan not\b",
|
|
182
|
+
r"\bcould not\b",
|
|
183
|
+
r"\b(?:did|do|does|is|are|was|were|has|have|had) not\b",
|
|
184
|
+
r"\b(?:didn t|don t|doesn t|isn t|aren t|wasn t|weren t)\b",
|
|
185
|
+
r"\b(?:fail|fails|failed) to\b",
|
|
186
|
+
r"\bunable to\b",
|
|
187
|
+
r"\bwithout\b",
|
|
188
|
+
r"\bnever\b",
|
|
189
|
+
r"\bno\b.*\b(?:observed|found|shown|showed|reported|measured|demonstrated)\b",
|
|
190
|
+
r"\bno "
|
|
191
|
+
r"(?:sample|samples|specimen|specimens|device|devices|case|cases|paper|papers|study|studies)\b",
|
|
192
|
+
r"\bnone of (?:the )?"
|
|
193
|
+
r"(?:sample|samples|specimen|specimens|device|devices|case|cases|paper|papers|study|studies)\b",
|
|
194
|
+
r"\bnone "
|
|
195
|
+
r"(?:show|shows|showed|had|has|have|observed|found|reported|reached|exceeded|met|demonstrated)\b",
|
|
196
|
+
r"\b(?:previous|prior|earlier) (?:work|study|studies|research)\b",
|
|
197
|
+
r"\b(?:may|might|would|should)\b",
|
|
198
|
+
r"\b(?:appear|appears|appeared|appearing) to\b",
|
|
199
|
+
r"\b(?:seem|seems|seemed|seeming) to\b",
|
|
200
|
+
r"\b(?:the )?(?:paper|article|study|work) "
|
|
201
|
+
r"(?:report|reports|reported|reporting)\b",
|
|
202
|
+
r"\b(?:the )?authors? "
|
|
203
|
+
r"(?:report|reports|reported|found|finds|observed|observes|noted|notes)\b",
|
|
204
|
+
r"\breportedly\b",
|
|
205
|
+
r"\b(?:claim|claims|claimed|claiming)\b",
|
|
206
|
+
r"\b(?:suggest|suggests|suggested|suggesting)\b",
|
|
207
|
+
r"\b(?:indicate|indicates|indicated|indicating)\b",
|
|
208
|
+
r"\b(?:imply|implies|implied|implying)\b",
|
|
209
|
+
r"\bsaid to\b",
|
|
210
|
+
r"\b(?:expect|expects|expected|expecting) to\b",
|
|
211
|
+
r"\b(?:project|projects|projected|projecting) to\b",
|
|
212
|
+
r"\b(?:estimate|estimates|estimated|estimating) to\b",
|
|
213
|
+
r"\bachievable\b",
|
|
214
|
+
r"\bpossible\b",
|
|
215
|
+
)
|
|
216
|
+
|
|
217
|
+
|
|
218
|
+
def check_claim_support(record: PaperRecord, claim: str) -> ClaimSupportResult:
|
|
219
|
+
if not record.abstract:
|
|
220
|
+
return ClaimSupportResult(
|
|
221
|
+
status="UNVERIFIABLE",
|
|
222
|
+
verdict="WARN",
|
|
223
|
+
reason="No abstract was available from the fetched record.",
|
|
224
|
+
evidence="",
|
|
225
|
+
paper=record,
|
|
226
|
+
claim=claim,
|
|
227
|
+
)
|
|
228
|
+
|
|
229
|
+
threshold = _claim_percentage_threshold(claim)
|
|
230
|
+
comparator = _claim_percentage_comparator(claim)
|
|
231
|
+
evidence_sentences = _ranked_evidence_sentences(record.abstract, claim)
|
|
232
|
+
evidence_sentence = evidence_sentences[0] if evidence_sentences else record.abstract.strip()
|
|
233
|
+
|
|
234
|
+
if threshold is None or not _is_strain_percentage_claim(claim):
|
|
235
|
+
numeric_result = check_numeric_claim_support(record.abstract, claim)
|
|
236
|
+
if numeric_result.status == "SUPPORTED":
|
|
237
|
+
return ClaimSupportResult(
|
|
238
|
+
status="SUPPORTED",
|
|
239
|
+
verdict="ACCEPT",
|
|
240
|
+
reason=numeric_result.reason,
|
|
241
|
+
evidence=numeric_result.evidence,
|
|
242
|
+
paper=record,
|
|
243
|
+
claim=claim,
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
if threshold is not None:
|
|
247
|
+
for sentence_index, sentence in enumerate(evidence_sentences):
|
|
248
|
+
supported = _sentence_supports_percentage_claim(
|
|
249
|
+
sentence,
|
|
250
|
+
threshold,
|
|
251
|
+
comparator,
|
|
252
|
+
claim,
|
|
253
|
+
)
|
|
254
|
+
if supported:
|
|
255
|
+
if _has_cross_sentence_contradictory_percentage_context(
|
|
256
|
+
evidence_sentences,
|
|
257
|
+
sentence_index,
|
|
258
|
+
threshold,
|
|
259
|
+
comparator,
|
|
260
|
+
claim,
|
|
261
|
+
):
|
|
262
|
+
continue
|
|
263
|
+
return ClaimSupportResult(
|
|
264
|
+
status="SUPPORTED",
|
|
265
|
+
verdict="ACCEPT",
|
|
266
|
+
reason="Fetched abstract explicitly reports a matching quantitative claim.",
|
|
267
|
+
evidence=sentence,
|
|
268
|
+
paper=record,
|
|
269
|
+
claim=claim,
|
|
270
|
+
)
|
|
271
|
+
|
|
272
|
+
for sentence in evidence_sentences:
|
|
273
|
+
if _all_percentage_evidence_is_prestrain(sentence):
|
|
274
|
+
return ClaimSupportResult(
|
|
275
|
+
status="PARTIAL",
|
|
276
|
+
verdict="WARN",
|
|
277
|
+
reason=(
|
|
278
|
+
"The abstract percentage appears in a pre-strain context, "
|
|
279
|
+
"not an actuation output."
|
|
280
|
+
),
|
|
281
|
+
evidence=sentence,
|
|
282
|
+
paper=record,
|
|
283
|
+
claim=claim,
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
if threshold is None:
|
|
287
|
+
for sentence in evidence_sentences:
|
|
288
|
+
if _sentence_supports_text_claim(sentence, claim):
|
|
289
|
+
return ClaimSupportResult(
|
|
290
|
+
status="SUPPORTED",
|
|
291
|
+
verdict="ACCEPT",
|
|
292
|
+
reason="Fetched abstract explicitly states the claim.",
|
|
293
|
+
evidence=sentence,
|
|
294
|
+
paper=record,
|
|
295
|
+
claim=claim,
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
if _term_overlap(claim, record.abstract) > 0:
|
|
299
|
+
return ClaimSupportResult(
|
|
300
|
+
status="PARTIAL",
|
|
301
|
+
verdict="WARN",
|
|
302
|
+
reason="The abstract is related, but does not explicitly support the specific claim.",
|
|
303
|
+
evidence=evidence_sentence,
|
|
304
|
+
paper=record,
|
|
305
|
+
claim=claim,
|
|
306
|
+
)
|
|
307
|
+
|
|
308
|
+
return ClaimSupportResult(
|
|
309
|
+
status="PARTIAL",
|
|
310
|
+
verdict="WARN",
|
|
311
|
+
reason="The abstract does not explicitly support the specific claim.",
|
|
312
|
+
evidence=evidence_sentence,
|
|
313
|
+
paper=record,
|
|
314
|
+
claim=claim,
|
|
315
|
+
)
|
|
316
|
+
|
|
317
|
+
|
|
318
|
+
def _claim_percentage_threshold(claim: str) -> float | None:
|
|
319
|
+
match = _PERCENTAGE_PATTERN.search(claim)
|
|
320
|
+
return _parse_percentage_value(match.group(1)) if match else None
|
|
321
|
+
|
|
322
|
+
|
|
323
|
+
def _claim_percentage_comparator(claim: str) -> str:
|
|
324
|
+
normalized = claim.lower()
|
|
325
|
+
if ">=" in normalized or "≥" in normalized:
|
|
326
|
+
return "gte"
|
|
327
|
+
if "<=" in normalized or "≤" in normalized:
|
|
328
|
+
return "lte"
|
|
329
|
+
if ">" in normalized:
|
|
330
|
+
return "gt"
|
|
331
|
+
if "<" in normalized:
|
|
332
|
+
return "lt"
|
|
333
|
+
if re.search(r"\b(?:more|greater|higher) than or equal to\b", normalized):
|
|
334
|
+
return "gte"
|
|
335
|
+
if re.search(r"\b(?:less|lower|fewer) than or equal to\b", normalized):
|
|
336
|
+
return "lte"
|
|
337
|
+
if re.search(r"\b(at least|not less than)\b", normalized):
|
|
338
|
+
return "gte"
|
|
339
|
+
if re.search(r"\b(at most|no more than)\b", normalized):
|
|
340
|
+
return "lte"
|
|
341
|
+
if re.search(r"\b(below|under|less than)\b", normalized):
|
|
342
|
+
return "lt"
|
|
343
|
+
if re.search(
|
|
344
|
+
r"\b(above|over|greater than|more than|exceeded|exceeds|exceeding)\b",
|
|
345
|
+
normalized,
|
|
346
|
+
):
|
|
347
|
+
return "gt"
|
|
348
|
+
return "exact"
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
def _is_strain_percentage_claim(claim: str) -> bool:
|
|
352
|
+
claim_terms = {_stem(token) for token in _tokens(claim)}
|
|
353
|
+
return "strain" in claim_terms
|
|
354
|
+
|
|
355
|
+
|
|
356
|
+
def _ranked_evidence_sentences(abstract: str, claim: str) -> list[str]:
|
|
357
|
+
sentences = re.split(r"(?<=[.!?])\s+", abstract.strip())
|
|
358
|
+
if not sentences:
|
|
359
|
+
return [abstract.strip()]
|
|
360
|
+
stripped = [sentence.strip() for sentence in sentences if sentence.strip()]
|
|
361
|
+
return sorted(stripped, key=lambda sentence: _term_overlap(claim, sentence), reverse=True)
|
|
362
|
+
|
|
363
|
+
|
|
364
|
+
def _sentence_supports_percentage_claim(
|
|
365
|
+
sentence: str,
|
|
366
|
+
threshold: float,
|
|
367
|
+
comparator: str,
|
|
368
|
+
claim: str,
|
|
369
|
+
) -> bool:
|
|
370
|
+
if _has_unsupported_claim_frame(sentence):
|
|
371
|
+
return False
|
|
372
|
+
if _has_sentence_scope_prefix(sentence):
|
|
373
|
+
return False
|
|
374
|
+
contexts = _percentage_contexts(sentence)
|
|
375
|
+
for index, (value, context, percentage_start, percentage_end) in enumerate(contexts):
|
|
376
|
+
if _mentions_prestrain_context(context):
|
|
377
|
+
continue
|
|
378
|
+
if _has_unsupported_claim_frame(context):
|
|
379
|
+
continue
|
|
380
|
+
if _has_percentage_scope_prefix(sentence, percentage_start):
|
|
381
|
+
continue
|
|
382
|
+
if _has_percentage_scope_suffix(sentence, percentage_end):
|
|
383
|
+
continue
|
|
384
|
+
if _has_approximate_percentage_context(
|
|
385
|
+
sentence,
|
|
386
|
+
percentage_start,
|
|
387
|
+
percentage_end,
|
|
388
|
+
):
|
|
389
|
+
continue
|
|
390
|
+
if not _has_percentage_subject_context(context, sentence, claim):
|
|
391
|
+
continue
|
|
392
|
+
evidence_comparator = _evidence_percentage_comparator(
|
|
393
|
+
context,
|
|
394
|
+
sentence[percentage_end:],
|
|
395
|
+
)
|
|
396
|
+
if _evidence_entails_claim(value, evidence_comparator, threshold, comparator):
|
|
397
|
+
if _has_contradictory_percentage_context(
|
|
398
|
+
contexts,
|
|
399
|
+
index,
|
|
400
|
+
threshold,
|
|
401
|
+
comparator,
|
|
402
|
+
claim,
|
|
403
|
+
sentence,
|
|
404
|
+
):
|
|
405
|
+
continue
|
|
406
|
+
return True
|
|
407
|
+
return False
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
def _evidence_entails_claim(
|
|
411
|
+
value: float,
|
|
412
|
+
evidence_comparator: str,
|
|
413
|
+
threshold: float,
|
|
414
|
+
claim_comparator: str,
|
|
415
|
+
) -> bool:
|
|
416
|
+
if evidence_comparator == "up_to":
|
|
417
|
+
if claim_comparator == "exact":
|
|
418
|
+
return False
|
|
419
|
+
return _compare_percentage(value, threshold, claim_comparator)
|
|
420
|
+
if evidence_comparator == "lt":
|
|
421
|
+
return claim_comparator in {"lt", "lte"} and value <= threshold
|
|
422
|
+
if evidence_comparator == "lte":
|
|
423
|
+
if claim_comparator == "lt":
|
|
424
|
+
return value < threshold
|
|
425
|
+
return claim_comparator == "lte" and value <= threshold
|
|
426
|
+
if evidence_comparator == "gt":
|
|
427
|
+
return claim_comparator in {"gt", "gte"} and value >= threshold
|
|
428
|
+
if evidence_comparator == "gte":
|
|
429
|
+
if claim_comparator == "gt":
|
|
430
|
+
return value > threshold
|
|
431
|
+
return claim_comparator == "gte" and value >= threshold
|
|
432
|
+
return _compare_percentage(value, threshold, claim_comparator)
|
|
433
|
+
|
|
434
|
+
|
|
435
|
+
def _compare_percentage(value: float, threshold: float, comparator: str) -> bool:
|
|
436
|
+
if comparator == "lt":
|
|
437
|
+
return value < threshold
|
|
438
|
+
if comparator == "lte":
|
|
439
|
+
return value <= threshold
|
|
440
|
+
if comparator == "gte":
|
|
441
|
+
return value >= threshold
|
|
442
|
+
if comparator == "gt":
|
|
443
|
+
return value > threshold
|
|
444
|
+
return value == threshold
|
|
445
|
+
|
|
446
|
+
|
|
447
|
+
def _has_contradictory_percentage_context(
|
|
448
|
+
contexts: list[tuple[float, str, int, int]],
|
|
449
|
+
supporting_index: int,
|
|
450
|
+
threshold: float,
|
|
451
|
+
claim_comparator: str,
|
|
452
|
+
claim: str,
|
|
453
|
+
sentence: str,
|
|
454
|
+
) -> bool:
|
|
455
|
+
supporting_context = contexts[supporting_index][1]
|
|
456
|
+
for index, (value, context, _, percentage_end) in enumerate(contexts):
|
|
457
|
+
if index == supporting_index:
|
|
458
|
+
continue
|
|
459
|
+
if _mentions_prestrain_context(context):
|
|
460
|
+
continue
|
|
461
|
+
if _has_unsupported_claim_frame(context):
|
|
462
|
+
continue
|
|
463
|
+
if not _has_percentage_subject_context(context, sentence, claim):
|
|
464
|
+
continue
|
|
465
|
+
if _has_distinct_percentage_qualifiers(
|
|
466
|
+
supporting_context,
|
|
467
|
+
context,
|
|
468
|
+
):
|
|
469
|
+
continue
|
|
470
|
+
evidence_comparator = _evidence_percentage_comparator(
|
|
471
|
+
context,
|
|
472
|
+
sentence[percentage_end:],
|
|
473
|
+
)
|
|
474
|
+
if _evidence_contradicts_claim(
|
|
475
|
+
value,
|
|
476
|
+
evidence_comparator,
|
|
477
|
+
threshold,
|
|
478
|
+
claim_comparator,
|
|
479
|
+
):
|
|
480
|
+
return True
|
|
481
|
+
return False
|
|
482
|
+
|
|
483
|
+
|
|
484
|
+
def _has_distinct_percentage_qualifiers(left: str, right: str) -> bool:
|
|
485
|
+
left_terms = _percentage_qualifier_terms(left)
|
|
486
|
+
right_terms = _percentage_qualifier_terms(right)
|
|
487
|
+
return bool(left_terms and right_terms and left_terms.isdisjoint(right_terms))
|
|
488
|
+
|
|
489
|
+
|
|
490
|
+
def _percentage_qualifier_terms(context: str) -> set[str]:
|
|
491
|
+
percentage = _PERCENTAGE_PATTERN.search(context)
|
|
492
|
+
if not percentage:
|
|
493
|
+
return set()
|
|
494
|
+
return {
|
|
495
|
+
_stem(token)
|
|
496
|
+
for token in _tokens(context[percentage.end() :])
|
|
497
|
+
if _stem(token) not in _GENERIC_PERCENTAGE_QUALIFIER_STEMS
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
|
|
501
|
+
def _has_cross_sentence_contradictory_percentage_context(
|
|
502
|
+
sentences: list[str],
|
|
503
|
+
supporting_sentence_index: int,
|
|
504
|
+
threshold: float,
|
|
505
|
+
claim_comparator: str,
|
|
506
|
+
claim: str,
|
|
507
|
+
) -> bool:
|
|
508
|
+
for sentence_index, sentence in enumerate(sentences):
|
|
509
|
+
if sentence_index == supporting_sentence_index:
|
|
510
|
+
continue
|
|
511
|
+
if _has_unsupported_claim_frame(sentence):
|
|
512
|
+
continue
|
|
513
|
+
contexts = _percentage_contexts(sentence)
|
|
514
|
+
for value, context, percentage_start, percentage_end in contexts:
|
|
515
|
+
if _mentions_prestrain_context(context):
|
|
516
|
+
continue
|
|
517
|
+
if _has_unsupported_claim_frame(context):
|
|
518
|
+
continue
|
|
519
|
+
if _has_percentage_scope_prefix(sentence, percentage_start):
|
|
520
|
+
continue
|
|
521
|
+
if _has_percentage_scope_suffix(sentence, percentage_end):
|
|
522
|
+
continue
|
|
523
|
+
if _has_approximate_percentage_context(
|
|
524
|
+
sentence,
|
|
525
|
+
percentage_start,
|
|
526
|
+
percentage_end,
|
|
527
|
+
):
|
|
528
|
+
continue
|
|
529
|
+
if not _has_percentage_subject_context(context, sentence, claim):
|
|
530
|
+
continue
|
|
531
|
+
evidence_comparator = _evidence_percentage_comparator(
|
|
532
|
+
context,
|
|
533
|
+
sentence[percentage_end:],
|
|
534
|
+
)
|
|
535
|
+
if _evidence_contradicts_claim(
|
|
536
|
+
value,
|
|
537
|
+
evidence_comparator,
|
|
538
|
+
threshold,
|
|
539
|
+
claim_comparator,
|
|
540
|
+
):
|
|
541
|
+
return True
|
|
542
|
+
return False
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def _has_percentage_subject_context(context: str, sentence: str, claim: str) -> bool:
|
|
546
|
+
if _has_actuation_strain_context(context, claim):
|
|
547
|
+
return True
|
|
548
|
+
if _inherits_actuation_strain_subject(context, sentence, claim):
|
|
549
|
+
return True
|
|
550
|
+
return _has_generic_percentage_subject_context(context, claim)
|
|
551
|
+
|
|
552
|
+
|
|
553
|
+
def _has_generic_percentage_subject_context(context: str, claim: str) -> bool:
|
|
554
|
+
claim_terms = _generic_percentage_subject_terms(claim)
|
|
555
|
+
if not claim_terms:
|
|
556
|
+
return False
|
|
557
|
+
context_terms = {_stem(token) for token in _tokens(context)}
|
|
558
|
+
return claim_terms <= context_terms
|
|
559
|
+
|
|
560
|
+
|
|
561
|
+
def _generic_percentage_subject_terms(claim: str) -> set[str]:
|
|
562
|
+
terms = {_stem(token) for token in _tokens(claim)}
|
|
563
|
+
if "actuat" in terms or "strain" in terms:
|
|
564
|
+
return set()
|
|
565
|
+
return terms - _GENERIC_PERCENTAGE_CLAIM_NON_SUBJECT_STEMS
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
def _inherits_actuation_strain_subject(context: str, sentence: str, claim: str) -> bool:
|
|
569
|
+
claim_terms = {_stem(token) for token in _tokens(claim)}
|
|
570
|
+
if "actuat" not in claim_terms:
|
|
571
|
+
return False
|
|
572
|
+
if _has_non_output_strain_compound(context):
|
|
573
|
+
return False
|
|
574
|
+
|
|
575
|
+
sentence_terms = {_stem(token) for token in _tokens(sentence)}
|
|
576
|
+
if not {"actuat", "strain"} <= sentence_terms:
|
|
577
|
+
return False
|
|
578
|
+
|
|
579
|
+
context_terms = {_stem(token) for token in _tokens(context)}
|
|
580
|
+
if context_terms & _STRAIN_QUALIFIER_STEMS:
|
|
581
|
+
return False
|
|
582
|
+
return not bool(context_terms & _UNRELATED_PERCENTAGE_SUBJECT_STEMS)
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def _evidence_contradicts_claim(
|
|
586
|
+
value: float,
|
|
587
|
+
evidence_comparator: str,
|
|
588
|
+
threshold: float,
|
|
589
|
+
claim_comparator: str,
|
|
590
|
+
) -> bool:
|
|
591
|
+
if claim_comparator == "lt":
|
|
592
|
+
return (
|
|
593
|
+
(evidence_comparator == "exact" and value >= threshold)
|
|
594
|
+
or (evidence_comparator == "gt" and value >= threshold)
|
|
595
|
+
or (evidence_comparator == "gte" and value >= threshold)
|
|
596
|
+
or (evidence_comparator == "up_to" and value >= threshold)
|
|
597
|
+
)
|
|
598
|
+
if claim_comparator == "lte":
|
|
599
|
+
return (
|
|
600
|
+
(evidence_comparator == "exact" and value > threshold)
|
|
601
|
+
or (evidence_comparator == "gt" and value >= threshold)
|
|
602
|
+
or (evidence_comparator == "gte" and value > threshold)
|
|
603
|
+
or (evidence_comparator == "up_to" and value > threshold)
|
|
604
|
+
)
|
|
605
|
+
if claim_comparator == "gt":
|
|
606
|
+
return (
|
|
607
|
+
(evidence_comparator == "exact" and value <= threshold)
|
|
608
|
+
or (evidence_comparator in {"lt", "lte", "up_to"} and value <= threshold)
|
|
609
|
+
)
|
|
610
|
+
if claim_comparator == "gte":
|
|
611
|
+
return (
|
|
612
|
+
(evidence_comparator == "exact" and value < threshold)
|
|
613
|
+
or (evidence_comparator == "lt" and value <= threshold)
|
|
614
|
+
or (evidence_comparator in {"lte", "up_to"} and value < threshold)
|
|
615
|
+
)
|
|
616
|
+
return (
|
|
617
|
+
(evidence_comparator == "exact" and value != threshold)
|
|
618
|
+
or (evidence_comparator == "lt" and value <= threshold)
|
|
619
|
+
or (evidence_comparator in {"lte", "up_to"} and value < threshold)
|
|
620
|
+
or (evidence_comparator == "gt" and value >= threshold)
|
|
621
|
+
or (evidence_comparator == "gte" and value > threshold)
|
|
622
|
+
)
|
|
623
|
+
|
|
624
|
+
|
|
625
|
+
def _sentence_supports_text_claim(sentence: str, claim: str) -> bool:
|
|
626
|
+
claim_numbers = set(_numbers(claim))
|
|
627
|
+
if claim_numbers and not claim_numbers <= set(_numbers(sentence)):
|
|
628
|
+
return False
|
|
629
|
+
|
|
630
|
+
claim_tokens = _phrase_tokens(claim)
|
|
631
|
+
if not claim_tokens:
|
|
632
|
+
return False
|
|
633
|
+
sentence_tokens = _phrase_tokens(sentence)
|
|
634
|
+
|
|
635
|
+
for start in _token_sequence_offsets(sentence_tokens, claim_tokens):
|
|
636
|
+
end = start + len(claim_tokens)
|
|
637
|
+
if _has_unsupported_claim_frame(sentence):
|
|
638
|
+
continue
|
|
639
|
+
if _has_scope_qualifier_prefix(sentence_tokens, start):
|
|
640
|
+
continue
|
|
641
|
+
if _has_comparative_suffix(sentence_tokens, end):
|
|
642
|
+
continue
|
|
643
|
+
return True
|
|
644
|
+
return False
|
|
645
|
+
|
|
646
|
+
|
|
647
|
+
def _all_percentage_evidence_is_prestrain(value: str) -> bool:
|
|
648
|
+
contexts = [context for _, context, _, _ in _percentage_contexts(value)]
|
|
649
|
+
return bool(contexts) and all(_mentions_prestrain_context(context) for context in contexts)
|
|
650
|
+
|
|
651
|
+
|
|
652
|
+
def _percentage_contexts(value: str) -> list[tuple[float, str, int, int]]:
|
|
653
|
+
contexts: list[tuple[float, str, int, int]] = []
|
|
654
|
+
for match in _PERCENTAGE_PATTERN.finditer(value):
|
|
655
|
+
start, end = _clause_bounds(value, match.start(), match.end())
|
|
656
|
+
contexts.append(
|
|
657
|
+
(
|
|
658
|
+
_parse_percentage_value(match.group(1)),
|
|
659
|
+
value[start:end],
|
|
660
|
+
match.start(),
|
|
661
|
+
match.end(),
|
|
662
|
+
)
|
|
663
|
+
)
|
|
664
|
+
return contexts
|
|
665
|
+
|
|
666
|
+
|
|
667
|
+
def _evidence_percentage_comparator(context: str, trailing_text: str = "") -> str:
|
|
668
|
+
percentage = _PERCENTAGE_PATTERN.search(context)
|
|
669
|
+
if not percentage:
|
|
670
|
+
return "exact"
|
|
671
|
+
|
|
672
|
+
prefix = context[: percentage.start()].lower()
|
|
673
|
+
suffix = f"{context[percentage.end() :]} {trailing_text}".lower()
|
|
674
|
+
stripped_prefix = prefix.rstrip()
|
|
675
|
+
stripped_suffix = suffix.lstrip()
|
|
676
|
+
if stripped_prefix.endswith((">=", "≥")):
|
|
677
|
+
return "gte"
|
|
678
|
+
if stripped_prefix.endswith(("<=", "≤")):
|
|
679
|
+
return "lte"
|
|
680
|
+
if stripped_prefix.endswith(">"):
|
|
681
|
+
return "gt"
|
|
682
|
+
if stripped_prefix.endswith("<"):
|
|
683
|
+
return "lt"
|
|
684
|
+
if re.search(
|
|
685
|
+
r"\b(?:no (?:more|greater|higher) than|"
|
|
686
|
+
r"(?:less|lower|fewer) than or equal to)\s*$",
|
|
687
|
+
prefix,
|
|
688
|
+
):
|
|
689
|
+
return "lte"
|
|
690
|
+
if re.search(
|
|
691
|
+
r"\b(?:no (?:less|lower|fewer) than|"
|
|
692
|
+
r"(?:more|greater|higher) than or equal to)\s*$",
|
|
693
|
+
prefix,
|
|
694
|
+
):
|
|
695
|
+
return "gte"
|
|
696
|
+
if re.search(r"\b(at least|not less than)\s*$", prefix):
|
|
697
|
+
return "gte"
|
|
698
|
+
if re.search(r"\b(at most|no more than)\s*$", prefix):
|
|
699
|
+
return "lte"
|
|
700
|
+
if stripped_suffix.startswith("+"):
|
|
701
|
+
return "gte"
|
|
702
|
+
if re.search(
|
|
703
|
+
r"^\s*[,;:]?\s*(?:or\s+)?(?:more|greater|higher|min|minimum)\b",
|
|
704
|
+
suffix,
|
|
705
|
+
):
|
|
706
|
+
return "gte"
|
|
707
|
+
if re.search(
|
|
708
|
+
r"^\s*[,;:]?\s*(?:or\s+)?(?:less|fewer|lower|max|maximum)\b",
|
|
709
|
+
suffix,
|
|
710
|
+
):
|
|
711
|
+
return "lte"
|
|
712
|
+
if re.search(r"\bup to\s*$", prefix):
|
|
713
|
+
return "up_to"
|
|
714
|
+
if re.search(r"\b(below|under|less than)\s*$", prefix):
|
|
715
|
+
return "lt"
|
|
716
|
+
if re.search(
|
|
717
|
+
r"\b(above|over|greater than|more than|exceeded|exceeds|exceeding)\s*$",
|
|
718
|
+
prefix,
|
|
719
|
+
):
|
|
720
|
+
return "gt"
|
|
721
|
+
return "exact"
|
|
722
|
+
|
|
723
|
+
|
|
724
|
+
def _has_sentence_scope_prefix(value: str) -> bool:
|
|
725
|
+
tokens = _phrase_tokens(value)
|
|
726
|
+
return _has_scope_qualifier_tokens(tokens[:2])
|
|
727
|
+
|
|
728
|
+
|
|
729
|
+
def _has_percentage_scope_suffix(sentence: str, percentage_end: int) -> bool:
|
|
730
|
+
suffix_tokens = _phrase_tokens(sentence[percentage_end:])
|
|
731
|
+
return _has_scope_qualifier_tokens(suffix_tokens)
|
|
732
|
+
|
|
733
|
+
|
|
734
|
+
def _has_percentage_scope_prefix(sentence: str, percentage_start: int) -> bool:
|
|
735
|
+
prefix_tokens = _phrase_tokens(sentence[:percentage_start])
|
|
736
|
+
return _has_scope_qualifier_tokens(prefix_tokens)
|
|
737
|
+
|
|
738
|
+
|
|
739
|
+
def _has_approximate_percentage_context(
|
|
740
|
+
sentence: str,
|
|
741
|
+
percentage_start: int,
|
|
742
|
+
percentage_end: int,
|
|
743
|
+
) -> bool:
|
|
744
|
+
prefix = sentence[:percentage_start]
|
|
745
|
+
if prefix.rstrip().endswith(_PERCENTAGE_APPROXIMATION_SYMBOLS):
|
|
746
|
+
return True
|
|
747
|
+
prefix_tokens = _phrase_tokens(prefix)
|
|
748
|
+
suffix_tokens = _phrase_tokens(sentence[percentage_end:])
|
|
749
|
+
nearby_tokens = prefix_tokens[-3:] + suffix_tokens[:3]
|
|
750
|
+
return any(token in _PERCENTAGE_APPROXIMATION_MODIFIERS for token in nearby_tokens)
|
|
751
|
+
|
|
752
|
+
|
|
753
|
+
def _parse_percentage_value(value: str) -> float:
|
|
754
|
+
return float(value.replace(",", ""))
|
|
755
|
+
|
|
756
|
+
|
|
757
|
+
def _clause_bounds(value: str, start: int, end: int) -> tuple[int, int]:
|
|
758
|
+
boundary = r"(?:[.;:]\s+|,\s+|,?\s+\b(?:and|but|while|whereas|although)\b\s+)"
|
|
759
|
+
context_start = 0
|
|
760
|
+
for match in re.finditer(boundary, value[:start]):
|
|
761
|
+
context_start = match.end()
|
|
762
|
+
|
|
763
|
+
next_boundary = re.search(boundary, value[end:])
|
|
764
|
+
context_end = end + next_boundary.start() if next_boundary else len(value)
|
|
765
|
+
return context_start, context_end
|
|
766
|
+
|
|
767
|
+
|
|
768
|
+
def _mentions_prestrain_context(value: str) -> bool:
|
|
769
|
+
normalized = value.lower()
|
|
770
|
+
return "pre-strain" in normalized or "prestrain" in normalized
|
|
771
|
+
|
|
772
|
+
|
|
773
|
+
def _has_actuation_strain_context(sentence: str, claim: str) -> bool:
|
|
774
|
+
terms = {_stem(token) for token in _tokens(sentence)}
|
|
775
|
+
claim_terms = {_stem(token) for token in _tokens(claim)}
|
|
776
|
+
if "strain" not in terms:
|
|
777
|
+
return False
|
|
778
|
+
if "actuat" in claim_terms:
|
|
779
|
+
if "actuat" not in terms:
|
|
780
|
+
return False
|
|
781
|
+
if _has_non_output_strain_compound(sentence):
|
|
782
|
+
return False
|
|
783
|
+
return not bool(terms & _STRAIN_QUALIFIER_STEMS)
|
|
784
|
+
claim_qualifiers = claim_terms & _STRAIN_QUALIFIER_STEMS
|
|
785
|
+
if claim_qualifiers and not claim_qualifiers <= terms:
|
|
786
|
+
return False
|
|
787
|
+
return True
|
|
788
|
+
|
|
789
|
+
|
|
790
|
+
def _has_non_output_strain_compound(value: str) -> bool:
|
|
791
|
+
tokens = [_stem(token) for token in _tokens(value)]
|
|
792
|
+
return any(
|
|
793
|
+
token == "strain"
|
|
794
|
+
and index + 1 < len(tokens)
|
|
795
|
+
and tokens[index + 1] in _NON_OUTPUT_STRAIN_FOLLOWER_STEMS
|
|
796
|
+
for index, token in enumerate(tokens)
|
|
797
|
+
)
|
|
798
|
+
|
|
799
|
+
|
|
800
|
+
def _term_overlap(left: str, right: str) -> int:
|
|
801
|
+
left_terms = {_stem(token) for token in _tokens(left)}
|
|
802
|
+
right_terms = {_stem(token) for token in _tokens(right)}
|
|
803
|
+
return len(left_terms & right_terms)
|
|
804
|
+
|
|
805
|
+
|
|
806
|
+
def _tokens(value: str) -> list[str]:
|
|
807
|
+
return [
|
|
808
|
+
token
|
|
809
|
+
for token in re.findall(r"[a-zA-Z]+", value.lower())
|
|
810
|
+
if token not in _STOPWORDS
|
|
811
|
+
]
|
|
812
|
+
|
|
813
|
+
|
|
814
|
+
def _numbers(value: str) -> list[str]:
|
|
815
|
+
return [
|
|
816
|
+
number.replace(",", "")
|
|
817
|
+
for number in re.findall(r"\d+(?:,\d{3})*(?:\.\d+)?", value)
|
|
818
|
+
]
|
|
819
|
+
|
|
820
|
+
|
|
821
|
+
def _phrase_tokens(value: str) -> list[str]:
|
|
822
|
+
return re.findall(r"[a-zA-Z0-9]+", value.lower())
|
|
823
|
+
|
|
824
|
+
|
|
825
|
+
def _token_sequence_offsets(tokens: list[str], target: list[str]) -> list[int]:
|
|
826
|
+
width = len(target)
|
|
827
|
+
return [
|
|
828
|
+
index
|
|
829
|
+
for index in range(0, len(tokens) - width + 1)
|
|
830
|
+
if tokens[index : index + width] == target
|
|
831
|
+
]
|
|
832
|
+
|
|
833
|
+
|
|
834
|
+
def _has_unsupported_claim_frame(value: str) -> bool:
|
|
835
|
+
normalized = " ".join(_phrase_tokens(value))
|
|
836
|
+
return any(
|
|
837
|
+
re.search(pattern, normalized)
|
|
838
|
+
for pattern in _UNSUPPORTED_CLAIM_FRAME_PATTERNS
|
|
839
|
+
)
|
|
840
|
+
|
|
841
|
+
|
|
842
|
+
def _has_comparative_suffix(tokens: list[str], claim_end: int) -> bool:
|
|
843
|
+
if claim_end >= len(tokens):
|
|
844
|
+
return False
|
|
845
|
+
|
|
846
|
+
suffix = tokens[claim_end:]
|
|
847
|
+
if any(token in _TEXT_CLAIM_COMPARATIVE_SUFFIXES for token in suffix):
|
|
848
|
+
return True
|
|
849
|
+
if _has_scope_qualifier_tokens(suffix):
|
|
850
|
+
return True
|
|
851
|
+
return False
|
|
852
|
+
|
|
853
|
+
|
|
854
|
+
def _has_scope_qualifier_prefix(tokens: list[str], claim_start: int) -> bool:
|
|
855
|
+
prefix = tokens[:claim_start]
|
|
856
|
+
return _has_scope_qualifier_tokens(prefix)
|
|
857
|
+
|
|
858
|
+
|
|
859
|
+
def _has_scope_qualifier_tokens(tokens: list[str]) -> bool:
|
|
860
|
+
for index, token in enumerate(tokens):
|
|
861
|
+
next_token = tokens[index + 1] if index + 1 < len(tokens) else ""
|
|
862
|
+
if token == "at" and next_token in {"least", "most"}:
|
|
863
|
+
continue
|
|
864
|
+
if token in _TEXT_CLAIM_SCOPE_SUFFIXES:
|
|
865
|
+
return True
|
|
866
|
+
return False
|
|
867
|
+
|
|
868
|
+
|
|
869
|
+
def _stem(token: str) -> str:
|
|
870
|
+
if token.startswith("actuat"):
|
|
871
|
+
return "actuat"
|
|
872
|
+
if token.startswith("strain"):
|
|
873
|
+
return "strain"
|
|
874
|
+
if token.endswith("s") and len(token) > 3:
|
|
875
|
+
return token[:-1]
|
|
876
|
+
return token
|