outis 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- outis/__init__.py +4 -0
- outis/analysis.py +334 -0
- outis/api.py +764 -0
- outis/cli.py +45 -0
- outis/data/__init__.py +1 -0
- outis/data/annotation_policy.py +121 -0
- outis/data/annotation_policy_ages.py +44 -0
- outis/data/annotation_policy_care_sites.py +108 -0
- outis/data/annotation_policy_eponyms.py +30 -0
- outis/data/annotation_policy_facility_names.py +25 -0
- outis/data/annotation_policy_fhir.py +47 -0
- outis/data/annotation_policy_places.py +60 -0
- outis/data/annotation_policy_reports.py +39 -0
- outis/data/annotation_policy_spoken.py +65 -0
- outis/data/annotation_policy_targeted.py +71 -0
- outis/data/annotation_policy_transcripts.py +33 -0
- outis/data/challenge.py +270 -0
- outis/data/corpus_quality.py +429 -0
- outis/data/generators/__init__.py +1 -0
- outis/data/generators/alias_contrastive.py +765 -0
- outis/data/generators/augment.py +521 -0
- outis/data/generators/base.py +547 -0
- outis/data/generators/compositional.py +570 -0
- outis/data/generators/context.py +267 -0
- outis/data/generators/coverage.py +398 -0
- outis/data/generators/generalization.py +1450 -0
- outis/data/generators/name_contrastive.py +902 -0
- outis/data/generators/narrative.py +520 -0
- outis/data/generators/nominal_alias.py +1224 -0
- outis/data/generators/place_sources.py +185 -0
- outis/data/generators/places.py +1160 -0
- outis/data/generators/places_contexts.py +622 -0
- outis/data/generators/remediation.py +1204 -0
- outis/data/generators/remediation_development.py +613 -0
- outis/data/v12/__init__.py +1 -0
- outis/data/v12/contexts.py +262 -0
- outis/data/v12/development.py +663 -0
- outis/data/v12/synthetic.py +555 -0
- outis/data/v13/__init__.py +1 -0
- outis/data/v13/contexts.py +735 -0
- outis/data/v13/replay.py +291 -0
- outis/data/v13/synthetic.py +284 -0
- outis/detection/__init__.py +1 -0
- outis/detection/deberta_attention.py +158 -0
- outis/detection/decoding.py +338 -0
- outis/detection/hf_encoder.py +217 -0
- outis/entities.py +68 -0
- outis/evaluation/__init__.py +1 -0
- outis/evaluation/evaluate.py +619 -0
- outis/evaluation/report.py +606 -0
- outis/evaluation/report_html.py +732 -0
- outis/evaluation/report_progress.py +150 -0
- outis/evaluation/runtime.py +73 -0
- outis/formats/__init__.py +1 -0
- outis/formats/files.py +861 -0
- outis/formats/structured.py +395 -0
- outis/operations.py +87 -0
- outis/paths.py +40 -0
- outis/policy.py +373 -0
- outis/replacement/__init__.py +1 -0
- outis/replacement/names/__init__.py +5 -0
- outis/replacement/names/age.py +63 -0
- outis/replacement/names/aliases.py +100 -0
- outis/replacement/names/identity.py +103 -0
- outis/replacement/names/nickname.py +521 -0
- outis/replacement/names/person.py +270 -0
- outis/replacement/names/record.py +105 -0
- outis/replacement/nicknames.py +154 -0
- outis/replacement/surrogates.py +856 -0
- outis/replacement/transform.py +596 -0
- outis/service.py +444 -0
- outis/types.py +40 -0
- outis-0.1.0.dist-info/METADATA +276 -0
- outis-0.1.0.dist-info/RECORD +79 -0
- outis-0.1.0.dist-info/WHEEL +5 -0
- outis-0.1.0.dist-info/entry_points.txt +2 -0
- outis-0.1.0.dist-info/licenses/LICENSE +201 -0
- outis-0.1.0.dist-info/licenses/NOTICE +25 -0
- outis-0.1.0.dist-info/top_level.txt +1 -0
outis/__init__.py
ADDED
outis/analysis.py
ADDED
|
@@ -0,0 +1,334 @@
|
|
|
1
|
+
"""Conservative, inspectable analysis of already detected English entities.
|
|
2
|
+
|
|
3
|
+
This module never expands detector coverage or infers clinical facts. Parsing
|
|
4
|
+
requires the complete detected value to match a supported grammar. Normalized
|
|
5
|
+
dates, ages and address components are sensitive values and require opt-in.
|
|
6
|
+
Relationships require an explicit supported predicate between adjacent spans;
|
|
7
|
+
they are experimental rule matches, not clinically validated assertions.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import re
|
|
13
|
+
from datetime import date
|
|
14
|
+
|
|
15
|
+
from outis.entities import checked_entities
|
|
16
|
+
|
|
17
|
+
ANALYSIS_VERSION = "1"
|
|
18
|
+
SUPPORTED_LOCALES = ("en-US", "en-GB")
|
|
19
|
+
_MONTHS = {
|
|
20
|
+
name: number
|
|
21
|
+
for number, names in enumerate(
|
|
22
|
+
(
|
|
23
|
+
("january", "jan"),
|
|
24
|
+
("february", "feb"),
|
|
25
|
+
("march", "mar"),
|
|
26
|
+
("april", "apr"),
|
|
27
|
+
("may",),
|
|
28
|
+
("june", "jun"),
|
|
29
|
+
("july", "jul"),
|
|
30
|
+
("august", "aug"),
|
|
31
|
+
("september", "sept", "sep"),
|
|
32
|
+
("october", "oct"),
|
|
33
|
+
("november", "nov"),
|
|
34
|
+
("december", "dec"),
|
|
35
|
+
),
|
|
36
|
+
1,
|
|
37
|
+
)
|
|
38
|
+
for name in names
|
|
39
|
+
}
|
|
40
|
+
_MONTH = "(?:" + "|".join(sorted(_MONTHS, key=len, reverse=True)) + r")\.?"
|
|
41
|
+
_STATES = frozenset(
|
|
42
|
+
"AL AK AZ AR CA CO CT DE DC FL GA HI ID IL IN IA KS KY LA ME MD MA MI MN MS MO MT NE NV NH NJ NM NY NC ND OH OK OR PA RI SC SD TN TX UT VT VA WA WV WI WY AS GU MP PR VI AA AE AP".split()
|
|
43
|
+
)
|
|
44
|
+
_ADDRESS = re.compile(
|
|
45
|
+
r"(?P<street_number>[0-9]{1,8}[A-Za-z]?(?:-[0-9]{1,8})?) "
|
|
46
|
+
r"(?P<street>[^,\r\n]{2,100}),\s*"
|
|
47
|
+
r"(?P<city>[A-Za-z][A-Za-z .'-]{0,63}),\s*"
|
|
48
|
+
r"(?P<state>[A-Z]{2})\s+(?P<postal_code>[0-9]{5}(?:-[0-9]{4})?)"
|
|
49
|
+
r"(?:,?\s+(?P<country>USA|US|United States))?",
|
|
50
|
+
re.I,
|
|
51
|
+
)
|
|
52
|
+
_UNIT = re.compile(
|
|
53
|
+
r"(?P<street>.+?)\s+(?P<unit>(?:apt\.?|suite|ste\.?|unit|#)\s*[A-Za-z0-9-]+)", re.I
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def _date_analysis(value, locale, include_sensitive):
|
|
58
|
+
value = value.strip()
|
|
59
|
+
result = {"kind": "date", "status": "unsupported"}
|
|
60
|
+
year = month = day = None
|
|
61
|
+
precision = "day"
|
|
62
|
+
match = re.fullmatch(r"([0-9]{4})([-/])([0-9]{1,2})\2([0-9]{1,2})", value)
|
|
63
|
+
if match:
|
|
64
|
+
year, month, day = int(match[1]), int(match[3]), int(match[4])
|
|
65
|
+
else:
|
|
66
|
+
match = re.fullmatch(r"([0-9]{1,2})([-/])([0-9]{1,2})\2([0-9]{4})", value)
|
|
67
|
+
if match:
|
|
68
|
+
first, second, year = int(match[1]), int(match[3]), int(match[4])
|
|
69
|
+
month, day = (first, second) if locale == "en-US" else (second, first)
|
|
70
|
+
result["interpretation"] = (
|
|
71
|
+
"month_first" if locale == "en-US" else "day_first"
|
|
72
|
+
)
|
|
73
|
+
result["locale_dependent"] = (
|
|
74
|
+
first <= 12 and second <= 12 and first != second
|
|
75
|
+
)
|
|
76
|
+
else:
|
|
77
|
+
match = re.fullmatch(
|
|
78
|
+
rf"({_MONTH})\s+([0-9]{{1,2}})(?:st|nd|rd|th)?(?:,?\s+)([0-9]{{4}})",
|
|
79
|
+
value,
|
|
80
|
+
re.I,
|
|
81
|
+
)
|
|
82
|
+
if match:
|
|
83
|
+
month, day, year = (
|
|
84
|
+
_MONTHS[match[1].rstrip(".").lower()],
|
|
85
|
+
int(match[2]),
|
|
86
|
+
int(match[3]),
|
|
87
|
+
)
|
|
88
|
+
else:
|
|
89
|
+
match = re.fullmatch(
|
|
90
|
+
rf"([0-9]{{1,2}})(?:st|nd|rd|th)?\s+({_MONTH})(?:,?\s+)([0-9]{{4}})",
|
|
91
|
+
value,
|
|
92
|
+
re.I,
|
|
93
|
+
)
|
|
94
|
+
if match:
|
|
95
|
+
day, month, year = (
|
|
96
|
+
int(match[1]),
|
|
97
|
+
_MONTHS[match[2].rstrip(".").lower()],
|
|
98
|
+
int(match[3]),
|
|
99
|
+
)
|
|
100
|
+
else:
|
|
101
|
+
match = re.fullmatch(rf"({_MONTH})\s+([0-9]{{4}})", value, re.I)
|
|
102
|
+
if match:
|
|
103
|
+
month, year = (
|
|
104
|
+
_MONTHS[match[1].rstrip(".").lower()],
|
|
105
|
+
int(match[2]),
|
|
106
|
+
)
|
|
107
|
+
precision = "month"
|
|
108
|
+
elif re.fullmatch(r"[0-9]{4}", value):
|
|
109
|
+
year, precision = int(value), "year"
|
|
110
|
+
elif re.fullmatch(
|
|
111
|
+
r"[0-9]{1,2}[-/][0-9]{1,2}(?:[-/][0-9]{2})?", value
|
|
112
|
+
):
|
|
113
|
+
return {
|
|
114
|
+
"kind": "date",
|
|
115
|
+
"status": "ambiguous",
|
|
116
|
+
"reason": "missing_four_digit_year",
|
|
117
|
+
}
|
|
118
|
+
else:
|
|
119
|
+
return result
|
|
120
|
+
ordinal = re.search(r"(?<![0-9])([0-9]{1,2})(st|nd|rd|th)\b", value, re.I)
|
|
121
|
+
if ordinal:
|
|
122
|
+
number = int(ordinal[1])
|
|
123
|
+
suffix = (
|
|
124
|
+
"th"
|
|
125
|
+
if number % 100 in {11, 12, 13}
|
|
126
|
+
else {1: "st", 2: "nd", 3: "rd"}.get(number % 10, "th")
|
|
127
|
+
)
|
|
128
|
+
if ordinal[2].lower() != suffix:
|
|
129
|
+
result["status"] = "invalid"
|
|
130
|
+
return result
|
|
131
|
+
try:
|
|
132
|
+
parsed = date(year, month or 1, day or 1)
|
|
133
|
+
# Zero is invalid, not an omitted calendar component.
|
|
134
|
+
if month == 0 or day == 0:
|
|
135
|
+
raise ValueError
|
|
136
|
+
except (ValueError, TypeError):
|
|
137
|
+
result["status"] = "invalid"
|
|
138
|
+
return result
|
|
139
|
+
result.update(status="parsed", precision=precision)
|
|
140
|
+
if include_sensitive:
|
|
141
|
+
result["normalized"] = parsed.isoformat()[
|
|
142
|
+
: {"day": 10, "month": 7, "year": 4}[precision]
|
|
143
|
+
]
|
|
144
|
+
return result
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def _age_analysis(value, include_sensitive):
|
|
148
|
+
value = value.strip().lower()
|
|
149
|
+
result = {"kind": "age", "status": "unsupported"}
|
|
150
|
+
older = re.fullmatch(r"([0-9]{1,3})\+\s*(?:years?(?: old)?)?", value)
|
|
151
|
+
match = re.fullmatch(
|
|
152
|
+
r"(?:aged?\s+)?([0-9]{1,4})(?:\s*[- ]?\s*"
|
|
153
|
+
r"(years?|yrs?|y/?o|months?|mos?|weeks?|wks?|days?)"
|
|
154
|
+
r"(?:\s*[- ]?\s*old)?)?",
|
|
155
|
+
value,
|
|
156
|
+
)
|
|
157
|
+
if older:
|
|
158
|
+
amount, unit = int(older[1]), "years"
|
|
159
|
+
normalized = {"minimum": amount, "unit": unit, "inclusive": True}
|
|
160
|
+
elif match:
|
|
161
|
+
amount = int(match[1])
|
|
162
|
+
raw_unit = match[2] or "years"
|
|
163
|
+
unit = next(
|
|
164
|
+
(
|
|
165
|
+
name
|
|
166
|
+
for prefix, name in (
|
|
167
|
+
("y", "years"),
|
|
168
|
+
("mo", "months"),
|
|
169
|
+
("w", "weeks"),
|
|
170
|
+
("d", "days"),
|
|
171
|
+
)
|
|
172
|
+
if raw_unit.startswith(prefix)
|
|
173
|
+
),
|
|
174
|
+
"years",
|
|
175
|
+
)
|
|
176
|
+
normalized = {"value": amount, "unit": unit}
|
|
177
|
+
else:
|
|
178
|
+
return result
|
|
179
|
+
if amount > {"years": 130, "months": 1560, "weeks": 6783, "days": 47483}[unit]:
|
|
180
|
+
result["status"] = "invalid"
|
|
181
|
+
return result
|
|
182
|
+
result.update(status="parsed", unit=unit)
|
|
183
|
+
if include_sensitive:
|
|
184
|
+
result["normalized"] = normalized
|
|
185
|
+
return result
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _address_analysis(value, include_sensitive):
|
|
189
|
+
result = {"kind": "address", "status": "unsupported"}
|
|
190
|
+
match = _ADDRESS.fullmatch(value.strip())
|
|
191
|
+
if not match or match["state"].upper() not in _STATES:
|
|
192
|
+
return result
|
|
193
|
+
components = {key: item.strip() for key, item in match.groupdict().items() if item}
|
|
194
|
+
components["state"] = components["state"].upper()
|
|
195
|
+
if "country" in components:
|
|
196
|
+
components["country"] = "US"
|
|
197
|
+
unit = _UNIT.fullmatch(components["street"])
|
|
198
|
+
if unit:
|
|
199
|
+
components["street"], components["unit"] = unit["street"], unit["unit"]
|
|
200
|
+
result.update(status="parsed", format="us_postal")
|
|
201
|
+
if include_sensitive:
|
|
202
|
+
result["components"] = components
|
|
203
|
+
return result
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def _luhn(value):
|
|
207
|
+
if not re.fullmatch(r"[0-9][0-9 -]*[0-9]", value):
|
|
208
|
+
return False
|
|
209
|
+
digits = [int(char) for char in value if char.isdigit()]
|
|
210
|
+
if not 13 <= len(digits) <= 19 or len(set(digits)) == 1:
|
|
211
|
+
return False
|
|
212
|
+
total = 0
|
|
213
|
+
for index, digit in enumerate(reversed(digits)):
|
|
214
|
+
if index % 2:
|
|
215
|
+
digit *= 2
|
|
216
|
+
if digit > 9:
|
|
217
|
+
digit -= 9
|
|
218
|
+
total += digit
|
|
219
|
+
return total % 10 == 0
|
|
220
|
+
|
|
221
|
+
|
|
222
|
+
def _routing_checksum(value):
|
|
223
|
+
if not re.fullmatch(r"[0-9]{9}", value) or value == "000000000":
|
|
224
|
+
return False
|
|
225
|
+
return (
|
|
226
|
+
sum(int(digit) * weight for digit, weight in zip(value, (3, 7, 1) * 3)) % 10
|
|
227
|
+
== 0
|
|
228
|
+
)
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
_RELATIONS = (
|
|
232
|
+
("NAME", {"DATE", "DOB"}, r"\s+(?:was\s+)?born\s+on\s+", "born_on"),
|
|
233
|
+
("NAME", {"ADDRESS"}, r"\s+(?:lives|resides)\s+at\s+", "resides_at"),
|
|
234
|
+
(
|
|
235
|
+
"NAME",
|
|
236
|
+
{"PHONE", "EMAIL"},
|
|
237
|
+
r"\s+(?:can|may)\s+be\s+reached\s+at\s+",
|
|
238
|
+
"contact_at",
|
|
239
|
+
),
|
|
240
|
+
("NAME", {"AGE"}, r"\s+(?:is\s+aged|is\s+age|is)\s+", "has_age"),
|
|
241
|
+
("NAME", {"CONDITION"}, r"\s+(?:has|was\s+diagnosed\s+with)\s+", "has_condition"),
|
|
242
|
+
("NAME", {"MEDICATION"}, r"\s+(?:takes|was\s+prescribed)\s+", "takes_medication"),
|
|
243
|
+
)
|
|
244
|
+
|
|
245
|
+
|
|
246
|
+
def _asserted_clause(text, left, right):
|
|
247
|
+
"""Reject visible uncertainty rather than guessing its syntactic scope."""
|
|
248
|
+
before = re.split(r"[.!?;\r\n]", text[max(0, left.start - 512) : left.start])[-1]
|
|
249
|
+
after = re.split(r"[.!;\r\n]", text[right.end : right.end + 256])[0]
|
|
250
|
+
if "?" in after:
|
|
251
|
+
return False
|
|
252
|
+
uncertain = r"\b(?:if|unless|whether|suppose|assuming|could|might|possible|suspected|uncertain|not|no|never|denies|denied|negative)\b"
|
|
253
|
+
if re.search(uncertain, before + " " + after, re.I):
|
|
254
|
+
return False
|
|
255
|
+
return not re.search(
|
|
256
|
+
r"\b(?:rule[ds]?\s+out|family|mother|father|sibling|brother|sister)\b",
|
|
257
|
+
before + " " + after,
|
|
258
|
+
re.I,
|
|
259
|
+
)
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
def analyze_entities(
|
|
263
|
+
text, entities, *, locale="en-US", include_sensitive=False
|
|
264
|
+
) -> dict:
|
|
265
|
+
"""Augment entities; input/output offsets are Python Unicode character offsets.
|
|
266
|
+
|
|
267
|
+
Entity values are omitted by default, including normalized sensitive values.
|
|
268
|
+
Validation results establish only syntax/checksum validity, never issuance,
|
|
269
|
+
identity, account ownership, or clinical truth.
|
|
270
|
+
"""
|
|
271
|
+
if not isinstance(text, str) or len(text) > 1_000_000:
|
|
272
|
+
raise ValueError("Invalid analysis text")
|
|
273
|
+
if locale not in SUPPORTED_LOCALES or type(include_sensitive) is not bool:
|
|
274
|
+
raise ValueError("Invalid analysis options")
|
|
275
|
+
entities = checked_entities(text, entities)
|
|
276
|
+
analyzed = []
|
|
277
|
+
for entity in entities:
|
|
278
|
+
item = entity.as_dict()
|
|
279
|
+
value = text[entity.start : entity.end]
|
|
280
|
+
if entity.label in {"DATE", "DOB"}:
|
|
281
|
+
item["analysis"] = _date_analysis(value, locale, include_sensitive)
|
|
282
|
+
elif entity.label == "AGE":
|
|
283
|
+
item["analysis"] = _age_analysis(value, include_sensitive)
|
|
284
|
+
elif entity.label == "ADDRESS":
|
|
285
|
+
item["analysis"] = _address_analysis(value, include_sensitive)
|
|
286
|
+
elif entity.label == "CREDIT_CARD":
|
|
287
|
+
item["analysis"] = {
|
|
288
|
+
"kind": "credit_card",
|
|
289
|
+
"status": "validated",
|
|
290
|
+
"checksum_valid": _luhn(value),
|
|
291
|
+
"algorithm": "luhn",
|
|
292
|
+
}
|
|
293
|
+
elif entity.label == "ROUTING_NUMBER":
|
|
294
|
+
item["analysis"] = {
|
|
295
|
+
"kind": "routing_number",
|
|
296
|
+
"status": "validated",
|
|
297
|
+
"checksum_valid": _routing_checksum(value),
|
|
298
|
+
"algorithm": "aba",
|
|
299
|
+
}
|
|
300
|
+
analyzed.append(item)
|
|
301
|
+
relationships = []
|
|
302
|
+
for index, (left, right) in enumerate(zip(entities, entities[1:])):
|
|
303
|
+
# Never infer cross-line associations or copy free text into metadata.
|
|
304
|
+
between = text[left.end : right.start]
|
|
305
|
+
if len(between) > 64 or "\n" in between or "\r" in between:
|
|
306
|
+
continue
|
|
307
|
+
for left_label, right_labels, predicate, relation in _RELATIONS:
|
|
308
|
+
if (
|
|
309
|
+
left.label == left_label
|
|
310
|
+
and right.label in right_labels
|
|
311
|
+
and re.fullmatch(predicate, between, re.I)
|
|
312
|
+
):
|
|
313
|
+
if not _asserted_clause(text, left, right):
|
|
314
|
+
continue
|
|
315
|
+
relationships.append(
|
|
316
|
+
{
|
|
317
|
+
"type": relation,
|
|
318
|
+
"source_entity": index,
|
|
319
|
+
"target_entity": index + 1,
|
|
320
|
+
"source": "rule:explicit_relation",
|
|
321
|
+
}
|
|
322
|
+
)
|
|
323
|
+
break
|
|
324
|
+
return {
|
|
325
|
+
"entities": analyzed,
|
|
326
|
+
"relationships": relationships,
|
|
327
|
+
"metadata": {
|
|
328
|
+
"analysis_version": ANALYSIS_VERSION,
|
|
329
|
+
"locale": locale,
|
|
330
|
+
"sensitive_values_included": include_sensitive,
|
|
331
|
+
"relationship_extraction": "explicit_templates",
|
|
332
|
+
"relationship_status": "experimental",
|
|
333
|
+
},
|
|
334
|
+
}
|