outis 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (79) hide show
  1. outis/__init__.py +4 -0
  2. outis/analysis.py +334 -0
  3. outis/api.py +764 -0
  4. outis/cli.py +45 -0
  5. outis/data/__init__.py +1 -0
  6. outis/data/annotation_policy.py +121 -0
  7. outis/data/annotation_policy_ages.py +44 -0
  8. outis/data/annotation_policy_care_sites.py +108 -0
  9. outis/data/annotation_policy_eponyms.py +30 -0
  10. outis/data/annotation_policy_facility_names.py +25 -0
  11. outis/data/annotation_policy_fhir.py +47 -0
  12. outis/data/annotation_policy_places.py +60 -0
  13. outis/data/annotation_policy_reports.py +39 -0
  14. outis/data/annotation_policy_spoken.py +65 -0
  15. outis/data/annotation_policy_targeted.py +71 -0
  16. outis/data/annotation_policy_transcripts.py +33 -0
  17. outis/data/challenge.py +270 -0
  18. outis/data/corpus_quality.py +429 -0
  19. outis/data/generators/__init__.py +1 -0
  20. outis/data/generators/alias_contrastive.py +765 -0
  21. outis/data/generators/augment.py +521 -0
  22. outis/data/generators/base.py +547 -0
  23. outis/data/generators/compositional.py +570 -0
  24. outis/data/generators/context.py +267 -0
  25. outis/data/generators/coverage.py +398 -0
  26. outis/data/generators/generalization.py +1450 -0
  27. outis/data/generators/name_contrastive.py +902 -0
  28. outis/data/generators/narrative.py +520 -0
  29. outis/data/generators/nominal_alias.py +1224 -0
  30. outis/data/generators/place_sources.py +185 -0
  31. outis/data/generators/places.py +1160 -0
  32. outis/data/generators/places_contexts.py +622 -0
  33. outis/data/generators/remediation.py +1204 -0
  34. outis/data/generators/remediation_development.py +613 -0
  35. outis/data/v12/__init__.py +1 -0
  36. outis/data/v12/contexts.py +262 -0
  37. outis/data/v12/development.py +663 -0
  38. outis/data/v12/synthetic.py +555 -0
  39. outis/data/v13/__init__.py +1 -0
  40. outis/data/v13/contexts.py +735 -0
  41. outis/data/v13/replay.py +291 -0
  42. outis/data/v13/synthetic.py +284 -0
  43. outis/detection/__init__.py +1 -0
  44. outis/detection/deberta_attention.py +158 -0
  45. outis/detection/decoding.py +338 -0
  46. outis/detection/hf_encoder.py +217 -0
  47. outis/entities.py +68 -0
  48. outis/evaluation/__init__.py +1 -0
  49. outis/evaluation/evaluate.py +619 -0
  50. outis/evaluation/report.py +606 -0
  51. outis/evaluation/report_html.py +732 -0
  52. outis/evaluation/report_progress.py +150 -0
  53. outis/evaluation/runtime.py +73 -0
  54. outis/formats/__init__.py +1 -0
  55. outis/formats/files.py +861 -0
  56. outis/formats/structured.py +395 -0
  57. outis/operations.py +87 -0
  58. outis/paths.py +40 -0
  59. outis/policy.py +373 -0
  60. outis/replacement/__init__.py +1 -0
  61. outis/replacement/names/__init__.py +5 -0
  62. outis/replacement/names/age.py +63 -0
  63. outis/replacement/names/aliases.py +100 -0
  64. outis/replacement/names/identity.py +103 -0
  65. outis/replacement/names/nickname.py +521 -0
  66. outis/replacement/names/person.py +270 -0
  67. outis/replacement/names/record.py +105 -0
  68. outis/replacement/nicknames.py +154 -0
  69. outis/replacement/surrogates.py +856 -0
  70. outis/replacement/transform.py +596 -0
  71. outis/service.py +444 -0
  72. outis/types.py +40 -0
  73. outis-0.1.0.dist-info/METADATA +276 -0
  74. outis-0.1.0.dist-info/RECORD +79 -0
  75. outis-0.1.0.dist-info/WHEEL +5 -0
  76. outis-0.1.0.dist-info/entry_points.txt +2 -0
  77. outis-0.1.0.dist-info/licenses/LICENSE +201 -0
  78. outis-0.1.0.dist-info/licenses/NOTICE +25 -0
  79. outis-0.1.0.dist-info/top_level.txt +1 -0
outis/__init__.py ADDED
@@ -0,0 +1,4 @@
1
+ from outis.service import Deidentifier
2
+ from outis.types import Span
3
+
4
+ __all__ = ["Span", "Deidentifier"]
outis/analysis.py ADDED
@@ -0,0 +1,334 @@
1
+ """Conservative, inspectable analysis of already detected English entities.
2
+
3
+ This module never expands detector coverage or infers clinical facts. Parsing
4
+ requires the complete detected value to match a supported grammar. Normalized
5
+ dates, ages and address components are sensitive values and require opt-in.
6
+ Relationships require an explicit supported predicate between adjacent spans;
7
+ they are experimental rule matches, not clinically validated assertions.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import re
13
+ from datetime import date
14
+
15
+ from outis.entities import checked_entities
16
+
17
+ ANALYSIS_VERSION = "1"
18
+ SUPPORTED_LOCALES = ("en-US", "en-GB")
19
+ _MONTHS = {
20
+ name: number
21
+ for number, names in enumerate(
22
+ (
23
+ ("january", "jan"),
24
+ ("february", "feb"),
25
+ ("march", "mar"),
26
+ ("april", "apr"),
27
+ ("may",),
28
+ ("june", "jun"),
29
+ ("july", "jul"),
30
+ ("august", "aug"),
31
+ ("september", "sept", "sep"),
32
+ ("october", "oct"),
33
+ ("november", "nov"),
34
+ ("december", "dec"),
35
+ ),
36
+ 1,
37
+ )
38
+ for name in names
39
+ }
40
+ _MONTH = "(?:" + "|".join(sorted(_MONTHS, key=len, reverse=True)) + r")\.?"
41
+ _STATES = frozenset(
42
+ "AL AK AZ AR CA CO CT DE DC FL GA HI ID IL IN IA KS KY LA ME MD MA MI MN MS MO MT NE NV NH NJ NM NY NC ND OH OK OR PA RI SC SD TN TX UT VT VA WA WV WI WY AS GU MP PR VI AA AE AP".split()
43
+ )
44
+ _ADDRESS = re.compile(
45
+ r"(?P<street_number>[0-9]{1,8}[A-Za-z]?(?:-[0-9]{1,8})?) "
46
+ r"(?P<street>[^,\r\n]{2,100}),\s*"
47
+ r"(?P<city>[A-Za-z][A-Za-z .'-]{0,63}),\s*"
48
+ r"(?P<state>[A-Z]{2})\s+(?P<postal_code>[0-9]{5}(?:-[0-9]{4})?)"
49
+ r"(?:,?\s+(?P<country>USA|US|United States))?",
50
+ re.I,
51
+ )
52
+ _UNIT = re.compile(
53
+ r"(?P<street>.+?)\s+(?P<unit>(?:apt\.?|suite|ste\.?|unit|#)\s*[A-Za-z0-9-]+)", re.I
54
+ )
55
+
56
+
57
+ def _date_analysis(value, locale, include_sensitive):
58
+ value = value.strip()
59
+ result = {"kind": "date", "status": "unsupported"}
60
+ year = month = day = None
61
+ precision = "day"
62
+ match = re.fullmatch(r"([0-9]{4})([-/])([0-9]{1,2})\2([0-9]{1,2})", value)
63
+ if match:
64
+ year, month, day = int(match[1]), int(match[3]), int(match[4])
65
+ else:
66
+ match = re.fullmatch(r"([0-9]{1,2})([-/])([0-9]{1,2})\2([0-9]{4})", value)
67
+ if match:
68
+ first, second, year = int(match[1]), int(match[3]), int(match[4])
69
+ month, day = (first, second) if locale == "en-US" else (second, first)
70
+ result["interpretation"] = (
71
+ "month_first" if locale == "en-US" else "day_first"
72
+ )
73
+ result["locale_dependent"] = (
74
+ first <= 12 and second <= 12 and first != second
75
+ )
76
+ else:
77
+ match = re.fullmatch(
78
+ rf"({_MONTH})\s+([0-9]{{1,2}})(?:st|nd|rd|th)?(?:,?\s+)([0-9]{{4}})",
79
+ value,
80
+ re.I,
81
+ )
82
+ if match:
83
+ month, day, year = (
84
+ _MONTHS[match[1].rstrip(".").lower()],
85
+ int(match[2]),
86
+ int(match[3]),
87
+ )
88
+ else:
89
+ match = re.fullmatch(
90
+ rf"([0-9]{{1,2}})(?:st|nd|rd|th)?\s+({_MONTH})(?:,?\s+)([0-9]{{4}})",
91
+ value,
92
+ re.I,
93
+ )
94
+ if match:
95
+ day, month, year = (
96
+ int(match[1]),
97
+ _MONTHS[match[2].rstrip(".").lower()],
98
+ int(match[3]),
99
+ )
100
+ else:
101
+ match = re.fullmatch(rf"({_MONTH})\s+([0-9]{{4}})", value, re.I)
102
+ if match:
103
+ month, year = (
104
+ _MONTHS[match[1].rstrip(".").lower()],
105
+ int(match[2]),
106
+ )
107
+ precision = "month"
108
+ elif re.fullmatch(r"[0-9]{4}", value):
109
+ year, precision = int(value), "year"
110
+ elif re.fullmatch(
111
+ r"[0-9]{1,2}[-/][0-9]{1,2}(?:[-/][0-9]{2})?", value
112
+ ):
113
+ return {
114
+ "kind": "date",
115
+ "status": "ambiguous",
116
+ "reason": "missing_four_digit_year",
117
+ }
118
+ else:
119
+ return result
120
+ ordinal = re.search(r"(?<![0-9])([0-9]{1,2})(st|nd|rd|th)\b", value, re.I)
121
+ if ordinal:
122
+ number = int(ordinal[1])
123
+ suffix = (
124
+ "th"
125
+ if number % 100 in {11, 12, 13}
126
+ else {1: "st", 2: "nd", 3: "rd"}.get(number % 10, "th")
127
+ )
128
+ if ordinal[2].lower() != suffix:
129
+ result["status"] = "invalid"
130
+ return result
131
+ try:
132
+ parsed = date(year, month or 1, day or 1)
133
+ # Zero is invalid, not an omitted calendar component.
134
+ if month == 0 or day == 0:
135
+ raise ValueError
136
+ except (ValueError, TypeError):
137
+ result["status"] = "invalid"
138
+ return result
139
+ result.update(status="parsed", precision=precision)
140
+ if include_sensitive:
141
+ result["normalized"] = parsed.isoformat()[
142
+ : {"day": 10, "month": 7, "year": 4}[precision]
143
+ ]
144
+ return result
145
+
146
+
147
+ def _age_analysis(value, include_sensitive):
148
+ value = value.strip().lower()
149
+ result = {"kind": "age", "status": "unsupported"}
150
+ older = re.fullmatch(r"([0-9]{1,3})\+\s*(?:years?(?: old)?)?", value)
151
+ match = re.fullmatch(
152
+ r"(?:aged?\s+)?([0-9]{1,4})(?:\s*[- ]?\s*"
153
+ r"(years?|yrs?|y/?o|months?|mos?|weeks?|wks?|days?)"
154
+ r"(?:\s*[- ]?\s*old)?)?",
155
+ value,
156
+ )
157
+ if older:
158
+ amount, unit = int(older[1]), "years"
159
+ normalized = {"minimum": amount, "unit": unit, "inclusive": True}
160
+ elif match:
161
+ amount = int(match[1])
162
+ raw_unit = match[2] or "years"
163
+ unit = next(
164
+ (
165
+ name
166
+ for prefix, name in (
167
+ ("y", "years"),
168
+ ("mo", "months"),
169
+ ("w", "weeks"),
170
+ ("d", "days"),
171
+ )
172
+ if raw_unit.startswith(prefix)
173
+ ),
174
+ "years",
175
+ )
176
+ normalized = {"value": amount, "unit": unit}
177
+ else:
178
+ return result
179
+ if amount > {"years": 130, "months": 1560, "weeks": 6783, "days": 47483}[unit]:
180
+ result["status"] = "invalid"
181
+ return result
182
+ result.update(status="parsed", unit=unit)
183
+ if include_sensitive:
184
+ result["normalized"] = normalized
185
+ return result
186
+
187
+
188
+ def _address_analysis(value, include_sensitive):
189
+ result = {"kind": "address", "status": "unsupported"}
190
+ match = _ADDRESS.fullmatch(value.strip())
191
+ if not match or match["state"].upper() not in _STATES:
192
+ return result
193
+ components = {key: item.strip() for key, item in match.groupdict().items() if item}
194
+ components["state"] = components["state"].upper()
195
+ if "country" in components:
196
+ components["country"] = "US"
197
+ unit = _UNIT.fullmatch(components["street"])
198
+ if unit:
199
+ components["street"], components["unit"] = unit["street"], unit["unit"]
200
+ result.update(status="parsed", format="us_postal")
201
+ if include_sensitive:
202
+ result["components"] = components
203
+ return result
204
+
205
+
206
+ def _luhn(value):
207
+ if not re.fullmatch(r"[0-9][0-9 -]*[0-9]", value):
208
+ return False
209
+ digits = [int(char) for char in value if char.isdigit()]
210
+ if not 13 <= len(digits) <= 19 or len(set(digits)) == 1:
211
+ return False
212
+ total = 0
213
+ for index, digit in enumerate(reversed(digits)):
214
+ if index % 2:
215
+ digit *= 2
216
+ if digit > 9:
217
+ digit -= 9
218
+ total += digit
219
+ return total % 10 == 0
220
+
221
+
222
+ def _routing_checksum(value):
223
+ if not re.fullmatch(r"[0-9]{9}", value) or value == "000000000":
224
+ return False
225
+ return (
226
+ sum(int(digit) * weight for digit, weight in zip(value, (3, 7, 1) * 3)) % 10
227
+ == 0
228
+ )
229
+
230
+
231
+ _RELATIONS = (
232
+ ("NAME", {"DATE", "DOB"}, r"\s+(?:was\s+)?born\s+on\s+", "born_on"),
233
+ ("NAME", {"ADDRESS"}, r"\s+(?:lives|resides)\s+at\s+", "resides_at"),
234
+ (
235
+ "NAME",
236
+ {"PHONE", "EMAIL"},
237
+ r"\s+(?:can|may)\s+be\s+reached\s+at\s+",
238
+ "contact_at",
239
+ ),
240
+ ("NAME", {"AGE"}, r"\s+(?:is\s+aged|is\s+age|is)\s+", "has_age"),
241
+ ("NAME", {"CONDITION"}, r"\s+(?:has|was\s+diagnosed\s+with)\s+", "has_condition"),
242
+ ("NAME", {"MEDICATION"}, r"\s+(?:takes|was\s+prescribed)\s+", "takes_medication"),
243
+ )
244
+
245
+
246
+ def _asserted_clause(text, left, right):
247
+ """Reject visible uncertainty rather than guessing its syntactic scope."""
248
+ before = re.split(r"[.!?;\r\n]", text[max(0, left.start - 512) : left.start])[-1]
249
+ after = re.split(r"[.!;\r\n]", text[right.end : right.end + 256])[0]
250
+ if "?" in after:
251
+ return False
252
+ uncertain = r"\b(?:if|unless|whether|suppose|assuming|could|might|possible|suspected|uncertain|not|no|never|denies|denied|negative)\b"
253
+ if re.search(uncertain, before + " " + after, re.I):
254
+ return False
255
+ return not re.search(
256
+ r"\b(?:rule[ds]?\s+out|family|mother|father|sibling|brother|sister)\b",
257
+ before + " " + after,
258
+ re.I,
259
+ )
260
+
261
+
262
+ def analyze_entities(
263
+ text, entities, *, locale="en-US", include_sensitive=False
264
+ ) -> dict:
265
+ """Augment entities; input/output offsets are Python Unicode character offsets.
266
+
267
+ Entity values are omitted by default, including normalized sensitive values.
268
+ Validation results establish only syntax/checksum validity, never issuance,
269
+ identity, account ownership, or clinical truth.
270
+ """
271
+ if not isinstance(text, str) or len(text) > 1_000_000:
272
+ raise ValueError("Invalid analysis text")
273
+ if locale not in SUPPORTED_LOCALES or type(include_sensitive) is not bool:
274
+ raise ValueError("Invalid analysis options")
275
+ entities = checked_entities(text, entities)
276
+ analyzed = []
277
+ for entity in entities:
278
+ item = entity.as_dict()
279
+ value = text[entity.start : entity.end]
280
+ if entity.label in {"DATE", "DOB"}:
281
+ item["analysis"] = _date_analysis(value, locale, include_sensitive)
282
+ elif entity.label == "AGE":
283
+ item["analysis"] = _age_analysis(value, include_sensitive)
284
+ elif entity.label == "ADDRESS":
285
+ item["analysis"] = _address_analysis(value, include_sensitive)
286
+ elif entity.label == "CREDIT_CARD":
287
+ item["analysis"] = {
288
+ "kind": "credit_card",
289
+ "status": "validated",
290
+ "checksum_valid": _luhn(value),
291
+ "algorithm": "luhn",
292
+ }
293
+ elif entity.label == "ROUTING_NUMBER":
294
+ item["analysis"] = {
295
+ "kind": "routing_number",
296
+ "status": "validated",
297
+ "checksum_valid": _routing_checksum(value),
298
+ "algorithm": "aba",
299
+ }
300
+ analyzed.append(item)
301
+ relationships = []
302
+ for index, (left, right) in enumerate(zip(entities, entities[1:])):
303
+ # Never infer cross-line associations or copy free text into metadata.
304
+ between = text[left.end : right.start]
305
+ if len(between) > 64 or "\n" in between or "\r" in between:
306
+ continue
307
+ for left_label, right_labels, predicate, relation in _RELATIONS:
308
+ if (
309
+ left.label == left_label
310
+ and right.label in right_labels
311
+ and re.fullmatch(predicate, between, re.I)
312
+ ):
313
+ if not _asserted_clause(text, left, right):
314
+ continue
315
+ relationships.append(
316
+ {
317
+ "type": relation,
318
+ "source_entity": index,
319
+ "target_entity": index + 1,
320
+ "source": "rule:explicit_relation",
321
+ }
322
+ )
323
+ break
324
+ return {
325
+ "entities": analyzed,
326
+ "relationships": relationships,
327
+ "metadata": {
328
+ "analysis_version": ANALYSIS_VERSION,
329
+ "locale": locale,
330
+ "sensitive_values_included": include_sensitive,
331
+ "relationship_extraction": "explicit_templates",
332
+ "relationship_status": "experimental",
333
+ },
334
+ }