sourcelock 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hc_source/__init__.py +5 -0
- hc_source/adapters/__init__.py +500 -0
- hc_source/adapters/_demo.py +258 -0
- hc_source/adapters/_demo_fixture.json +25 -0
- hc_source/adapters/_leie_sample.csv +15 -0
- hc_source/adapters/codes.py +1232 -0
- hc_source/adapters/coverage.py +1569 -0
- hc_source/adapters/hcc.py +1450 -0
- hc_source/adapters/leie.py +1310 -0
- hc_source/adapters/provider.py +1159 -0
- hc_source/cache.py +664 -0
- hc_source/cli.py +959 -0
- hc_source/cli_manifest.py +207 -0
- hc_source/data/codes/hcpcs_2026q3.csv.gz +0 -0
- hc_source/data/codes/icd10cm_fy2026.csv.gz +0 -0
- hc_source/data/codes/icd10cm_fy2027.csv.gz +0 -0
- hc_source/data/codes/manifest.json +75 -0
- hc_source/data/codes/regenerate.py +291 -0
- hc_source/data/hcc/hcc_data.json.zlib +0 -0
- hc_source/doctor.py +472 -0
- hc_source/guard.py +877 -0
- hc_source/http.py +541 -0
- hc_source/interfaces.py +395 -0
- hc_source/lockfile.py +236 -0
- hc_source/manifest.py +422 -0
- hc_source/mcp_server.py +203 -0
- hc_source/npi.py +50 -0
- hc_source/receipts.py +74 -0
- hc_source/schemas.py +339 -0
- sourcelock-0.1.0.dist-info/METADATA +272 -0
- sourcelock-0.1.0.dist-info/RECORD +34 -0
- sourcelock-0.1.0.dist-info/WHEEL +4 -0
- sourcelock-0.1.0.dist-info/entry_points.txt +2 -0
- sourcelock-0.1.0.dist-info/licenses/LICENSE +21 -0
hc_source/guard.py
ADDED
|
@@ -0,0 +1,877 @@
|
|
|
1
|
+
"""Zero-PHI guard.
|
|
2
|
+
|
|
3
|
+
Every tool call passes through :func:`assert_public_params` before any upstream
|
|
4
|
+
request is made and before anything is logged. The guard's job is narrow and
|
|
5
|
+
adversarial: refuse anything shaped like a patient record.
|
|
6
|
+
|
|
7
|
+
**What this is, honestly.** A structural, best-effort refusal of patient-shaped
|
|
8
|
+
input. It is NOT a HIPAA compliance control, not a certification, and not a
|
|
9
|
+
guarantee that PHI cannot reach this process. It raises the cost of an accident
|
|
10
|
+
-- a pasted HL7 segment, a CSV column that turned out to hold MRNs, an agent
|
|
11
|
+
that decided a patient name was a search term -- and it will not stop a
|
|
12
|
+
determined sender. Treat a clean scan as "nothing patient-shaped was detected",
|
|
13
|
+
never as "this input contains no PHI". The receipt says ``PHI_NOT_EXPECTED``
|
|
14
|
+
for exactly that reason: it is a statement of design intent, not a finding.
|
|
15
|
+
|
|
16
|
+
Design rules that matter to adapter authors:
|
|
17
|
+
|
|
18
|
+
* The guard runs on the RAW parameter mapping, before pydantic validation, so
|
|
19
|
+
patient-shaped input is refused even when it would have been dropped anyway.
|
|
20
|
+
* Rejection messages never echo the offending value -- the whole point is that
|
|
21
|
+
the value must not reach a log line. They name the path and the detector.
|
|
22
|
+
* Detection is structural (FHIR/X12/HL7v2 envelopes, SSN and MBI patterns,
|
|
23
|
+
DOB+name combinations, patient-record field names), not a guess about content.
|
|
24
|
+
* Every budget is a REFUSAL when exceeded, never a silent pass. A guard that
|
|
25
|
+
gives up quietly is worse than no guard.
|
|
26
|
+
|
|
27
|
+
Where the line sits, and why
|
|
28
|
+
----------------------------
|
|
29
|
+
Two shipped tools take genuinely free text. ``leie.candidate_search`` takes a
|
|
30
|
+
person's NAME -- that is the entire tool -- and ``coverage.search_documents``
|
|
31
|
+
takes clinical prose like "excision of malignant skin lesions". So:
|
|
32
|
+
|
|
33
|
+
* an ambiguous-but-plausible value PASSES: a bare surname, a clinical phrase, a
|
|
34
|
+
date on its own, an NPI;
|
|
35
|
+
* a structured patient-record SHAPE does not: a name next to an SSN, an HL7
|
|
36
|
+
segment header, an MBI, a DOB beside a name, a FHIR object.
|
|
37
|
+
|
|
38
|
+
Field NAMES are matched on token boundaries, never as bare substrings. Bare
|
|
39
|
+
substring matching refuses "geographic" (contains "phi"), "adobe" and
|
|
40
|
+
"endobronchial" (contain "dob"), and "outpatient" (contains "patient") -- all of
|
|
41
|
+
which are ordinary product input. The two corpora in ``tests/test_guard_bypass.py``
|
|
42
|
+
and ``tests/test_guard_must_pass.py`` are the executable form of this paragraph;
|
|
43
|
+
change one and you must change the other.
|
|
44
|
+
|
|
45
|
+
Known evasion classes NOT closed
|
|
46
|
+
--------------------------------
|
|
47
|
+
Stated plainly because honest limits beat silent gaps:
|
|
48
|
+
|
|
49
|
+
* **Semantic-only PHI.** "the patient in room 4 who came in Tuesday" has no
|
|
50
|
+
structural marker. Nothing here will catch it.
|
|
51
|
+
* **A single unstructured name or address.** Refusing those would break
|
|
52
|
+
``leie.candidate_search``, whose purpose is to search by name. A name alone is
|
|
53
|
+
accepted by design.
|
|
54
|
+
* **Integer-typed values.** The guard scans string leaves. A caller invoking the
|
|
55
|
+
Python API with ``{"contractor_id": 123456789}`` as an int is not scanned for
|
|
56
|
+
SSN shape; over the CLI and MCP everything arrives as a string and is.
|
|
57
|
+
* **Encryption and unknown encodings.** Base64 and percent-encoding are decoded
|
|
58
|
+
and rescanned to a bounded depth; anything encrypted, compressed, or in an
|
|
59
|
+
encoding we do not probe passes as opaque bytes.
|
|
60
|
+
* **Free-text identifiers that are not SSN/MBI-shaped.** Local chart numbers,
|
|
61
|
+
payer-specific member ids and account numbers have no universal shape.
|
|
62
|
+
* **Homoglyph coverage is partial.** A confusables table for the common Cyrillic
|
|
63
|
+
and Greek Latin-lookalikes is applied; it is not the full Unicode set.
|
|
64
|
+
|
|
65
|
+
Operator escape hatch
|
|
66
|
+
---------------------
|
|
67
|
+
``HC_SOURCE_GUARD_ALLOW`` takes a comma-separated list of detector codes
|
|
68
|
+
(e.g. ``SSN_PATTERN,MBI_PATTERN``) that :func:`assert_public_params` will not
|
|
69
|
+
refuse on. It exists so a false positive is a config change rather than an
|
|
70
|
+
abandoned pipeline. It suppresses named detectors only -- never the whole guard
|
|
71
|
+
-- and :func:`scan_for_phi` itself is never silenced, so audits and tests always
|
|
72
|
+
see the truth.
|
|
73
|
+
"""
|
|
74
|
+
|
|
75
|
+
from __future__ import annotations
|
|
76
|
+
|
|
77
|
+
import base64
|
|
78
|
+
import binascii
|
|
79
|
+
import json
|
|
80
|
+
import os
|
|
81
|
+
import re
|
|
82
|
+
import unicodedata
|
|
83
|
+
from typing import Any, Iterable, Mapping
|
|
84
|
+
from urllib.parse import unquote
|
|
85
|
+
|
|
86
|
+
__all__ = [
|
|
87
|
+
"GUARD_ALLOW_ENV",
|
|
88
|
+
"PHI_NON_CLAIM",
|
|
89
|
+
"PHIRejected",
|
|
90
|
+
"ParamsRejected",
|
|
91
|
+
"REDACTED",
|
|
92
|
+
"allowed_detectors",
|
|
93
|
+
"assert_public_params",
|
|
94
|
+
"normalize_for_matching",
|
|
95
|
+
"redact_input",
|
|
96
|
+
"safe_label",
|
|
97
|
+
"safe_location",
|
|
98
|
+
"safe_tool_name",
|
|
99
|
+
"safe_validation_message",
|
|
100
|
+
"safe_validation_reasons",
|
|
101
|
+
"scan_for_phi",
|
|
102
|
+
]
|
|
103
|
+
|
|
104
|
+
PHI_NON_CLAIM = "PHI_NOT_EXPECTED: this tool accepts only public typed parameters."
|
|
105
|
+
|
|
106
|
+
GUARD_ALLOW_ENV = "HC_SOURCE_GUARD_ALLOW"
|
|
107
|
+
|
|
108
|
+
MAX_DEPTH = 12
|
|
109
|
+
MAX_SERIALIZED_BYTES = 64 * 1024
|
|
110
|
+
|
|
111
|
+
#: Work budgets. Exceeding any of these is a refusal, not a pass -- an input
|
|
112
|
+
#: large or convoluted enough to exhaust the guard is by definition not "a small
|
|
113
|
+
#: typed public parameter".
|
|
114
|
+
MAX_NODES = 4096
|
|
115
|
+
MAX_DECODE_DEPTH = 4
|
|
116
|
+
MAX_EMBEDDED_CANDIDATES = 8
|
|
117
|
+
MAX_ENCODED_RUNS = 4
|
|
118
|
+
MIN_ENCODED_RUN = 12 # 'MTIzLTQ1LTY3ODk=' (a base64 SSN) is 15 chars of payload
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
# ---------------------------------------------------------------------------
|
|
122
|
+
# Normalisation
|
|
123
|
+
# ---------------------------------------------------------------------------
|
|
124
|
+
#
|
|
125
|
+
# Round-1 finding C3: `123-45-6789` was caught but `123.45.6789`, `123–45–6789`
|
|
126
|
+
# (en dash), `123 45 6789` (NBSP) and `123-<ZWSP>45-6789` all walked past,
|
|
127
|
+
# because the pattern hard-coded ASCII hyphen and space. Matching a normalised
|
|
128
|
+
# copy is the fix that generalises; adding one more separator to one more regex
|
|
129
|
+
# is the fix that does not.
|
|
130
|
+
|
|
131
|
+
_ZERO_WIDTH = dict.fromkeys(
|
|
132
|
+
[
|
|
133
|
+
0x00AD, # soft hyphen
|
|
134
|
+
0x200B, # zero width space
|
|
135
|
+
0x200C, # zero width non-joiner
|
|
136
|
+
0x200D, # zero width joiner
|
|
137
|
+
0x2060, # word joiner
|
|
138
|
+
0xFEFF, # zero width no-break space
|
|
139
|
+
]
|
|
140
|
+
)
|
|
141
|
+
|
|
142
|
+
#: Latin lookalikes from the Cyrillic and Greek blocks. Partial by design; see
|
|
143
|
+
#: the module docstring's list of limits.
|
|
144
|
+
_CONFUSABLES = str.maketrans(
|
|
145
|
+
{
|
|
146
|
+
"а": "a", "е": "e", "о": "o", "р": "p", "с": "c",
|
|
147
|
+
"у": "y", "х": "x", "ѕ": "s", "і": "i", "ј": "j",
|
|
148
|
+
"ԁ": "d", "һ": "h", "ӏ": "l", "м": "m", "т": "t",
|
|
149
|
+
"в": "b", "н": "h", "к": "k", "А": "A", "Е": "E",
|
|
150
|
+
"О": "O", "Р": "P", "С": "C", "Х": "X", "Ѕ": "S",
|
|
151
|
+
"І": "I", "М": "M", "Т": "T", "В": "B", "Н": "H",
|
|
152
|
+
"К": "K", "ο": "o", "α": "a", "ρ": "p", "ν": "v",
|
|
153
|
+
"τ": "t", "ι": "i", "κ": "k", "μ": "u", "χ": "x",
|
|
154
|
+
"Α": "A", "Β": "B", "Ε": "E", "Η": "H", "Ι": "I",
|
|
155
|
+
"Κ": "K", "Μ": "M", "Ν": "N", "Ο": "O", "Ρ": "P",
|
|
156
|
+
"Τ": "T", "Χ": "X",
|
|
157
|
+
}
|
|
158
|
+
)
|
|
159
|
+
|
|
160
|
+
#: Every dash and space variant folded onto the ASCII pair the patterns use.
|
|
161
|
+
_SEPARATORS = str.maketrans(
|
|
162
|
+
{
|
|
163
|
+
"‐": "-", "‑": "-", "‒": "-", "–": "-", "—": "-",
|
|
164
|
+
"―": "-", "−": "-", "﹘": "-", "﹣": "-", "-": "-",
|
|
165
|
+
" ": " ", " ": " ", " ": " ", " ": " ", " ": " ",
|
|
166
|
+
" ": " ", " ": " ", " ": " ", " ": " ", " ": " ",
|
|
167
|
+
" ": " ", " ": " ", " ": " ", " ": " ", " ": " ",
|
|
168
|
+
" ": " ",
|
|
169
|
+
}
|
|
170
|
+
)
|
|
171
|
+
|
|
172
|
+
|
|
173
|
+
def normalize_for_matching(value: str) -> str:
|
|
174
|
+
"""NFKC, strip zero-width, fold homoglyphs and separator variants.
|
|
175
|
+
|
|
176
|
+
Applied to a COPY used only for matching. The original value is never
|
|
177
|
+
altered and never echoed.
|
|
178
|
+
"""
|
|
179
|
+
text = unicodedata.normalize("NFKC", value)
|
|
180
|
+
text = text.translate(_ZERO_WIDTH)
|
|
181
|
+
text = text.translate(_CONFUSABLES)
|
|
182
|
+
return text.translate(_SEPARATORS)
|
|
183
|
+
|
|
184
|
+
|
|
185
|
+
# ---------------------------------------------------------------------------
|
|
186
|
+
# Detectors
|
|
187
|
+
# ---------------------------------------------------------------------------
|
|
188
|
+
|
|
189
|
+
_FHIR_PATIENT_RESOURCES = {
|
|
190
|
+
"Patient",
|
|
191
|
+
"Bundle",
|
|
192
|
+
"Coverage",
|
|
193
|
+
"Claim",
|
|
194
|
+
"ClaimResponse",
|
|
195
|
+
"ExplanationOfBenefit",
|
|
196
|
+
"Encounter",
|
|
197
|
+
"Observation",
|
|
198
|
+
"Condition",
|
|
199
|
+
"MedicationRequest",
|
|
200
|
+
"DocumentReference",
|
|
201
|
+
"RelatedPerson",
|
|
202
|
+
"Person",
|
|
203
|
+
}
|
|
204
|
+
|
|
205
|
+
#: FHIR field names that are distinctive on their own...
|
|
206
|
+
_FHIR_STRONG_KEYS = {
|
|
207
|
+
"resourcetype", "birthdate", "gender", "telecom", "maritalstatus",
|
|
208
|
+
"given", "family", "deceasedboolean", "multiplebirthboolean",
|
|
209
|
+
"managingorganization",
|
|
210
|
+
}
|
|
211
|
+
#: ...and ones that only mean something in company.
|
|
212
|
+
_FHIR_WEAK_KEYS = {"name", "identifier", "address", "subject", "patient", "contact"}
|
|
213
|
+
|
|
214
|
+
# SSN: separated (any dialect, post-normalisation) or a bare nine-digit run.
|
|
215
|
+
# The bare-run rule is a deliberate over-block: no shipped parameter legitimately
|
|
216
|
+
# carries a nine-digit string (an NPI is ten, a MAC contractor id is five), and
|
|
217
|
+
# an SSN typed without separators was the single most common bypass in round 1.
|
|
218
|
+
_SSN_SEPARATED_RE = re.compile(r"\b\d{3}[-. ]\d{2}[-. ]\d{4}\b")
|
|
219
|
+
_SSN_BARE_RE = re.compile(r"(?<!\d)\d{9}(?!\d)")
|
|
220
|
+
|
|
221
|
+
# Medicare Beneficiary Identifier: 11 characters, C-A-AN-N-A-AN-N-A-A-N-N, where
|
|
222
|
+
# A excludes S, L, O, I, B and Z (CMS drops them to avoid digit confusion) and
|
|
223
|
+
# the first character is 1-9. Hyphens are allowed where CMS prints them.
|
|
224
|
+
_MBI_ALPHA = "ACDEFGHJKMNPQRTUVWXY"
|
|
225
|
+
_MBI_RE = re.compile(
|
|
226
|
+
rf"(?<![A-Za-z0-9])"
|
|
227
|
+
rf"[1-9][{_MBI_ALPHA}][{_MBI_ALPHA}0-9]\d-?"
|
|
228
|
+
rf"[{_MBI_ALPHA}][{_MBI_ALPHA}0-9]\d-?"
|
|
229
|
+
rf"[{_MBI_ALPHA}][{_MBI_ALPHA}]\d\d"
|
|
230
|
+
rf"(?![A-Za-z0-9])",
|
|
231
|
+
re.IGNORECASE,
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
# X12 and HL7 v2: a SINGLE strong marker now trips. Requiring two ANDed markers
|
|
235
|
+
# was finding C3's biggest hole -- a bare `PID|` segment is the most common leak
|
|
236
|
+
# shape there is, and it carried name, MRN, DOB, sex and address.
|
|
237
|
+
_X12_RE = re.compile(
|
|
238
|
+
r"(?:\bISA\*|\bGS\*(HC|HB|HI|HN|HP|HR|HS)\b|\bST\*(837|835|270|271|276|277|278)\b"
|
|
239
|
+
r"|\bNM1\*IL\*|\bDMG\*D8\*)"
|
|
240
|
+
)
|
|
241
|
+
_HL7V2_SEGMENTS = "MSH|PID|PV1|OBX|OBR|NK1|IN1|IN2|DG1|EVN|ORC|AL1|GT1"
|
|
242
|
+
_HL7V2_RE = re.compile(rf"(?:\A|[\r\n\s>:])(?:{_HL7V2_SEGMENTS})\|")
|
|
243
|
+
|
|
244
|
+
# Free-text patient record: a patient-record word AND a date, together.
|
|
245
|
+
_FREE_TEXT_SUBJECT_TOKENS = {
|
|
246
|
+
"mrn", "dob", "ssn", "patient", "member", "beneficiary", "subscriber", "hicn", "mbi",
|
|
247
|
+
}
|
|
248
|
+
_FREE_TEXT_DATE_RE = re.compile(
|
|
249
|
+
r"(?<!\d)(?:\d{1,2}[/.-]\d{1,2}[/.-]\d{2,4}|\d{4}[/.-]\d{1,2}[/.-]\d{1,2}|\d{8})(?!\d)"
|
|
250
|
+
)
|
|
251
|
+
|
|
252
|
+
# Field-name patterns. Matched against TOKENS of a parameter key, never as bare
|
|
253
|
+
# substrings: "geographic" contains "phi", "adobe" contains "dob", "outpatient"
|
|
254
|
+
# contains "patient", and all three are ordinary product input.
|
|
255
|
+
_STRONG_KEY_TOKENS: dict[str, str] = {
|
|
256
|
+
"ssn": "SSN",
|
|
257
|
+
"mrn": "MEDICAL_RECORD_NUMBER",
|
|
258
|
+
"phi": "PHI",
|
|
259
|
+
"dob": "DOB",
|
|
260
|
+
"birthdate": "DOB",
|
|
261
|
+
"birthday": "DOB",
|
|
262
|
+
"birthyear": "DOB",
|
|
263
|
+
"yob": "DOB",
|
|
264
|
+
"patient": "PATIENT_RECORD",
|
|
265
|
+
"hicn": "MEMBER_IDENTIFIER",
|
|
266
|
+
"mbi": "MEMBER_IDENTIFIER",
|
|
267
|
+
}
|
|
268
|
+
#: Token PAIRS. "beneficiary notice of noncoverage" is a real CMS document type,
|
|
269
|
+
#: so "beneficiary" alone must not refuse -- "beneficiary_id" must.
|
|
270
|
+
_SUBJECT_TOKENS = {"patient", "member", "subscriber", "beneficiary", "enrollee", "insured"}
|
|
271
|
+
_IDENTIFIER_TOKENS = {
|
|
272
|
+
"id", "ids", "identifier", "number", "num", "no", "nbr", "mrn", "ssn", "dob",
|
|
273
|
+
"name", "record", "chart", "account", "acct",
|
|
274
|
+
}
|
|
275
|
+
_PAIR_LABELS = {
|
|
276
|
+
"patient": "PATIENT_RECORD",
|
|
277
|
+
"member": "MEMBER_IDENTIFIER",
|
|
278
|
+
"subscriber": "MEMBER_IDENTIFIER",
|
|
279
|
+
"beneficiary": "MEMBER_IDENTIFIER",
|
|
280
|
+
"enrollee": "MEMBER_IDENTIFIER",
|
|
281
|
+
"insured": "MEMBER_IDENTIFIER",
|
|
282
|
+
}
|
|
283
|
+
#: Multi-word field names, matched against the key's token SEQUENCE.
|
|
284
|
+
_PHRASE_KEY_TOKENS: tuple[tuple[tuple[str, ...], str], ...] = (
|
|
285
|
+
(("social", "security"), "SSN"),
|
|
286
|
+
(("medical", "record"), "MEDICAL_RECORD_NUMBER"),
|
|
287
|
+
(("date", "of", "birth"), "DOB"),
|
|
288
|
+
(("birth", "date"), "DOB"),
|
|
289
|
+
(("birth", "year"), "DOB"),
|
|
290
|
+
(("date", "of", "service"), "PATIENT_RECORD"),
|
|
291
|
+
)
|
|
292
|
+
|
|
293
|
+
_DOB_TOKENS = {"dob", "birthdate", "birthday", "birthyear", "yob"}
|
|
294
|
+
_DOB_PHRASES = ((("date", "of", "birth"),), (("birth", "date"),), (("birth", "year"),))
|
|
295
|
+
_NAME_TOKENS = {"firstname", "lastname", "familyname", "givenname", "middlename", "fullname"}
|
|
296
|
+
_NAME_PHRASES = (
|
|
297
|
+
("first", "name"),
|
|
298
|
+
("last", "name"),
|
|
299
|
+
("family", "name"),
|
|
300
|
+
("given", "name"),
|
|
301
|
+
("middle", "name"),
|
|
302
|
+
("full", "name"),
|
|
303
|
+
("patient", "name"),
|
|
304
|
+
)
|
|
305
|
+
_DATE_VALUE_RE = re.compile(r"^\s*(\d{4}-\d{2}-\d{2}|\d{2}/\d{2}/\d{4}|\d{4})\s*$")
|
|
306
|
+
|
|
307
|
+
_TOKEN_SPLIT_RE = re.compile(r"[^a-z0-9]+")
|
|
308
|
+
_CAMEL_SPLIT_RE = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
|
|
309
|
+
_BASE64_RUN_RE = re.compile(r"[A-Za-z0-9+/]{%d,}={0,2}" % MIN_ENCODED_RUN)
|
|
310
|
+
_PERCENT_RE = re.compile(r"%[0-9A-Fa-f]{2}")
|
|
311
|
+
|
|
312
|
+
|
|
313
|
+
def _key_tokens(key: str) -> tuple[list[str], str]:
|
|
314
|
+
"""Return ``(tokens, condensed)`` for a parameter key.
|
|
315
|
+
|
|
316
|
+
``tokens`` splits on separators and camelCase humps, so ``patientMRN`` and
|
|
317
|
+
``patient_mrn`` both yield ``["patient", "mrn"]`` while ``outpatient`` yields
|
|
318
|
+
``["outpatient"]`` and is left alone. ``condensed`` is the alphanumeric-only
|
|
319
|
+
form, used only to defeat separator obfuscation (``p_a_t_i_e_n_t``).
|
|
320
|
+
"""
|
|
321
|
+
normalized = normalize_for_matching(key)
|
|
322
|
+
humped = _CAMEL_SPLIT_RE.sub(" ", normalized).lower()
|
|
323
|
+
tokens = [t for t in _TOKEN_SPLIT_RE.split(humped) if t]
|
|
324
|
+
condensed = "".join(ch for ch in normalized.lower() if ch.isalnum())
|
|
325
|
+
return tokens, condensed
|
|
326
|
+
|
|
327
|
+
|
|
328
|
+
class PHIRejected(ValueError):
|
|
329
|
+
"""Raised when a tool call's parameters look like a patient record."""
|
|
330
|
+
|
|
331
|
+
def __init__(self, tool: str, reasons: list[str]) -> None:
|
|
332
|
+
self.tool = tool
|
|
333
|
+
self.reasons = list(reasons)
|
|
334
|
+
joined = "; ".join(self.reasons)
|
|
335
|
+
super().__init__(
|
|
336
|
+
f"SourceLock refused the call to {tool!r}: parameters look like protected health "
|
|
337
|
+
f"information. {PHI_NON_CLAIM} Reasons: {joined}"
|
|
338
|
+
)
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
# ---------------------------------------------------------------------------
|
|
342
|
+
# Value-free parameter rejection
|
|
343
|
+
# ---------------------------------------------------------------------------
|
|
344
|
+
#
|
|
345
|
+
# Round-1 finding C1: a custom pydantic validator that quoted its input put the
|
|
346
|
+
# raw value into ``error["msg"]``, and every consumer built its message from
|
|
347
|
+
# ``msg``. The consumers were value-blind by assumption, not by construction.
|
|
348
|
+
#
|
|
349
|
+
# The fix is a boundary, not a patch: ``ToolSpec.invoke`` converts every
|
|
350
|
+
# ``ValidationError`` into a :class:`ParamsRejected` whose text is built from
|
|
351
|
+
# ``loc`` plus a message that has been scrubbed of the input value. Nothing
|
|
352
|
+
# downstream -- CLI, MCP, or a library consumer that simply does ``str(exc)`` --
|
|
353
|
+
# can reach the raw value, and a future adapter that quotes its input is
|
|
354
|
+
# scrubbed on the way out instead of reopening the hole.
|
|
355
|
+
|
|
356
|
+
REDACTED = "[value withheld]"
|
|
357
|
+
|
|
358
|
+
#: Below this length a value is not distinctive enough to be worth redacting,
|
|
359
|
+
#: and redacting it would mangle ordinary words in rule text ("ncd", "lcd").
|
|
360
|
+
_MIN_REDACTABLE = 3
|
|
361
|
+
|
|
362
|
+
#: A label safe to print back at whoever sent it: an ordinary identifier. Field
|
|
363
|
+
#: names, tool names and pydantic error locations all take this shape, and
|
|
364
|
+
#: nothing that carries data does.
|
|
365
|
+
_SAFE_LABEL_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,39}$")
|
|
366
|
+
|
|
367
|
+
#: Six digits in a row. Long enough to exclude every field name and version
|
|
368
|
+
#: suffix this product uses, short enough to catch anything that identifies a
|
|
369
|
+
#: person or an entity.
|
|
370
|
+
_LONG_DIGIT_RUN_RE = re.compile(r"\d{6,}")
|
|
371
|
+
|
|
372
|
+
|
|
373
|
+
def safe_label(text: str, index: int | None = None) -> str:
|
|
374
|
+
"""Echo ``text`` only if it is an identifier; otherwise a positional stand-in.
|
|
375
|
+
|
|
376
|
+
Round-1 closed the value-echo hole for parameter VALUES. Keys, tool names and
|
|
377
|
+
pydantic ``loc`` entries were still printed verbatim -- and every one of them
|
|
378
|
+
is caller-controlled. ``{"123-45-6789": "x"}`` against a model with
|
|
379
|
+
``extra='forbid'`` produced a rejection reason and a pydantic ``loc`` that
|
|
380
|
+
both quoted the SSN: the refusal itself became the leak.
|
|
381
|
+
|
|
382
|
+
An identifier-shaped label is kept because a caller genuinely needs to know
|
|
383
|
+
WHICH parameter was wrong. Anything else is data wearing a name, and it is
|
|
384
|
+
replaced with its position.
|
|
385
|
+
"""
|
|
386
|
+
if _SAFE_LABEL_RE.match(text) and not _looks_like_an_identifier_value(text):
|
|
387
|
+
return text
|
|
388
|
+
return f"<key #{index}>" if index is not None else "<withheld>"
|
|
389
|
+
|
|
390
|
+
|
|
391
|
+
def _looks_like_an_identifier_value(text: str) -> bool:
|
|
392
|
+
"""True when an identifier-shaped string still carries an identifier inside it.
|
|
393
|
+
|
|
394
|
+
``ssn123456789`` passes ``_SAFE_LABEL_RE`` and is still nine digits of
|
|
395
|
+
somebody's identity. The long-digit-run rule generalises past SSN and MBI:
|
|
396
|
+
no field name in this product carries six consecutive digits, while every
|
|
397
|
+
identifier worth withholding does -- SSN, NPI, MBI, a chart number.
|
|
398
|
+
``icd10cm_fy2026`` and ``v28`` are unaffected.
|
|
399
|
+
"""
|
|
400
|
+
normalized = normalize_for_matching(text)
|
|
401
|
+
return bool(
|
|
402
|
+
_LONG_DIGIT_RUN_RE.search(normalized)
|
|
403
|
+
or _SSN_SEPARATED_RE.search(normalized)
|
|
404
|
+
or _MBI_RE.search(normalized)
|
|
405
|
+
)
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
#: A route name: ``source.tool``, both halves lowercase slugs. Every shipped
|
|
409
|
+
#: tool matches; nothing that carries data does.
|
|
410
|
+
_TOOL_NAME_RE = re.compile(r"^[a-z][a-z0-9_]{0,31}\.[a-z][a-z0-9_]{0,31}$")
|
|
411
|
+
|
|
412
|
+
|
|
413
|
+
def safe_tool_name(name: str) -> str:
|
|
414
|
+
"""Echo a requested tool name only if it is shaped like one.
|
|
415
|
+
|
|
416
|
+
An MCP client picks the string, and "no tool named X" reflected X back
|
|
417
|
+
verbatim -- so a client could get an arbitrary value written into a server
|
|
418
|
+
log by asking for it as a tool. A real name is useful in the error; anything
|
|
419
|
+
else tells the caller nothing they did not already know.
|
|
420
|
+
"""
|
|
421
|
+
return name if _TOOL_NAME_RE.match(name) else "<name withheld>"
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def safe_location(loc: Any) -> str:
|
|
425
|
+
"""A pydantic ``loc`` tuple rendered without echoing caller-controlled names.
|
|
426
|
+
|
|
427
|
+
With ``extra='forbid'`` the ``loc`` of an unexpected-field error IS the key
|
|
428
|
+
the caller invented, so this is the same hole as :func:`safe_label` in a
|
|
429
|
+
different coat.
|
|
430
|
+
"""
|
|
431
|
+
parts = loc if isinstance(loc, (tuple, list)) else (loc,)
|
|
432
|
+
rendered = [
|
|
433
|
+
str(p) if isinstance(p, int) else safe_label(str(p), i)
|
|
434
|
+
for i, p in enumerate(parts)
|
|
435
|
+
]
|
|
436
|
+
return ".".join(rendered) or "<params>"
|
|
437
|
+
|
|
438
|
+
|
|
439
|
+
class ParamsRejected(ValueError):
|
|
440
|
+
"""Parameters did not fit a tool's declared model.
|
|
441
|
+
|
|
442
|
+
Carries field locations and rule text only. The offending value is never
|
|
443
|
+
stored on the exception, so no consumer can echo it -- not by printing the
|
|
444
|
+
exception, not by walking ``errors()``, and not from a traceback (the
|
|
445
|
+
originating ``ValidationError`` is deliberately not chained).
|
|
446
|
+
"""
|
|
447
|
+
|
|
448
|
+
def __init__(self, tool: str, problems: list[tuple[str, str]]) -> None:
|
|
449
|
+
self.tool = tool
|
|
450
|
+
self.problems = list(problems)
|
|
451
|
+
joined = "; ".join(f"{loc}: {msg}" for loc, msg in self.problems) or "invalid parameters"
|
|
452
|
+
super().__init__(f"{tool}: invalid parameters -- {joined}")
|
|
453
|
+
|
|
454
|
+
def errors(self) -> list[dict[str, str]]:
|
|
455
|
+
"""Value-free stand-in for ``ValidationError.errors()``."""
|
|
456
|
+
return [{"loc": loc, "msg": msg} for loc, msg in self.problems]
|
|
457
|
+
|
|
458
|
+
|
|
459
|
+
def _leaf_strings(value: Any, depth: int = 0) -> Iterable[str]:
|
|
460
|
+
"""Yield every string-ish leaf of ``value`` that is worth redacting."""
|
|
461
|
+
if depth > MAX_DEPTH:
|
|
462
|
+
return
|
|
463
|
+
if isinstance(value, str):
|
|
464
|
+
if len(value) >= _MIN_REDACTABLE:
|
|
465
|
+
yield value
|
|
466
|
+
elif isinstance(value, Mapping):
|
|
467
|
+
for k, v in value.items():
|
|
468
|
+
yield from _leaf_strings(k, depth + 1)
|
|
469
|
+
yield from _leaf_strings(v, depth + 1)
|
|
470
|
+
elif isinstance(value, (list, tuple, set)):
|
|
471
|
+
for v in value:
|
|
472
|
+
yield from _leaf_strings(v, depth + 1)
|
|
473
|
+
elif isinstance(value, (int, float)) and not isinstance(value, bool):
|
|
474
|
+
text = str(value)
|
|
475
|
+
if len(text) >= _MIN_REDACTABLE:
|
|
476
|
+
yield text
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
def redact_input(text: str, value: Any) -> str:
|
|
480
|
+
"""Remove every occurrence of ``value`` (or any of its leaves) from ``text``.
|
|
481
|
+
|
|
482
|
+
Case-insensitive, because a value is often upper/lower-cased on its way into
|
|
483
|
+
a message. Longest leaves first so a nested value is not half-redacted.
|
|
484
|
+
"""
|
|
485
|
+
leaves = sorted(set(_leaf_strings(value)), key=len, reverse=True)
|
|
486
|
+
for leaf in leaves:
|
|
487
|
+
text = re.sub(re.escape(leaf), REDACTED, text, flags=re.IGNORECASE)
|
|
488
|
+
return text
|
|
489
|
+
|
|
490
|
+
|
|
491
|
+
def safe_validation_reasons(exc: Any) -> list[tuple[str, str]]:
|
|
492
|
+
"""Reduce a pydantic ``ValidationError`` to ``(location, rule)`` pairs.
|
|
493
|
+
|
|
494
|
+
``msg`` is used because pydantic's own rule text is genuinely useful ("String
|
|
495
|
+
should match pattern ..."), but it is scrubbed against ``input`` first: a
|
|
496
|
+
custom validator is free to quote its value and still cannot leak it.
|
|
497
|
+
"""
|
|
498
|
+
errors = getattr(exc, "errors", None)
|
|
499
|
+
if not callable(errors):
|
|
500
|
+
return [("<params>", type(exc).__name__)]
|
|
501
|
+
problems: list[tuple[str, str]] = []
|
|
502
|
+
for error in errors():
|
|
503
|
+
location = safe_location(error.get("loc", ()))
|
|
504
|
+
message = str(error.get("msg", "invalid"))
|
|
505
|
+
if "input" in error:
|
|
506
|
+
message = redact_input(message, error["input"])
|
|
507
|
+
problems.append((location, message))
|
|
508
|
+
return problems
|
|
509
|
+
|
|
510
|
+
|
|
511
|
+
def safe_validation_message(exc: Any) -> str:
|
|
512
|
+
"""One line of ``field: rule`` pairs, with no input values in it."""
|
|
513
|
+
return "; ".join(f"{loc}: {msg}" for loc, msg in safe_validation_reasons(exc)) or (
|
|
514
|
+
"invalid parameters"
|
|
515
|
+
)
|
|
516
|
+
|
|
517
|
+
|
|
518
|
+
# ---------------------------------------------------------------------------
|
|
519
|
+
# Entry points
|
|
520
|
+
# ---------------------------------------------------------------------------
|
|
521
|
+
|
|
522
|
+
|
|
523
|
+
def allowed_detectors() -> set[str]:
|
|
524
|
+
"""Detector codes the operator has chosen to accept, from the environment."""
|
|
525
|
+
raw = os.environ.get(GUARD_ALLOW_ENV, "")
|
|
526
|
+
return {part.strip().upper() for part in raw.split(",") if part.strip()}
|
|
527
|
+
|
|
528
|
+
|
|
529
|
+
def assert_public_params(params: Mapping[str, Any] | Any, *, tool: str) -> None:
|
|
530
|
+
"""Raise :class:`PHIRejected` unless ``params`` are public typed parameters.
|
|
531
|
+
|
|
532
|
+
Honours ``HC_SOURCE_GUARD_ALLOW``: reasons whose detector code appears there
|
|
533
|
+
are not grounds for refusal. Everything else still is -- suppressing one
|
|
534
|
+
detector never disables the guard.
|
|
535
|
+
"""
|
|
536
|
+
reasons = scan_for_phi(params)
|
|
537
|
+
allowed = allowed_detectors()
|
|
538
|
+
if allowed:
|
|
539
|
+
reasons = [r for r in reasons if r.split(":", 1)[0].strip().upper() not in allowed]
|
|
540
|
+
if reasons:
|
|
541
|
+
raise PHIRejected(tool, reasons)
|
|
542
|
+
return None
|
|
543
|
+
|
|
544
|
+
|
|
545
|
+
def scan_for_phi(params: Mapping[str, Any] | Any) -> list[str]:
|
|
546
|
+
"""Return a list of reasons ``params`` look patient-shaped (empty == clean).
|
|
547
|
+
|
|
548
|
+
A pure detector: never reads the environment, never suppresses anything.
|
|
549
|
+
The operator allowlist is applied by :func:`assert_public_params`, so this
|
|
550
|
+
function always tells an auditor the truth.
|
|
551
|
+
|
|
552
|
+
Reasons never contain the offending value.
|
|
553
|
+
"""
|
|
554
|
+
reasons: list[str] = []
|
|
555
|
+
_check_size(params, reasons)
|
|
556
|
+
if reasons:
|
|
557
|
+
return reasons
|
|
558
|
+
|
|
559
|
+
budget = _Budget()
|
|
560
|
+
for path, key, value in _walk(params, reasons, budget=budget):
|
|
561
|
+
if key is not None:
|
|
562
|
+
_check_key(key, path, reasons)
|
|
563
|
+
if isinstance(value, Mapping):
|
|
564
|
+
_check_dob_name_combo(value, path, reasons)
|
|
565
|
+
_check_value(value, path, reasons, budget=budget)
|
|
566
|
+
|
|
567
|
+
# Stable, de-duplicated order.
|
|
568
|
+
seen: set[str] = set()
|
|
569
|
+
unique: list[str] = []
|
|
570
|
+
for r in reasons:
|
|
571
|
+
if r not in seen:
|
|
572
|
+
seen.add(r)
|
|
573
|
+
unique.append(r)
|
|
574
|
+
return unique
|
|
575
|
+
|
|
576
|
+
|
|
577
|
+
class _Budget:
|
|
578
|
+
"""Shared work counters for one scan. Exhaustion is a refusal, not a pass."""
|
|
579
|
+
|
|
580
|
+
def __init__(self) -> None:
|
|
581
|
+
self.nodes = 0
|
|
582
|
+
self.exhausted = False
|
|
583
|
+
|
|
584
|
+
|
|
585
|
+
def _check_size(params: Any, reasons: list[str]) -> None:
|
|
586
|
+
try:
|
|
587
|
+
size = len(json.dumps(params, default=str))
|
|
588
|
+
except (TypeError, ValueError):
|
|
589
|
+
return
|
|
590
|
+
if size > MAX_SERIALIZED_BYTES:
|
|
591
|
+
reasons.append(
|
|
592
|
+
f"OVERSIZED_PARAMS: parameters serialize to {size} bytes (limit "
|
|
593
|
+
f"{MAX_SERIALIZED_BYTES}); SourceLock tools take small typed public parameters, "
|
|
594
|
+
"not documents"
|
|
595
|
+
)
|
|
596
|
+
|
|
597
|
+
|
|
598
|
+
def _walk(
|
|
599
|
+
node: Any,
|
|
600
|
+
reasons: list[str],
|
|
601
|
+
path: str = "params",
|
|
602
|
+
depth: int = 0,
|
|
603
|
+
*,
|
|
604
|
+
budget: _Budget,
|
|
605
|
+
) -> Iterable[tuple[str, str | None, Any]]:
|
|
606
|
+
"""Iteratively yield ``(path, key, value)`` for every node in the structure."""
|
|
607
|
+
stack: list[tuple[str, str | None, Any, int]] = [(path, None, node, depth)]
|
|
608
|
+
while stack:
|
|
609
|
+
cur_path, cur_key, cur_value, cur_depth = stack.pop()
|
|
610
|
+
if cur_depth > MAX_DEPTH:
|
|
611
|
+
reasons.append(
|
|
612
|
+
f"UNEXPECTED_STRUCTURE: parameters nest deeper than {MAX_DEPTH} levels at "
|
|
613
|
+
f"{cur_path}; public typed parameters are flat"
|
|
614
|
+
)
|
|
615
|
+
continue
|
|
616
|
+
|
|
617
|
+
budget.nodes += 1
|
|
618
|
+
if budget.nodes > MAX_NODES:
|
|
619
|
+
if not budget.exhausted:
|
|
620
|
+
budget.exhausted = True
|
|
621
|
+
reasons.append(
|
|
622
|
+
f"UNEXPECTED_STRUCTURE: parameters contain more than {MAX_NODES} values; "
|
|
623
|
+
"SourceLock tools take a handful of small typed public parameters, so an "
|
|
624
|
+
"input this large is refused rather than partially inspected"
|
|
625
|
+
)
|
|
626
|
+
return
|
|
627
|
+
|
|
628
|
+
yield cur_path, cur_key, cur_value
|
|
629
|
+
|
|
630
|
+
if isinstance(cur_value, Mapping):
|
|
631
|
+
for i, (k, v) in enumerate(cur_value.items()):
|
|
632
|
+
# The KEY is caller-controlled too. `{"123-45-6789": "x"}` used
|
|
633
|
+
# to produce a refusal reason that quoted the SSN back in the
|
|
634
|
+
# path -- the guard refusing a value by repeating it.
|
|
635
|
+
stack.append(
|
|
636
|
+
(f"{cur_path}.{safe_label(str(k), i)}", str(k), v, cur_depth + 1)
|
|
637
|
+
)
|
|
638
|
+
elif isinstance(cur_value, (list, tuple, set)):
|
|
639
|
+
for i, v in enumerate(cur_value):
|
|
640
|
+
stack.append((f"{cur_path}[{i}]", None, v, cur_depth + 1))
|
|
641
|
+
|
|
642
|
+
|
|
643
|
+
def _check_key(key: str, path: str, reasons: list[str]) -> None:
|
|
644
|
+
tokens, condensed = _key_tokens(key)
|
|
645
|
+
token_set = set(tokens)
|
|
646
|
+
|
|
647
|
+
for token, label in _STRONG_KEY_TOKENS.items():
|
|
648
|
+
if token in token_set:
|
|
649
|
+
reasons.append(
|
|
650
|
+
f"PATIENT_SHAPED_KEY: parameter at {path} is named after a patient-record "
|
|
651
|
+
f"field ({label})"
|
|
652
|
+
)
|
|
653
|
+
|
|
654
|
+
for subject in _SUBJECT_TOKENS & token_set:
|
|
655
|
+
if token_set & _IDENTIFIER_TOKENS:
|
|
656
|
+
reasons.append(
|
|
657
|
+
f"PATIENT_SHAPED_KEY: parameter at {path} names a person plus an identifier "
|
|
658
|
+
f"({_PAIR_LABELS[subject]})"
|
|
659
|
+
)
|
|
660
|
+
|
|
661
|
+
joined = " ".join(tokens)
|
|
662
|
+
for phrase, label in _PHRASE_KEY_TOKENS:
|
|
663
|
+
if " ".join(phrase) in joined:
|
|
664
|
+
reasons.append(
|
|
665
|
+
f"PATIENT_SHAPED_KEY: parameter at {path} is named after a patient-record "
|
|
666
|
+
f"field ({label})"
|
|
667
|
+
)
|
|
668
|
+
|
|
669
|
+
# Separator obfuscation only: `p_a_t_i_e_n_t` splits into single characters,
|
|
670
|
+
# which no ordinary field name does. Condensing `outpatient_setting` would
|
|
671
|
+
# false-positive, so the condensed form is consulted for THIS shape alone.
|
|
672
|
+
if sum(1 for t in tokens if len(t) == 1) >= 3:
|
|
673
|
+
for token, label in _STRONG_KEY_TOKENS.items():
|
|
674
|
+
if token in condensed:
|
|
675
|
+
reasons.append(
|
|
676
|
+
f"OBFUSCATED_KEY: parameter at {path} spells a patient-record field name "
|
|
677
|
+
f"through separators ({label})"
|
|
678
|
+
)
|
|
679
|
+
|
|
680
|
+
|
|
681
|
+
def _check_value(value: Any, path: str, reasons: list[str], *, budget: _Budget) -> None:
|
|
682
|
+
if isinstance(value, Mapping):
|
|
683
|
+
_check_fhir_object(value, path, reasons)
|
|
684
|
+
return
|
|
685
|
+
|
|
686
|
+
if not isinstance(value, str):
|
|
687
|
+
return
|
|
688
|
+
|
|
689
|
+
_scan_text(value, path, reasons, budget=budget, decode_depth=0)
|
|
690
|
+
|
|
691
|
+
|
|
692
|
+
def _check_fhir_object(value: Mapping[str, Any], path: str, reasons: list[str]) -> None:
|
|
693
|
+
"""FHIR detection that does not depend on ``resourceType``.
|
|
694
|
+
|
|
695
|
+
Finding C3: an object carrying ``name`` and ``gender`` but no
|
|
696
|
+
``resourceType`` was invisible, because detection keyed entirely off that
|
|
697
|
+
one field.
|
|
698
|
+
"""
|
|
699
|
+
resource = value.get("resourceType")
|
|
700
|
+
if isinstance(resource, str):
|
|
701
|
+
reasons.append(
|
|
702
|
+
f"FHIR_RESOURCE: object at {path} carries resourceType "
|
|
703
|
+
f"{'(patient-shaped)' if resource in _FHIR_PATIENT_RESOURCES else '(FHIR)'}; "
|
|
704
|
+
"SourceLock does not accept FHIR payloads"
|
|
705
|
+
)
|
|
706
|
+
return
|
|
707
|
+
|
|
708
|
+
keys = {str(k).lower() for k in value}
|
|
709
|
+
strong = keys & _FHIR_STRONG_KEYS
|
|
710
|
+
weak = keys & _FHIR_WEAK_KEYS
|
|
711
|
+
if strong and len(strong | weak) >= 2:
|
|
712
|
+
reasons.append(
|
|
713
|
+
f"PATIENT_RESOURCE_SHAPE: object at {path} carries person-record fields "
|
|
714
|
+
"(demographics plus an identifier or name) without declaring a resourceType; "
|
|
715
|
+
"SourceLock does not accept patient records"
|
|
716
|
+
)
|
|
717
|
+
|
|
718
|
+
|
|
719
|
+
def _scan_text(value: str, path: str, reasons: list[str], *, budget: _Budget, decode_depth: int) -> None:
|
|
720
|
+
"""Run every text detector, then probe encodings and embedded structures."""
|
|
721
|
+
text = normalize_for_matching(value)
|
|
722
|
+
|
|
723
|
+
if _SSN_SEPARATED_RE.search(text) or _SSN_BARE_RE.search(text):
|
|
724
|
+
reasons.append(f"SSN_PATTERN: value at {path} contains an SSN-shaped token")
|
|
725
|
+
|
|
726
|
+
if _MBI_RE.search(text):
|
|
727
|
+
reasons.append(
|
|
728
|
+
f"MBI_PATTERN: value at {path} contains a Medicare Beneficiary Identifier-shaped "
|
|
729
|
+
"token (11 characters in the CMS C-A-AN-N-A-AN-N-A-A-N-N form)"
|
|
730
|
+
)
|
|
731
|
+
|
|
732
|
+
if _X12_RE.search(text):
|
|
733
|
+
reasons.append(f"X12_ENVELOPE: value at {path} contains an X12 segment marker")
|
|
734
|
+
|
|
735
|
+
if _HL7V2_RE.search(text):
|
|
736
|
+
reasons.append(f"HL7V2_MESSAGE: value at {path} contains an HL7 v2 segment header")
|
|
737
|
+
|
|
738
|
+
_check_free_text_record(text, path, reasons)
|
|
739
|
+
|
|
740
|
+
if decode_depth >= MAX_DECODE_DEPTH:
|
|
741
|
+
# Do not give up quietly: if there is still something decodable here, the
|
|
742
|
+
# input is more encoded than any public typed parameter has cause to be.
|
|
743
|
+
if _looks_encoded(text):
|
|
744
|
+
reasons.append(
|
|
745
|
+
f"ENCODED_PAYLOAD: value at {path} is encoded more than {MAX_DECODE_DEPTH} "
|
|
746
|
+
"layers deep; SourceLock takes small typed public parameters, not encoded "
|
|
747
|
+
"documents"
|
|
748
|
+
)
|
|
749
|
+
return
|
|
750
|
+
|
|
751
|
+
_scan_embedded_json(value, path, reasons, budget=budget, decode_depth=decode_depth)
|
|
752
|
+
_scan_encoded(value, path, reasons, budget=budget, decode_depth=decode_depth)
|
|
753
|
+
|
|
754
|
+
|
|
755
|
+
def _check_free_text_record(text: str, path: str, reasons: list[str]) -> None:
|
|
756
|
+
"""A patient-record word next to a date is a patient record in prose.
|
|
757
|
+
|
|
758
|
+
``MRN 88213 DOE JOHN DOB 01/02/1980`` had no structural envelope at all and
|
|
759
|
+
walked straight through round 1. Either half alone passes: "dobutamine" is
|
|
760
|
+
not "dob", and a date on its own is just a date.
|
|
761
|
+
"""
|
|
762
|
+
tokens = {t for t in _TOKEN_SPLIT_RE.split(text.lower()) if t}
|
|
763
|
+
if not (tokens & _FREE_TEXT_SUBJECT_TOKENS):
|
|
764
|
+
return
|
|
765
|
+
if _FREE_TEXT_DATE_RE.search(text):
|
|
766
|
+
reasons.append(
|
|
767
|
+
f"PATIENT_RECORD_TEXT: value at {path} combines a patient-record term "
|
|
768
|
+
"(MRN/DOB/SSN/patient/member) with a date, which identifies an individual"
|
|
769
|
+
)
|
|
770
|
+
|
|
771
|
+
|
|
772
|
+
def _looks_encoded(text: str) -> bool:
|
|
773
|
+
return bool(_PERCENT_RE.search(text)) or bool(_BASE64_RUN_RE.search(text))
|
|
774
|
+
|
|
775
|
+
|
|
776
|
+
def _scan_embedded_json(
|
|
777
|
+
value: str, path: str, reasons: list[str], *, budget: _Budget, decode_depth: int
|
|
778
|
+
) -> None:
|
|
779
|
+
"""Parse JSON found ANYWHERE in the string, not only at position 0.
|
|
780
|
+
|
|
781
|
+
Finding C3: ``"payload: " + <FHIR Bundle>`` passed because the scan only
|
|
782
|
+
fired when the first non-space character was ``{`` or ``[``.
|
|
783
|
+
"""
|
|
784
|
+
if len(value) > MAX_SERIALIZED_BYTES:
|
|
785
|
+
return
|
|
786
|
+
|
|
787
|
+
stripped = value.strip()
|
|
788
|
+
if stripped[:1] == '"':
|
|
789
|
+
# Double-encoded JSON: the outer parse yields a string that is itself
|
|
790
|
+
# JSON. One `json.loads` and a rescan, rather than one and done.
|
|
791
|
+
try:
|
|
792
|
+
inner = json.loads(stripped)
|
|
793
|
+
except (ValueError, RecursionError):
|
|
794
|
+
inner = None
|
|
795
|
+
if isinstance(inner, str):
|
|
796
|
+
_scan_text(inner, path, reasons, budget=budget, decode_depth=decode_depth + 1)
|
|
797
|
+
return
|
|
798
|
+
|
|
799
|
+
decoder = json.JSONDecoder()
|
|
800
|
+
candidates = 0
|
|
801
|
+
for match in re.finditer(r"[{\[]", value):
|
|
802
|
+
candidates += 1
|
|
803
|
+
if candidates > MAX_EMBEDDED_CANDIDATES:
|
|
804
|
+
return
|
|
805
|
+
try:
|
|
806
|
+
parsed, _ = decoder.raw_decode(value, match.start())
|
|
807
|
+
except (ValueError, RecursionError):
|
|
808
|
+
continue
|
|
809
|
+
if not isinstance(parsed, (Mapping, list)):
|
|
810
|
+
continue
|
|
811
|
+
for reason in scan_for_phi(parsed):
|
|
812
|
+
reasons.append(f"EMBEDDED_JSON at {path}: {reason}")
|
|
813
|
+
return
|
|
814
|
+
|
|
815
|
+
|
|
816
|
+
def _scan_encoded(
|
|
817
|
+
value: str, path: str, reasons: list[str], *, budget: _Budget, decode_depth: int
|
|
818
|
+
) -> None:
|
|
819
|
+
"""Decode base64 and percent-encoded runs and rescan what comes out.
|
|
820
|
+
|
|
821
|
+
Only findings from the DECODED text are reported, so a name that happens to
|
|
822
|
+
look like base64 costs one decode attempt and nothing else.
|
|
823
|
+
"""
|
|
824
|
+
if _PERCENT_RE.search(value):
|
|
825
|
+
decoded = unquote(value)
|
|
826
|
+
if decoded != value:
|
|
827
|
+
_scan_text(decoded, path, reasons, budget=budget, decode_depth=decode_depth + 1)
|
|
828
|
+
|
|
829
|
+
runs = 0
|
|
830
|
+
for match in _BASE64_RUN_RE.finditer(value):
|
|
831
|
+
runs += 1
|
|
832
|
+
if runs > MAX_ENCODED_RUNS:
|
|
833
|
+
return
|
|
834
|
+
blob = match.group(0)
|
|
835
|
+
try:
|
|
836
|
+
raw = base64.b64decode(blob + "=" * (-len(blob) % 4), validate=True)
|
|
837
|
+
except (binascii.Error, ValueError):
|
|
838
|
+
continue
|
|
839
|
+
try:
|
|
840
|
+
decoded = raw.decode("utf-8")
|
|
841
|
+
except UnicodeDecodeError:
|
|
842
|
+
continue
|
|
843
|
+
if not decoded.isprintable() and not any(c in decoded for c in "\r\n\t"):
|
|
844
|
+
continue
|
|
845
|
+
_scan_text(decoded, path, reasons, budget=budget, decode_depth=decode_depth + 1)
|
|
846
|
+
|
|
847
|
+
|
|
848
|
+
def _check_dob_name_combo(params: Mapping[str, Any], path: str, reasons: list[str]) -> None:
|
|
849
|
+
"""A date of birth beside a personal name identifies an individual.
|
|
850
|
+
|
|
851
|
+
Runs on every mapping the walk reaches, not just the top level, so it also
|
|
852
|
+
fires inside decoded or embedded JSON.
|
|
853
|
+
"""
|
|
854
|
+
has_dob = False
|
|
855
|
+
has_name = False
|
|
856
|
+
for key, value in params.items():
|
|
857
|
+
tokens, _ = _key_tokens(str(key))
|
|
858
|
+
token_set = set(tokens)
|
|
859
|
+
joined = " ".join(tokens)
|
|
860
|
+
|
|
861
|
+
if token_set & _DOB_TOKENS or any(
|
|
862
|
+
" ".join(phrase) in joined for group in _DOB_PHRASES for phrase in group
|
|
863
|
+
):
|
|
864
|
+
has_dob = True
|
|
865
|
+
elif isinstance(value, str) and _DATE_VALUE_RE.match(value) and (
|
|
866
|
+
"birth" in token_set or "dob" in token_set
|
|
867
|
+
):
|
|
868
|
+
has_dob = True
|
|
869
|
+
|
|
870
|
+
if token_set & _NAME_TOKENS or any(" ".join(p) in joined for p in _NAME_PHRASES):
|
|
871
|
+
has_name = True
|
|
872
|
+
|
|
873
|
+
if has_dob and has_name:
|
|
874
|
+
reasons.append(
|
|
875
|
+
f"DOB_PLUS_NAME: parameters at {path} combine a date-of-birth field with a "
|
|
876
|
+
"personal name field, which identifies an individual"
|
|
877
|
+
)
|