sourcelock 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hc_source/guard.py ADDED
@@ -0,0 +1,877 @@
1
+ """Zero-PHI guard.
2
+
3
+ Every tool call passes through :func:`assert_public_params` before any upstream
4
+ request is made and before anything is logged. The guard's job is narrow and
5
+ adversarial: refuse anything shaped like a patient record.
6
+
7
+ **What this is, honestly.** A structural, best-effort refusal of patient-shaped
8
+ input. It is NOT a HIPAA compliance control, not a certification, and not a
9
+ guarantee that PHI cannot reach this process. It raises the cost of an accident
10
+ -- a pasted HL7 segment, a CSV column that turned out to hold MRNs, an agent
11
+ that decided a patient name was a search term -- and it will not stop a
12
+ determined sender. Treat a clean scan as "nothing patient-shaped was detected",
13
+ never as "this input contains no PHI". The receipt says ``PHI_NOT_EXPECTED``
14
+ for exactly that reason: it is a statement of design intent, not a finding.
15
+
16
+ Design rules that matter to adapter authors:
17
+
18
+ * The guard runs on the RAW parameter mapping, before pydantic validation, so
19
+ patient-shaped input is refused even when it would have been dropped anyway.
20
+ * Rejection messages never echo the offending value -- the whole point is that
21
+ the value must not reach a log line. They name the path and the detector.
22
+ * Detection is structural (FHIR/X12/HL7v2 envelopes, SSN and MBI patterns,
23
+ DOB+name combinations, patient-record field names), not a guess about content.
24
+ * Every budget is a REFUSAL when exceeded, never a silent pass. A guard that
25
+ gives up quietly is worse than no guard.
26
+
27
+ Where the line sits, and why
28
+ ----------------------------
29
+ Two shipped tools take genuinely free text. ``leie.candidate_search`` takes a
30
+ person's NAME -- that is the entire tool -- and ``coverage.search_documents``
31
+ takes clinical prose like "excision of malignant skin lesions". So:
32
+
33
+ * an ambiguous-but-plausible value PASSES: a bare surname, a clinical phrase, a
34
+ date on its own, an NPI;
35
+ * a structured patient-record SHAPE does not: a name next to an SSN, an HL7
36
+ segment header, an MBI, a DOB beside a name, a FHIR object.
37
+
38
+ Field NAMES are matched on token boundaries, never as bare substrings. Bare
39
+ substring matching refuses "geographic" (contains "phi"), "adobe" and
40
+ "endobronchial" (contain "dob"), and "outpatient" (contains "patient") -- all of
41
+ which are ordinary product input. The two corpora in ``tests/test_guard_bypass.py``
42
+ and ``tests/test_guard_must_pass.py`` are the executable form of this paragraph;
43
+ change one and you must change the other.
44
+
45
+ Known evasion classes NOT closed
46
+ --------------------------------
47
+ Stated plainly because honest limits beat silent gaps:
48
+
49
+ * **Semantic-only PHI.** "the patient in room 4 who came in Tuesday" has no
50
+ structural marker. Nothing here will catch it.
51
+ * **A single unstructured name or address.** Refusing those would break
52
+ ``leie.candidate_search``, whose purpose is to search by name. A name alone is
53
+ accepted by design.
54
+ * **Integer-typed values.** The guard scans string leaves. A caller invoking the
55
+ Python API with ``{"contractor_id": 123456789}`` as an int is not scanned for
56
+ SSN shape; over the CLI and MCP everything arrives as a string and is.
57
+ * **Encryption and unknown encodings.** Base64 and percent-encoding are decoded
58
+ and rescanned to a bounded depth; anything encrypted, compressed, or in an
59
+ encoding we do not probe passes as opaque bytes.
60
+ * **Free-text identifiers that are not SSN/MBI-shaped.** Local chart numbers,
61
+ payer-specific member ids and account numbers have no universal shape.
62
+ * **Homoglyph coverage is partial.** A confusables table for the common Cyrillic
63
+ and Greek Latin-lookalikes is applied; it is not the full Unicode set.
64
+
65
+ Operator escape hatch
66
+ ---------------------
67
+ ``HC_SOURCE_GUARD_ALLOW`` takes a comma-separated list of detector codes
68
+ (e.g. ``SSN_PATTERN,MBI_PATTERN``) that :func:`assert_public_params` will not
69
+ refuse on. It exists so a false positive is a config change rather than an
70
+ abandoned pipeline. It suppresses named detectors only -- never the whole guard
71
+ -- and :func:`scan_for_phi` itself is never silenced, so audits and tests always
72
+ see the truth.
73
+ """
74
+
75
+ from __future__ import annotations
76
+
77
+ import base64
78
+ import binascii
79
+ import json
80
+ import os
81
+ import re
82
+ import unicodedata
83
+ from typing import Any, Iterable, Mapping
84
+ from urllib.parse import unquote
85
+
86
+ __all__ = [
87
+ "GUARD_ALLOW_ENV",
88
+ "PHI_NON_CLAIM",
89
+ "PHIRejected",
90
+ "ParamsRejected",
91
+ "REDACTED",
92
+ "allowed_detectors",
93
+ "assert_public_params",
94
+ "normalize_for_matching",
95
+ "redact_input",
96
+ "safe_label",
97
+ "safe_location",
98
+ "safe_tool_name",
99
+ "safe_validation_message",
100
+ "safe_validation_reasons",
101
+ "scan_for_phi",
102
+ ]
103
+
104
+ PHI_NON_CLAIM = "PHI_NOT_EXPECTED: this tool accepts only public typed parameters."
105
+
106
+ GUARD_ALLOW_ENV = "HC_SOURCE_GUARD_ALLOW"
107
+
108
+ MAX_DEPTH = 12
109
+ MAX_SERIALIZED_BYTES = 64 * 1024
110
+
111
+ #: Work budgets. Exceeding any of these is a refusal, not a pass -- an input
112
+ #: large or convoluted enough to exhaust the guard is by definition not "a small
113
+ #: typed public parameter".
114
+ MAX_NODES = 4096
115
+ MAX_DECODE_DEPTH = 4
116
+ MAX_EMBEDDED_CANDIDATES = 8
117
+ MAX_ENCODED_RUNS = 4
118
+ MIN_ENCODED_RUN = 12 # 'MTIzLTQ1LTY3ODk=' (a base64 SSN) is 15 chars of payload
119
+
120
+
121
+ # ---------------------------------------------------------------------------
122
+ # Normalisation
123
+ # ---------------------------------------------------------------------------
124
+ #
125
+ # Round-1 finding C3: `123-45-6789` was caught but `123.45.6789`, `123–45–6789`
126
+ # (en dash), `123 45 6789` (NBSP) and `123-<ZWSP>45-6789` all walked past,
127
+ # because the pattern hard-coded ASCII hyphen and space. Matching a normalised
128
+ # copy is the fix that generalises; adding one more separator to one more regex
129
+ # is the fix that does not.
130
+
131
+ _ZERO_WIDTH = dict.fromkeys(
132
+ [
133
+ 0x00AD, # soft hyphen
134
+ 0x200B, # zero width space
135
+ 0x200C, # zero width non-joiner
136
+ 0x200D, # zero width joiner
137
+ 0x2060, # word joiner
138
+ 0xFEFF, # zero width no-break space
139
+ ]
140
+ )
141
+
142
+ #: Latin lookalikes from the Cyrillic and Greek blocks. Partial by design; see
143
+ #: the module docstring's list of limits.
144
+ _CONFUSABLES = str.maketrans(
145
+ {
146
+ "а": "a", "е": "e", "о": "o", "р": "p", "с": "c",
147
+ "у": "y", "х": "x", "ѕ": "s", "і": "i", "ј": "j",
148
+ "ԁ": "d", "һ": "h", "ӏ": "l", "м": "m", "т": "t",
149
+ "в": "b", "н": "h", "к": "k", "А": "A", "Е": "E",
150
+ "О": "O", "Р": "P", "С": "C", "Х": "X", "Ѕ": "S",
151
+ "І": "I", "М": "M", "Т": "T", "В": "B", "Н": "H",
152
+ "К": "K", "ο": "o", "α": "a", "ρ": "p", "ν": "v",
153
+ "τ": "t", "ι": "i", "κ": "k", "μ": "u", "χ": "x",
154
+ "Α": "A", "Β": "B", "Ε": "E", "Η": "H", "Ι": "I",
155
+ "Κ": "K", "Μ": "M", "Ν": "N", "Ο": "O", "Ρ": "P",
156
+ "Τ": "T", "Χ": "X",
157
+ }
158
+ )
159
+
160
+ #: Every dash and space variant folded onto the ASCII pair the patterns use.
161
+ _SEPARATORS = str.maketrans(
162
+ {
163
+ "‐": "-", "‑": "-", "‒": "-", "–": "-", "—": "-",
164
+ "―": "-", "−": "-", "﹘": "-", "﹣": "-", "-": "-",
165
+ " ": " ", " ": " ", " ": " ", " ": " ", " ": " ",
166
+ " ": " ", " ": " ", " ": " ", " ": " ", " ": " ",
167
+ " ": " ", " ": " ", " ": " ", " ": " ", " ": " ",
168
+ " ": " ",
169
+ }
170
+ )
171
+
172
+
173
+ def normalize_for_matching(value: str) -> str:
174
+ """NFKC, strip zero-width, fold homoglyphs and separator variants.
175
+
176
+ Applied to a COPY used only for matching. The original value is never
177
+ altered and never echoed.
178
+ """
179
+ text = unicodedata.normalize("NFKC", value)
180
+ text = text.translate(_ZERO_WIDTH)
181
+ text = text.translate(_CONFUSABLES)
182
+ return text.translate(_SEPARATORS)
183
+
184
+
185
+ # ---------------------------------------------------------------------------
186
+ # Detectors
187
+ # ---------------------------------------------------------------------------
188
+
189
+ _FHIR_PATIENT_RESOURCES = {
190
+ "Patient",
191
+ "Bundle",
192
+ "Coverage",
193
+ "Claim",
194
+ "ClaimResponse",
195
+ "ExplanationOfBenefit",
196
+ "Encounter",
197
+ "Observation",
198
+ "Condition",
199
+ "MedicationRequest",
200
+ "DocumentReference",
201
+ "RelatedPerson",
202
+ "Person",
203
+ }
204
+
205
+ #: FHIR field names that are distinctive on their own...
206
+ _FHIR_STRONG_KEYS = {
207
+ "resourcetype", "birthdate", "gender", "telecom", "maritalstatus",
208
+ "given", "family", "deceasedboolean", "multiplebirthboolean",
209
+ "managingorganization",
210
+ }
211
+ #: ...and ones that only mean something in company.
212
+ _FHIR_WEAK_KEYS = {"name", "identifier", "address", "subject", "patient", "contact"}
213
+
214
+ # SSN: separated (any dialect, post-normalisation) or a bare nine-digit run.
215
+ # The bare-run rule is a deliberate over-block: no shipped parameter legitimately
216
+ # carries a nine-digit string (an NPI is ten, a MAC contractor id is five), and
217
+ # an SSN typed without separators was the single most common bypass in round 1.
218
+ _SSN_SEPARATED_RE = re.compile(r"\b\d{3}[-. ]\d{2}[-. ]\d{4}\b")
219
+ _SSN_BARE_RE = re.compile(r"(?<!\d)\d{9}(?!\d)")
220
+
221
+ # Medicare Beneficiary Identifier: 11 characters, C-A-AN-N-A-AN-N-A-A-N-N, where
222
+ # A excludes S, L, O, I, B and Z (CMS drops them to avoid digit confusion) and
223
+ # the first character is 1-9. Hyphens are allowed where CMS prints them.
224
+ _MBI_ALPHA = "ACDEFGHJKMNPQRTUVWXY"
225
+ _MBI_RE = re.compile(
226
+ rf"(?<![A-Za-z0-9])"
227
+ rf"[1-9][{_MBI_ALPHA}][{_MBI_ALPHA}0-9]\d-?"
228
+ rf"[{_MBI_ALPHA}][{_MBI_ALPHA}0-9]\d-?"
229
+ rf"[{_MBI_ALPHA}][{_MBI_ALPHA}]\d\d"
230
+ rf"(?![A-Za-z0-9])",
231
+ re.IGNORECASE,
232
+ )
233
+
234
+ # X12 and HL7 v2: a SINGLE strong marker now trips. Requiring two ANDed markers
235
+ # was finding C3's biggest hole -- a bare `PID|` segment is the most common leak
236
+ # shape there is, and it carried name, MRN, DOB, sex and address.
237
+ _X12_RE = re.compile(
238
+ r"(?:\bISA\*|\bGS\*(HC|HB|HI|HN|HP|HR|HS)\b|\bST\*(837|835|270|271|276|277|278)\b"
239
+ r"|\bNM1\*IL\*|\bDMG\*D8\*)"
240
+ )
241
+ _HL7V2_SEGMENTS = "MSH|PID|PV1|OBX|OBR|NK1|IN1|IN2|DG1|EVN|ORC|AL1|GT1"
242
+ _HL7V2_RE = re.compile(rf"(?:\A|[\r\n\s>:])(?:{_HL7V2_SEGMENTS})\|")
243
+
244
+ # Free-text patient record: a patient-record word AND a date, together.
245
+ _FREE_TEXT_SUBJECT_TOKENS = {
246
+ "mrn", "dob", "ssn", "patient", "member", "beneficiary", "subscriber", "hicn", "mbi",
247
+ }
248
+ _FREE_TEXT_DATE_RE = re.compile(
249
+ r"(?<!\d)(?:\d{1,2}[/.-]\d{1,2}[/.-]\d{2,4}|\d{4}[/.-]\d{1,2}[/.-]\d{1,2}|\d{8})(?!\d)"
250
+ )
251
+
252
+ # Field-name patterns. Matched against TOKENS of a parameter key, never as bare
253
+ # substrings: "geographic" contains "phi", "adobe" contains "dob", "outpatient"
254
+ # contains "patient", and all three are ordinary product input.
255
+ _STRONG_KEY_TOKENS: dict[str, str] = {
256
+ "ssn": "SSN",
257
+ "mrn": "MEDICAL_RECORD_NUMBER",
258
+ "phi": "PHI",
259
+ "dob": "DOB",
260
+ "birthdate": "DOB",
261
+ "birthday": "DOB",
262
+ "birthyear": "DOB",
263
+ "yob": "DOB",
264
+ "patient": "PATIENT_RECORD",
265
+ "hicn": "MEMBER_IDENTIFIER",
266
+ "mbi": "MEMBER_IDENTIFIER",
267
+ }
268
+ #: Token PAIRS. "beneficiary notice of noncoverage" is a real CMS document type,
269
+ #: so "beneficiary" alone must not refuse -- "beneficiary_id" must.
270
+ _SUBJECT_TOKENS = {"patient", "member", "subscriber", "beneficiary", "enrollee", "insured"}
271
+ _IDENTIFIER_TOKENS = {
272
+ "id", "ids", "identifier", "number", "num", "no", "nbr", "mrn", "ssn", "dob",
273
+ "name", "record", "chart", "account", "acct",
274
+ }
275
+ _PAIR_LABELS = {
276
+ "patient": "PATIENT_RECORD",
277
+ "member": "MEMBER_IDENTIFIER",
278
+ "subscriber": "MEMBER_IDENTIFIER",
279
+ "beneficiary": "MEMBER_IDENTIFIER",
280
+ "enrollee": "MEMBER_IDENTIFIER",
281
+ "insured": "MEMBER_IDENTIFIER",
282
+ }
283
+ #: Multi-word field names, matched against the key's token SEQUENCE.
284
+ _PHRASE_KEY_TOKENS: tuple[tuple[tuple[str, ...], str], ...] = (
285
+ (("social", "security"), "SSN"),
286
+ (("medical", "record"), "MEDICAL_RECORD_NUMBER"),
287
+ (("date", "of", "birth"), "DOB"),
288
+ (("birth", "date"), "DOB"),
289
+ (("birth", "year"), "DOB"),
290
+ (("date", "of", "service"), "PATIENT_RECORD"),
291
+ )
292
+
293
+ _DOB_TOKENS = {"dob", "birthdate", "birthday", "birthyear", "yob"}
294
+ _DOB_PHRASES = ((("date", "of", "birth"),), (("birth", "date"),), (("birth", "year"),))
295
+ _NAME_TOKENS = {"firstname", "lastname", "familyname", "givenname", "middlename", "fullname"}
296
+ _NAME_PHRASES = (
297
+ ("first", "name"),
298
+ ("last", "name"),
299
+ ("family", "name"),
300
+ ("given", "name"),
301
+ ("middle", "name"),
302
+ ("full", "name"),
303
+ ("patient", "name"),
304
+ )
305
+ _DATE_VALUE_RE = re.compile(r"^\s*(\d{4}-\d{2}-\d{2}|\d{2}/\d{2}/\d{4}|\d{4})\s*$")
306
+
307
+ _TOKEN_SPLIT_RE = re.compile(r"[^a-z0-9]+")
308
+ _CAMEL_SPLIT_RE = re.compile(r"(?<=[a-z0-9])(?=[A-Z])")
309
+ _BASE64_RUN_RE = re.compile(r"[A-Za-z0-9+/]{%d,}={0,2}" % MIN_ENCODED_RUN)
310
+ _PERCENT_RE = re.compile(r"%[0-9A-Fa-f]{2}")
311
+
312
+
313
+ def _key_tokens(key: str) -> tuple[list[str], str]:
314
+ """Return ``(tokens, condensed)`` for a parameter key.
315
+
316
+ ``tokens`` splits on separators and camelCase humps, so ``patientMRN`` and
317
+ ``patient_mrn`` both yield ``["patient", "mrn"]`` while ``outpatient`` yields
318
+ ``["outpatient"]`` and is left alone. ``condensed`` is the alphanumeric-only
319
+ form, used only to defeat separator obfuscation (``p_a_t_i_e_n_t``).
320
+ """
321
+ normalized = normalize_for_matching(key)
322
+ humped = _CAMEL_SPLIT_RE.sub(" ", normalized).lower()
323
+ tokens = [t for t in _TOKEN_SPLIT_RE.split(humped) if t]
324
+ condensed = "".join(ch for ch in normalized.lower() if ch.isalnum())
325
+ return tokens, condensed
326
+
327
+
328
+ class PHIRejected(ValueError):
329
+ """Raised when a tool call's parameters look like a patient record."""
330
+
331
+ def __init__(self, tool: str, reasons: list[str]) -> None:
332
+ self.tool = tool
333
+ self.reasons = list(reasons)
334
+ joined = "; ".join(self.reasons)
335
+ super().__init__(
336
+ f"SourceLock refused the call to {tool!r}: parameters look like protected health "
337
+ f"information. {PHI_NON_CLAIM} Reasons: {joined}"
338
+ )
339
+
340
+
341
+ # ---------------------------------------------------------------------------
342
+ # Value-free parameter rejection
343
+ # ---------------------------------------------------------------------------
344
+ #
345
+ # Round-1 finding C1: a custom pydantic validator that quoted its input put the
346
+ # raw value into ``error["msg"]``, and every consumer built its message from
347
+ # ``msg``. The consumers were value-blind by assumption, not by construction.
348
+ #
349
+ # The fix is a boundary, not a patch: ``ToolSpec.invoke`` converts every
350
+ # ``ValidationError`` into a :class:`ParamsRejected` whose text is built from
351
+ # ``loc`` plus a message that has been scrubbed of the input value. Nothing
352
+ # downstream -- CLI, MCP, or a library consumer that simply does ``str(exc)`` --
353
+ # can reach the raw value, and a future adapter that quotes its input is
354
+ # scrubbed on the way out instead of reopening the hole.
355
+
356
+ REDACTED = "[value withheld]"
357
+
358
+ #: Below this length a value is not distinctive enough to be worth redacting,
359
+ #: and redacting it would mangle ordinary words in rule text ("ncd", "lcd").
360
+ _MIN_REDACTABLE = 3
361
+
362
+ #: A label safe to print back at whoever sent it: an ordinary identifier. Field
363
+ #: names, tool names and pydantic error locations all take this shape, and
364
+ #: nothing that carries data does.
365
+ _SAFE_LABEL_RE = re.compile(r"^[A-Za-z][A-Za-z0-9_]{0,39}$")
366
+
367
+ #: Six digits in a row. Long enough to exclude every field name and version
368
+ #: suffix this product uses, short enough to catch anything that identifies a
369
+ #: person or an entity.
370
+ _LONG_DIGIT_RUN_RE = re.compile(r"\d{6,}")
371
+
372
+
373
+ def safe_label(text: str, index: int | None = None) -> str:
374
+ """Echo ``text`` only if it is an identifier; otherwise a positional stand-in.
375
+
376
+ Round-1 closed the value-echo hole for parameter VALUES. Keys, tool names and
377
+ pydantic ``loc`` entries were still printed verbatim -- and every one of them
378
+ is caller-controlled. ``{"123-45-6789": "x"}`` against a model with
379
+ ``extra='forbid'`` produced a rejection reason and a pydantic ``loc`` that
380
+ both quoted the SSN: the refusal itself became the leak.
381
+
382
+ An identifier-shaped label is kept because a caller genuinely needs to know
383
+ WHICH parameter was wrong. Anything else is data wearing a name, and it is
384
+ replaced with its position.
385
+ """
386
+ if _SAFE_LABEL_RE.match(text) and not _looks_like_an_identifier_value(text):
387
+ return text
388
+ return f"<key #{index}>" if index is not None else "<withheld>"
389
+
390
+
391
+ def _looks_like_an_identifier_value(text: str) -> bool:
392
+ """True when an identifier-shaped string still carries an identifier inside it.
393
+
394
+ ``ssn123456789`` passes ``_SAFE_LABEL_RE`` and is still nine digits of
395
+ somebody's identity. The long-digit-run rule generalises past SSN and MBI:
396
+ no field name in this product carries six consecutive digits, while every
397
+ identifier worth withholding does -- SSN, NPI, MBI, a chart number.
398
+ ``icd10cm_fy2026`` and ``v28`` are unaffected.
399
+ """
400
+ normalized = normalize_for_matching(text)
401
+ return bool(
402
+ _LONG_DIGIT_RUN_RE.search(normalized)
403
+ or _SSN_SEPARATED_RE.search(normalized)
404
+ or _MBI_RE.search(normalized)
405
+ )
406
+
407
+
408
+ #: A route name: ``source.tool``, both halves lowercase slugs. Every shipped
409
+ #: tool matches; nothing that carries data does.
410
+ _TOOL_NAME_RE = re.compile(r"^[a-z][a-z0-9_]{0,31}\.[a-z][a-z0-9_]{0,31}$")
411
+
412
+
413
+ def safe_tool_name(name: str) -> str:
414
+ """Echo a requested tool name only if it is shaped like one.
415
+
416
+ An MCP client picks the string, and "no tool named X" reflected X back
417
+ verbatim -- so a client could get an arbitrary value written into a server
418
+ log by asking for it as a tool. A real name is useful in the error; anything
419
+ else tells the caller nothing they did not already know.
420
+ """
421
+ return name if _TOOL_NAME_RE.match(name) else "<name withheld>"
422
+
423
+
424
+ def safe_location(loc: Any) -> str:
425
+ """A pydantic ``loc`` tuple rendered without echoing caller-controlled names.
426
+
427
+ With ``extra='forbid'`` the ``loc`` of an unexpected-field error IS the key
428
+ the caller invented, so this is the same hole as :func:`safe_label` in a
429
+ different coat.
430
+ """
431
+ parts = loc if isinstance(loc, (tuple, list)) else (loc,)
432
+ rendered = [
433
+ str(p) if isinstance(p, int) else safe_label(str(p), i)
434
+ for i, p in enumerate(parts)
435
+ ]
436
+ return ".".join(rendered) or "<params>"
437
+
438
+
439
+ class ParamsRejected(ValueError):
440
+ """Parameters did not fit a tool's declared model.
441
+
442
+ Carries field locations and rule text only. The offending value is never
443
+ stored on the exception, so no consumer can echo it -- not by printing the
444
+ exception, not by walking ``errors()``, and not from a traceback (the
445
+ originating ``ValidationError`` is deliberately not chained).
446
+ """
447
+
448
+ def __init__(self, tool: str, problems: list[tuple[str, str]]) -> None:
449
+ self.tool = tool
450
+ self.problems = list(problems)
451
+ joined = "; ".join(f"{loc}: {msg}" for loc, msg in self.problems) or "invalid parameters"
452
+ super().__init__(f"{tool}: invalid parameters -- {joined}")
453
+
454
+ def errors(self) -> list[dict[str, str]]:
455
+ """Value-free stand-in for ``ValidationError.errors()``."""
456
+ return [{"loc": loc, "msg": msg} for loc, msg in self.problems]
457
+
458
+
459
+ def _leaf_strings(value: Any, depth: int = 0) -> Iterable[str]:
460
+ """Yield every string-ish leaf of ``value`` that is worth redacting."""
461
+ if depth > MAX_DEPTH:
462
+ return
463
+ if isinstance(value, str):
464
+ if len(value) >= _MIN_REDACTABLE:
465
+ yield value
466
+ elif isinstance(value, Mapping):
467
+ for k, v in value.items():
468
+ yield from _leaf_strings(k, depth + 1)
469
+ yield from _leaf_strings(v, depth + 1)
470
+ elif isinstance(value, (list, tuple, set)):
471
+ for v in value:
472
+ yield from _leaf_strings(v, depth + 1)
473
+ elif isinstance(value, (int, float)) and not isinstance(value, bool):
474
+ text = str(value)
475
+ if len(text) >= _MIN_REDACTABLE:
476
+ yield text
477
+
478
+
479
+ def redact_input(text: str, value: Any) -> str:
480
+ """Remove every occurrence of ``value`` (or any of its leaves) from ``text``.
481
+
482
+ Case-insensitive, because a value is often upper/lower-cased on its way into
483
+ a message. Longest leaves first so a nested value is not half-redacted.
484
+ """
485
+ leaves = sorted(set(_leaf_strings(value)), key=len, reverse=True)
486
+ for leaf in leaves:
487
+ text = re.sub(re.escape(leaf), REDACTED, text, flags=re.IGNORECASE)
488
+ return text
489
+
490
+
491
+ def safe_validation_reasons(exc: Any) -> list[tuple[str, str]]:
492
+ """Reduce a pydantic ``ValidationError`` to ``(location, rule)`` pairs.
493
+
494
+ ``msg`` is used because pydantic's own rule text is genuinely useful ("String
495
+ should match pattern ..."), but it is scrubbed against ``input`` first: a
496
+ custom validator is free to quote its value and still cannot leak it.
497
+ """
498
+ errors = getattr(exc, "errors", None)
499
+ if not callable(errors):
500
+ return [("<params>", type(exc).__name__)]
501
+ problems: list[tuple[str, str]] = []
502
+ for error in errors():
503
+ location = safe_location(error.get("loc", ()))
504
+ message = str(error.get("msg", "invalid"))
505
+ if "input" in error:
506
+ message = redact_input(message, error["input"])
507
+ problems.append((location, message))
508
+ return problems
509
+
510
+
511
+ def safe_validation_message(exc: Any) -> str:
512
+ """One line of ``field: rule`` pairs, with no input values in it."""
513
+ return "; ".join(f"{loc}: {msg}" for loc, msg in safe_validation_reasons(exc)) or (
514
+ "invalid parameters"
515
+ )
516
+
517
+
518
+ # ---------------------------------------------------------------------------
519
+ # Entry points
520
+ # ---------------------------------------------------------------------------
521
+
522
+
523
+ def allowed_detectors() -> set[str]:
524
+ """Detector codes the operator has chosen to accept, from the environment."""
525
+ raw = os.environ.get(GUARD_ALLOW_ENV, "")
526
+ return {part.strip().upper() for part in raw.split(",") if part.strip()}
527
+
528
+
529
+ def assert_public_params(params: Mapping[str, Any] | Any, *, tool: str) -> None:
530
+ """Raise :class:`PHIRejected` unless ``params`` are public typed parameters.
531
+
532
+ Honours ``HC_SOURCE_GUARD_ALLOW``: reasons whose detector code appears there
533
+ are not grounds for refusal. Everything else still is -- suppressing one
534
+ detector never disables the guard.
535
+ """
536
+ reasons = scan_for_phi(params)
537
+ allowed = allowed_detectors()
538
+ if allowed:
539
+ reasons = [r for r in reasons if r.split(":", 1)[0].strip().upper() not in allowed]
540
+ if reasons:
541
+ raise PHIRejected(tool, reasons)
542
+ return None
543
+
544
+
545
+ def scan_for_phi(params: Mapping[str, Any] | Any) -> list[str]:
546
+ """Return a list of reasons ``params`` look patient-shaped (empty == clean).
547
+
548
+ A pure detector: never reads the environment, never suppresses anything.
549
+ The operator allowlist is applied by :func:`assert_public_params`, so this
550
+ function always tells an auditor the truth.
551
+
552
+ Reasons never contain the offending value.
553
+ """
554
+ reasons: list[str] = []
555
+ _check_size(params, reasons)
556
+ if reasons:
557
+ return reasons
558
+
559
+ budget = _Budget()
560
+ for path, key, value in _walk(params, reasons, budget=budget):
561
+ if key is not None:
562
+ _check_key(key, path, reasons)
563
+ if isinstance(value, Mapping):
564
+ _check_dob_name_combo(value, path, reasons)
565
+ _check_value(value, path, reasons, budget=budget)
566
+
567
+ # Stable, de-duplicated order.
568
+ seen: set[str] = set()
569
+ unique: list[str] = []
570
+ for r in reasons:
571
+ if r not in seen:
572
+ seen.add(r)
573
+ unique.append(r)
574
+ return unique
575
+
576
+
577
+ class _Budget:
578
+ """Shared work counters for one scan. Exhaustion is a refusal, not a pass."""
579
+
580
+ def __init__(self) -> None:
581
+ self.nodes = 0
582
+ self.exhausted = False
583
+
584
+
585
+ def _check_size(params: Any, reasons: list[str]) -> None:
586
+ try:
587
+ size = len(json.dumps(params, default=str))
588
+ except (TypeError, ValueError):
589
+ return
590
+ if size > MAX_SERIALIZED_BYTES:
591
+ reasons.append(
592
+ f"OVERSIZED_PARAMS: parameters serialize to {size} bytes (limit "
593
+ f"{MAX_SERIALIZED_BYTES}); SourceLock tools take small typed public parameters, "
594
+ "not documents"
595
+ )
596
+
597
+
598
+ def _walk(
599
+ node: Any,
600
+ reasons: list[str],
601
+ path: str = "params",
602
+ depth: int = 0,
603
+ *,
604
+ budget: _Budget,
605
+ ) -> Iterable[tuple[str, str | None, Any]]:
606
+ """Iteratively yield ``(path, key, value)`` for every node in the structure."""
607
+ stack: list[tuple[str, str | None, Any, int]] = [(path, None, node, depth)]
608
+ while stack:
609
+ cur_path, cur_key, cur_value, cur_depth = stack.pop()
610
+ if cur_depth > MAX_DEPTH:
611
+ reasons.append(
612
+ f"UNEXPECTED_STRUCTURE: parameters nest deeper than {MAX_DEPTH} levels at "
613
+ f"{cur_path}; public typed parameters are flat"
614
+ )
615
+ continue
616
+
617
+ budget.nodes += 1
618
+ if budget.nodes > MAX_NODES:
619
+ if not budget.exhausted:
620
+ budget.exhausted = True
621
+ reasons.append(
622
+ f"UNEXPECTED_STRUCTURE: parameters contain more than {MAX_NODES} values; "
623
+ "SourceLock tools take a handful of small typed public parameters, so an "
624
+ "input this large is refused rather than partially inspected"
625
+ )
626
+ return
627
+
628
+ yield cur_path, cur_key, cur_value
629
+
630
+ if isinstance(cur_value, Mapping):
631
+ for i, (k, v) in enumerate(cur_value.items()):
632
+ # The KEY is caller-controlled too. `{"123-45-6789": "x"}` used
633
+ # to produce a refusal reason that quoted the SSN back in the
634
+ # path -- the guard refusing a value by repeating it.
635
+ stack.append(
636
+ (f"{cur_path}.{safe_label(str(k), i)}", str(k), v, cur_depth + 1)
637
+ )
638
+ elif isinstance(cur_value, (list, tuple, set)):
639
+ for i, v in enumerate(cur_value):
640
+ stack.append((f"{cur_path}[{i}]", None, v, cur_depth + 1))
641
+
642
+
643
+ def _check_key(key: str, path: str, reasons: list[str]) -> None:
644
+ tokens, condensed = _key_tokens(key)
645
+ token_set = set(tokens)
646
+
647
+ for token, label in _STRONG_KEY_TOKENS.items():
648
+ if token in token_set:
649
+ reasons.append(
650
+ f"PATIENT_SHAPED_KEY: parameter at {path} is named after a patient-record "
651
+ f"field ({label})"
652
+ )
653
+
654
+ for subject in _SUBJECT_TOKENS & token_set:
655
+ if token_set & _IDENTIFIER_TOKENS:
656
+ reasons.append(
657
+ f"PATIENT_SHAPED_KEY: parameter at {path} names a person plus an identifier "
658
+ f"({_PAIR_LABELS[subject]})"
659
+ )
660
+
661
+ joined = " ".join(tokens)
662
+ for phrase, label in _PHRASE_KEY_TOKENS:
663
+ if " ".join(phrase) in joined:
664
+ reasons.append(
665
+ f"PATIENT_SHAPED_KEY: parameter at {path} is named after a patient-record "
666
+ f"field ({label})"
667
+ )
668
+
669
+ # Separator obfuscation only: `p_a_t_i_e_n_t` splits into single characters,
670
+ # which no ordinary field name does. Condensing `outpatient_setting` would
671
+ # false-positive, so the condensed form is consulted for THIS shape alone.
672
+ if sum(1 for t in tokens if len(t) == 1) >= 3:
673
+ for token, label in _STRONG_KEY_TOKENS.items():
674
+ if token in condensed:
675
+ reasons.append(
676
+ f"OBFUSCATED_KEY: parameter at {path} spells a patient-record field name "
677
+ f"through separators ({label})"
678
+ )
679
+
680
+
681
+ def _check_value(value: Any, path: str, reasons: list[str], *, budget: _Budget) -> None:
682
+ if isinstance(value, Mapping):
683
+ _check_fhir_object(value, path, reasons)
684
+ return
685
+
686
+ if not isinstance(value, str):
687
+ return
688
+
689
+ _scan_text(value, path, reasons, budget=budget, decode_depth=0)
690
+
691
+
692
+ def _check_fhir_object(value: Mapping[str, Any], path: str, reasons: list[str]) -> None:
693
+ """FHIR detection that does not depend on ``resourceType``.
694
+
695
+ Finding C3: an object carrying ``name`` and ``gender`` but no
696
+ ``resourceType`` was invisible, because detection keyed entirely off that
697
+ one field.
698
+ """
699
+ resource = value.get("resourceType")
700
+ if isinstance(resource, str):
701
+ reasons.append(
702
+ f"FHIR_RESOURCE: object at {path} carries resourceType "
703
+ f"{'(patient-shaped)' if resource in _FHIR_PATIENT_RESOURCES else '(FHIR)'}; "
704
+ "SourceLock does not accept FHIR payloads"
705
+ )
706
+ return
707
+
708
+ keys = {str(k).lower() for k in value}
709
+ strong = keys & _FHIR_STRONG_KEYS
710
+ weak = keys & _FHIR_WEAK_KEYS
711
+ if strong and len(strong | weak) >= 2:
712
+ reasons.append(
713
+ f"PATIENT_RESOURCE_SHAPE: object at {path} carries person-record fields "
714
+ "(demographics plus an identifier or name) without declaring a resourceType; "
715
+ "SourceLock does not accept patient records"
716
+ )
717
+
718
+
719
+ def _scan_text(value: str, path: str, reasons: list[str], *, budget: _Budget, decode_depth: int) -> None:
720
+ """Run every text detector, then probe encodings and embedded structures."""
721
+ text = normalize_for_matching(value)
722
+
723
+ if _SSN_SEPARATED_RE.search(text) or _SSN_BARE_RE.search(text):
724
+ reasons.append(f"SSN_PATTERN: value at {path} contains an SSN-shaped token")
725
+
726
+ if _MBI_RE.search(text):
727
+ reasons.append(
728
+ f"MBI_PATTERN: value at {path} contains a Medicare Beneficiary Identifier-shaped "
729
+ "token (11 characters in the CMS C-A-AN-N-A-AN-N-A-A-N-N form)"
730
+ )
731
+
732
+ if _X12_RE.search(text):
733
+ reasons.append(f"X12_ENVELOPE: value at {path} contains an X12 segment marker")
734
+
735
+ if _HL7V2_RE.search(text):
736
+ reasons.append(f"HL7V2_MESSAGE: value at {path} contains an HL7 v2 segment header")
737
+
738
+ _check_free_text_record(text, path, reasons)
739
+
740
+ if decode_depth >= MAX_DECODE_DEPTH:
741
+ # Do not give up quietly: if there is still something decodable here, the
742
+ # input is more encoded than any public typed parameter has cause to be.
743
+ if _looks_encoded(text):
744
+ reasons.append(
745
+ f"ENCODED_PAYLOAD: value at {path} is encoded more than {MAX_DECODE_DEPTH} "
746
+ "layers deep; SourceLock takes small typed public parameters, not encoded "
747
+ "documents"
748
+ )
749
+ return
750
+
751
+ _scan_embedded_json(value, path, reasons, budget=budget, decode_depth=decode_depth)
752
+ _scan_encoded(value, path, reasons, budget=budget, decode_depth=decode_depth)
753
+
754
+
755
+ def _check_free_text_record(text: str, path: str, reasons: list[str]) -> None:
756
+ """A patient-record word next to a date is a patient record in prose.
757
+
758
+ ``MRN 88213 DOE JOHN DOB 01/02/1980`` had no structural envelope at all and
759
+ walked straight through round 1. Either half alone passes: "dobutamine" is
760
+ not "dob", and a date on its own is just a date.
761
+ """
762
+ tokens = {t for t in _TOKEN_SPLIT_RE.split(text.lower()) if t}
763
+ if not (tokens & _FREE_TEXT_SUBJECT_TOKENS):
764
+ return
765
+ if _FREE_TEXT_DATE_RE.search(text):
766
+ reasons.append(
767
+ f"PATIENT_RECORD_TEXT: value at {path} combines a patient-record term "
768
+ "(MRN/DOB/SSN/patient/member) with a date, which identifies an individual"
769
+ )
770
+
771
+
772
+ def _looks_encoded(text: str) -> bool:
773
+ return bool(_PERCENT_RE.search(text)) or bool(_BASE64_RUN_RE.search(text))
774
+
775
+
776
+ def _scan_embedded_json(
777
+ value: str, path: str, reasons: list[str], *, budget: _Budget, decode_depth: int
778
+ ) -> None:
779
+ """Parse JSON found ANYWHERE in the string, not only at position 0.
780
+
781
+ Finding C3: ``"payload: " + <FHIR Bundle>`` passed because the scan only
782
+ fired when the first non-space character was ``{`` or ``[``.
783
+ """
784
+ if len(value) > MAX_SERIALIZED_BYTES:
785
+ return
786
+
787
+ stripped = value.strip()
788
+ if stripped[:1] == '"':
789
+ # Double-encoded JSON: the outer parse yields a string that is itself
790
+ # JSON. One `json.loads` and a rescan, rather than one and done.
791
+ try:
792
+ inner = json.loads(stripped)
793
+ except (ValueError, RecursionError):
794
+ inner = None
795
+ if isinstance(inner, str):
796
+ _scan_text(inner, path, reasons, budget=budget, decode_depth=decode_depth + 1)
797
+ return
798
+
799
+ decoder = json.JSONDecoder()
800
+ candidates = 0
801
+ for match in re.finditer(r"[{\[]", value):
802
+ candidates += 1
803
+ if candidates > MAX_EMBEDDED_CANDIDATES:
804
+ return
805
+ try:
806
+ parsed, _ = decoder.raw_decode(value, match.start())
807
+ except (ValueError, RecursionError):
808
+ continue
809
+ if not isinstance(parsed, (Mapping, list)):
810
+ continue
811
+ for reason in scan_for_phi(parsed):
812
+ reasons.append(f"EMBEDDED_JSON at {path}: {reason}")
813
+ return
814
+
815
+
816
+ def _scan_encoded(
817
+ value: str, path: str, reasons: list[str], *, budget: _Budget, decode_depth: int
818
+ ) -> None:
819
+ """Decode base64 and percent-encoded runs and rescan what comes out.
820
+
821
+ Only findings from the DECODED text are reported, so a name that happens to
822
+ look like base64 costs one decode attempt and nothing else.
823
+ """
824
+ if _PERCENT_RE.search(value):
825
+ decoded = unquote(value)
826
+ if decoded != value:
827
+ _scan_text(decoded, path, reasons, budget=budget, decode_depth=decode_depth + 1)
828
+
829
+ runs = 0
830
+ for match in _BASE64_RUN_RE.finditer(value):
831
+ runs += 1
832
+ if runs > MAX_ENCODED_RUNS:
833
+ return
834
+ blob = match.group(0)
835
+ try:
836
+ raw = base64.b64decode(blob + "=" * (-len(blob) % 4), validate=True)
837
+ except (binascii.Error, ValueError):
838
+ continue
839
+ try:
840
+ decoded = raw.decode("utf-8")
841
+ except UnicodeDecodeError:
842
+ continue
843
+ if not decoded.isprintable() and not any(c in decoded for c in "\r\n\t"):
844
+ continue
845
+ _scan_text(decoded, path, reasons, budget=budget, decode_depth=decode_depth + 1)
846
+
847
+
848
+ def _check_dob_name_combo(params: Mapping[str, Any], path: str, reasons: list[str]) -> None:
849
+ """A date of birth beside a personal name identifies an individual.
850
+
851
+ Runs on every mapping the walk reaches, not just the top level, so it also
852
+ fires inside decoded or embedded JSON.
853
+ """
854
+ has_dob = False
855
+ has_name = False
856
+ for key, value in params.items():
857
+ tokens, _ = _key_tokens(str(key))
858
+ token_set = set(tokens)
859
+ joined = " ".join(tokens)
860
+
861
+ if token_set & _DOB_TOKENS or any(
862
+ " ".join(phrase) in joined for group in _DOB_PHRASES for phrase in group
863
+ ):
864
+ has_dob = True
865
+ elif isinstance(value, str) and _DATE_VALUE_RE.match(value) and (
866
+ "birth" in token_set or "dob" in token_set
867
+ ):
868
+ has_dob = True
869
+
870
+ if token_set & _NAME_TOKENS or any(" ".join(p) in joined for p in _NAME_PHRASES):
871
+ has_name = True
872
+
873
+ if has_dob and has_name:
874
+ reasons.append(
875
+ f"DOB_PLUS_NAME: parameters at {path} combine a date-of-birth field with a "
876
+ "personal name field, which identifies an individual"
877
+ )