mrfkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mrfkit/codes.py ADDED
@@ -0,0 +1,1607 @@
1
+ """Normalize billing codes and code types.
2
+
3
+ Hospitals label the same code system many ways ("CPT4", "HCPCS/CPT", "REV",
4
+ "AP-DRG"), file standard codes under their own chargemaster buckets, wrap
5
+ codes in prefixes ("HCPCS C1776", "DRG100") and bake modifiers into the code
6
+ ("73721TC"). ``normalize_code`` sorts all of that out; the helpers around it
7
+ reject rows that are too broken to keep and infer the billing class when the
8
+ file does not say.
9
+
10
+ Nothing here touches a database. A few guards get sharper when you pass a
11
+ ``ReferenceData`` with published code lists; without one they fall back to the
12
+ shape-only rules.
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ import re
18
+ from typing import TYPE_CHECKING, Dict, Optional, Tuple
19
+
20
+ if TYPE_CHECKING:
21
+ from mrfkit.reference import ParseStats, ReferenceData
22
+
23
+ # ============================================================================
24
+ # CODE & CODE_TYPE NORMALIZATION
25
+ # ============================================================================
26
+ # Standard US medical code systems and their formats:
27
+ # CPT - 5 digits (Level I HCPCS), e.g. 99213, 27447
28
+ # HCPCS - letter + 4 digits (Level II HCPCS), e.g. J0690, C1776, A4550
29
+ # Also includes Category III: 5 digits + 'T', e.g. 0237T
30
+ # CDT - Current Dental Terminology, D + 4 digits, e.g. D0120, D7509
31
+ # RC - Revenue Code, 4 digits zero-padded (UB-04), e.g. 0360, 0750
32
+ # MS-DRG - 3 digits, e.g. 469, 470
33
+ # APC - Ambulatory Payment Classification, 4 digits or 'N'+3 digits
34
+ # NDC - National Drug Code, 10-11 digits with dashes, e.g. 12345-6789-01
35
+ # CDM - Charge Description Master (hospital-internal), non-standard formats
36
+ # LOCAL - Hospital-internal codes that don't match any standard system
37
+
38
+ # code_type synonyms: normalize various source labels to canonical types
39
+ _CODE_TYPE_NORMALIZE = {
40
+ 'CPT': 'CPT',
41
+ 'CPT4': 'CPT',
42
+ 'CPT-4': 'CPT',
43
+ 'HCPCS': 'HCPCS',
44
+ 'HCPCS LEVEL II': 'HCPCS',
45
+ 'HCPCS II': 'HCPCS',
46
+ 'HCPCS2': 'HCPCS',
47
+ # Hedge labels used by hospitals that don't know whether a code is
48
+ # CPT (Level I) or HCPCS (Level II). Normalize to HCPCS; Step 5 will
49
+ # reclassify 5-digit numeric codes to CPT based on shape.
50
+ 'CPT/HCPCS': 'HCPCS',
51
+ 'HCPCS/CPT': 'HCPCS',
52
+ 'CPT / HCPCS': 'HCPCS',
53
+ 'HCPCS / CPT': 'HCPCS',
54
+ 'CPT-HCPCS': 'HCPCS',
55
+ 'HCPCS-CPT': 'HCPCS',
56
+ 'CPTHCPCS': 'HCPCS',
57
+ 'HCPCSCPT': 'HCPCS',
58
+ 'RC': 'RC',
59
+ 'REV': 'RC',
60
+ 'REVENUE': 'RC',
61
+ 'REVENUE CODE': 'RC',
62
+ 'REVCODE': 'RC',
63
+ 'MS-DRG': 'MS-DRG',
64
+ 'MSDRG': 'MS-DRG',
65
+ 'DRG': 'MS-DRG',
66
+ 'TRIS-DRG': 'MS-DRG', # Tenet Revenue Integrity System - same codes as MS-DRG
67
+ 'APR-DRG': 'APR-DRG', # All Patient Refined DRG (3M) - distinct from MS-DRG
68
+ 'APRDRG': 'APR-DRG',
69
+ 'APR DRG': 'APR-DRG',
70
+ 'APR_DRG': 'APR-DRG',
71
+ 'AP-DRG': 'APR-DRG', # All Patient DRG variant: same code set
72
+ 'APC': 'APC',
73
+ 'CMG': 'CMG', # Case Mix Group (CMS inpatient rehab grouper)
74
+ 'NDC': 'NDC',
75
+ 'CDT': 'CDT',
76
+ 'DENTAL': 'CDT',
77
+ 'CDM': 'CDM',
78
+ 'CHARGEMASTER': 'CDM',
79
+ 'LOCAL': 'LOCAL',
80
+ # Partners Healthcare internal types
81
+ 'SUP': 'CDM', # supplies
82
+ 'EAP': 'CDM', # enterprise-assigned procedure
83
+ 'ERX': 'CDM', # enterprise pharmacy
84
+ # Compound NDC+system types: disambiguated in normalize_code step 2A
85
+ 'NDCHCPCS': 'NDCHCPCS',
86
+ 'NDCCPT': 'NDCCPT',
87
+ }
88
+
89
+ # Non-system junk code_type labels that some hospitals emit. When
90
+ # the canonicalized code_type is one of these we discard it (treat as missing)
91
+ # and re-derive from code shape, exactly as for a blank code_type.
92
+ _JUNK_CODE_TYPES = frozenset({'TYPE', 'MODIFIER', 'MODIFIERS', 'OBS', 'DOT', 'UN'})
93
+
94
+ # Canonical validity regexes for controlled-vocabulary numeric types.
95
+ # Applied in two places: cross-system reclassify (step 6A) and residue gating
96
+ # (step 12). APR-DRG base-severity form (775-1) and CMG group-tier form
97
+ # (1902-D) are both canonical - do NOT gate them.
98
+ _RE_VALID_RC = re.compile(r'^\d{3,4}$')
99
+ _RE_VALID_MS_DRG = re.compile(r'^\d{1,3}$')
100
+ _RE_VALID_APR_DRG = re.compile(r'^\d{1,4}(-\d{1,2})?$')
101
+ _RE_VALID_APC = re.compile(r'^(?:\d{4,5}|[Nn]\d{3,4})$')
102
+ _RE_VALID_CMG = re.compile(r'^\d{4}(-[A-Z])?$')
103
+ _RE_VALID_EAPG = re.compile(r'^\d{3,5}$')
104
+
105
+ _CONTROLLED_VOCAB_VALIDATORS = {
106
+ 'RC': _RE_VALID_RC,
107
+ 'MS-DRG': _RE_VALID_MS_DRG,
108
+ 'APR-DRG': _RE_VALID_APR_DRG,
109
+ 'APC': _RE_VALID_APC,
110
+ 'CMG': _RE_VALID_CMG,
111
+ 'EAPG': _RE_VALID_EAPG,
112
+ }
113
+
114
+ # Revenue codes are 1-4 digit numbers in the range 0001-0999.
115
+ # Some hospitals report them under code_type='HCPCS' using the
116
+ # 3-digit UB-04 revenue center number (e.g., 272 for sterile supply).
117
+ _KNOWN_REVENUE_CODES = {
118
+ # Commonly misclassified as HCPCS
119
+ '250', '0250', '255', '0255', '258', '0258', # Pharmacy
120
+ '270', '0270', '271', '0271', '272', '0272', # Supplies
121
+ '275', '0275', '276', '0276', '278', '0278', # Implants, supplies
122
+ '300', '0300', '301', '0301', '302', '0302', # Lab
123
+ '305', '0305', '306', '0306', '309', '0309',
124
+ '310', '0310', '311', '0311', '312', '0312',
125
+ '320', '0320', '323', '0323', '333', '0333', # Radiology
126
+ '335', '0335', '341', '0341', '342', '0342',
127
+ '343', '0343', '350', '0350', '351', '0351',
128
+ '352', '0352', '360', '0360', '361', '0361', # OR
129
+ '370', '0370', '390', '0390', # Anesthesia
130
+ '402', '0402', '403', '0403', '404', '0404', # Other imaging
131
+ '410', '0410', '412', '0412', '420', '0420', # Physical therapy
132
+ '424', '0424', '430', '0430', '440', '0440',
133
+ '450', '0450', '460', '0460', '470', '0470', # ER, ambulance
134
+ '471', '0471', '480', '0480', '481', '0481',
135
+ '483', '0483', '489', '0489', '510', '0510',
136
+ '521', '0521', '610', '0610', '611', '0611',
137
+ '615', '0615', '618', '0618', '636', '0636', # Pharmacy IV
138
+ '637', '0637', '710', '0710', '721', '0721',
139
+ '722', '0722', '730', '0730', '731', '0731',
140
+ '740', '0740', '750', '0750', '760', '0760',
141
+ '812', '0812', '906', '0906', '916', '0916',
142
+ '918', '0918', '920', '0920', '921', '0921',
143
+ '922', '0922', '940', '0940', '942', '0942',
144
+ '949', '0949', '987', '0987', '998', '0998',
145
+ }
146
+
147
+
148
+ def _strip_non_ascii(s: str) -> str:
149
+ """Remove non-ASCII characters (mojibake, control chars) from a string."""
150
+ return ''.join(c for c in s if ord(c) < 128)
151
+
152
+
153
+ # Regex patterns for valid code formats
154
+ _RE_CPT = re.compile(r'\d{5}') # 5 digits: 99213
155
+ _RE_CPT_PLA = re.compile(r'\d{4}[A-Z]') # PLA codes: 0202U, 0003M
156
+ _RE_HCPCS = re.compile(r'[A-Z]\d{4}') # letter + 4 digits: J0591
157
+ _RE_CAT3 = re.compile(r'\d{4}T') # Category III: 0237T
158
+
159
+ # ── NDC normalization ─────────────────────────────────────────────────
160
+ # Produces canonical 11-digit no-dash NDCs directly.
161
+ _RE_NDC_LEADING_ALPHA = re.compile(r'^[A-Za-z]')
162
+ _RE_NDC_CANON_11 = re.compile(r'^\d{11}$')
163
+ _RE_NDC_NINE_DIGIT = re.compile(r'^\d{9}$')
164
+ # Bare dash-less 10-digit NDC10 → NDC11 via leading '0'.
165
+ _RE_NDC_TEN_DIGIT = re.compile(r'^\d{10}$')
166
+ _RE_NDC_SETTING_SUFFIX = re.compile(
167
+ r'^(\d{10,11})_(?:ip|op)$', re.IGNORECASE
168
+ )
169
+ _RE_NDC_TRAILING_LETTER = re.compile(r'^(\d{10,11})[A-Za-z]$')
170
+ _RE_NDC_PKG_DIGIT_SUFFIX = re.compile(r'^(\d{10,11})_\d+$')
171
+ # Package + setting double suffix, e.g. "39822105505_4_ip".
172
+ _RE_NDC_PKG_SETTING_SUFFIX = re.compile(
173
+ r'^(\d{10,11})_\d+_(?:ip|op)$', re.IGNORECASE
174
+ )
175
+ # 3-segment dashed NDC, optionally followed by an alpha packaging suffix
176
+ # (e.g. "RL1", "OSPT") or "-<digits>" extra tail (e.g. "-50"). Used for
177
+ # both shape detection and canonicalization.
178
+ _RE_NDC_3SEG = re.compile(
179
+ r'^(\d{4,5})-(\d{3,5})-(\d{1,2})(?:[A-Za-z][A-Za-z0-9]*|-\d+)?$'
180
+ )
181
+
182
+
183
+ def _ndc_canonicalize_base(base: str) -> Optional[str]:
184
+ """10/11-digit numeric base → canonical 11-digit string. None if neither."""
185
+ n = len(base)
186
+ if n == 11:
187
+ return base
188
+ if n == 10:
189
+ return '0' + base
190
+ return None
191
+
192
+
193
+ def _normalize_ndc(
194
+ code: Optional[str],
195
+ ) -> Tuple[Optional[str], str]:
196
+ """Normalize an NDC code to canonical 11-digit no-dash form.
197
+
198
+ Returns ``(canonical_code, code_type)`` where ``code_type`` is either
199
+ ``'NDC'`` (default) or ``'LOCAL'`` (when the value starts with an
200
+ ASCII letter: those are charge-master codes, not NDCs).
201
+
202
+ Rules, first match wins:
203
+
204
+ 1. ``None`` → ``(None, 'NDC')``.
205
+ 2. Empty after strip → ``(raw, 'NDC')``.
206
+ 3. Leading ASCII letter → ``(raw, 'LOCAL')``.
207
+ 4. Already canonical 11-digit → ``(raw, 'NDC')``.
208
+ 5. 9-digit all-numeric → ``('00' + raw, 'NDC')``.
209
+ 6. Bare 10-digit all-numeric → ``('0' + raw, 'NDC')``.
210
+ 7. 10/11-digit base + ``_ip``/``_op`` setting suffix → canonical base.
211
+ 8. 10/11-digit base + single trailing ASCII letter → canonical base.
212
+ 9. 10/11-digit base + ``_<digits>`` package suffix → canonical base.
213
+ 10. 10/11-digit base + ``_<digits>_<ip|op>`` package+setting double
214
+ suffix → canonical base.
215
+ 11. 3-segment dashed shape (with optional alpha/dash packaging
216
+ suffix): strip non-digits; 11 or 10 digits canonicalize, else
217
+ leave as-is.
218
+ 12. Anything else → ``(raw, 'NDC')`` unchanged.
219
+ """
220
+ if code is None:
221
+ return None, 'NDC'
222
+ s = code.strip()
223
+ if not s:
224
+ return code, 'NDC'
225
+
226
+ # 3. Leading-letter codes are local charge-master codes, not NDCs.
227
+ if _RE_NDC_LEADING_ALPHA.match(s):
228
+ return code, 'LOCAL'
229
+
230
+ # 4. Already canonical.
231
+ if _RE_NDC_CANON_11.match(s):
232
+ return s, 'NDC'
233
+
234
+ # 5. 9-digit pad with '00'.
235
+ if _RE_NDC_NINE_DIGIT.match(s):
236
+ return '00' + s, 'NDC'
237
+
238
+ # 6. Bare 10-digit pad with '0'.
239
+ if _RE_NDC_TEN_DIGIT.match(s):
240
+ return '0' + s, 'NDC'
241
+
242
+ # 7-10. 10/11-digit base with various trailing suffixes (setting,
243
+ # letter, package, and package+setting).
244
+ for rx in (
245
+ _RE_NDC_SETTING_SUFFIX,
246
+ _RE_NDC_TRAILING_LETTER,
247
+ _RE_NDC_PKG_DIGIT_SUFFIX,
248
+ _RE_NDC_PKG_SETTING_SUFFIX,
249
+ ):
250
+ m = rx.match(s)
251
+ if m:
252
+ canon = _ndc_canonicalize_base(m.group(1))
253
+ if canon is not None:
254
+ return canon, 'NDC'
255
+
256
+ # 9. 3-segment dashed shape with optional packaging suffix
257
+ # (alpha tail like "RL1" or "-50"). Canonicalize via segment
258
+ # alignment: seg1→5, seg2→4, seg3→2. The packaging
259
+ # suffix is intentionally discarded - it is not part of the
260
+ # canonical NDC product code. Requires seg1≤5, seg2≤4,
261
+ # seg3∈[1,2] digits so total = 11; otherwise leave alone.
262
+ m = _RE_NDC_3SEG.match(s)
263
+ if m:
264
+ seg1, seg2, seg3 = m.group(1), m.group(2), m.group(3)
265
+ if len(seg1) <= 5 and len(seg2) <= 4 and len(seg3) <= 2:
266
+ canon = seg1.zfill(5) + seg2.zfill(4) + seg3.zfill(2)
267
+ if len(canon) == 11:
268
+ return canon, 'NDC'
269
+ # Doesn't fit canonical layout (e.g. shifted-dash 5-5-1):
270
+ # leave as-is rather than guess.
271
+ return s, 'NDC'
272
+
273
+ # 10. Anything else: leave untouched.
274
+ return s, 'NDC'
275
+
276
+
277
+ # Placeholder/test patterns that should be classified as LOCAL
278
+ _RE_PLACEHOLDER = re.compile(
279
+ r'^(?:X{3,}|TEST\b|N/?A$|NONE$|TBD$|UNKNOWN$)',
280
+ re.IGNORECASE,
281
+ )
282
+
283
+ # Composite code prefixes: code values like "HCPCS C1776" or "CPT 87339"
284
+ # embed the code type as a prefix. Keys are uppercase prefixes that can
285
+ # appear at the start of a code value (followed by a space).
286
+ _COMPOSITE_CODE_PREFIXES = {
287
+ 'HCPCS', 'CPT', 'CPT4', 'MS-DRG', 'MSDRG', 'DRG',
288
+ 'NDC', 'REV', 'RC', 'REVENUE', 'REVENUE CODE', 'REVCODE', 'APC', 'CDM',
289
+ 'CDT', 'APR-DRG', 'APRDRG', 'APR DRG', 'APR_DRG', 'CMG',
290
+ }
291
+
292
+ _COMPOSITE_CODE_PREFIXES_BY_LENGTH = sorted(
293
+ _COMPOSITE_CODE_PREFIXES,
294
+ key=len,
295
+ reverse=True,
296
+ )
297
+
298
+ # code_types whose numeric codes may arrive with a spurious trailing
299
+ # ".0" due to Excel / ETL float coercion (e.g. "320.0" → "320"). The strip
300
+ # is scoped to these numeric controlled-vocabulary types only - ICD codes like
301
+ # "250.0" are valid decimal-coded ICD-9 entries and must NOT be altered.
302
+ _FLOAT_COERCE_STRIP_TYPES = frozenset({'RC', 'MS-DRG', 'APR-DRG', 'APC', 'CMG', 'EAPG'})
303
+
304
+
305
+ def _valid_prefixed_code_remainder(rest: str, norm_type: str) -> bool:
306
+ """Return whether a composite prefix leaves a plausible code value."""
307
+ s = rest.strip()
308
+ if norm_type == 'CDT':
309
+ return bool(re.fullmatch(r'[Dd]\d{4}', s))
310
+ if norm_type == 'APR-DRG':
311
+ return bool(re.match(r'\d{3}\s*-\s*\d', s) or re.fullmatch(r'\d{3,4}', s))
312
+ if norm_type == 'APC':
313
+ return bool(re.fullmatch(r'(?:\d{3,4}|[Nn]\d{3})', s))
314
+ if norm_type == 'RC':
315
+ return bool(re.fullmatch(r'\d{1,4}', s))
316
+ if norm_type == 'CMG':
317
+ # 4-digit group number, optionally followed by '-' + comorbidity tier letter
318
+ return bool(re.fullmatch(r'\d{4}(-[A-Z])?', s))
319
+ return True
320
+
321
+ # MS-DRG composite: "MS-DRG V41.0 (FY 2024) 155" - the actual DRG number
322
+ # is the last group of 1-3 digits in the string.
323
+ _RE_DRG_TRAILING_NUM = re.compile(r'(\d{1,3})\s*$')
324
+
325
+
326
+ def _split_attached_composite_code(
327
+ code: str,
328
+ code_type: Optional[str],
329
+ ) -> Tuple[str, Optional[str]]:
330
+ """Split tightly attached code-system prefixes at the start of a value.
331
+
332
+ This is deliberately narrower than the space-delimited composite splitter:
333
+ the full code value must be just the prefix plus a structurally valid code.
334
+ That lets us recover values like ``DRG100`` while avoiding infix false
335
+ positives such as ``SUP-D224DRG`` or free-text markers like ``NEED CPT``.
336
+ """
337
+ stripped = code.strip()
338
+
339
+ rules = (
340
+ ('MS-DRG', r'(?:MS-DRG|MSDRG|DRG)(\d{1,3})'),
341
+ # "MS-012" abbreviated form (MS- + 1-3 digits, no 'DRG' suffix).
342
+ # Distinct from the full "MS-DRG470" pattern above. Only split when the
343
+ # declared code_type is MS-DRG-family, CDM, LOCAL, or absent - never
344
+ # override a conflicting standard type (same guard as the other rules).
345
+ ('MS-DRG', r'MS-(\d{1,3})'),
346
+ ('APR-DRG', r'(?:APR-DRG|APRDRG)(\d{3}(?:-\d|\d)?)'),
347
+ ('APC', r'APC((?:\d{3,4}|[Nn]\d{3}))'),
348
+ ('RC', r'RC(\d{1,4})'),
349
+ ('NDC', r'NDC(\d{9,11}|\d{4,5}-\d{3,5}-\d{1,2}(?:[A-Za-z][A-Za-z0-9]*|-\d+)?)'),
350
+ ('CPT', r'CPT(\d{5}|\d{4}[FTUftu])'),
351
+ ('HCPCS', r'HCPCS([A-CE-Va-ce-v]\d{4})'),
352
+ ('CDT', r'CDT([Dd]\d{4})'),
353
+ # "CMG-1902-D" dash-attached form: CMG + dash + 4-digit group
354
+ # + optional comorbidity tier letter. The space form ("CMG 1902-D") is
355
+ # handled by the space-prefix splitter; this rule covers the dash form.
356
+ ('CMG', r'CMG-(\d{4}(?:-[A-Z])?)'),
357
+ )
358
+
359
+ for prefix_type, pattern in rules:
360
+ m = re.fullmatch(pattern, stripped, flags=re.IGNORECASE)
361
+ if not m:
362
+ continue
363
+
364
+ if code_type:
365
+ norm_ct = _CODE_TYPE_NORMALIZE.get(code_type.strip().upper())
366
+ norm_prefix = _CODE_TYPE_NORMALIZE.get(prefix_type, prefix_type)
367
+ if norm_ct and norm_ct not in ('CDM', 'LOCAL', norm_prefix):
368
+ return code, code_type
369
+
370
+ return m.group(1), prefix_type
371
+
372
+ return code, code_type
373
+
374
+
375
+ def _split_composite_code(
376
+ code: str,
377
+ code_type: Optional[str],
378
+ ) -> Tuple[str, Optional[str]]:
379
+ """
380
+ Split composite code values where code_type is embedded as a prefix.
381
+
382
+ Examples:
383
+ "HCPCS C1776" → code="C1776", code_type="HCPCS"
384
+ "CPT 87339" → code="87339", code_type="CPT"
385
+ "MS-DRG V41.0 (FY 2024) 155" → code="155", code_type="MS-DRG"
386
+ "HCPCS 25009999" → code="25009999", code_type="HCPCS"
387
+
388
+ Only splits when:
389
+ - The code contains a space
390
+ - The part before the first space is a known code type prefix
391
+ - There is no separately declared code_type, OR the declared code_type
392
+ is a non-standard value (not in _CODE_TYPE_NORMALIZE)
393
+
394
+ Returns (code, code_type) - possibly unchanged.
395
+ """
396
+ stripped = code.strip()
397
+ if ' ' not in stripped:
398
+ return _split_attached_composite_code(code, code_type)
399
+
400
+ def _match_composite_prefix(value: str) -> Tuple[Optional[str], Optional[str]]:
401
+ upper_value = value.upper()
402
+ for known_prefix in _COMPOSITE_CODE_PREFIXES_BY_LENGTH:
403
+ if upper_value.startswith(known_prefix + ' '):
404
+ prefix_text = value[:len(known_prefix)].upper().rstrip('®')
405
+ rest_text = value[len(known_prefix) + 1:].strip()
406
+ return prefix_text, rest_text
407
+ return None, None
408
+
409
+ # Don't override a valid, recognized code_type - with exceptions
410
+ if code_type:
411
+ ct_upper = code_type.strip().upper()
412
+ norm_ct = _CODE_TYPE_NORMALIZE.get(ct_upper)
413
+ if norm_ct:
414
+ prefix, rest = _match_composite_prefix(stripped)
415
+ if not prefix:
416
+ first_space = stripped.index(' ')
417
+ prefix = stripped[:first_space].upper().rstrip('®')
418
+ rest = stripped[first_space + 1:].strip()
419
+ # If code_type is hospital-internal (CDM, LOCAL) but the code
420
+ # starts with a standard code type prefix, prefer the standard type
421
+ if norm_ct in ('CDM', 'LOCAL') and prefix in _COMPOSITE_CODE_PREFIXES:
422
+ pass # fall through to split logic below
423
+ # If the prefix matches the declared code_type (redundant), strip it
424
+ elif prefix == ct_upper or _CODE_TYPE_NORMALIZE.get(prefix) == norm_ct:
425
+ if rest:
426
+ if not _valid_prefixed_code_remainder(rest, norm_ct):
427
+ return code, code_type
428
+ # For DRG-family types, extract the trailing DRG number
429
+ if norm_ct == 'MS-DRG':
430
+ m = _RE_DRG_TRAILING_NUM.search(rest)
431
+ if m:
432
+ return m.group(1), code_type
433
+ return code, code_type
434
+ return rest, code_type
435
+ return code, code_type
436
+ else:
437
+ # Code type is standard (CPT, HCPCS, etc.) and prefix doesn't
438
+ # match - don't split (the space is part of the code/description)
439
+ return code, code_type
440
+
441
+ prefix, rest = _match_composite_prefix(stripped)
442
+ if not prefix:
443
+ return code, code_type
444
+
445
+ if not rest:
446
+ return code, code_type
447
+
448
+ # For DRG-family prefixes, the actual DRG number is the trailing digits
449
+ # (e.g. "V41.0 (FY 2024) 155" → "155")
450
+ norm_prefix = _CODE_TYPE_NORMALIZE.get(prefix, prefix)
451
+ if not _valid_prefixed_code_remainder(rest, norm_prefix):
452
+ return code, code_type
453
+ if norm_prefix == 'MS-DRG':
454
+ m = _RE_DRG_TRAILING_NUM.search(rest)
455
+ if m:
456
+ return m.group(1), prefix
457
+ return code, code_type
458
+
459
+ return rest, prefix
460
+
461
+
462
+ # ── Split modifiers baked into the code field ──────────────────────────
463
+ #
464
+ # Hospitals sometimes fuse a modifier onto the procedure code in their
465
+ # MRF: `73721TC`, `87077QW`, `36415CP`, `82274SC`. These don't parse as a
466
+ # valid 5-char CPT, so they land under `code_type='CDM'` or `'LOCAL'` and
467
+ # the modifier is trapped in the code string.
468
+ #
469
+ # The helper takes the known-modifier set as a parameter so its rules can
470
+ # be tested on their own; `apply_baked_modifier_split` below wires in the
471
+ # curated set.
472
+
473
+ # Allowed prefix shapes the splitter recognizes - must be structurally
474
+ # valid CPT or HCPCS Level II. Anything else (ICD-10-PCS 7-char, NDC,
475
+ # legitimate 6-char numerics) MUST NOT be touched.
476
+ #
477
+ # Why these patterns specifically:
478
+ # * `\d{5}` - plain CPT (73721, 87077, 36415, 82274)
479
+ # * `\d{4}[FTU]` - CPT Cat-II/III/PLA (`0001F`, `0202U`, `0237T`).
480
+ # Currently the trailing letter is part of the code, not a modifier;
481
+ # but a baked-modifier composite like `0001FTC` is theoretically
482
+ # possible - exclude for now to avoid ambiguity.
483
+ # * `[A-CE-V]\d{4}` - HCPCS Level II (J0591, A4253). Excludes D
484
+ # (CDT - different code system) and W/X/Y/Z (not assigned).
485
+ #
486
+ # Modifier suffix shape: `[A-Z0-9]{2}` matches the 2-char atomic-modifier
487
+ # format that covers every curated atomic modifier (TC, SG, QW, LT, RT,
488
+ # JW, JZ, 22, 26, 50, etc.). The actual must-be-a-known-modifier check is
489
+ # the caller's responsibility: that's what guards against false-positive
490
+ # splits.
491
+ _RE_BAKED_MODIFIER_CPT5 = re.compile(r'^(\d{5})([A-Z0-9]{2})$')
492
+ _RE_BAKED_MODIFIER_HCPCS = re.compile(r'^([A-CE-V]\d{4})([A-Z0-9]{2})$')
493
+
494
+
495
+ def _split_attached_modifier_code(
496
+ code: str,
497
+ code_type: Optional[str],
498
+ known_modifiers: frozenset,
499
+ canonical_prefixes_by_type: Optional[Dict[str, frozenset]] = None,
500
+ digit_modifiers: Optional[frozenset] = None,
501
+ modifier_validity_by_code: Optional[Dict[str, frozenset]] = None,
502
+ descriptions_by_code: Optional[Dict[str, str]] = None,
503
+ row_description: Optional[str] = None,
504
+ ) -> Tuple[str, Optional[str], Optional[str]]:
505
+ """Detect a `<structurally-valid CPT/HCPCS><known atomic modifier>`
506
+ cell and split it.
507
+
508
+ Returns ``(clean_code, new_code_type, extracted_modifier)``:
509
+ * On a successful split (`73721TC` with `code_type='CDM'` and `TC`
510
+ in ``known_modifiers``): returns ``('73721', 'CPT', 'TC')``.
511
+ * On any guard failure (unrecognized suffix, wrong source
512
+ ``code_type``, NDC/ICD shape, already-classified-as-CPT): returns
513
+ ``(code, code_type, None)`` - caller treats the cell as-is.
514
+
515
+ Guards (in priority order):
516
+
517
+ 1. ``code_type`` must be one of `CDM`, `LOCAL`, or `None` - codes
518
+ already filed under a real coding system are not in scope. A
519
+ `73721TC` already filed as `CPT` would mean the hospital encoded
520
+ it as a 7-char CPT, which is invalid; that's a separate bug.
521
+ 2. The cell must match `_RE_BAKED_MODIFIER_CPT5` or
522
+ `_RE_BAKED_MODIFIER_HCPCS` exactly. ICD-10-PCS (7-char
523
+ alphanumeric like `02573ZZ` where `ZZ` is a legitimate part of
524
+ the code) does NOT match these regexes because the prefix shape
525
+ requires `\\d{5}` or `[A-CE-V]\\d{4}` - ICD codes use a
526
+ different layout. NDC codes are 9–11+ digits with no letter
527
+ suffix, also no match.
528
+ 3. The 2-char suffix must be in ``known_modifiers``: the caller
529
+ passes the curated modifier set.
530
+ 4. If ``canonical_prefixes_by_type`` is provided, the extracted prefix
531
+ must be in the canonical set for the resolved ``target_type``. Kills
532
+ the over-match where 7-digit hospital CDM IDs (`4954052`
533
+ "azithromycin tab") factor structurally as `<5-digit><2-char
534
+ modifier>` but the 5-digit prefix isn't a real published CPT/HCPCS,
535
+ only a coincidence. When ``None``, this guard is skipped.
536
+ 5. Digit-suffix gate: if ``digit_modifiers`` is provided and the suffix
537
+ is in that set (i.e. it is a digit-only modifier token), then
538
+ ``modifier_validity_by_code`` is consulted as the final guard.
539
+ - ``modifier_validity_by_code`` is ``None`` → default-deny.
540
+ - ``modifier_validity_by_code`` is a dict but the (prefix, suffix)
541
+ pair is absent → reject (not CMS-validated).
542
+ - ``modifier_validity_by_code`` has the prefix and suffix in its
543
+ frozenset → accept.
544
+ Letter suffixes never enter this branch.
545
+ 6. Description-consistency guard: applies ONLY inside the digit-suffix
546
+ branch, AFTER Guard 5 passes.
547
+ - ``descriptions_by_code`` is ``None`` → skip the guard.
548
+ - ``descriptions_by_code`` is a dict → look up the reference
549
+ description for ``prefix`` and require
550
+ ``_description_matches(row_description, canonical_desc)`` → True.
551
+ If no reference description is found, or the descriptions don't
552
+ match → reject.
553
+ Letter suffixes are NEVER subject to Guard 6.
554
+
555
+ Casing: the regexes accept uppercase only. Callers should already
556
+ have uppercased the cell via the existing pipeline (`_strip_non_ascii`
557
+ + the `_CODE_TYPE_NORMALIZE` upper).
558
+ """
559
+ if not code:
560
+ return code, code_type, None
561
+
562
+ # Guard 1: source code_type must be a hospital-internal bucket.
563
+ if code_type not in (None, 'CDM', 'LOCAL'):
564
+ return code, code_type, None
565
+
566
+ stripped = code.strip()
567
+ if not stripped:
568
+ return code, code_type, None
569
+
570
+ # Guard 2: structural match. Try CPT first (more common at the
571
+ # prevalence we measured), then HCPCS.
572
+ m = _RE_BAKED_MODIFIER_CPT5.match(stripped)
573
+ target_type: Optional[str] = None
574
+ if m:
575
+ target_type = 'CPT'
576
+ else:
577
+ m = _RE_BAKED_MODIFIER_HCPCS.match(stripped)
578
+ if m:
579
+ target_type = 'HCPCS'
580
+
581
+ if m is None:
582
+ return code, code_type, None
583
+
584
+ prefix, suffix = m.group(1), m.group(2)
585
+
586
+ # Guard 3: the suffix must be a real modifier per the caller's
587
+ # curated set.
588
+ if suffix not in known_modifiers:
589
+ return code, code_type, None
590
+
591
+ # Guard 4: the prefix must be a canonical published code. Skipped when
592
+ # the caller passes None.
593
+ if canonical_prefixes_by_type is not None:
594
+ prefixes = canonical_prefixes_by_type.get(target_type)
595
+ if not prefixes or prefix not in prefixes:
596
+ return code, code_type, None
597
+
598
+ # Guard 5: digit-suffix CMS-validity gate.
599
+ # Only digit tokens enter this branch; letter suffixes are unaffected.
600
+ # `modifier_validity_by_code` is keyed by bare CPT code only, so HCPCS
601
+ # digit composites have no entries and fall through to default-deny
602
+ # below. The frozensets hold only CMS-valid modifiers, so membership
603
+ # here IS the validity check.
604
+ if digit_modifiers and suffix in digit_modifiers:
605
+ # Default-deny when no validity matrix is provided.
606
+ if modifier_validity_by_code is None:
607
+ return code, code_type, None
608
+ valid = modifier_validity_by_code.get(prefix)
609
+ if not valid or suffix not in valid:
610
+ return code, code_type, None
611
+
612
+ # Guard 6: description-consistency guard. Skipped when
613
+ # descriptions_by_code is None. When provided, the row's free-text
614
+ # description must match the reference description via
615
+ # `_description_matches`: prevents coincidental CDM IDs (drugs,
616
+ # devices, generics) from splitting just because their 5-digit
617
+ # prefix happens to be a valid CPT.
618
+ if descriptions_by_code is not None:
619
+ canonical_desc = descriptions_by_code.get(prefix)
620
+ if not _description_matches(row_description, canonical_desc):
621
+ return code, code_type, None
622
+
623
+ return prefix, target_type, suffix
624
+
625
+
626
+ # ---------------------------------------------------------------------------
627
+ # The curated modifier set `apply_baked_modifier_split` splits on.
628
+ #
629
+ # Only LETTER suffixes split on shape alone. Digit suffixes are far riskier:
630
+ # 7-digit hospital CDM IDs often factor as `<5-digit><2-digit>` by
631
+ # coincidence (`4954052` "azithromycin tab" looks like `49540` + `52`), and
632
+ # splitting them on shape alone produced a flood of false positives. So the
633
+ # digit tokens split only when a CMS (code, modifier) validity matrix is
634
+ # supplied and confirms the pair (PC/TC indicator, bilateral indicator,
635
+ # CLIA-waived list).
636
+ #
637
+ # Other common baked modifiers (`SG`, `CL`, `AS`, `80-82`, `62/66`,
638
+ # `53/73/74`) are deliberately not in the set yet: the set is kept equal to
639
+ # the curated modifier dictionary so every split modifier is one the rest of
640
+ # a pipeline knows how to classify.
641
+ _BAKED_MODIFIER_LETTER_KEYS = frozenset({
642
+ # ------- component -------
643
+ "TC",
644
+ # ------- drug_supply -------
645
+ "JW", "JZ", "TB", "SL",
646
+ # ------- distinct -------
647
+ "XE", "XP", "XS", "XU",
648
+ # ------- enhancement / oversight -------
649
+ "QW",
650
+ # ------- anatomy / supply -------
651
+ "LT", "RT",
652
+ # ------- admin_noise -------
653
+ "GY", "FY", "GO", "GN", "GP", "NU", "PO",
654
+ })
655
+
656
+ # Numeric-suffix tokens. Split only when a CMS validity matrix confirms the
657
+ # (CPT, modifier) pair.
658
+ _BAKED_MODIFIER_DIGIT_KEYS = frozenset({
659
+ # ------- component -------
660
+ "26",
661
+ # ------- surgical_phase -------
662
+ "54", "55", "56", "58", "78", "79", "24",
663
+ # ------- repeat / distinct -------
664
+ "76", "77", "91", "59",
665
+ # ------- enhancement / oversight -------
666
+ "22", "25", "50", "52", "90", "95",
667
+ })
668
+
669
+ _BAKED_MODIFIER_SPLIT_KEYS = _BAKED_MODIFIER_LETTER_KEYS | _BAKED_MODIFIER_DIGIT_KEYS
670
+
671
+
672
+ # ---------------------------------------------------------------------------
673
+ # Description-consistency helper.
674
+ #
675
+ # Decides whether a CDM row's free-text description is consistent with a
676
+ # reference description for the code (``ReferenceData.code_descriptions``).
677
+ # Both strings are normalized before comparison:
678
+ #
679
+ # 1. Uppercase; strip surrounding quotes; replace non-alphanumeric runs
680
+ # with spaces; collapse whitespace.
681
+ # 2. Expand a small abbreviation map so common radiography short-forms
682
+ # align with the reference descriptions.
683
+ # 3. Reject on a denylist of generic/non-specific descriptions that match
684
+ # virtually anything (e.g. "OTHER OUTPATIENT", "MISC").
685
+ # 4. Tokenize both sides; drop 1-char tokens (noise). Compute overlap of
686
+ # the canonical token set C against the row token set R.
687
+ # Match condition: |C ∩ R| / |C| >= 0.5 AND |C ∩ R| >= 1
688
+ # AND at least one shared token has len >= 4 (avoids matching on tiny
689
+ # words like "OF", "OR", "BY", "MG" only).
690
+ #
691
+ # Rationale for the 0.5 threshold: reference short descriptions are often
692
+ # brief (2-4 meaningful tokens); requiring only 50 % coverage admits slight synonymy
693
+ # while blocking completely unrelated descriptions (drugs, devices).
694
+ # ---------------------------------------------------------------------------
695
+
696
+ # Small abbreviation expansion map.
697
+ # Applied in two passes:
698
+ # Pass 1 (pre-normalization): whole-string regex substitutions that must
699
+ # fire BEFORE the general non-alphanumeric → space replacement. Handles
700
+ # hyphenated forms like "X-RAY" → "XRAY" which would otherwise be split
701
+ # into "X" + "RAY" by the normalizer.
702
+ # Pass 2 (post-normalization): token-level lookup on the already-uppercased,
703
+ # whitespace-collapsed string. Keys are standalone uppercase tokens.
704
+ # Keep the map small and well-commented - it's a precision tuning knob.
705
+
706
+ # Pass 1: pre-normalization whole-word substitutions (case-insensitive regex).
707
+ # Each entry is (pattern, replacement) where pattern is matched against the
708
+ # uppercased raw string before non-alpha stripping.
709
+ _DESC_PRENORM_SUBS: list = [
710
+ # "X-RAY" / "X-RAYS" → "XRAY" / "XRAYS" (hyphen removed before normalization
711
+ # strips all non-alphanumeric, so the compound becomes a single token).
712
+ (re.compile(r"\bX-RAYS?\b"), "XRAY"),
713
+ ]
714
+
715
+ # Pass 2: post-normalization token → canonical expansion.
716
+ # Applied after uppercasing + non-alpha strip + collapse.
717
+ _DESC_ABBREV: Dict[str, str] = {
718
+ "XR": "XRAY", # "XR FEMUR" → "XRAY FEMUR"
719
+ "BIL": "BILATERAL",
720
+ "BILAT": "BILATERAL",
721
+ "W": "WITH", # "W/" becomes "W" after non-alpha strip
722
+ "WO": "WITHOUT", # "W/O" becomes "WO" after non-alpha strip
723
+ "BX": "BIOPSY",
724
+ }
725
+
726
+ # Normalized forms of the generic denylist (applied after normalization step).
727
+ _DESC_GENERIC_DENYLIST: frozenset = frozenset({
728
+ "OTHER OUTPATIENT",
729
+ "OUTPATIENT",
730
+ "OTHER",
731
+ "MISC",
732
+ "MISCELLANEOUS",
733
+ })
734
+
735
+ # Non-discriminating filler words. Removed from BOTH token sets before the
736
+ # coverage calc so they don't dilute the canonical denominator: short
737
+ # reference descriptions often carry "AND"/"OF", which would otherwise drop
738
+ # a real composite below the threshold.
739
+ # Dropping them never lowers precision (they carry no clinical signal).
740
+ _DESC_STOPWORDS: frozenset = frozenset({
741
+ "AND", "OF", "OR", "THE", "WITH", "WITHOUT", "FOR", "TO", "IN", "ON",
742
+ "BY", "A", "AN",
743
+ })
744
+
745
+ _RE_DESC_NONALNUM = re.compile(r"[^A-Z0-9]+")
746
+
747
+
748
+ def _normalize_desc(s: str) -> str:
749
+ """Normalize a description for comparison:
750
+ upper-case → pre-norm substitutions → strip surrounding quotes →
751
+ replace non-alphanumeric with spaces → collapse whitespace.
752
+ """
753
+ s = s.upper().strip()
754
+ # Pass 1: pre-normalization substitutions (hyphenated compounds).
755
+ for pattern, repl in _DESC_PRENORM_SUBS:
756
+ s = pattern.sub(repl, s)
757
+ s = s.strip('"').strip("'")
758
+ s = _RE_DESC_NONALNUM.sub(" ", s).strip()
759
+ return s
760
+
761
+
762
+ def _description_matches(
763
+ row_desc: Optional[str],
764
+ canonical_desc: Optional[str],
765
+ ) -> bool:
766
+ """Return True when the CDM row description is consistent with the
767
+ reference description for the code.
768
+
769
+ Both strings must be non-empty after normalization; empty-after-normalize
770
+ → False (default-deny). Generic/non-specific row descriptions → False.
771
+
772
+ Matching rule (token-set overlap):
773
+ Let C = canonical token set (tokens len >= 2), R = row token set
774
+ (tokens len >= 2).
775
+ Match when:
776
+ |C ∩ R| / |C| >= 0.5 (at least half of canonical tokens covered)
777
+ AND |C ∩ R| >= 1
778
+ AND at least one shared token has len >= 4 (no trivial-word match)
779
+ """
780
+ if not row_desc or not canonical_desc:
781
+ return False
782
+
783
+ row_norm = _normalize_desc(row_desc)
784
+ can_norm = _normalize_desc(canonical_desc)
785
+
786
+ if not row_norm or not can_norm:
787
+ return False
788
+
789
+ # Reject generic row descriptions outright.
790
+ if row_norm in _DESC_GENERIC_DENYLIST:
791
+ return False
792
+
793
+ # Pass 2: token-level abbreviation expansion.
794
+ def _expand(text: str) -> str:
795
+ tokens = text.split()
796
+ return " ".join(_DESC_ABBREV.get(t, t) for t in tokens)
797
+
798
+ row_norm = _expand(row_norm)
799
+ can_norm = _expand(can_norm)
800
+
801
+ # Tokenize; drop 1-char noise tokens and non-discriminating stop-words
802
+ # (stop-words in the reference description would otherwise inflate the
803
+ # coverage denominator and reject real composites).
804
+ row_tokens = {t for t in row_norm.split()
805
+ if len(t) >= 2 and t not in _DESC_STOPWORDS}
806
+ can_tokens = {t for t in can_norm.split()
807
+ if len(t) >= 2 and t not in _DESC_STOPWORDS}
808
+
809
+ if not can_tokens:
810
+ return False
811
+
812
+ overlap = can_tokens & row_tokens
813
+ if not overlap:
814
+ return False
815
+
816
+ # Coverage: at least half of canonical tokens must be present.
817
+ coverage = len(overlap) / len(can_tokens)
818
+ if coverage < 0.5:
819
+ return False
820
+
821
+ # At least one shared token must be "substantial" (len >= 4) so two
822
+ # descriptions that share only tiny words ("OF", "MG", "BY") don't match.
823
+ if not any(len(t) >= 4 for t in overlap):
824
+ return False
825
+
826
+ return True
827
+
828
+
829
+ def apply_baked_modifier_split(
830
+ code: Optional[str],
831
+ code_type: Optional[str],
832
+ description: Optional[str] = None,
833
+ *,
834
+ ref: Optional[ReferenceData] = None,
835
+ stats: Optional[ParseStats] = None,
836
+ ) -> Tuple[Optional[str], Optional[str], Optional[str]]:
837
+ """Split a modifier baked into the code, using the curated modifier set.
838
+
839
+ Returns ``(code, code_type, baked_modifier)``:
840
+ * Successful split (`'73721TC' + 'CDM'` → `('73721', 'CPT', 'TC')`):
841
+ caller updates code/code_type AND merges ``baked_modifier`` into
842
+ the row's `modifiers` field (via `merge_modifier_into_field`).
843
+ * No split (any guard failure): returns the original ``(code,
844
+ code_type, None)`` unchanged. Caller proceeds as before.
845
+
846
+ Safe to call with ``code = None`` or ``code = ''`` (returns inputs
847
+ unchanged with ``baked_modifier = None``).
848
+
849
+ Letter suffixes split on shape alone. With ``ref``:
850
+ * ``ref.code_prefixes`` makes the prefix a published CPT/HCPCS code
851
+ (Guard 4);
852
+ * ``ref.modifier_validity`` lets digit suffixes split when CMS marks
853
+ the (code, modifier) pair valid (Guard 5). Without it digit
854
+ suffixes never split;
855
+ * ``ref.code_descriptions`` additionally requires the row
856
+ description to match the code's reference description before a
857
+ digit suffix splits (Guard 6).
858
+
859
+ ``stats``, when given, counts each split.
860
+ """
861
+ if not code:
862
+ return code, code_type, None
863
+ result = _split_attached_modifier_code(
864
+ code, code_type,
865
+ _BAKED_MODIFIER_SPLIT_KEYS,
866
+ canonical_prefixes_by_type=ref.code_prefixes if ref else None,
867
+ digit_modifiers=_BAKED_MODIFIER_DIGIT_KEYS,
868
+ modifier_validity_by_code=ref.modifier_validity if ref else None,
869
+ descriptions_by_code=ref.code_descriptions if ref else None,
870
+ row_description=description,
871
+ )
872
+ if stats is not None and result[2]:
873
+ stats.record_baked_modifier_split(result[2], code_type)
874
+ return result
875
+
876
+
877
+ def merge_modifier_into_field(
878
+ existing: Optional[str],
879
+ baked: Optional[str],
880
+ ) -> Optional[str]:
881
+ """Merge a modifier split out of the code into a row's modifiers field.
882
+
883
+ Used after ``apply_baked_modifier_split`` returns a non-None
884
+ ``baked_modifier``: the caller adds it to whatever the MRF row already
885
+ has in its `modifiers` column. Token-set semantics: order is
886
+ preserved (existing tokens first, baked appended last) but a duplicate
887
+ is dropped so a fixture row with both `'73721TC'` AND a `'TC'` in its
888
+ modifiers column doesn't end up with `'TC,TC'`.
889
+
890
+ Tokens are split on `[|, ;]+`, the delimiter set modifier fields use
891
+ in the wild. The output uses a single comma separator so it
892
+ round-trips through the same tokenizer cleanly.
893
+
894
+ Returns the merged string, or ``None`` if both inputs are empty.
895
+ """
896
+ if not baked:
897
+ return existing
898
+ baked = baked.strip()
899
+ if not baked:
900
+ return existing
901
+ if not existing or not existing.strip():
902
+ return baked
903
+
904
+ # Use the same delimiter regex readers split on so we don't invent a
905
+ # new token boundary here.
906
+ tokens = [t for t in re.split(r"[|, ;]+", existing) if t]
907
+ if baked in tokens:
908
+ # Already present - return canonicalized form (comma-joined, no
909
+ # leading/trailing whitespace) so we don't drift the field shape.
910
+ return ",".join(tokens)
911
+ tokens.append(baked)
912
+ return ",".join(tokens)
913
+
914
+
915
+ def normalize_code(
916
+ code: Optional[str],
917
+ code_type: Optional[str],
918
+ ) -> Tuple[Optional[str], Optional[str]]:
919
+ """
920
+ Normalize code and code_type to standard formats.
921
+
922
+ Returns: (normalized_code, normalized_code_type)
923
+
924
+ Rules applied:
925
+ 1. Strip non-ASCII characters (mojibake, encoding artifacts)
926
+ 2. Uppercase and canonicalize code_type via _CODE_TYPE_NORMALIZE
927
+ 2A. NDCHCPCS / NDCCPT disambiguation: pick real type from code shape
928
+ 2B. Junk code_type discard (TYPE/MODIFIER/MODIFIERS/OBS/DOT/UN → treat
929
+ as missing, re-derive from code shape)
930
+ 3. Detect hospital-internal codes (SUP-*, PX-*, RX-*) → CDM
931
+ 4. Detect misclassified Revenue Codes stored as HCPCS → RC
932
+ 4.5. Reclassify unambiguous standard code shapes mislabeled CDM/LOCAL
933
+ 5. Normalize HCPCS Level I → CPT for 5-digit numeric codes
934
+ 5.5. Reclassify D-prefix codes (CDT dental) → CDT
935
+ 5.6. HCPCS Level II forward fix: letter-prefix [A-CE-V]\\d{4} that
936
+ hospitals labeled CPT → HCPCS (excludes D handled by 5.5)
937
+ 5.7. CPT Cat II/III/PLA reverse fix: \\d{4}[FTU] that hospitals
938
+ labeled HCPCS → CPT
939
+ 6. Zero-pad Revenue Codes to 4 digits
940
+ 6A. Cross-system reclassify: controlled-vocab numeric type holding a
941
+ standard-shape code (e.g. CPT 99232 declared as RC) → reclassify
942
+ to correct standard type before residue gating.
943
+ 7. Normalize NDC to canonical 11-digit no-dash form (see
944
+ _normalize_ndc): leading-letter NDCs reclassify to LOCAL;
945
+ 3-segment dashed forms with optional packaging suffix collapse to
946
+ 11 digits; 9-digit numeric pads to 11 with '00'; 10/11-digit
947
+ bases with _ip/_op, trailing letter, or _<digits> package suffix
948
+ canonicalize to 11.
949
+ 8. Infer code_type from code format when code_type is missing (then
950
+ re-apply NDC normalization if NDC was inferred)
951
+ 9. Validate code against declared code_type format; extract valid code
952
+ from garbage wrapping when possible (e.g. XJ0591X → J0591/HCPCS)
953
+ 10. Classify placeholder/test codes as LOCAL
954
+ 11. APR-DRG qualifier stripping
955
+ 12. Residue gating: controlled-vocab numeric type with a code that still
956
+ fails the canonical regex → reclassify to LOCAL (code preserved).
957
+ """
958
+ if not code and not code_type:
959
+ return code, code_type
960
+
961
+ # --- Step 1: Strip non-ASCII characters ---
962
+ if code:
963
+ cleaned = _strip_non_ascii(code.strip())
964
+ if cleaned != code.strip():
965
+ code = cleaned
966
+ if not code:
967
+ return code, code_type
968
+
969
+ # --- Step 1.5: Split composite code values ---
970
+ # Some hospitals (e.g. Partners Healthcare / Brigham & Women's) embed the
971
+ # code type as a prefix in the code column: "HCPCS C1776", "CPT 87339",
972
+ # "MS-DRG V41.0 (FY 2024) 155". Split these so the code type flows into
973
+ # Step 2 for normalization.
974
+ if code:
975
+ code, code_type = _split_composite_code(code, code_type)
976
+
977
+ # --- Step 2: Normalize code_type ---
978
+ norm_type = None
979
+ if code_type:
980
+ raw_upper = code_type.strip().upper()
981
+ norm_type = _CODE_TYPE_NORMALIZE.get(raw_upper, raw_upper)
982
+
983
+ # --- Step 2A: NDCHCPCS / NDCCPT disambiguation ---
984
+ # Some CMS/hospital data formats emit compound types like "NDCHCPCS" or
985
+ # "NDCCPT" to indicate the code column may contain either an NDC or a
986
+ # procedure code. Pick the real type by code shape:
987
+ # - NDC shape → normalize through _normalize_ndc() → NDC or LOCAL
988
+ # - [A-CE-V]\d{4} → HCPCS (Level II letter-prefix)
989
+ # - \d{5} or \d{4}[FTU] → CPT (for NDCCPT; also acceptable for NDCHCPCS)
990
+ # - else → LOCAL
991
+ if code and norm_type in ('NDCHCPCS', 'NDCCPT'):
992
+ s = code.strip()
993
+ # NDC shape: 11-digit, 9-digit, 10-digit, or 3-segment dashed form
994
+ if (re.fullmatch(r'\d{9,11}', s)
995
+ or re.fullmatch(r'\d{4,5}-\d{3,4}-\d{1,2}', s)):
996
+ new_code, new_type = _normalize_ndc(code)
997
+ return new_code, new_type
998
+ elif re.fullmatch(r'[A-CE-Va-ce-v]\d{4}', s):
999
+ norm_type = 'HCPCS'
1000
+ elif (re.fullmatch(r'\d{5}', s)
1001
+ or re.fullmatch(r'\d{4}[FTUftu]', s)):
1002
+ norm_type = 'CPT'
1003
+ else:
1004
+ norm_type = 'LOCAL'
1005
+
1006
+ # --- Step 2B: Junk code_type discard ---
1007
+ # Non-system labels (TYPE, MODIFIER, MODIFIERS, OBS, DOT, UN) carry no
1008
+ # system information. Discard them so the blank-type path (steps 4.5,
1009
+ # 5, 5.5–5.7, 8) can re-derive the type from code shape.
1010
+ if norm_type in _JUNK_CODE_TYPES:
1011
+ norm_type = None
1012
+
1013
+ # --- Step 2.5: Strip trailing float-coercion artifact (.0) ---
1014
+ # Excel / ETL tools sometimes coerce integer revenue codes and DRG numbers
1015
+ # to floats, writing "320.0" instead of "320". Strip a trailing "\.0+$"
1016
+ # ONLY for numeric controlled-vocabulary types where the decimal is always
1017
+ # spurious. MUST NOT apply to ICD (250.0 is a valid ICD-9 code), CPT,
1018
+ # HCPCS, NDC, CDM, LOCAL, or any other type.
1019
+ if code and norm_type in _FLOAT_COERCE_STRIP_TYPES:
1020
+ stripped_code = code.strip()
1021
+ stripped_code = re.sub(r'\.0+$', '', stripped_code)
1022
+ if stripped_code != code.strip():
1023
+ code = stripped_code
1024
+
1025
+ # --- Step 3: Detect hospital-internal codes by prefix ---
1026
+ if code:
1027
+ code_upper = code.strip()
1028
+ # SUP- (supplies), PX- (procedures), RX- (pharmacy) are CDM conventions
1029
+ if code_upper.startswith(('SUP-', 'PX-', 'RX-')):
1030
+ return code_upper, 'CDM'
1031
+
1032
+ # --- Step 4: Detect misclassified codes ---
1033
+ # Some hospitals report RC values under code_type='HCPCS'
1034
+ if code and norm_type in ('HCPCS', None) and code.strip() in _KNOWN_REVENUE_CODES:
1035
+ norm_type = 'RC'
1036
+ # Numeric codes > 5 digits labeled as CPT/HCPCS are really CDM codes
1037
+ elif (code and norm_type in ('CPT', 'HCPCS')
1038
+ and code.strip().isdigit() and len(code.strip()) > 5):
1039
+ norm_type = 'CDM'
1040
+
1041
+ # --- Step 4.5: Recover standard codes mislabeled as CDM/LOCAL ---
1042
+ # CDM/LOCAL are hospital-internal buckets, but many files put standard
1043
+ # codes there. Only reclassify shapes that are unambiguous. Bare 3-digit
1044
+ # DRGs and bare 4-digit APCs are intentionally not inferred because they
1045
+ # collide with revenue codes and hospital-local identifiers. Prefixed
1046
+ # values such as "MS-DRG 470" and "APC 5191" are handled by Step 1.5.
1047
+ if code and norm_type in ('CDM', 'LOCAL'):
1048
+ stripped = code.strip()
1049
+ if re.fullmatch(r'[Dd]\d{4}', stripped):
1050
+ norm_type = 'CDT'
1051
+ elif re.fullmatch(r'[A-CE-Va-ce-v]\d{4}', stripped):
1052
+ norm_type = 'HCPCS'
1053
+ elif (re.fullmatch(r'\d{5}', stripped)
1054
+ or re.fullmatch(r'\d{4}[FTUftu]', stripped)):
1055
+ norm_type = 'CPT'
1056
+
1057
+ # --- Step 5: Normalize HCPCS Level I → CPT ---
1058
+ # Many hospitals label all procedure codes as 'HCPCS' even when they are
1059
+ # 5-digit numeric CPT codes. HCPCS Level I ≡ CPT, so normalize them.
1060
+ # True HCPCS Level II codes have a letter prefix (A0000-V9999) and are
1061
+ # already correctly classified.
1062
+ if (code and norm_type == 'HCPCS'
1063
+ and code.strip().isdigit() and len(code.strip()) == 5):
1064
+ norm_type = 'CPT'
1065
+
1066
+ # --- Step 5.5: Reclassify D-prefix codes as CDT ---
1067
+ # CDT dental codes (D0120, D7509, etc.) are often misreported as HCPCS or
1068
+ # CPT by hospitals. Reclassify them so dental codes have a consistent type.
1069
+ if (code and norm_type in ('HCPCS', 'CPT')
1070
+ and re.fullmatch(r'[Dd]\d{4}', code.strip())):
1071
+ norm_type = 'CDT'
1072
+
1073
+ # --- Step 5.6: HCPCS Level II forward fix ---
1074
+ # Letter-prefix Level II codes (A0000-V9999, excluding D) are HCPCS by
1075
+ # definition. Some hospital MRFs misclassify them as CPT (e.g. Q5128
1076
+ # under code_type='CPT'). CPT codes are unambiguously 5-digit numerics
1077
+ # or `\d{4}[FTU]` Cat II/III/PLA codes, never letter-prefix.
1078
+ # Excludes D (CDT) which Step 5.5 already handled.
1079
+ if (code and norm_type == 'CPT'
1080
+ and re.fullmatch(r'[A-CE-Va-ce-v]\d{4}', code.strip())):
1081
+ norm_type = 'HCPCS'
1082
+
1083
+ # --- Step 5.7: CPT Cat II/III/PLA reverse fix ---
1084
+ # Cat II (\d{4}F), Cat III (\d{4}T), and PLA (\d{4}U) codes are CPT
1085
+ # by definition (HCPCS Level II never has a trailing F/T/U). Some
1086
+ # hospital MRFs put these under code_type='HCPCS', and some omit the
1087
+ # type entirely; normalize both cases to CPT.
1088
+ if (code and norm_type in ('HCPCS', None)
1089
+ and re.fullmatch(r'\d{4}[FTUftu]', code.strip())):
1090
+ norm_type = 'CPT'
1091
+
1092
+ # --- Step 6A: Cross-system reclassify ---
1093
+ # A controlled-vocab numeric type (RC / MS-DRG / APR-DRG / APC / CMG /
1094
+ # EAPG) may hold a standard-shape procedure code that a hospital
1095
+ # misrouted (e.g. CPT 99232 declared as RC). If the code fails the
1096
+ # type's canonical regex AND matches a standard shape, reclassify:
1097
+ # \d{5} / \d{4}[FTU] → CPT
1098
+ # [A-CE-V]\d{4} → HCPCS
1099
+ # D\d{4} → CDT
1100
+ # Run this BEFORE residue gating (step 12) so reclassified codes never
1101
+ # reach the LOCAL fallback. RC zero-padding (step 6) runs after this so
1102
+ # we don't accidentally treat a 5-digit CPT as a 5-digit revenue code.
1103
+ if code and norm_type in _CONTROLLED_VOCAB_VALIDATORS:
1104
+ validator = _CONTROLLED_VOCAB_VALIDATORS[norm_type]
1105
+ stripped = code.strip()
1106
+ if not validator.fullmatch(stripped):
1107
+ if re.fullmatch(r'\d{5}', stripped):
1108
+ norm_type = 'CPT'
1109
+ elif re.fullmatch(r'\d{4}[FTUftu]', stripped):
1110
+ norm_type = 'CPT'
1111
+ elif re.fullmatch(r'[A-CE-Va-ce-v]\d{4}', stripped):
1112
+ norm_type = 'HCPCS'
1113
+ elif re.fullmatch(r'[Dd]\d{4}', stripped):
1114
+ norm_type = 'CDT'
1115
+
1116
+ # --- Step 6: Zero-pad Revenue Codes ---
1117
+ if code and norm_type == 'RC':
1118
+ stripped = code.strip()
1119
+ if stripped.isdigit():
1120
+ code = stripped.zfill(4)
1121
+ return code, norm_type
1122
+
1123
+ # --- Step 7: Normalize NDC to canonical 11-digit no-dash form ---
1124
+ # See _normalize_ndc(). Handles leading-letter→LOCAL reclassification,
1125
+ # 9- and 10-digit padding, setting/letter/package (incl. double)
1126
+ # suffixes, and 3-segment dashed shapes.
1127
+ if code and norm_type == 'NDC':
1128
+ new_code, new_type = _normalize_ndc(code)
1129
+ return new_code, new_type
1130
+
1131
+ # --- Step 8: Infer code_type from code format when missing ---
1132
+ if code and not norm_type:
1133
+ stripped = code.strip()
1134
+ # CDT dental codes: D + 4 digits (D0120, D7509, etc.)
1135
+ if re.fullmatch(r'[Dd]\d{4}', stripped):
1136
+ norm_type = 'CDT'
1137
+ # HCPCS Level II: letter + 4 digits (J0690, C1776, A4550, etc.)
1138
+ elif re.fullmatch(r'[A-Za-z]\d{4}', stripped):
1139
+ norm_type = 'HCPCS'
1140
+ # CPT: exactly 5 digits
1141
+ elif re.fullmatch(r'\d{5}', stripped):
1142
+ norm_type = 'CPT'
1143
+ # CPT Category II/III/PLA: 4 digits + F/T/U (1036F, 0237T, 0016U)
1144
+ elif re.fullmatch(r'\d{4}[FTUftu]', stripped):
1145
+ norm_type = 'CPT'
1146
+ # NDC: digits with dashes (10-11 digit segments). Run the full
1147
+ # NDC normalizer so the inferred NDC is also canonicalized.
1148
+ elif re.fullmatch(r'\d{4,5}-\d{3,4}-\d{1,2}', stripped):
1149
+ new_code, new_type = _normalize_ndc(code)
1150
+ return new_code, new_type
1151
+ # MS-DRG: exactly 3 digits
1152
+ elif re.fullmatch(r'\d{3}', stripped):
1153
+ # Ambiguous: could be RC or MS-DRG. Don't guess - leave as-is.
1154
+ pass
1155
+
1156
+ # --- Step 9: Validate code against declared code_type ---
1157
+ # If the code doesn't match the expected format for its declared type,
1158
+ # try to extract a valid code from the raw value. Handles cases like
1159
+ # "XJ0591X" labeled as CPT → extract "J0591" and reclassify as HCPCS.
1160
+ if code and norm_type in ('CPT', 'HCPCS'):
1161
+ stripped = code.strip().upper()
1162
+ is_valid = (
1163
+ _RE_CPT.fullmatch(stripped) # CPT: 99213
1164
+ or _RE_CPT_PLA.fullmatch(stripped) # PLA: 0202U, 0003M
1165
+ or _RE_HCPCS.fullmatch(stripped) # HCPCS: J0591
1166
+ or _RE_CAT3.fullmatch(stripped) # Category III: 0237T
1167
+ )
1168
+ if not is_valid:
1169
+ # Try to extract a valid code from the garbage
1170
+ extracted_code, extracted_type = _extract_code_from_garbage(stripped)
1171
+ if extracted_code:
1172
+ code = extracted_code
1173
+ norm_type = extracted_type
1174
+ else:
1175
+ # Can't salvage - downgrade to LOCAL
1176
+ norm_type = 'LOCAL'
1177
+
1178
+ # --- Step 10: Classify placeholder/test codes ---
1179
+ if code and norm_type not in ('RC', 'NDC', 'CDM', 'MS-DRG', 'APC'):
1180
+ stripped = code.strip()
1181
+ if _RE_PLACEHOLDER.match(stripped):
1182
+ norm_type = 'LOCAL'
1183
+
1184
+ # --- Step 11: APR-DRG qualifier stripping ---
1185
+ # APR-DRG codes are formatted as 'NNN-S' where NNN is 3-digit DRG and S is
1186
+ # a 1-digit severity (0-4). Many hospitals append qualifier text:
1187
+ # '001-1 Short Stay' -> '001-1'
1188
+ # '001-1- IP LOS Greater Than 50' -> '001-1'
1189
+ # '001 - 1' -> '001-1' (normalise whitespace)
1190
+ if code and norm_type == 'APR-DRG':
1191
+ code = _strip_apr_drg_qualifier(code)
1192
+
1193
+ if code:
1194
+ code = code.strip()
1195
+
1196
+ # --- Step 12: Residue gating → LOCAL ---
1197
+ # After all recovery/reclassify steps above, if the final code_type is a
1198
+ # controlled-vocab numeric system and the final code still fails that
1199
+ # system's canonical regex, route code_type to LOCAL. The code value is
1200
+ # preserved unchanged - this is a classification fix, not a deletion.
1201
+ # Examples that land here: MS-DRG '1622' (4-digit, valid range is 1-3
1202
+ # digits), EAPG wrong-length numeric, miscellaneous numeric hospital IDs.
1203
+ # Do NOT gate CPT or HCPCS: validating those needs the licensed code
1204
+ # lists, and shape checks already ran in steps 5-9.
1205
+ #
1206
+ # Scope: only numeric-looking codes (all-digit, or digit-dash for
1207
+ # APR-DRG/CMG subtypes). Non-numeric or mixed values (e.g. 'MSCODE',
1208
+ # 'MS-012', 'NONE' under RC) are preserved in their declared type -
1209
+ # they are hospital chargemaster labels that cannot be safely reclassified.
1210
+ if code and norm_type in _CONTROLLED_VOCAB_VALIDATORS:
1211
+ validator = _CONTROLLED_VOCAB_VALIDATORS[norm_type]
1212
+ if not validator.fullmatch(code):
1213
+ # Only gate codes that are purely numeric (or digit+dash, which is
1214
+ # the canonical form for APR-DRG/CMG). Letter-prefix or mixed codes
1215
+ # stay in their declared type.
1216
+ if re.fullmatch(r'[\d-]+', code):
1217
+ norm_type = 'LOCAL'
1218
+
1219
+ return code, norm_type
1220
+
1221
+
1222
+ # Regex used by _strip_apr_drg_qualifier: extracts the NNN-S base from the
1223
+ # start of an APR-DRG code, tolerating whitespace and dashes.
1224
+ _RE_APR_DRG_BASE = re.compile(r'^\s*(\d{3})\s*-\s*(\d)')
1225
+
1226
+ # Period-separated form: '48.4' → '048-4', '532.2' → '532-2'
1227
+ # Must precede _RE_APR_DRG_NNNS/base fallback to avoid '48.4' → '048'.
1228
+ _RE_APR_DRG_PERIOD = re.compile(r'^\s*(\d{1,3})\.(\d)\s*$')
1229
+
1230
+ # Verbose 'APRnnn SOI s' form (case-insensitive; double-space tolerated).
1231
+ # Example: 'APR052 SOI 1' → '052-1'.
1232
+ _RE_APR_DRG_VERBOSE = re.compile(r'^\s*APR(\d{3})\s+SOI\s+(\d)\s*$', re.IGNORECASE)
1233
+
1234
+ # Matches the legacy bare 4-digit NNNS form (no separator). Only applied
1235
+ # when no separator is present at all, so it can't misfire on codes like
1236
+ # '0012 Short Stay' (those are handled by _RE_APR_DRG_BASE first).
1237
+ _RE_APR_DRG_NNNS = re.compile(r'^\s*(\d{3})(\d)\s*$')
1238
+
1239
+ # Short leading-zero hyphen form: '48-3' → '048-3', '5-2' → '005-2'
1240
+ # Matches only 1-2 digit base with a single severity digit (no ambiguity with
1241
+ # the canonical NNN-S form handled by _RE_APR_DRG_BASE).
1242
+ _RE_APR_DRG_SHORT_HYPHEN = re.compile(r'^\s*(\d{1,2})-(\d)\s*$')
1243
+
1244
+ # Bare short base (no severity): '48' → '048', '5' → '005'.
1245
+ # Base-only without severity; passes _RE_VALID_APR_DRG which allows ^\d{1,4}(-\d{1,2})?$.
1246
+ _RE_APR_DRG_SHORT_BASE = re.compile(r'^\s*(\d{1,2})\s*$')
1247
+
1248
+ # Compact 'APRnnnns' form (case-insensitive): 'APR0011' → '001-1'.
1249
+ # DISABLED: the only known source of this form is a single file that lists
1250
+ # 1,340 all-distinct values, which looks more like an enumeration dump than
1251
+ # real APR-DRG codes. Gated behind _APR_COMPACT_ENABLED until that is
1252
+ # confirmed either way.
1253
+ _RE_APR_DRG_COMPACT = re.compile(r'^\s*APR(\d{3})(\d)\s*$', re.IGNORECASE)
1254
+ _APR_COMPACT_ENABLED = False
1255
+
1256
+ # Helper: zero-pad a string to 3 digits.
1257
+ _pad3 = lambda s: s.zfill(3)
1258
+
1259
+
1260
+ def _strip_apr_drg_qualifier(code: str) -> str:
1261
+ """Return the canonical APR-DRG form of *code*.
1262
+
1263
+ Handles the following source formats, tested in order (ORDER IS CORRECTNESS):
1264
+
1265
+ 1. 'NNN-S' with optional qualifier text - '001-1 Short Stay' → '001-1'
1266
+ '001 - 1' (whitespace-padded dash) → '001-1'
1267
+ 2. Period-separated 'N.S' / 'NN.S' / 'NNN.S' - '48.4'→'048-4', '532.2'→'532-2'
1268
+ MUST precede bare-3-digit fallback to avoid eating the severity digit.
1269
+ 3. Verbose 'APRnnn SOI s' - 'APR052 SOI 1' → '052-1'
1270
+ 4. Compact 'APRnnns' (DARK/GATED) - 'APR0011' → '001-1'
1271
+ Disabled: see _APR_COMPACT_ENABLED.
1272
+ 5. Bare 4-digit NNNS - '0011' → '001-1'
1273
+ 6. Short leading-zero hyphen 'NN-S'/'N-S' - '48-3'→'048-3', '5-2'→'005-2'
1274
+ 7. Short bare base (no severity) - '48'→'048', '5'→'005'
1275
+ Base-only is intentional; passes _RE_VALID_APR_DRG.
1276
+ 8. Bare 3-digit fallback (no severity) - '139 Unspecified' → '139'
1277
+
1278
+ If none match, return the original value unchanged.
1279
+ """
1280
+ if not code:
1281
+ return code
1282
+
1283
+ # 1. Canonical NNN-S (with optional qualifier / spacing).
1284
+ m = _RE_APR_DRG_BASE.match(code)
1285
+ if m:
1286
+ return f"{m.group(1)}-{m.group(2)}"
1287
+
1288
+ # 2. Period-separated: NNN.S (MUST be before bare-3-digit fallback).
1289
+ m = _RE_APR_DRG_PERIOD.match(code)
1290
+ if m:
1291
+ return f"{_pad3(m.group(1))}-{m.group(2)}"
1292
+
1293
+ # 3. Verbose 'APRnnn SOI s'.
1294
+ m = _RE_APR_DRG_VERBOSE.match(code)
1295
+ if m:
1296
+ return f"{m.group(1)}-{m.group(2)}"
1297
+
1298
+ # 4. Compact 'APRnnns' - DARK until precheck confirms non-enumeration source.
1299
+ if _APR_COMPACT_ENABLED:
1300
+ m = _RE_APR_DRG_COMPACT.match(code)
1301
+ if m:
1302
+ return f"{m.group(1)}-{m.group(2)}"
1303
+
1304
+ # 5. Legacy bare 4-digit NNNS form: '0011' → '001-1'.
1305
+ m = _RE_APR_DRG_NNNS.match(code)
1306
+ if m:
1307
+ return f"{m.group(1)}-{m.group(2)}"
1308
+
1309
+ # 6. Short leading-zero hyphen: '48-3' → '048-3', '5-2' → '005-2'.
1310
+ m = _RE_APR_DRG_SHORT_HYPHEN.match(code)
1311
+ if m:
1312
+ return f"{_pad3(m.group(1))}-{m.group(2)}"
1313
+
1314
+ # 7. Short bare base (no severity): '48' → '048', '5' → '005'.
1315
+ m = _RE_APR_DRG_SHORT_BASE.match(code)
1316
+ if m:
1317
+ return _pad3(m.group(1))
1318
+
1319
+ # 8. Last resort: extract leading 3-digit base (partial/unknown qualifier text).
1320
+ m = re.match(r'^\s*(\d{3})', code)
1321
+ if m:
1322
+ return m.group(1)
1323
+
1324
+ return code
1325
+
1326
+
1327
+ # ============================================================================
1328
+ # CODE REJECTION: detect corrupt rows that should be skipped
1329
+ # ============================================================================
1330
+ # After normalize_code() has done its best, some rows are still irreparably
1331
+ # corrupt and should be dropped. Examples seen in real files:
1332
+ # - code_type is a bare numeric literal like '1' or '2' (column mis-parse)
1333
+ # - MS-DRG rows with 'N.NNNN' code values (decimal DRG, not a real format)
1334
+ # Return a short machine-readable reason string, or None if the row is OK.
1335
+
1336
+
1337
+ _RE_NUMERIC_ONLY = re.compile(r'^\d+$')
1338
+ _RE_MS_DRG_DECIMAL = re.compile(r'^\d+\.\d+$')
1339
+ # Structurally valid code_types are short uppercase tokens composed of
1340
+ # letters, digits, '-' and '/' only. Real-world examples span CPT, HCPCS,
1341
+ # MS-DRG, APR-DRG, TRIS-DRG, CPT/HCPCS, CDM, NDC, ICD, APC, EAPG, RC, CDT,
1342
+ # CMG, R-DRG, LOCAL, HIPPS. We allow unknown-but-plausible tokens through
1343
+ # (e.g. a future 'XYZ-DRG') while rejecting obvious column-shift garbage
1344
+ # like stray descriptions, decimals, or sentence fragments.
1345
+ _RE_VALID_CODE_TYPE_SHAPE = re.compile(r'^[A-Z][A-Z0-9]*(?:[-/][A-Z0-9]+)*$')
1346
+ _MAX_CODE_TYPE_LEN = 16
1347
+ # When code_type itself matches a HCPCS/CDT Level II pattern (letter +
1348
+ # 4 digits), the columns have almost certainly shifted: what looks like
1349
+ # the 'type' is actually a code. 5-digit numeric values (e.g. a CPT
1350
+ # leaking into the type column) are caught separately by the numeric-only
1351
+ # rule. Seen in the wild: code='CDM', code_type='C1887' - where CDM is
1352
+ # the real type and C1887 is the real code.
1353
+ _RE_CODE_TYPE_LOOKS_LIKE_HCPCS = re.compile(r'^[A-Z]\d{4}$')
1354
+
1355
+
1356
+ def is_rejected_code(
1357
+ code: Optional[str],
1358
+ code_type: Optional[str],
1359
+ *,
1360
+ stats: Optional[ParseStats] = None,
1361
+ ) -> Optional[str]:
1362
+ """Decide whether a (code, code_type) pair is too corrupt to keep.
1363
+
1364
+ Returns a short reason string (e.g. 'numeric_code_type',
1365
+ 'malformed_code_type', 'ms_drg_decimal') when the row must be
1366
+ rejected, or None when the row is acceptable. Callers should drop
1367
+ rejected rows rather than emit them.
1368
+
1369
+ This is intentionally conservative: only clearly broken inputs are
1370
+ rejected; everything borderline is accepted and classified as LOCAL by
1371
+ normalize_code() so a human can still review it downstream.
1372
+
1373
+ ``stats``, when given, counts each rejection.
1374
+ """
1375
+ reason = _rejection_reason(code, code_type)
1376
+ if reason and stats is not None:
1377
+ stats.record_rejected_code(code, code_type)
1378
+ return reason
1379
+
1380
+
1381
+ def _rejection_reason(
1382
+ code: Optional[str],
1383
+ code_type: Optional[str],
1384
+ ) -> Optional[str]:
1385
+ # Empty rows are not "corrupt", just uninteresting - let callers decide.
1386
+ if not code and not code_type:
1387
+ return None
1388
+
1389
+ # code_type is purely numeric (e.g. '1', '2', '12') - almost always a
1390
+ # column mis-parse where a price or count leaked into the type column.
1391
+ if code_type:
1392
+ ct_stripped = code_type.strip()
1393
+ if ct_stripped and _RE_NUMERIC_ONLY.fullmatch(ct_stripped):
1394
+ return 'numeric_code_type'
1395
+
1396
+ # HCPCS/CDT Level II lookalike: a single uppercase letter followed by
1397
+ # exactly 4 digits (e.g. 'C1887', 'J0591', 'D9999') is a HCPCS code,
1398
+ # not a code_type. Seeing this in the code_type column means the
1399
+ # columns have shifted and the real type sits elsewhere.
1400
+ if ct_stripped and _RE_CODE_TYPE_LOOKS_LIKE_HCPCS.fullmatch(ct_stripped.upper()):
1401
+ return 'malformed_code_type'
1402
+
1403
+ # Structural sanity: real code_types are short token-like values.
1404
+ # Anything containing whitespace, decimals, or punctuation other
1405
+ # than '-' / '/' is almost certainly a CSV column-shift artefact
1406
+ # (e.g. an unescaped inch-mark in a description causes
1407
+ # csv.DictReader to swallow the separator and shift every
1408
+ # subsequent column by one). We validate the uppercased form, so
1409
+ # lowercase-only tokens are allowed if uppercasing yields a valid
1410
+ # shape, while prose like 'as remittances do not itemize...' is
1411
+ # rejected.
1412
+ if ct_stripped and len(ct_stripped) <= _MAX_CODE_TYPE_LEN:
1413
+ if not _RE_VALID_CODE_TYPE_SHAPE.fullmatch(ct_stripped.upper()):
1414
+ return 'malformed_code_type'
1415
+ elif ct_stripped:
1416
+ return 'malformed_code_type'
1417
+
1418
+ # MS-DRG with a decimal value (e.g. '0.1234') is not a valid DRG format.
1419
+ # Real MS-DRGs are 3-digit integers. These rows came from weird
1420
+ # column-swap bugs in some chargemasters.
1421
+ if code and code_type and code_type.upper() == 'MS-DRG':
1422
+ if _RE_MS_DRG_DECIMAL.fullmatch(code.strip()):
1423
+ return 'ms_drg_decimal'
1424
+
1425
+ return None
1426
+
1427
+
1428
+ def _extract_code_from_garbage(raw: str) -> Tuple[Optional[str], Optional[str]]:
1429
+ """
1430
+ Try to extract a valid CPT/HCPCS/CDT code from a malformed code string.
1431
+
1432
+ Handles patterns like:
1433
+ - XJ0591X → J0591 (HCPCS wrapped in garbage)
1434
+ - AD9999 → D9999 (CDT dental code with A-prefix)
1435
+ - A29999.41 → 29999 (A-prefix CPT with decimal sub-code)
1436
+ - A0237T → 0237T (A-prefix Category III)
1437
+
1438
+ Returns: (extracted_code, code_type) or (None, None) if nothing found.
1439
+ """
1440
+ # Pattern 1: A-prefix + Category III code (A0237T → 0237T)
1441
+ # Must check before HCPCS extraction to avoid matching the A as HCPCS prefix
1442
+ m = re.fullmatch(r'A(\d{4}T)', raw)
1443
+ if m:
1444
+ return m.group(1), 'HCPCS'
1445
+
1446
+ # Pattern 2: A-prefix + 5-digit CPT with optional decimal (A29999.41 → 29999)
1447
+ # Stanford convention: A = facility component prefix on CPT codes
1448
+ m = re.fullmatch(r'A(\d{5})(?:\.\d+)?', raw)
1449
+ if m:
1450
+ return m.group(1), 'CPT'
1451
+
1452
+ # Pattern 3: A-prefix + HCPCS with optional decimal (A4649.0099 → A4649)
1453
+ # Real HCPCS code A4649 with sub-code suffix
1454
+ m = re.fullmatch(r'([A-Z]\d{4})\.\d+', raw)
1455
+ if m:
1456
+ return m.group(1), 'HCPCS'
1457
+
1458
+ # Pattern 4: Embedded HCPCS/CDT code in garbage (XJ0591X → J0591, AD9999 → D9999)
1459
+ m = re.search(r'(?<!\d)([A-Z]\d{4})(?!\d)', raw)
1460
+ if m:
1461
+ extracted = m.group(1)
1462
+ # D-prefix codes are CDT (dental), all others are HCPCS Level II
1463
+ ext_type = 'CDT' if extracted[0] == 'D' else 'HCPCS'
1464
+ return extracted, ext_type
1465
+
1466
+ # Pattern 5: Embedded CPT code in garbage
1467
+ m = re.search(r'(?<![A-Z\d])(\d{5})(?!\d)', raw)
1468
+ if m:
1469
+ return m.group(1), 'CPT'
1470
+
1471
+ # Pattern 6: Embedded PLA code in garbage (digits + letter suffix)
1472
+ m = re.search(r'(?<!\d)(\d{4}[A-Z])(?![A-Z\d])', raw)
1473
+ if m:
1474
+ return m.group(1), 'CPT'
1475
+
1476
+ return None, None
1477
+
1478
+
1479
+ def apply_code_extraction(
1480
+ code: Optional[str],
1481
+ code_type: Optional[str],
1482
+ config: Optional[Dict],
1483
+ ) -> Tuple[Optional[str], Optional[str]]:
1484
+ """
1485
+ Extract standard codes from composite code strings based on config rules.
1486
+
1487
+ Some hospitals (e.g., UCLA) encode multiple fields into a single composite
1488
+ "code" value like ``RRUCLA-7616710100-1000-67101-0761-8580-Y`` where a
1489
+ real CPT code (67101) is embedded at a known position.
1490
+
1491
+ This function splits such composite codes and extracts the embedded
1492
+ standard code. The caller should then pass the result through
1493
+ ``normalize_code()`` for classification.
1494
+
1495
+ Config format (the per-source config dict)::
1496
+
1497
+ {
1498
+ "code_extraction": {
1499
+ "pattern": "^RRUCLA", # regex - only codes matching this are processed
1500
+ "separator": "-", # split character
1501
+ "code_position": 3, # 0-indexed segment with the standard code
1502
+ "fallback_position": 1 # segment to use as CDM code when code_position is empty
1503
+ }
1504
+ }
1505
+
1506
+ Returns: (extracted_code, extracted_code_type)
1507
+ - If extraction succeeds: the embedded code with code_type=None
1508
+ (caller should pass through ``normalize_code()`` for classification)
1509
+ - If code_position is empty: (parts[fallback_position], 'CDM')
1510
+ - If no config or no match: (code, code_type) unchanged
1511
+ """
1512
+ if not config or not code:
1513
+ return code, code_type
1514
+
1515
+ extraction = config.get('code_extraction')
1516
+ if not extraction:
1517
+ return code, code_type
1518
+
1519
+ pattern = extraction.get('pattern')
1520
+ if not pattern:
1521
+ return code, code_type
1522
+
1523
+ code_stripped = code.strip()
1524
+ if not re.search(pattern, code_stripped):
1525
+ return code, code_type
1526
+
1527
+ separator = extraction.get('separator', '-')
1528
+ parts = code_stripped.split(separator)
1529
+ pos = extraction.get('code_position', 3)
1530
+
1531
+ # Extract the standard code from the expected position
1532
+ if pos < len(parts) and parts[pos]:
1533
+ return parts[pos], None
1534
+
1535
+ # Fallback: use CDM ID from another position
1536
+ fb_pos = extraction.get('fallback_position', 1)
1537
+ if fb_pos < len(parts) and parts[fb_pos]:
1538
+ return parts[fb_pos], 'CDM'
1539
+
1540
+ return code, code_type
1541
+
1542
+
1543
+ def infer_billing_class(
1544
+ code: Optional[str],
1545
+ code_type: Optional[str],
1546
+ description: Optional[str],
1547
+ billing_class: Optional[str] = None,
1548
+ ) -> Tuple[Optional[str], Optional[str], Optional[str]]:
1549
+ """
1550
+ Infer billing_class and normalize code when not explicitly provided.
1551
+
1552
+ Returns: (billing_class, normalized_code, code_type)
1553
+
1554
+ Heuristic rules (applied only when billing_class is None):
1555
+ 1. A-prefix on CPT/HCPCS codes: 'A' + a full 5-digit CPT = facility
1556
+ component. Strip the 'A' prefix, set billing_class = 'facility', fix
1557
+ code_type. Does NOT fire on 'A' + 4 digits - that shape is a genuine
1558
+ HCPCS Level II code, not a prefixed CPT.
1559
+ 2. Description prefix 'PR ': professional component.
1560
+ 3. Description prefix 'HC ' on RC items: facility component.
1561
+ 4. Revenue codes (code_type='RC'): facility by definition.
1562
+ """
1563
+ normalized_code = code
1564
+ norm_type = code_type
1565
+
1566
+ # If billing_class already set (from source data), don't override
1567
+ if billing_class:
1568
+ return billing_class, normalized_code, norm_type
1569
+
1570
+ if not code and not description:
1571
+ return None, normalized_code, norm_type
1572
+
1573
+ # Rule 1: A-prefix on CPT/HCPCS codes → facility, strip prefix
1574
+ # e.g., A19325 (HCPCS) → code=19325, code_type=CPT, billing_class=facility
1575
+ # e.g., A0237T (HCPCS) → code=0237T, code_type=HCPCS, billing_class=facility
1576
+ # The SUFFIX, not the whole code, has to be a valid CPT. A genuine HCPCS
1577
+ # Level II code is exactly letter + 4 digits (length 5), so the old
1578
+ # `len(code) >= 5` guard swallowed every A-prefixed Level II supply code:
1579
+ # A4322 (irrigation syringe) became CPT '4322', which is not a valid CPT
1580
+ # at all. Nothing re-ran normalize_code() on the result, so the corrupt
1581
+ # pair went straight into the output.
1582
+ if (code and code_type in ('CPT', 'HCPCS', None)
1583
+ and len(code) >= 6 and code[0] == 'A'):
1584
+ suffix = code[1:]
1585
+ if re.fullmatch(r'\d{5}', suffix):
1586
+ # A + 5-digit CPT → CPT (strip A, reclassify)
1587
+ return 'facility', suffix, 'CPT'
1588
+ if re.fullmatch(r'\d{4}T', suffix):
1589
+ # A + Category III → HCPCS (strip A, keep HCPCS)
1590
+ return 'facility', suffix, 'HCPCS'
1591
+
1592
+ # Rule 2: Description starts with 'PR ' → professional
1593
+ # Common at O'Connor, John Muir: "PR CT Thorax Diag W Con"
1594
+ if description and description.startswith('PR '):
1595
+ return 'professional', normalized_code, norm_type
1596
+
1597
+ # Rule 3: Revenue codes → facility (they are inherently facility charges)
1598
+ if code_type == 'RC':
1599
+ return 'facility', normalized_code, norm_type
1600
+
1601
+ # Rule 4: Description starts with 'HC ' on non-RC items → facility
1602
+ # (RC items already caught by Rule 3)
1603
+ if description and description.startswith('HC '):
1604
+ return 'facility', normalized_code, norm_type
1605
+
1606
+ return None, normalized_code, norm_type
1607
+