mrfkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mrfkit/__init__.py +68 -0
- mrfkit/__main__.py +3 -0
- mrfkit/cli.py +79 -0
- mrfkit/codes.py +1607 -0
- mrfkit/csv_reader.py +324 -0
- mrfkit/files.py +651 -0
- mrfkit/headers.py +737 -0
- mrfkit/json_reader.py +603 -0
- mrfkit/payers.py +2755 -0
- mrfkit/records.py +260 -0
- mrfkit/reference.py +136 -0
- mrfkit/sinks.py +126 -0
- mrfkit/tabular.py +651 -0
- mrfkit/tic.py +660 -0
- mrfkit/values.py +627 -0
- mrfkit-0.1.0.dist-info/METADATA +136 -0
- mrfkit-0.1.0.dist-info/RECORD +21 -0
- mrfkit-0.1.0.dist-info/WHEEL +4 -0
- mrfkit-0.1.0.dist-info/entry_points.txt +2 -0
- mrfkit-0.1.0.dist-info/licenses/LICENSE +201 -0
- mrfkit-0.1.0.dist-info/licenses/NOTICE +4 -0
mrfkit/codes.py
ADDED
|
@@ -0,0 +1,1607 @@
|
|
|
1
|
+
"""Normalize billing codes and code types.
|
|
2
|
+
|
|
3
|
+
Hospitals label the same code system many ways ("CPT4", "HCPCS/CPT", "REV",
|
|
4
|
+
"AP-DRG"), file standard codes under their own chargemaster buckets, wrap
|
|
5
|
+
codes in prefixes ("HCPCS C1776", "DRG100") and bake modifiers into the code
|
|
6
|
+
("73721TC"). ``normalize_code`` sorts all of that out; the helpers around it
|
|
7
|
+
reject rows that are too broken to keep and infer the billing class when the
|
|
8
|
+
file does not say.
|
|
9
|
+
|
|
10
|
+
Nothing here touches a database. A few guards get sharper when you pass a
|
|
11
|
+
``ReferenceData`` with published code lists; without one they fall back to the
|
|
12
|
+
shape-only rules.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
import re
|
|
18
|
+
from typing import TYPE_CHECKING, Dict, Optional, Tuple
|
|
19
|
+
|
|
20
|
+
if TYPE_CHECKING:
|
|
21
|
+
from mrfkit.reference import ParseStats, ReferenceData
|
|
22
|
+
|
|
23
|
+
# ============================================================================
|
|
24
|
+
# CODE & CODE_TYPE NORMALIZATION
|
|
25
|
+
# ============================================================================
|
|
26
|
+
# Standard US medical code systems and their formats:
|
|
27
|
+
# CPT - 5 digits (Level I HCPCS), e.g. 99213, 27447
|
|
28
|
+
# HCPCS - letter + 4 digits (Level II HCPCS), e.g. J0690, C1776, A4550
|
|
29
|
+
# Also includes Category III: 5 digits + 'T', e.g. 0237T
|
|
30
|
+
# CDT - Current Dental Terminology, D + 4 digits, e.g. D0120, D7509
|
|
31
|
+
# RC - Revenue Code, 4 digits zero-padded (UB-04), e.g. 0360, 0750
|
|
32
|
+
# MS-DRG - 3 digits, e.g. 469, 470
|
|
33
|
+
# APC - Ambulatory Payment Classification, 4 digits or 'N'+3 digits
|
|
34
|
+
# NDC - National Drug Code, 10-11 digits with dashes, e.g. 12345-6789-01
|
|
35
|
+
# CDM - Charge Description Master (hospital-internal), non-standard formats
|
|
36
|
+
# LOCAL - Hospital-internal codes that don't match any standard system
|
|
37
|
+
|
|
38
|
+
# code_type synonyms: normalize various source labels to canonical types
|
|
39
|
+
_CODE_TYPE_NORMALIZE = {
|
|
40
|
+
'CPT': 'CPT',
|
|
41
|
+
'CPT4': 'CPT',
|
|
42
|
+
'CPT-4': 'CPT',
|
|
43
|
+
'HCPCS': 'HCPCS',
|
|
44
|
+
'HCPCS LEVEL II': 'HCPCS',
|
|
45
|
+
'HCPCS II': 'HCPCS',
|
|
46
|
+
'HCPCS2': 'HCPCS',
|
|
47
|
+
# Hedge labels used by hospitals that don't know whether a code is
|
|
48
|
+
# CPT (Level I) or HCPCS (Level II). Normalize to HCPCS; Step 5 will
|
|
49
|
+
# reclassify 5-digit numeric codes to CPT based on shape.
|
|
50
|
+
'CPT/HCPCS': 'HCPCS',
|
|
51
|
+
'HCPCS/CPT': 'HCPCS',
|
|
52
|
+
'CPT / HCPCS': 'HCPCS',
|
|
53
|
+
'HCPCS / CPT': 'HCPCS',
|
|
54
|
+
'CPT-HCPCS': 'HCPCS',
|
|
55
|
+
'HCPCS-CPT': 'HCPCS',
|
|
56
|
+
'CPTHCPCS': 'HCPCS',
|
|
57
|
+
'HCPCSCPT': 'HCPCS',
|
|
58
|
+
'RC': 'RC',
|
|
59
|
+
'REV': 'RC',
|
|
60
|
+
'REVENUE': 'RC',
|
|
61
|
+
'REVENUE CODE': 'RC',
|
|
62
|
+
'REVCODE': 'RC',
|
|
63
|
+
'MS-DRG': 'MS-DRG',
|
|
64
|
+
'MSDRG': 'MS-DRG',
|
|
65
|
+
'DRG': 'MS-DRG',
|
|
66
|
+
'TRIS-DRG': 'MS-DRG', # Tenet Revenue Integrity System - same codes as MS-DRG
|
|
67
|
+
'APR-DRG': 'APR-DRG', # All Patient Refined DRG (3M) - distinct from MS-DRG
|
|
68
|
+
'APRDRG': 'APR-DRG',
|
|
69
|
+
'APR DRG': 'APR-DRG',
|
|
70
|
+
'APR_DRG': 'APR-DRG',
|
|
71
|
+
'AP-DRG': 'APR-DRG', # All Patient DRG variant: same code set
|
|
72
|
+
'APC': 'APC',
|
|
73
|
+
'CMG': 'CMG', # Case Mix Group (CMS inpatient rehab grouper)
|
|
74
|
+
'NDC': 'NDC',
|
|
75
|
+
'CDT': 'CDT',
|
|
76
|
+
'DENTAL': 'CDT',
|
|
77
|
+
'CDM': 'CDM',
|
|
78
|
+
'CHARGEMASTER': 'CDM',
|
|
79
|
+
'LOCAL': 'LOCAL',
|
|
80
|
+
# Partners Healthcare internal types
|
|
81
|
+
'SUP': 'CDM', # supplies
|
|
82
|
+
'EAP': 'CDM', # enterprise-assigned procedure
|
|
83
|
+
'ERX': 'CDM', # enterprise pharmacy
|
|
84
|
+
# Compound NDC+system types: disambiguated in normalize_code step 2A
|
|
85
|
+
'NDCHCPCS': 'NDCHCPCS',
|
|
86
|
+
'NDCCPT': 'NDCCPT',
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
# Non-system junk code_type labels that some hospitals emit. When
|
|
90
|
+
# the canonicalized code_type is one of these we discard it (treat as missing)
|
|
91
|
+
# and re-derive from code shape, exactly as for a blank code_type.
|
|
92
|
+
_JUNK_CODE_TYPES = frozenset({'TYPE', 'MODIFIER', 'MODIFIERS', 'OBS', 'DOT', 'UN'})
|
|
93
|
+
|
|
94
|
+
# Canonical validity regexes for controlled-vocabulary numeric types.
|
|
95
|
+
# Applied in two places: cross-system reclassify (step 6A) and residue gating
|
|
96
|
+
# (step 12). APR-DRG base-severity form (775-1) and CMG group-tier form
|
|
97
|
+
# (1902-D) are both canonical - do NOT gate them.
|
|
98
|
+
_RE_VALID_RC = re.compile(r'^\d{3,4}$')
|
|
99
|
+
_RE_VALID_MS_DRG = re.compile(r'^\d{1,3}$')
|
|
100
|
+
_RE_VALID_APR_DRG = re.compile(r'^\d{1,4}(-\d{1,2})?$')
|
|
101
|
+
_RE_VALID_APC = re.compile(r'^(?:\d{4,5}|[Nn]\d{3,4})$')
|
|
102
|
+
_RE_VALID_CMG = re.compile(r'^\d{4}(-[A-Z])?$')
|
|
103
|
+
_RE_VALID_EAPG = re.compile(r'^\d{3,5}$')
|
|
104
|
+
|
|
105
|
+
_CONTROLLED_VOCAB_VALIDATORS = {
|
|
106
|
+
'RC': _RE_VALID_RC,
|
|
107
|
+
'MS-DRG': _RE_VALID_MS_DRG,
|
|
108
|
+
'APR-DRG': _RE_VALID_APR_DRG,
|
|
109
|
+
'APC': _RE_VALID_APC,
|
|
110
|
+
'CMG': _RE_VALID_CMG,
|
|
111
|
+
'EAPG': _RE_VALID_EAPG,
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
# Revenue codes are 1-4 digit numbers in the range 0001-0999.
|
|
115
|
+
# Some hospitals report them under code_type='HCPCS' using the
|
|
116
|
+
# 3-digit UB-04 revenue center number (e.g., 272 for sterile supply).
|
|
117
|
+
_KNOWN_REVENUE_CODES = {
|
|
118
|
+
# Commonly misclassified as HCPCS
|
|
119
|
+
'250', '0250', '255', '0255', '258', '0258', # Pharmacy
|
|
120
|
+
'270', '0270', '271', '0271', '272', '0272', # Supplies
|
|
121
|
+
'275', '0275', '276', '0276', '278', '0278', # Implants, supplies
|
|
122
|
+
'300', '0300', '301', '0301', '302', '0302', # Lab
|
|
123
|
+
'305', '0305', '306', '0306', '309', '0309',
|
|
124
|
+
'310', '0310', '311', '0311', '312', '0312',
|
|
125
|
+
'320', '0320', '323', '0323', '333', '0333', # Radiology
|
|
126
|
+
'335', '0335', '341', '0341', '342', '0342',
|
|
127
|
+
'343', '0343', '350', '0350', '351', '0351',
|
|
128
|
+
'352', '0352', '360', '0360', '361', '0361', # OR
|
|
129
|
+
'370', '0370', '390', '0390', # Anesthesia
|
|
130
|
+
'402', '0402', '403', '0403', '404', '0404', # Other imaging
|
|
131
|
+
'410', '0410', '412', '0412', '420', '0420', # Physical therapy
|
|
132
|
+
'424', '0424', '430', '0430', '440', '0440',
|
|
133
|
+
'450', '0450', '460', '0460', '470', '0470', # ER, ambulance
|
|
134
|
+
'471', '0471', '480', '0480', '481', '0481',
|
|
135
|
+
'483', '0483', '489', '0489', '510', '0510',
|
|
136
|
+
'521', '0521', '610', '0610', '611', '0611',
|
|
137
|
+
'615', '0615', '618', '0618', '636', '0636', # Pharmacy IV
|
|
138
|
+
'637', '0637', '710', '0710', '721', '0721',
|
|
139
|
+
'722', '0722', '730', '0730', '731', '0731',
|
|
140
|
+
'740', '0740', '750', '0750', '760', '0760',
|
|
141
|
+
'812', '0812', '906', '0906', '916', '0916',
|
|
142
|
+
'918', '0918', '920', '0920', '921', '0921',
|
|
143
|
+
'922', '0922', '940', '0940', '942', '0942',
|
|
144
|
+
'949', '0949', '987', '0987', '998', '0998',
|
|
145
|
+
}
|
|
146
|
+
|
|
147
|
+
|
|
148
|
+
def _strip_non_ascii(s: str) -> str:
|
|
149
|
+
"""Remove non-ASCII characters (mojibake, control chars) from a string."""
|
|
150
|
+
return ''.join(c for c in s if ord(c) < 128)
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
# Regex patterns for valid code formats
|
|
154
|
+
_RE_CPT = re.compile(r'\d{5}') # 5 digits: 99213
|
|
155
|
+
_RE_CPT_PLA = re.compile(r'\d{4}[A-Z]') # PLA codes: 0202U, 0003M
|
|
156
|
+
_RE_HCPCS = re.compile(r'[A-Z]\d{4}') # letter + 4 digits: J0591
|
|
157
|
+
_RE_CAT3 = re.compile(r'\d{4}T') # Category III: 0237T
|
|
158
|
+
|
|
159
|
+
# ── NDC normalization ─────────────────────────────────────────────────
|
|
160
|
+
# Produces canonical 11-digit no-dash NDCs directly.
|
|
161
|
+
_RE_NDC_LEADING_ALPHA = re.compile(r'^[A-Za-z]')
|
|
162
|
+
_RE_NDC_CANON_11 = re.compile(r'^\d{11}$')
|
|
163
|
+
_RE_NDC_NINE_DIGIT = re.compile(r'^\d{9}$')
|
|
164
|
+
# Bare dash-less 10-digit NDC10 → NDC11 via leading '0'.
|
|
165
|
+
_RE_NDC_TEN_DIGIT = re.compile(r'^\d{10}$')
|
|
166
|
+
_RE_NDC_SETTING_SUFFIX = re.compile(
|
|
167
|
+
r'^(\d{10,11})_(?:ip|op)$', re.IGNORECASE
|
|
168
|
+
)
|
|
169
|
+
_RE_NDC_TRAILING_LETTER = re.compile(r'^(\d{10,11})[A-Za-z]$')
|
|
170
|
+
_RE_NDC_PKG_DIGIT_SUFFIX = re.compile(r'^(\d{10,11})_\d+$')
|
|
171
|
+
# Package + setting double suffix, e.g. "39822105505_4_ip".
|
|
172
|
+
_RE_NDC_PKG_SETTING_SUFFIX = re.compile(
|
|
173
|
+
r'^(\d{10,11})_\d+_(?:ip|op)$', re.IGNORECASE
|
|
174
|
+
)
|
|
175
|
+
# 3-segment dashed NDC, optionally followed by an alpha packaging suffix
|
|
176
|
+
# (e.g. "RL1", "OSPT") or "-<digits>" extra tail (e.g. "-50"). Used for
|
|
177
|
+
# both shape detection and canonicalization.
|
|
178
|
+
_RE_NDC_3SEG = re.compile(
|
|
179
|
+
r'^(\d{4,5})-(\d{3,5})-(\d{1,2})(?:[A-Za-z][A-Za-z0-9]*|-\d+)?$'
|
|
180
|
+
)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def _ndc_canonicalize_base(base: str) -> Optional[str]:
|
|
184
|
+
"""10/11-digit numeric base → canonical 11-digit string. None if neither."""
|
|
185
|
+
n = len(base)
|
|
186
|
+
if n == 11:
|
|
187
|
+
return base
|
|
188
|
+
if n == 10:
|
|
189
|
+
return '0' + base
|
|
190
|
+
return None
|
|
191
|
+
|
|
192
|
+
|
|
193
|
+
def _normalize_ndc(
|
|
194
|
+
code: Optional[str],
|
|
195
|
+
) -> Tuple[Optional[str], str]:
|
|
196
|
+
"""Normalize an NDC code to canonical 11-digit no-dash form.
|
|
197
|
+
|
|
198
|
+
Returns ``(canonical_code, code_type)`` where ``code_type`` is either
|
|
199
|
+
``'NDC'`` (default) or ``'LOCAL'`` (when the value starts with an
|
|
200
|
+
ASCII letter: those are charge-master codes, not NDCs).
|
|
201
|
+
|
|
202
|
+
Rules, first match wins:
|
|
203
|
+
|
|
204
|
+
1. ``None`` → ``(None, 'NDC')``.
|
|
205
|
+
2. Empty after strip → ``(raw, 'NDC')``.
|
|
206
|
+
3. Leading ASCII letter → ``(raw, 'LOCAL')``.
|
|
207
|
+
4. Already canonical 11-digit → ``(raw, 'NDC')``.
|
|
208
|
+
5. 9-digit all-numeric → ``('00' + raw, 'NDC')``.
|
|
209
|
+
6. Bare 10-digit all-numeric → ``('0' + raw, 'NDC')``.
|
|
210
|
+
7. 10/11-digit base + ``_ip``/``_op`` setting suffix → canonical base.
|
|
211
|
+
8. 10/11-digit base + single trailing ASCII letter → canonical base.
|
|
212
|
+
9. 10/11-digit base + ``_<digits>`` package suffix → canonical base.
|
|
213
|
+
10. 10/11-digit base + ``_<digits>_<ip|op>`` package+setting double
|
|
214
|
+
suffix → canonical base.
|
|
215
|
+
11. 3-segment dashed shape (with optional alpha/dash packaging
|
|
216
|
+
suffix): strip non-digits; 11 or 10 digits canonicalize, else
|
|
217
|
+
leave as-is.
|
|
218
|
+
12. Anything else → ``(raw, 'NDC')`` unchanged.
|
|
219
|
+
"""
|
|
220
|
+
if code is None:
|
|
221
|
+
return None, 'NDC'
|
|
222
|
+
s = code.strip()
|
|
223
|
+
if not s:
|
|
224
|
+
return code, 'NDC'
|
|
225
|
+
|
|
226
|
+
# 3. Leading-letter codes are local charge-master codes, not NDCs.
|
|
227
|
+
if _RE_NDC_LEADING_ALPHA.match(s):
|
|
228
|
+
return code, 'LOCAL'
|
|
229
|
+
|
|
230
|
+
# 4. Already canonical.
|
|
231
|
+
if _RE_NDC_CANON_11.match(s):
|
|
232
|
+
return s, 'NDC'
|
|
233
|
+
|
|
234
|
+
# 5. 9-digit pad with '00'.
|
|
235
|
+
if _RE_NDC_NINE_DIGIT.match(s):
|
|
236
|
+
return '00' + s, 'NDC'
|
|
237
|
+
|
|
238
|
+
# 6. Bare 10-digit pad with '0'.
|
|
239
|
+
if _RE_NDC_TEN_DIGIT.match(s):
|
|
240
|
+
return '0' + s, 'NDC'
|
|
241
|
+
|
|
242
|
+
# 7-10. 10/11-digit base with various trailing suffixes (setting,
|
|
243
|
+
# letter, package, and package+setting).
|
|
244
|
+
for rx in (
|
|
245
|
+
_RE_NDC_SETTING_SUFFIX,
|
|
246
|
+
_RE_NDC_TRAILING_LETTER,
|
|
247
|
+
_RE_NDC_PKG_DIGIT_SUFFIX,
|
|
248
|
+
_RE_NDC_PKG_SETTING_SUFFIX,
|
|
249
|
+
):
|
|
250
|
+
m = rx.match(s)
|
|
251
|
+
if m:
|
|
252
|
+
canon = _ndc_canonicalize_base(m.group(1))
|
|
253
|
+
if canon is not None:
|
|
254
|
+
return canon, 'NDC'
|
|
255
|
+
|
|
256
|
+
# 9. 3-segment dashed shape with optional packaging suffix
|
|
257
|
+
# (alpha tail like "RL1" or "-50"). Canonicalize via segment
|
|
258
|
+
# alignment: seg1→5, seg2→4, seg3→2. The packaging
|
|
259
|
+
# suffix is intentionally discarded - it is not part of the
|
|
260
|
+
# canonical NDC product code. Requires seg1≤5, seg2≤4,
|
|
261
|
+
# seg3∈[1,2] digits so total = 11; otherwise leave alone.
|
|
262
|
+
m = _RE_NDC_3SEG.match(s)
|
|
263
|
+
if m:
|
|
264
|
+
seg1, seg2, seg3 = m.group(1), m.group(2), m.group(3)
|
|
265
|
+
if len(seg1) <= 5 and len(seg2) <= 4 and len(seg3) <= 2:
|
|
266
|
+
canon = seg1.zfill(5) + seg2.zfill(4) + seg3.zfill(2)
|
|
267
|
+
if len(canon) == 11:
|
|
268
|
+
return canon, 'NDC'
|
|
269
|
+
# Doesn't fit canonical layout (e.g. shifted-dash 5-5-1):
|
|
270
|
+
# leave as-is rather than guess.
|
|
271
|
+
return s, 'NDC'
|
|
272
|
+
|
|
273
|
+
# 10. Anything else: leave untouched.
|
|
274
|
+
return s, 'NDC'
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
# Placeholder/test patterns that should be classified as LOCAL
|
|
278
|
+
_RE_PLACEHOLDER = re.compile(
|
|
279
|
+
r'^(?:X{3,}|TEST\b|N/?A$|NONE$|TBD$|UNKNOWN$)',
|
|
280
|
+
re.IGNORECASE,
|
|
281
|
+
)
|
|
282
|
+
|
|
283
|
+
# Composite code prefixes: code values like "HCPCS C1776" or "CPT 87339"
|
|
284
|
+
# embed the code type as a prefix. Keys are uppercase prefixes that can
|
|
285
|
+
# appear at the start of a code value (followed by a space).
|
|
286
|
+
_COMPOSITE_CODE_PREFIXES = {
|
|
287
|
+
'HCPCS', 'CPT', 'CPT4', 'MS-DRG', 'MSDRG', 'DRG',
|
|
288
|
+
'NDC', 'REV', 'RC', 'REVENUE', 'REVENUE CODE', 'REVCODE', 'APC', 'CDM',
|
|
289
|
+
'CDT', 'APR-DRG', 'APRDRG', 'APR DRG', 'APR_DRG', 'CMG',
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
_COMPOSITE_CODE_PREFIXES_BY_LENGTH = sorted(
|
|
293
|
+
_COMPOSITE_CODE_PREFIXES,
|
|
294
|
+
key=len,
|
|
295
|
+
reverse=True,
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
# code_types whose numeric codes may arrive with a spurious trailing
|
|
299
|
+
# ".0" due to Excel / ETL float coercion (e.g. "320.0" → "320"). The strip
|
|
300
|
+
# is scoped to these numeric controlled-vocabulary types only - ICD codes like
|
|
301
|
+
# "250.0" are valid decimal-coded ICD-9 entries and must NOT be altered.
|
|
302
|
+
_FLOAT_COERCE_STRIP_TYPES = frozenset({'RC', 'MS-DRG', 'APR-DRG', 'APC', 'CMG', 'EAPG'})
|
|
303
|
+
|
|
304
|
+
|
|
305
|
+
def _valid_prefixed_code_remainder(rest: str, norm_type: str) -> bool:
|
|
306
|
+
"""Return whether a composite prefix leaves a plausible code value."""
|
|
307
|
+
s = rest.strip()
|
|
308
|
+
if norm_type == 'CDT':
|
|
309
|
+
return bool(re.fullmatch(r'[Dd]\d{4}', s))
|
|
310
|
+
if norm_type == 'APR-DRG':
|
|
311
|
+
return bool(re.match(r'\d{3}\s*-\s*\d', s) or re.fullmatch(r'\d{3,4}', s))
|
|
312
|
+
if norm_type == 'APC':
|
|
313
|
+
return bool(re.fullmatch(r'(?:\d{3,4}|[Nn]\d{3})', s))
|
|
314
|
+
if norm_type == 'RC':
|
|
315
|
+
return bool(re.fullmatch(r'\d{1,4}', s))
|
|
316
|
+
if norm_type == 'CMG':
|
|
317
|
+
# 4-digit group number, optionally followed by '-' + comorbidity tier letter
|
|
318
|
+
return bool(re.fullmatch(r'\d{4}(-[A-Z])?', s))
|
|
319
|
+
return True
|
|
320
|
+
|
|
321
|
+
# MS-DRG composite: "MS-DRG V41.0 (FY 2024) 155" - the actual DRG number
|
|
322
|
+
# is the last group of 1-3 digits in the string.
|
|
323
|
+
_RE_DRG_TRAILING_NUM = re.compile(r'(\d{1,3})\s*$')
|
|
324
|
+
|
|
325
|
+
|
|
326
|
+
def _split_attached_composite_code(
|
|
327
|
+
code: str,
|
|
328
|
+
code_type: Optional[str],
|
|
329
|
+
) -> Tuple[str, Optional[str]]:
|
|
330
|
+
"""Split tightly attached code-system prefixes at the start of a value.
|
|
331
|
+
|
|
332
|
+
This is deliberately narrower than the space-delimited composite splitter:
|
|
333
|
+
the full code value must be just the prefix plus a structurally valid code.
|
|
334
|
+
That lets us recover values like ``DRG100`` while avoiding infix false
|
|
335
|
+
positives such as ``SUP-D224DRG`` or free-text markers like ``NEED CPT``.
|
|
336
|
+
"""
|
|
337
|
+
stripped = code.strip()
|
|
338
|
+
|
|
339
|
+
rules = (
|
|
340
|
+
('MS-DRG', r'(?:MS-DRG|MSDRG|DRG)(\d{1,3})'),
|
|
341
|
+
# "MS-012" abbreviated form (MS- + 1-3 digits, no 'DRG' suffix).
|
|
342
|
+
# Distinct from the full "MS-DRG470" pattern above. Only split when the
|
|
343
|
+
# declared code_type is MS-DRG-family, CDM, LOCAL, or absent - never
|
|
344
|
+
# override a conflicting standard type (same guard as the other rules).
|
|
345
|
+
('MS-DRG', r'MS-(\d{1,3})'),
|
|
346
|
+
('APR-DRG', r'(?:APR-DRG|APRDRG)(\d{3}(?:-\d|\d)?)'),
|
|
347
|
+
('APC', r'APC((?:\d{3,4}|[Nn]\d{3}))'),
|
|
348
|
+
('RC', r'RC(\d{1,4})'),
|
|
349
|
+
('NDC', r'NDC(\d{9,11}|\d{4,5}-\d{3,5}-\d{1,2}(?:[A-Za-z][A-Za-z0-9]*|-\d+)?)'),
|
|
350
|
+
('CPT', r'CPT(\d{5}|\d{4}[FTUftu])'),
|
|
351
|
+
('HCPCS', r'HCPCS([A-CE-Va-ce-v]\d{4})'),
|
|
352
|
+
('CDT', r'CDT([Dd]\d{4})'),
|
|
353
|
+
# "CMG-1902-D" dash-attached form: CMG + dash + 4-digit group
|
|
354
|
+
# + optional comorbidity tier letter. The space form ("CMG 1902-D") is
|
|
355
|
+
# handled by the space-prefix splitter; this rule covers the dash form.
|
|
356
|
+
('CMG', r'CMG-(\d{4}(?:-[A-Z])?)'),
|
|
357
|
+
)
|
|
358
|
+
|
|
359
|
+
for prefix_type, pattern in rules:
|
|
360
|
+
m = re.fullmatch(pattern, stripped, flags=re.IGNORECASE)
|
|
361
|
+
if not m:
|
|
362
|
+
continue
|
|
363
|
+
|
|
364
|
+
if code_type:
|
|
365
|
+
norm_ct = _CODE_TYPE_NORMALIZE.get(code_type.strip().upper())
|
|
366
|
+
norm_prefix = _CODE_TYPE_NORMALIZE.get(prefix_type, prefix_type)
|
|
367
|
+
if norm_ct and norm_ct not in ('CDM', 'LOCAL', norm_prefix):
|
|
368
|
+
return code, code_type
|
|
369
|
+
|
|
370
|
+
return m.group(1), prefix_type
|
|
371
|
+
|
|
372
|
+
return code, code_type
|
|
373
|
+
|
|
374
|
+
|
|
375
|
+
def _split_composite_code(
|
|
376
|
+
code: str,
|
|
377
|
+
code_type: Optional[str],
|
|
378
|
+
) -> Tuple[str, Optional[str]]:
|
|
379
|
+
"""
|
|
380
|
+
Split composite code values where code_type is embedded as a prefix.
|
|
381
|
+
|
|
382
|
+
Examples:
|
|
383
|
+
"HCPCS C1776" → code="C1776", code_type="HCPCS"
|
|
384
|
+
"CPT 87339" → code="87339", code_type="CPT"
|
|
385
|
+
"MS-DRG V41.0 (FY 2024) 155" → code="155", code_type="MS-DRG"
|
|
386
|
+
"HCPCS 25009999" → code="25009999", code_type="HCPCS"
|
|
387
|
+
|
|
388
|
+
Only splits when:
|
|
389
|
+
- The code contains a space
|
|
390
|
+
- The part before the first space is a known code type prefix
|
|
391
|
+
- There is no separately declared code_type, OR the declared code_type
|
|
392
|
+
is a non-standard value (not in _CODE_TYPE_NORMALIZE)
|
|
393
|
+
|
|
394
|
+
Returns (code, code_type) - possibly unchanged.
|
|
395
|
+
"""
|
|
396
|
+
stripped = code.strip()
|
|
397
|
+
if ' ' not in stripped:
|
|
398
|
+
return _split_attached_composite_code(code, code_type)
|
|
399
|
+
|
|
400
|
+
def _match_composite_prefix(value: str) -> Tuple[Optional[str], Optional[str]]:
|
|
401
|
+
upper_value = value.upper()
|
|
402
|
+
for known_prefix in _COMPOSITE_CODE_PREFIXES_BY_LENGTH:
|
|
403
|
+
if upper_value.startswith(known_prefix + ' '):
|
|
404
|
+
prefix_text = value[:len(known_prefix)].upper().rstrip('®')
|
|
405
|
+
rest_text = value[len(known_prefix) + 1:].strip()
|
|
406
|
+
return prefix_text, rest_text
|
|
407
|
+
return None, None
|
|
408
|
+
|
|
409
|
+
# Don't override a valid, recognized code_type - with exceptions
|
|
410
|
+
if code_type:
|
|
411
|
+
ct_upper = code_type.strip().upper()
|
|
412
|
+
norm_ct = _CODE_TYPE_NORMALIZE.get(ct_upper)
|
|
413
|
+
if norm_ct:
|
|
414
|
+
prefix, rest = _match_composite_prefix(stripped)
|
|
415
|
+
if not prefix:
|
|
416
|
+
first_space = stripped.index(' ')
|
|
417
|
+
prefix = stripped[:first_space].upper().rstrip('®')
|
|
418
|
+
rest = stripped[first_space + 1:].strip()
|
|
419
|
+
# If code_type is hospital-internal (CDM, LOCAL) but the code
|
|
420
|
+
# starts with a standard code type prefix, prefer the standard type
|
|
421
|
+
if norm_ct in ('CDM', 'LOCAL') and prefix in _COMPOSITE_CODE_PREFIXES:
|
|
422
|
+
pass # fall through to split logic below
|
|
423
|
+
# If the prefix matches the declared code_type (redundant), strip it
|
|
424
|
+
elif prefix == ct_upper or _CODE_TYPE_NORMALIZE.get(prefix) == norm_ct:
|
|
425
|
+
if rest:
|
|
426
|
+
if not _valid_prefixed_code_remainder(rest, norm_ct):
|
|
427
|
+
return code, code_type
|
|
428
|
+
# For DRG-family types, extract the trailing DRG number
|
|
429
|
+
if norm_ct == 'MS-DRG':
|
|
430
|
+
m = _RE_DRG_TRAILING_NUM.search(rest)
|
|
431
|
+
if m:
|
|
432
|
+
return m.group(1), code_type
|
|
433
|
+
return code, code_type
|
|
434
|
+
return rest, code_type
|
|
435
|
+
return code, code_type
|
|
436
|
+
else:
|
|
437
|
+
# Code type is standard (CPT, HCPCS, etc.) and prefix doesn't
|
|
438
|
+
# match - don't split (the space is part of the code/description)
|
|
439
|
+
return code, code_type
|
|
440
|
+
|
|
441
|
+
prefix, rest = _match_composite_prefix(stripped)
|
|
442
|
+
if not prefix:
|
|
443
|
+
return code, code_type
|
|
444
|
+
|
|
445
|
+
if not rest:
|
|
446
|
+
return code, code_type
|
|
447
|
+
|
|
448
|
+
# For DRG-family prefixes, the actual DRG number is the trailing digits
|
|
449
|
+
# (e.g. "V41.0 (FY 2024) 155" → "155")
|
|
450
|
+
norm_prefix = _CODE_TYPE_NORMALIZE.get(prefix, prefix)
|
|
451
|
+
if not _valid_prefixed_code_remainder(rest, norm_prefix):
|
|
452
|
+
return code, code_type
|
|
453
|
+
if norm_prefix == 'MS-DRG':
|
|
454
|
+
m = _RE_DRG_TRAILING_NUM.search(rest)
|
|
455
|
+
if m:
|
|
456
|
+
return m.group(1), prefix
|
|
457
|
+
return code, code_type
|
|
458
|
+
|
|
459
|
+
return rest, prefix
|
|
460
|
+
|
|
461
|
+
|
|
462
|
+
# ── Split modifiers baked into the code field ──────────────────────────
|
|
463
|
+
#
|
|
464
|
+
# Hospitals sometimes fuse a modifier onto the procedure code in their
|
|
465
|
+
# MRF: `73721TC`, `87077QW`, `36415CP`, `82274SC`. These don't parse as a
|
|
466
|
+
# valid 5-char CPT, so they land under `code_type='CDM'` or `'LOCAL'` and
|
|
467
|
+
# the modifier is trapped in the code string.
|
|
468
|
+
#
|
|
469
|
+
# The helper takes the known-modifier set as a parameter so its rules can
|
|
470
|
+
# be tested on their own; `apply_baked_modifier_split` below wires in the
|
|
471
|
+
# curated set.
|
|
472
|
+
|
|
473
|
+
# Allowed prefix shapes the splitter recognizes - must be structurally
|
|
474
|
+
# valid CPT or HCPCS Level II. Anything else (ICD-10-PCS 7-char, NDC,
|
|
475
|
+
# legitimate 6-char numerics) MUST NOT be touched.
|
|
476
|
+
#
|
|
477
|
+
# Why these patterns specifically:
|
|
478
|
+
# * `\d{5}` - plain CPT (73721, 87077, 36415, 82274)
|
|
479
|
+
# * `\d{4}[FTU]` - CPT Cat-II/III/PLA (`0001F`, `0202U`, `0237T`).
|
|
480
|
+
# Currently the trailing letter is part of the code, not a modifier;
|
|
481
|
+
# but a baked-modifier composite like `0001FTC` is theoretically
|
|
482
|
+
# possible - exclude for now to avoid ambiguity.
|
|
483
|
+
# * `[A-CE-V]\d{4}` - HCPCS Level II (J0591, A4253). Excludes D
|
|
484
|
+
# (CDT - different code system) and W/X/Y/Z (not assigned).
|
|
485
|
+
#
|
|
486
|
+
# Modifier suffix shape: `[A-Z0-9]{2}` matches the 2-char atomic-modifier
|
|
487
|
+
# format that covers every curated atomic modifier (TC, SG, QW, LT, RT,
|
|
488
|
+
# JW, JZ, 22, 26, 50, etc.). The actual must-be-a-known-modifier check is
|
|
489
|
+
# the caller's responsibility: that's what guards against false-positive
|
|
490
|
+
# splits.
|
|
491
|
+
_RE_BAKED_MODIFIER_CPT5 = re.compile(r'^(\d{5})([A-Z0-9]{2})$')
|
|
492
|
+
_RE_BAKED_MODIFIER_HCPCS = re.compile(r'^([A-CE-V]\d{4})([A-Z0-9]{2})$')
|
|
493
|
+
|
|
494
|
+
|
|
495
|
+
def _split_attached_modifier_code(
|
|
496
|
+
code: str,
|
|
497
|
+
code_type: Optional[str],
|
|
498
|
+
known_modifiers: frozenset,
|
|
499
|
+
canonical_prefixes_by_type: Optional[Dict[str, frozenset]] = None,
|
|
500
|
+
digit_modifiers: Optional[frozenset] = None,
|
|
501
|
+
modifier_validity_by_code: Optional[Dict[str, frozenset]] = None,
|
|
502
|
+
descriptions_by_code: Optional[Dict[str, str]] = None,
|
|
503
|
+
row_description: Optional[str] = None,
|
|
504
|
+
) -> Tuple[str, Optional[str], Optional[str]]:
|
|
505
|
+
"""Detect a `<structurally-valid CPT/HCPCS><known atomic modifier>`
|
|
506
|
+
cell and split it.
|
|
507
|
+
|
|
508
|
+
Returns ``(clean_code, new_code_type, extracted_modifier)``:
|
|
509
|
+
* On a successful split (`73721TC` with `code_type='CDM'` and `TC`
|
|
510
|
+
in ``known_modifiers``): returns ``('73721', 'CPT', 'TC')``.
|
|
511
|
+
* On any guard failure (unrecognized suffix, wrong source
|
|
512
|
+
``code_type``, NDC/ICD shape, already-classified-as-CPT): returns
|
|
513
|
+
``(code, code_type, None)`` - caller treats the cell as-is.
|
|
514
|
+
|
|
515
|
+
Guards (in priority order):
|
|
516
|
+
|
|
517
|
+
1. ``code_type`` must be one of `CDM`, `LOCAL`, or `None` - codes
|
|
518
|
+
already filed under a real coding system are not in scope. A
|
|
519
|
+
`73721TC` already filed as `CPT` would mean the hospital encoded
|
|
520
|
+
it as a 7-char CPT, which is invalid; that's a separate bug.
|
|
521
|
+
2. The cell must match `_RE_BAKED_MODIFIER_CPT5` or
|
|
522
|
+
`_RE_BAKED_MODIFIER_HCPCS` exactly. ICD-10-PCS (7-char
|
|
523
|
+
alphanumeric like `02573ZZ` where `ZZ` is a legitimate part of
|
|
524
|
+
the code) does NOT match these regexes because the prefix shape
|
|
525
|
+
requires `\\d{5}` or `[A-CE-V]\\d{4}` - ICD codes use a
|
|
526
|
+
different layout. NDC codes are 9–11+ digits with no letter
|
|
527
|
+
suffix, also no match.
|
|
528
|
+
3. The 2-char suffix must be in ``known_modifiers``: the caller
|
|
529
|
+
passes the curated modifier set.
|
|
530
|
+
4. If ``canonical_prefixes_by_type`` is provided, the extracted prefix
|
|
531
|
+
must be in the canonical set for the resolved ``target_type``. Kills
|
|
532
|
+
the over-match where 7-digit hospital CDM IDs (`4954052`
|
|
533
|
+
"azithromycin tab") factor structurally as `<5-digit><2-char
|
|
534
|
+
modifier>` but the 5-digit prefix isn't a real published CPT/HCPCS,
|
|
535
|
+
only a coincidence. When ``None``, this guard is skipped.
|
|
536
|
+
5. Digit-suffix gate: if ``digit_modifiers`` is provided and the suffix
|
|
537
|
+
is in that set (i.e. it is a digit-only modifier token), then
|
|
538
|
+
``modifier_validity_by_code`` is consulted as the final guard.
|
|
539
|
+
- ``modifier_validity_by_code`` is ``None`` → default-deny.
|
|
540
|
+
- ``modifier_validity_by_code`` is a dict but the (prefix, suffix)
|
|
541
|
+
pair is absent → reject (not CMS-validated).
|
|
542
|
+
- ``modifier_validity_by_code`` has the prefix and suffix in its
|
|
543
|
+
frozenset → accept.
|
|
544
|
+
Letter suffixes never enter this branch.
|
|
545
|
+
6. Description-consistency guard: applies ONLY inside the digit-suffix
|
|
546
|
+
branch, AFTER Guard 5 passes.
|
|
547
|
+
- ``descriptions_by_code`` is ``None`` → skip the guard.
|
|
548
|
+
- ``descriptions_by_code`` is a dict → look up the reference
|
|
549
|
+
description for ``prefix`` and require
|
|
550
|
+
``_description_matches(row_description, canonical_desc)`` → True.
|
|
551
|
+
If no reference description is found, or the descriptions don't
|
|
552
|
+
match → reject.
|
|
553
|
+
Letter suffixes are NEVER subject to Guard 6.
|
|
554
|
+
|
|
555
|
+
Casing: the regexes accept uppercase only. Callers should already
|
|
556
|
+
have uppercased the cell via the existing pipeline (`_strip_non_ascii`
|
|
557
|
+
+ the `_CODE_TYPE_NORMALIZE` upper).
|
|
558
|
+
"""
|
|
559
|
+
if not code:
|
|
560
|
+
return code, code_type, None
|
|
561
|
+
|
|
562
|
+
# Guard 1: source code_type must be a hospital-internal bucket.
|
|
563
|
+
if code_type not in (None, 'CDM', 'LOCAL'):
|
|
564
|
+
return code, code_type, None
|
|
565
|
+
|
|
566
|
+
stripped = code.strip()
|
|
567
|
+
if not stripped:
|
|
568
|
+
return code, code_type, None
|
|
569
|
+
|
|
570
|
+
# Guard 2: structural match. Try CPT first (more common at the
|
|
571
|
+
# prevalence we measured), then HCPCS.
|
|
572
|
+
m = _RE_BAKED_MODIFIER_CPT5.match(stripped)
|
|
573
|
+
target_type: Optional[str] = None
|
|
574
|
+
if m:
|
|
575
|
+
target_type = 'CPT'
|
|
576
|
+
else:
|
|
577
|
+
m = _RE_BAKED_MODIFIER_HCPCS.match(stripped)
|
|
578
|
+
if m:
|
|
579
|
+
target_type = 'HCPCS'
|
|
580
|
+
|
|
581
|
+
if m is None:
|
|
582
|
+
return code, code_type, None
|
|
583
|
+
|
|
584
|
+
prefix, suffix = m.group(1), m.group(2)
|
|
585
|
+
|
|
586
|
+
# Guard 3: the suffix must be a real modifier per the caller's
|
|
587
|
+
# curated set.
|
|
588
|
+
if suffix not in known_modifiers:
|
|
589
|
+
return code, code_type, None
|
|
590
|
+
|
|
591
|
+
# Guard 4: the prefix must be a canonical published code. Skipped when
|
|
592
|
+
# the caller passes None.
|
|
593
|
+
if canonical_prefixes_by_type is not None:
|
|
594
|
+
prefixes = canonical_prefixes_by_type.get(target_type)
|
|
595
|
+
if not prefixes or prefix not in prefixes:
|
|
596
|
+
return code, code_type, None
|
|
597
|
+
|
|
598
|
+
# Guard 5: digit-suffix CMS-validity gate.
|
|
599
|
+
# Only digit tokens enter this branch; letter suffixes are unaffected.
|
|
600
|
+
# `modifier_validity_by_code` is keyed by bare CPT code only, so HCPCS
|
|
601
|
+
# digit composites have no entries and fall through to default-deny
|
|
602
|
+
# below. The frozensets hold only CMS-valid modifiers, so membership
|
|
603
|
+
# here IS the validity check.
|
|
604
|
+
if digit_modifiers and suffix in digit_modifiers:
|
|
605
|
+
# Default-deny when no validity matrix is provided.
|
|
606
|
+
if modifier_validity_by_code is None:
|
|
607
|
+
return code, code_type, None
|
|
608
|
+
valid = modifier_validity_by_code.get(prefix)
|
|
609
|
+
if not valid or suffix not in valid:
|
|
610
|
+
return code, code_type, None
|
|
611
|
+
|
|
612
|
+
# Guard 6: description-consistency guard. Skipped when
|
|
613
|
+
# descriptions_by_code is None. When provided, the row's free-text
|
|
614
|
+
# description must match the reference description via
|
|
615
|
+
# `_description_matches`: prevents coincidental CDM IDs (drugs,
|
|
616
|
+
# devices, generics) from splitting just because their 5-digit
|
|
617
|
+
# prefix happens to be a valid CPT.
|
|
618
|
+
if descriptions_by_code is not None:
|
|
619
|
+
canonical_desc = descriptions_by_code.get(prefix)
|
|
620
|
+
if not _description_matches(row_description, canonical_desc):
|
|
621
|
+
return code, code_type, None
|
|
622
|
+
|
|
623
|
+
return prefix, target_type, suffix
|
|
624
|
+
|
|
625
|
+
|
|
626
|
+
# ---------------------------------------------------------------------------
|
|
627
|
+
# The curated modifier set `apply_baked_modifier_split` splits on.
|
|
628
|
+
#
|
|
629
|
+
# Only LETTER suffixes split on shape alone. Digit suffixes are far riskier:
|
|
630
|
+
# 7-digit hospital CDM IDs often factor as `<5-digit><2-digit>` by
|
|
631
|
+
# coincidence (`4954052` "azithromycin tab" looks like `49540` + `52`), and
|
|
632
|
+
# splitting them on shape alone produced a flood of false positives. So the
|
|
633
|
+
# digit tokens split only when a CMS (code, modifier) validity matrix is
|
|
634
|
+
# supplied and confirms the pair (PC/TC indicator, bilateral indicator,
|
|
635
|
+
# CLIA-waived list).
|
|
636
|
+
#
|
|
637
|
+
# Other common baked modifiers (`SG`, `CL`, `AS`, `80-82`, `62/66`,
|
|
638
|
+
# `53/73/74`) are deliberately not in the set yet: the set is kept equal to
|
|
639
|
+
# the curated modifier dictionary so every split modifier is one the rest of
|
|
640
|
+
# a pipeline knows how to classify.
|
|
641
|
+
_BAKED_MODIFIER_LETTER_KEYS = frozenset({
|
|
642
|
+
# ------- component -------
|
|
643
|
+
"TC",
|
|
644
|
+
# ------- drug_supply -------
|
|
645
|
+
"JW", "JZ", "TB", "SL",
|
|
646
|
+
# ------- distinct -------
|
|
647
|
+
"XE", "XP", "XS", "XU",
|
|
648
|
+
# ------- enhancement / oversight -------
|
|
649
|
+
"QW",
|
|
650
|
+
# ------- anatomy / supply -------
|
|
651
|
+
"LT", "RT",
|
|
652
|
+
# ------- admin_noise -------
|
|
653
|
+
"GY", "FY", "GO", "GN", "GP", "NU", "PO",
|
|
654
|
+
})
|
|
655
|
+
|
|
656
|
+
# Numeric-suffix tokens. Split only when a CMS validity matrix confirms the
|
|
657
|
+
# (CPT, modifier) pair.
|
|
658
|
+
_BAKED_MODIFIER_DIGIT_KEYS = frozenset({
|
|
659
|
+
# ------- component -------
|
|
660
|
+
"26",
|
|
661
|
+
# ------- surgical_phase -------
|
|
662
|
+
"54", "55", "56", "58", "78", "79", "24",
|
|
663
|
+
# ------- repeat / distinct -------
|
|
664
|
+
"76", "77", "91", "59",
|
|
665
|
+
# ------- enhancement / oversight -------
|
|
666
|
+
"22", "25", "50", "52", "90", "95",
|
|
667
|
+
})
|
|
668
|
+
|
|
669
|
+
_BAKED_MODIFIER_SPLIT_KEYS = _BAKED_MODIFIER_LETTER_KEYS | _BAKED_MODIFIER_DIGIT_KEYS
|
|
670
|
+
|
|
671
|
+
|
|
672
|
+
# ---------------------------------------------------------------------------
|
|
673
|
+
# Description-consistency helper.
|
|
674
|
+
#
|
|
675
|
+
# Decides whether a CDM row's free-text description is consistent with a
|
|
676
|
+
# reference description for the code (``ReferenceData.code_descriptions``).
|
|
677
|
+
# Both strings are normalized before comparison:
|
|
678
|
+
#
|
|
679
|
+
# 1. Uppercase; strip surrounding quotes; replace non-alphanumeric runs
|
|
680
|
+
# with spaces; collapse whitespace.
|
|
681
|
+
# 2. Expand a small abbreviation map so common radiography short-forms
|
|
682
|
+
# align with the reference descriptions.
|
|
683
|
+
# 3. Reject on a denylist of generic/non-specific descriptions that match
|
|
684
|
+
# virtually anything (e.g. "OTHER OUTPATIENT", "MISC").
|
|
685
|
+
# 4. Tokenize both sides; drop 1-char tokens (noise). Compute overlap of
|
|
686
|
+
# the canonical token set C against the row token set R.
|
|
687
|
+
# Match condition: |C ∩ R| / |C| >= 0.5 AND |C ∩ R| >= 1
|
|
688
|
+
# AND at least one shared token has len >= 4 (avoids matching on tiny
|
|
689
|
+
# words like "OF", "OR", "BY", "MG" only).
|
|
690
|
+
#
|
|
691
|
+
# Rationale for the 0.5 threshold: reference short descriptions are often
|
|
692
|
+
# brief (2-4 meaningful tokens); requiring only 50 % coverage admits slight synonymy
|
|
693
|
+
# while blocking completely unrelated descriptions (drugs, devices).
|
|
694
|
+
# ---------------------------------------------------------------------------
|
|
695
|
+
|
|
696
|
+
# Small abbreviation expansion map.
|
|
697
|
+
# Applied in two passes:
|
|
698
|
+
# Pass 1 (pre-normalization): whole-string regex substitutions that must
|
|
699
|
+
# fire BEFORE the general non-alphanumeric → space replacement. Handles
|
|
700
|
+
# hyphenated forms like "X-RAY" → "XRAY" which would otherwise be split
|
|
701
|
+
# into "X" + "RAY" by the normalizer.
|
|
702
|
+
# Pass 2 (post-normalization): token-level lookup on the already-uppercased,
|
|
703
|
+
# whitespace-collapsed string. Keys are standalone uppercase tokens.
|
|
704
|
+
# Keep the map small and well-commented - it's a precision tuning knob.
|
|
705
|
+
|
|
706
|
+
# Pass 1: pre-normalization whole-word substitutions (case-insensitive regex).
|
|
707
|
+
# Each entry is (pattern, replacement) where pattern is matched against the
|
|
708
|
+
# uppercased raw string before non-alpha stripping.
|
|
709
|
+
_DESC_PRENORM_SUBS: list = [
|
|
710
|
+
# "X-RAY" / "X-RAYS" → "XRAY" / "XRAYS" (hyphen removed before normalization
|
|
711
|
+
# strips all non-alphanumeric, so the compound becomes a single token).
|
|
712
|
+
(re.compile(r"\bX-RAYS?\b"), "XRAY"),
|
|
713
|
+
]
|
|
714
|
+
|
|
715
|
+
# Pass 2: post-normalization token → canonical expansion.
|
|
716
|
+
# Applied after uppercasing + non-alpha strip + collapse.
|
|
717
|
+
_DESC_ABBREV: Dict[str, str] = {
|
|
718
|
+
"XR": "XRAY", # "XR FEMUR" → "XRAY FEMUR"
|
|
719
|
+
"BIL": "BILATERAL",
|
|
720
|
+
"BILAT": "BILATERAL",
|
|
721
|
+
"W": "WITH", # "W/" becomes "W" after non-alpha strip
|
|
722
|
+
"WO": "WITHOUT", # "W/O" becomes "WO" after non-alpha strip
|
|
723
|
+
"BX": "BIOPSY",
|
|
724
|
+
}
|
|
725
|
+
|
|
726
|
+
# Normalized forms of the generic denylist (applied after normalization step).
|
|
727
|
+
_DESC_GENERIC_DENYLIST: frozenset = frozenset({
|
|
728
|
+
"OTHER OUTPATIENT",
|
|
729
|
+
"OUTPATIENT",
|
|
730
|
+
"OTHER",
|
|
731
|
+
"MISC",
|
|
732
|
+
"MISCELLANEOUS",
|
|
733
|
+
})
|
|
734
|
+
|
|
735
|
+
# Non-discriminating filler words. Removed from BOTH token sets before the
|
|
736
|
+
# coverage calc so they don't dilute the canonical denominator: short
|
|
737
|
+
# reference descriptions often carry "AND"/"OF", which would otherwise drop
|
|
738
|
+
# a real composite below the threshold.
|
|
739
|
+
# Dropping them never lowers precision (they carry no clinical signal).
|
|
740
|
+
_DESC_STOPWORDS: frozenset = frozenset({
|
|
741
|
+
"AND", "OF", "OR", "THE", "WITH", "WITHOUT", "FOR", "TO", "IN", "ON",
|
|
742
|
+
"BY", "A", "AN",
|
|
743
|
+
})
|
|
744
|
+
|
|
745
|
+
_RE_DESC_NONALNUM = re.compile(r"[^A-Z0-9]+")
|
|
746
|
+
|
|
747
|
+
|
|
748
|
+
def _normalize_desc(s: str) -> str:
|
|
749
|
+
"""Normalize a description for comparison:
|
|
750
|
+
upper-case → pre-norm substitutions → strip surrounding quotes →
|
|
751
|
+
replace non-alphanumeric with spaces → collapse whitespace.
|
|
752
|
+
"""
|
|
753
|
+
s = s.upper().strip()
|
|
754
|
+
# Pass 1: pre-normalization substitutions (hyphenated compounds).
|
|
755
|
+
for pattern, repl in _DESC_PRENORM_SUBS:
|
|
756
|
+
s = pattern.sub(repl, s)
|
|
757
|
+
s = s.strip('"').strip("'")
|
|
758
|
+
s = _RE_DESC_NONALNUM.sub(" ", s).strip()
|
|
759
|
+
return s
|
|
760
|
+
|
|
761
|
+
|
|
762
|
+
def _description_matches(
|
|
763
|
+
row_desc: Optional[str],
|
|
764
|
+
canonical_desc: Optional[str],
|
|
765
|
+
) -> bool:
|
|
766
|
+
"""Return True when the CDM row description is consistent with the
|
|
767
|
+
reference description for the code.
|
|
768
|
+
|
|
769
|
+
Both strings must be non-empty after normalization; empty-after-normalize
|
|
770
|
+
→ False (default-deny). Generic/non-specific row descriptions → False.
|
|
771
|
+
|
|
772
|
+
Matching rule (token-set overlap):
|
|
773
|
+
Let C = canonical token set (tokens len >= 2), R = row token set
|
|
774
|
+
(tokens len >= 2).
|
|
775
|
+
Match when:
|
|
776
|
+
|C ∩ R| / |C| >= 0.5 (at least half of canonical tokens covered)
|
|
777
|
+
AND |C ∩ R| >= 1
|
|
778
|
+
AND at least one shared token has len >= 4 (no trivial-word match)
|
|
779
|
+
"""
|
|
780
|
+
if not row_desc or not canonical_desc:
|
|
781
|
+
return False
|
|
782
|
+
|
|
783
|
+
row_norm = _normalize_desc(row_desc)
|
|
784
|
+
can_norm = _normalize_desc(canonical_desc)
|
|
785
|
+
|
|
786
|
+
if not row_norm or not can_norm:
|
|
787
|
+
return False
|
|
788
|
+
|
|
789
|
+
# Reject generic row descriptions outright.
|
|
790
|
+
if row_norm in _DESC_GENERIC_DENYLIST:
|
|
791
|
+
return False
|
|
792
|
+
|
|
793
|
+
# Pass 2: token-level abbreviation expansion.
|
|
794
|
+
def _expand(text: str) -> str:
|
|
795
|
+
tokens = text.split()
|
|
796
|
+
return " ".join(_DESC_ABBREV.get(t, t) for t in tokens)
|
|
797
|
+
|
|
798
|
+
row_norm = _expand(row_norm)
|
|
799
|
+
can_norm = _expand(can_norm)
|
|
800
|
+
|
|
801
|
+
# Tokenize; drop 1-char noise tokens and non-discriminating stop-words
|
|
802
|
+
# (stop-words in the reference description would otherwise inflate the
|
|
803
|
+
# coverage denominator and reject real composites).
|
|
804
|
+
row_tokens = {t for t in row_norm.split()
|
|
805
|
+
if len(t) >= 2 and t not in _DESC_STOPWORDS}
|
|
806
|
+
can_tokens = {t for t in can_norm.split()
|
|
807
|
+
if len(t) >= 2 and t not in _DESC_STOPWORDS}
|
|
808
|
+
|
|
809
|
+
if not can_tokens:
|
|
810
|
+
return False
|
|
811
|
+
|
|
812
|
+
overlap = can_tokens & row_tokens
|
|
813
|
+
if not overlap:
|
|
814
|
+
return False
|
|
815
|
+
|
|
816
|
+
# Coverage: at least half of canonical tokens must be present.
|
|
817
|
+
coverage = len(overlap) / len(can_tokens)
|
|
818
|
+
if coverage < 0.5:
|
|
819
|
+
return False
|
|
820
|
+
|
|
821
|
+
# At least one shared token must be "substantial" (len >= 4) so two
|
|
822
|
+
# descriptions that share only tiny words ("OF", "MG", "BY") don't match.
|
|
823
|
+
if not any(len(t) >= 4 for t in overlap):
|
|
824
|
+
return False
|
|
825
|
+
|
|
826
|
+
return True
|
|
827
|
+
|
|
828
|
+
|
|
829
|
+
def apply_baked_modifier_split(
|
|
830
|
+
code: Optional[str],
|
|
831
|
+
code_type: Optional[str],
|
|
832
|
+
description: Optional[str] = None,
|
|
833
|
+
*,
|
|
834
|
+
ref: Optional[ReferenceData] = None,
|
|
835
|
+
stats: Optional[ParseStats] = None,
|
|
836
|
+
) -> Tuple[Optional[str], Optional[str], Optional[str]]:
|
|
837
|
+
"""Split a modifier baked into the code, using the curated modifier set.
|
|
838
|
+
|
|
839
|
+
Returns ``(code, code_type, baked_modifier)``:
|
|
840
|
+
* Successful split (`'73721TC' + 'CDM'` → `('73721', 'CPT', 'TC')`):
|
|
841
|
+
caller updates code/code_type AND merges ``baked_modifier`` into
|
|
842
|
+
the row's `modifiers` field (via `merge_modifier_into_field`).
|
|
843
|
+
* No split (any guard failure): returns the original ``(code,
|
|
844
|
+
code_type, None)`` unchanged. Caller proceeds as before.
|
|
845
|
+
|
|
846
|
+
Safe to call with ``code = None`` or ``code = ''`` (returns inputs
|
|
847
|
+
unchanged with ``baked_modifier = None``).
|
|
848
|
+
|
|
849
|
+
Letter suffixes split on shape alone. With ``ref``:
|
|
850
|
+
* ``ref.code_prefixes`` makes the prefix a published CPT/HCPCS code
|
|
851
|
+
(Guard 4);
|
|
852
|
+
* ``ref.modifier_validity`` lets digit suffixes split when CMS marks
|
|
853
|
+
the (code, modifier) pair valid (Guard 5). Without it digit
|
|
854
|
+
suffixes never split;
|
|
855
|
+
* ``ref.code_descriptions`` additionally requires the row
|
|
856
|
+
description to match the code's reference description before a
|
|
857
|
+
digit suffix splits (Guard 6).
|
|
858
|
+
|
|
859
|
+
``stats``, when given, counts each split.
|
|
860
|
+
"""
|
|
861
|
+
if not code:
|
|
862
|
+
return code, code_type, None
|
|
863
|
+
result = _split_attached_modifier_code(
|
|
864
|
+
code, code_type,
|
|
865
|
+
_BAKED_MODIFIER_SPLIT_KEYS,
|
|
866
|
+
canonical_prefixes_by_type=ref.code_prefixes if ref else None,
|
|
867
|
+
digit_modifiers=_BAKED_MODIFIER_DIGIT_KEYS,
|
|
868
|
+
modifier_validity_by_code=ref.modifier_validity if ref else None,
|
|
869
|
+
descriptions_by_code=ref.code_descriptions if ref else None,
|
|
870
|
+
row_description=description,
|
|
871
|
+
)
|
|
872
|
+
if stats is not None and result[2]:
|
|
873
|
+
stats.record_baked_modifier_split(result[2], code_type)
|
|
874
|
+
return result
|
|
875
|
+
|
|
876
|
+
|
|
877
|
+
def merge_modifier_into_field(
|
|
878
|
+
existing: Optional[str],
|
|
879
|
+
baked: Optional[str],
|
|
880
|
+
) -> Optional[str]:
|
|
881
|
+
"""Merge a modifier split out of the code into a row's modifiers field.
|
|
882
|
+
|
|
883
|
+
Used after ``apply_baked_modifier_split`` returns a non-None
|
|
884
|
+
``baked_modifier``: the caller adds it to whatever the MRF row already
|
|
885
|
+
has in its `modifiers` column. Token-set semantics: order is
|
|
886
|
+
preserved (existing tokens first, baked appended last) but a duplicate
|
|
887
|
+
is dropped so a fixture row with both `'73721TC'` AND a `'TC'` in its
|
|
888
|
+
modifiers column doesn't end up with `'TC,TC'`.
|
|
889
|
+
|
|
890
|
+
Tokens are split on `[|, ;]+`, the delimiter set modifier fields use
|
|
891
|
+
in the wild. The output uses a single comma separator so it
|
|
892
|
+
round-trips through the same tokenizer cleanly.
|
|
893
|
+
|
|
894
|
+
Returns the merged string, or ``None`` if both inputs are empty.
|
|
895
|
+
"""
|
|
896
|
+
if not baked:
|
|
897
|
+
return existing
|
|
898
|
+
baked = baked.strip()
|
|
899
|
+
if not baked:
|
|
900
|
+
return existing
|
|
901
|
+
if not existing or not existing.strip():
|
|
902
|
+
return baked
|
|
903
|
+
|
|
904
|
+
# Use the same delimiter regex readers split on so we don't invent a
|
|
905
|
+
# new token boundary here.
|
|
906
|
+
tokens = [t for t in re.split(r"[|, ;]+", existing) if t]
|
|
907
|
+
if baked in tokens:
|
|
908
|
+
# Already present - return canonicalized form (comma-joined, no
|
|
909
|
+
# leading/trailing whitespace) so we don't drift the field shape.
|
|
910
|
+
return ",".join(tokens)
|
|
911
|
+
tokens.append(baked)
|
|
912
|
+
return ",".join(tokens)
|
|
913
|
+
|
|
914
|
+
|
|
915
|
+
def normalize_code(
|
|
916
|
+
code: Optional[str],
|
|
917
|
+
code_type: Optional[str],
|
|
918
|
+
) -> Tuple[Optional[str], Optional[str]]:
|
|
919
|
+
"""
|
|
920
|
+
Normalize code and code_type to standard formats.
|
|
921
|
+
|
|
922
|
+
Returns: (normalized_code, normalized_code_type)
|
|
923
|
+
|
|
924
|
+
Rules applied:
|
|
925
|
+
1. Strip non-ASCII characters (mojibake, encoding artifacts)
|
|
926
|
+
2. Uppercase and canonicalize code_type via _CODE_TYPE_NORMALIZE
|
|
927
|
+
2A. NDCHCPCS / NDCCPT disambiguation: pick real type from code shape
|
|
928
|
+
2B. Junk code_type discard (TYPE/MODIFIER/MODIFIERS/OBS/DOT/UN → treat
|
|
929
|
+
as missing, re-derive from code shape)
|
|
930
|
+
3. Detect hospital-internal codes (SUP-*, PX-*, RX-*) → CDM
|
|
931
|
+
4. Detect misclassified Revenue Codes stored as HCPCS → RC
|
|
932
|
+
4.5. Reclassify unambiguous standard code shapes mislabeled CDM/LOCAL
|
|
933
|
+
5. Normalize HCPCS Level I → CPT for 5-digit numeric codes
|
|
934
|
+
5.5. Reclassify D-prefix codes (CDT dental) → CDT
|
|
935
|
+
5.6. HCPCS Level II forward fix: letter-prefix [A-CE-V]\\d{4} that
|
|
936
|
+
hospitals labeled CPT → HCPCS (excludes D handled by 5.5)
|
|
937
|
+
5.7. CPT Cat II/III/PLA reverse fix: \\d{4}[FTU] that hospitals
|
|
938
|
+
labeled HCPCS → CPT
|
|
939
|
+
6. Zero-pad Revenue Codes to 4 digits
|
|
940
|
+
6A. Cross-system reclassify: controlled-vocab numeric type holding a
|
|
941
|
+
standard-shape code (e.g. CPT 99232 declared as RC) → reclassify
|
|
942
|
+
to correct standard type before residue gating.
|
|
943
|
+
7. Normalize NDC to canonical 11-digit no-dash form (see
|
|
944
|
+
_normalize_ndc): leading-letter NDCs reclassify to LOCAL;
|
|
945
|
+
3-segment dashed forms with optional packaging suffix collapse to
|
|
946
|
+
11 digits; 9-digit numeric pads to 11 with '00'; 10/11-digit
|
|
947
|
+
bases with _ip/_op, trailing letter, or _<digits> package suffix
|
|
948
|
+
canonicalize to 11.
|
|
949
|
+
8. Infer code_type from code format when code_type is missing (then
|
|
950
|
+
re-apply NDC normalization if NDC was inferred)
|
|
951
|
+
9. Validate code against declared code_type format; extract valid code
|
|
952
|
+
from garbage wrapping when possible (e.g. XJ0591X → J0591/HCPCS)
|
|
953
|
+
10. Classify placeholder/test codes as LOCAL
|
|
954
|
+
11. APR-DRG qualifier stripping
|
|
955
|
+
12. Residue gating: controlled-vocab numeric type with a code that still
|
|
956
|
+
fails the canonical regex → reclassify to LOCAL (code preserved).
|
|
957
|
+
"""
|
|
958
|
+
if not code and not code_type:
|
|
959
|
+
return code, code_type
|
|
960
|
+
|
|
961
|
+
# --- Step 1: Strip non-ASCII characters ---
|
|
962
|
+
if code:
|
|
963
|
+
cleaned = _strip_non_ascii(code.strip())
|
|
964
|
+
if cleaned != code.strip():
|
|
965
|
+
code = cleaned
|
|
966
|
+
if not code:
|
|
967
|
+
return code, code_type
|
|
968
|
+
|
|
969
|
+
# --- Step 1.5: Split composite code values ---
|
|
970
|
+
# Some hospitals (e.g. Partners Healthcare / Brigham & Women's) embed the
|
|
971
|
+
# code type as a prefix in the code column: "HCPCS C1776", "CPT 87339",
|
|
972
|
+
# "MS-DRG V41.0 (FY 2024) 155". Split these so the code type flows into
|
|
973
|
+
# Step 2 for normalization.
|
|
974
|
+
if code:
|
|
975
|
+
code, code_type = _split_composite_code(code, code_type)
|
|
976
|
+
|
|
977
|
+
# --- Step 2: Normalize code_type ---
|
|
978
|
+
norm_type = None
|
|
979
|
+
if code_type:
|
|
980
|
+
raw_upper = code_type.strip().upper()
|
|
981
|
+
norm_type = _CODE_TYPE_NORMALIZE.get(raw_upper, raw_upper)
|
|
982
|
+
|
|
983
|
+
# --- Step 2A: NDCHCPCS / NDCCPT disambiguation ---
|
|
984
|
+
# Some CMS/hospital data formats emit compound types like "NDCHCPCS" or
|
|
985
|
+
# "NDCCPT" to indicate the code column may contain either an NDC or a
|
|
986
|
+
# procedure code. Pick the real type by code shape:
|
|
987
|
+
# - NDC shape → normalize through _normalize_ndc() → NDC or LOCAL
|
|
988
|
+
# - [A-CE-V]\d{4} → HCPCS (Level II letter-prefix)
|
|
989
|
+
# - \d{5} or \d{4}[FTU] → CPT (for NDCCPT; also acceptable for NDCHCPCS)
|
|
990
|
+
# - else → LOCAL
|
|
991
|
+
if code and norm_type in ('NDCHCPCS', 'NDCCPT'):
|
|
992
|
+
s = code.strip()
|
|
993
|
+
# NDC shape: 11-digit, 9-digit, 10-digit, or 3-segment dashed form
|
|
994
|
+
if (re.fullmatch(r'\d{9,11}', s)
|
|
995
|
+
or re.fullmatch(r'\d{4,5}-\d{3,4}-\d{1,2}', s)):
|
|
996
|
+
new_code, new_type = _normalize_ndc(code)
|
|
997
|
+
return new_code, new_type
|
|
998
|
+
elif re.fullmatch(r'[A-CE-Va-ce-v]\d{4}', s):
|
|
999
|
+
norm_type = 'HCPCS'
|
|
1000
|
+
elif (re.fullmatch(r'\d{5}', s)
|
|
1001
|
+
or re.fullmatch(r'\d{4}[FTUftu]', s)):
|
|
1002
|
+
norm_type = 'CPT'
|
|
1003
|
+
else:
|
|
1004
|
+
norm_type = 'LOCAL'
|
|
1005
|
+
|
|
1006
|
+
# --- Step 2B: Junk code_type discard ---
|
|
1007
|
+
# Non-system labels (TYPE, MODIFIER, MODIFIERS, OBS, DOT, UN) carry no
|
|
1008
|
+
# system information. Discard them so the blank-type path (steps 4.5,
|
|
1009
|
+
# 5, 5.5–5.7, 8) can re-derive the type from code shape.
|
|
1010
|
+
if norm_type in _JUNK_CODE_TYPES:
|
|
1011
|
+
norm_type = None
|
|
1012
|
+
|
|
1013
|
+
# --- Step 2.5: Strip trailing float-coercion artifact (.0) ---
|
|
1014
|
+
# Excel / ETL tools sometimes coerce integer revenue codes and DRG numbers
|
|
1015
|
+
# to floats, writing "320.0" instead of "320". Strip a trailing "\.0+$"
|
|
1016
|
+
# ONLY for numeric controlled-vocabulary types where the decimal is always
|
|
1017
|
+
# spurious. MUST NOT apply to ICD (250.0 is a valid ICD-9 code), CPT,
|
|
1018
|
+
# HCPCS, NDC, CDM, LOCAL, or any other type.
|
|
1019
|
+
if code and norm_type in _FLOAT_COERCE_STRIP_TYPES:
|
|
1020
|
+
stripped_code = code.strip()
|
|
1021
|
+
stripped_code = re.sub(r'\.0+$', '', stripped_code)
|
|
1022
|
+
if stripped_code != code.strip():
|
|
1023
|
+
code = stripped_code
|
|
1024
|
+
|
|
1025
|
+
# --- Step 3: Detect hospital-internal codes by prefix ---
|
|
1026
|
+
if code:
|
|
1027
|
+
code_upper = code.strip()
|
|
1028
|
+
# SUP- (supplies), PX- (procedures), RX- (pharmacy) are CDM conventions
|
|
1029
|
+
if code_upper.startswith(('SUP-', 'PX-', 'RX-')):
|
|
1030
|
+
return code_upper, 'CDM'
|
|
1031
|
+
|
|
1032
|
+
# --- Step 4: Detect misclassified codes ---
|
|
1033
|
+
# Some hospitals report RC values under code_type='HCPCS'
|
|
1034
|
+
if code and norm_type in ('HCPCS', None) and code.strip() in _KNOWN_REVENUE_CODES:
|
|
1035
|
+
norm_type = 'RC'
|
|
1036
|
+
# Numeric codes > 5 digits labeled as CPT/HCPCS are really CDM codes
|
|
1037
|
+
elif (code and norm_type in ('CPT', 'HCPCS')
|
|
1038
|
+
and code.strip().isdigit() and len(code.strip()) > 5):
|
|
1039
|
+
norm_type = 'CDM'
|
|
1040
|
+
|
|
1041
|
+
# --- Step 4.5: Recover standard codes mislabeled as CDM/LOCAL ---
|
|
1042
|
+
# CDM/LOCAL are hospital-internal buckets, but many files put standard
|
|
1043
|
+
# codes there. Only reclassify shapes that are unambiguous. Bare 3-digit
|
|
1044
|
+
# DRGs and bare 4-digit APCs are intentionally not inferred because they
|
|
1045
|
+
# collide with revenue codes and hospital-local identifiers. Prefixed
|
|
1046
|
+
# values such as "MS-DRG 470" and "APC 5191" are handled by Step 1.5.
|
|
1047
|
+
if code and norm_type in ('CDM', 'LOCAL'):
|
|
1048
|
+
stripped = code.strip()
|
|
1049
|
+
if re.fullmatch(r'[Dd]\d{4}', stripped):
|
|
1050
|
+
norm_type = 'CDT'
|
|
1051
|
+
elif re.fullmatch(r'[A-CE-Va-ce-v]\d{4}', stripped):
|
|
1052
|
+
norm_type = 'HCPCS'
|
|
1053
|
+
elif (re.fullmatch(r'\d{5}', stripped)
|
|
1054
|
+
or re.fullmatch(r'\d{4}[FTUftu]', stripped)):
|
|
1055
|
+
norm_type = 'CPT'
|
|
1056
|
+
|
|
1057
|
+
# --- Step 5: Normalize HCPCS Level I → CPT ---
|
|
1058
|
+
# Many hospitals label all procedure codes as 'HCPCS' even when they are
|
|
1059
|
+
# 5-digit numeric CPT codes. HCPCS Level I ≡ CPT, so normalize them.
|
|
1060
|
+
# True HCPCS Level II codes have a letter prefix (A0000-V9999) and are
|
|
1061
|
+
# already correctly classified.
|
|
1062
|
+
if (code and norm_type == 'HCPCS'
|
|
1063
|
+
and code.strip().isdigit() and len(code.strip()) == 5):
|
|
1064
|
+
norm_type = 'CPT'
|
|
1065
|
+
|
|
1066
|
+
# --- Step 5.5: Reclassify D-prefix codes as CDT ---
|
|
1067
|
+
# CDT dental codes (D0120, D7509, etc.) are often misreported as HCPCS or
|
|
1068
|
+
# CPT by hospitals. Reclassify them so dental codes have a consistent type.
|
|
1069
|
+
if (code and norm_type in ('HCPCS', 'CPT')
|
|
1070
|
+
and re.fullmatch(r'[Dd]\d{4}', code.strip())):
|
|
1071
|
+
norm_type = 'CDT'
|
|
1072
|
+
|
|
1073
|
+
# --- Step 5.6: HCPCS Level II forward fix ---
|
|
1074
|
+
# Letter-prefix Level II codes (A0000-V9999, excluding D) are HCPCS by
|
|
1075
|
+
# definition. Some hospital MRFs misclassify them as CPT (e.g. Q5128
|
|
1076
|
+
# under code_type='CPT'). CPT codes are unambiguously 5-digit numerics
|
|
1077
|
+
# or `\d{4}[FTU]` Cat II/III/PLA codes, never letter-prefix.
|
|
1078
|
+
# Excludes D (CDT) which Step 5.5 already handled.
|
|
1079
|
+
if (code and norm_type == 'CPT'
|
|
1080
|
+
and re.fullmatch(r'[A-CE-Va-ce-v]\d{4}', code.strip())):
|
|
1081
|
+
norm_type = 'HCPCS'
|
|
1082
|
+
|
|
1083
|
+
# --- Step 5.7: CPT Cat II/III/PLA reverse fix ---
|
|
1084
|
+
# Cat II (\d{4}F), Cat III (\d{4}T), and PLA (\d{4}U) codes are CPT
|
|
1085
|
+
# by definition (HCPCS Level II never has a trailing F/T/U). Some
|
|
1086
|
+
# hospital MRFs put these under code_type='HCPCS', and some omit the
|
|
1087
|
+
# type entirely; normalize both cases to CPT.
|
|
1088
|
+
if (code and norm_type in ('HCPCS', None)
|
|
1089
|
+
and re.fullmatch(r'\d{4}[FTUftu]', code.strip())):
|
|
1090
|
+
norm_type = 'CPT'
|
|
1091
|
+
|
|
1092
|
+
# --- Step 6A: Cross-system reclassify ---
|
|
1093
|
+
# A controlled-vocab numeric type (RC / MS-DRG / APR-DRG / APC / CMG /
|
|
1094
|
+
# EAPG) may hold a standard-shape procedure code that a hospital
|
|
1095
|
+
# misrouted (e.g. CPT 99232 declared as RC). If the code fails the
|
|
1096
|
+
# type's canonical regex AND matches a standard shape, reclassify:
|
|
1097
|
+
# \d{5} / \d{4}[FTU] → CPT
|
|
1098
|
+
# [A-CE-V]\d{4} → HCPCS
|
|
1099
|
+
# D\d{4} → CDT
|
|
1100
|
+
# Run this BEFORE residue gating (step 12) so reclassified codes never
|
|
1101
|
+
# reach the LOCAL fallback. RC zero-padding (step 6) runs after this so
|
|
1102
|
+
# we don't accidentally treat a 5-digit CPT as a 5-digit revenue code.
|
|
1103
|
+
if code and norm_type in _CONTROLLED_VOCAB_VALIDATORS:
|
|
1104
|
+
validator = _CONTROLLED_VOCAB_VALIDATORS[norm_type]
|
|
1105
|
+
stripped = code.strip()
|
|
1106
|
+
if not validator.fullmatch(stripped):
|
|
1107
|
+
if re.fullmatch(r'\d{5}', stripped):
|
|
1108
|
+
norm_type = 'CPT'
|
|
1109
|
+
elif re.fullmatch(r'\d{4}[FTUftu]', stripped):
|
|
1110
|
+
norm_type = 'CPT'
|
|
1111
|
+
elif re.fullmatch(r'[A-CE-Va-ce-v]\d{4}', stripped):
|
|
1112
|
+
norm_type = 'HCPCS'
|
|
1113
|
+
elif re.fullmatch(r'[Dd]\d{4}', stripped):
|
|
1114
|
+
norm_type = 'CDT'
|
|
1115
|
+
|
|
1116
|
+
# --- Step 6: Zero-pad Revenue Codes ---
|
|
1117
|
+
if code and norm_type == 'RC':
|
|
1118
|
+
stripped = code.strip()
|
|
1119
|
+
if stripped.isdigit():
|
|
1120
|
+
code = stripped.zfill(4)
|
|
1121
|
+
return code, norm_type
|
|
1122
|
+
|
|
1123
|
+
# --- Step 7: Normalize NDC to canonical 11-digit no-dash form ---
|
|
1124
|
+
# See _normalize_ndc(). Handles leading-letter→LOCAL reclassification,
|
|
1125
|
+
# 9- and 10-digit padding, setting/letter/package (incl. double)
|
|
1126
|
+
# suffixes, and 3-segment dashed shapes.
|
|
1127
|
+
if code and norm_type == 'NDC':
|
|
1128
|
+
new_code, new_type = _normalize_ndc(code)
|
|
1129
|
+
return new_code, new_type
|
|
1130
|
+
|
|
1131
|
+
# --- Step 8: Infer code_type from code format when missing ---
|
|
1132
|
+
if code and not norm_type:
|
|
1133
|
+
stripped = code.strip()
|
|
1134
|
+
# CDT dental codes: D + 4 digits (D0120, D7509, etc.)
|
|
1135
|
+
if re.fullmatch(r'[Dd]\d{4}', stripped):
|
|
1136
|
+
norm_type = 'CDT'
|
|
1137
|
+
# HCPCS Level II: letter + 4 digits (J0690, C1776, A4550, etc.)
|
|
1138
|
+
elif re.fullmatch(r'[A-Za-z]\d{4}', stripped):
|
|
1139
|
+
norm_type = 'HCPCS'
|
|
1140
|
+
# CPT: exactly 5 digits
|
|
1141
|
+
elif re.fullmatch(r'\d{5}', stripped):
|
|
1142
|
+
norm_type = 'CPT'
|
|
1143
|
+
# CPT Category II/III/PLA: 4 digits + F/T/U (1036F, 0237T, 0016U)
|
|
1144
|
+
elif re.fullmatch(r'\d{4}[FTUftu]', stripped):
|
|
1145
|
+
norm_type = 'CPT'
|
|
1146
|
+
# NDC: digits with dashes (10-11 digit segments). Run the full
|
|
1147
|
+
# NDC normalizer so the inferred NDC is also canonicalized.
|
|
1148
|
+
elif re.fullmatch(r'\d{4,5}-\d{3,4}-\d{1,2}', stripped):
|
|
1149
|
+
new_code, new_type = _normalize_ndc(code)
|
|
1150
|
+
return new_code, new_type
|
|
1151
|
+
# MS-DRG: exactly 3 digits
|
|
1152
|
+
elif re.fullmatch(r'\d{3}', stripped):
|
|
1153
|
+
# Ambiguous: could be RC or MS-DRG. Don't guess - leave as-is.
|
|
1154
|
+
pass
|
|
1155
|
+
|
|
1156
|
+
# --- Step 9: Validate code against declared code_type ---
|
|
1157
|
+
# If the code doesn't match the expected format for its declared type,
|
|
1158
|
+
# try to extract a valid code from the raw value. Handles cases like
|
|
1159
|
+
# "XJ0591X" labeled as CPT → extract "J0591" and reclassify as HCPCS.
|
|
1160
|
+
if code and norm_type in ('CPT', 'HCPCS'):
|
|
1161
|
+
stripped = code.strip().upper()
|
|
1162
|
+
is_valid = (
|
|
1163
|
+
_RE_CPT.fullmatch(stripped) # CPT: 99213
|
|
1164
|
+
or _RE_CPT_PLA.fullmatch(stripped) # PLA: 0202U, 0003M
|
|
1165
|
+
or _RE_HCPCS.fullmatch(stripped) # HCPCS: J0591
|
|
1166
|
+
or _RE_CAT3.fullmatch(stripped) # Category III: 0237T
|
|
1167
|
+
)
|
|
1168
|
+
if not is_valid:
|
|
1169
|
+
# Try to extract a valid code from the garbage
|
|
1170
|
+
extracted_code, extracted_type = _extract_code_from_garbage(stripped)
|
|
1171
|
+
if extracted_code:
|
|
1172
|
+
code = extracted_code
|
|
1173
|
+
norm_type = extracted_type
|
|
1174
|
+
else:
|
|
1175
|
+
# Can't salvage - downgrade to LOCAL
|
|
1176
|
+
norm_type = 'LOCAL'
|
|
1177
|
+
|
|
1178
|
+
# --- Step 10: Classify placeholder/test codes ---
|
|
1179
|
+
if code and norm_type not in ('RC', 'NDC', 'CDM', 'MS-DRG', 'APC'):
|
|
1180
|
+
stripped = code.strip()
|
|
1181
|
+
if _RE_PLACEHOLDER.match(stripped):
|
|
1182
|
+
norm_type = 'LOCAL'
|
|
1183
|
+
|
|
1184
|
+
# --- Step 11: APR-DRG qualifier stripping ---
|
|
1185
|
+
# APR-DRG codes are formatted as 'NNN-S' where NNN is 3-digit DRG and S is
|
|
1186
|
+
# a 1-digit severity (0-4). Many hospitals append qualifier text:
|
|
1187
|
+
# '001-1 Short Stay' -> '001-1'
|
|
1188
|
+
# '001-1- IP LOS Greater Than 50' -> '001-1'
|
|
1189
|
+
# '001 - 1' -> '001-1' (normalise whitespace)
|
|
1190
|
+
if code and norm_type == 'APR-DRG':
|
|
1191
|
+
code = _strip_apr_drg_qualifier(code)
|
|
1192
|
+
|
|
1193
|
+
if code:
|
|
1194
|
+
code = code.strip()
|
|
1195
|
+
|
|
1196
|
+
# --- Step 12: Residue gating → LOCAL ---
|
|
1197
|
+
# After all recovery/reclassify steps above, if the final code_type is a
|
|
1198
|
+
# controlled-vocab numeric system and the final code still fails that
|
|
1199
|
+
# system's canonical regex, route code_type to LOCAL. The code value is
|
|
1200
|
+
# preserved unchanged - this is a classification fix, not a deletion.
|
|
1201
|
+
# Examples that land here: MS-DRG '1622' (4-digit, valid range is 1-3
|
|
1202
|
+
# digits), EAPG wrong-length numeric, miscellaneous numeric hospital IDs.
|
|
1203
|
+
# Do NOT gate CPT or HCPCS: validating those needs the licensed code
|
|
1204
|
+
# lists, and shape checks already ran in steps 5-9.
|
|
1205
|
+
#
|
|
1206
|
+
# Scope: only numeric-looking codes (all-digit, or digit-dash for
|
|
1207
|
+
# APR-DRG/CMG subtypes). Non-numeric or mixed values (e.g. 'MSCODE',
|
|
1208
|
+
# 'MS-012', 'NONE' under RC) are preserved in their declared type -
|
|
1209
|
+
# they are hospital chargemaster labels that cannot be safely reclassified.
|
|
1210
|
+
if code and norm_type in _CONTROLLED_VOCAB_VALIDATORS:
|
|
1211
|
+
validator = _CONTROLLED_VOCAB_VALIDATORS[norm_type]
|
|
1212
|
+
if not validator.fullmatch(code):
|
|
1213
|
+
# Only gate codes that are purely numeric (or digit+dash, which is
|
|
1214
|
+
# the canonical form for APR-DRG/CMG). Letter-prefix or mixed codes
|
|
1215
|
+
# stay in their declared type.
|
|
1216
|
+
if re.fullmatch(r'[\d-]+', code):
|
|
1217
|
+
norm_type = 'LOCAL'
|
|
1218
|
+
|
|
1219
|
+
return code, norm_type
|
|
1220
|
+
|
|
1221
|
+
|
|
1222
|
+
# Regex used by _strip_apr_drg_qualifier: extracts the NNN-S base from the
|
|
1223
|
+
# start of an APR-DRG code, tolerating whitespace and dashes.
|
|
1224
|
+
_RE_APR_DRG_BASE = re.compile(r'^\s*(\d{3})\s*-\s*(\d)')
|
|
1225
|
+
|
|
1226
|
+
# Period-separated form: '48.4' → '048-4', '532.2' → '532-2'
|
|
1227
|
+
# Must precede _RE_APR_DRG_NNNS/base fallback to avoid '48.4' → '048'.
|
|
1228
|
+
_RE_APR_DRG_PERIOD = re.compile(r'^\s*(\d{1,3})\.(\d)\s*$')
|
|
1229
|
+
|
|
1230
|
+
# Verbose 'APRnnn SOI s' form (case-insensitive; double-space tolerated).
|
|
1231
|
+
# Example: 'APR052 SOI 1' → '052-1'.
|
|
1232
|
+
_RE_APR_DRG_VERBOSE = re.compile(r'^\s*APR(\d{3})\s+SOI\s+(\d)\s*$', re.IGNORECASE)
|
|
1233
|
+
|
|
1234
|
+
# Matches the legacy bare 4-digit NNNS form (no separator). Only applied
|
|
1235
|
+
# when no separator is present at all, so it can't misfire on codes like
|
|
1236
|
+
# '0012 Short Stay' (those are handled by _RE_APR_DRG_BASE first).
|
|
1237
|
+
_RE_APR_DRG_NNNS = re.compile(r'^\s*(\d{3})(\d)\s*$')
|
|
1238
|
+
|
|
1239
|
+
# Short leading-zero hyphen form: '48-3' → '048-3', '5-2' → '005-2'
|
|
1240
|
+
# Matches only 1-2 digit base with a single severity digit (no ambiguity with
|
|
1241
|
+
# the canonical NNN-S form handled by _RE_APR_DRG_BASE).
|
|
1242
|
+
_RE_APR_DRG_SHORT_HYPHEN = re.compile(r'^\s*(\d{1,2})-(\d)\s*$')
|
|
1243
|
+
|
|
1244
|
+
# Bare short base (no severity): '48' → '048', '5' → '005'.
|
|
1245
|
+
# Base-only without severity; passes _RE_VALID_APR_DRG which allows ^\d{1,4}(-\d{1,2})?$.
|
|
1246
|
+
_RE_APR_DRG_SHORT_BASE = re.compile(r'^\s*(\d{1,2})\s*$')
|
|
1247
|
+
|
|
1248
|
+
# Compact 'APRnnnns' form (case-insensitive): 'APR0011' → '001-1'.
|
|
1249
|
+
# DISABLED: the only known source of this form is a single file that lists
|
|
1250
|
+
# 1,340 all-distinct values, which looks more like an enumeration dump than
|
|
1251
|
+
# real APR-DRG codes. Gated behind _APR_COMPACT_ENABLED until that is
|
|
1252
|
+
# confirmed either way.
|
|
1253
|
+
_RE_APR_DRG_COMPACT = re.compile(r'^\s*APR(\d{3})(\d)\s*$', re.IGNORECASE)
|
|
1254
|
+
_APR_COMPACT_ENABLED = False
|
|
1255
|
+
|
|
1256
|
+
# Helper: zero-pad a string to 3 digits.
|
|
1257
|
+
_pad3 = lambda s: s.zfill(3)
|
|
1258
|
+
|
|
1259
|
+
|
|
1260
|
+
def _strip_apr_drg_qualifier(code: str) -> str:
|
|
1261
|
+
"""Return the canonical APR-DRG form of *code*.
|
|
1262
|
+
|
|
1263
|
+
Handles the following source formats, tested in order (ORDER IS CORRECTNESS):
|
|
1264
|
+
|
|
1265
|
+
1. 'NNN-S' with optional qualifier text - '001-1 Short Stay' → '001-1'
|
|
1266
|
+
'001 - 1' (whitespace-padded dash) → '001-1'
|
|
1267
|
+
2. Period-separated 'N.S' / 'NN.S' / 'NNN.S' - '48.4'→'048-4', '532.2'→'532-2'
|
|
1268
|
+
MUST precede bare-3-digit fallback to avoid eating the severity digit.
|
|
1269
|
+
3. Verbose 'APRnnn SOI s' - 'APR052 SOI 1' → '052-1'
|
|
1270
|
+
4. Compact 'APRnnns' (DARK/GATED) - 'APR0011' → '001-1'
|
|
1271
|
+
Disabled: see _APR_COMPACT_ENABLED.
|
|
1272
|
+
5. Bare 4-digit NNNS - '0011' → '001-1'
|
|
1273
|
+
6. Short leading-zero hyphen 'NN-S'/'N-S' - '48-3'→'048-3', '5-2'→'005-2'
|
|
1274
|
+
7. Short bare base (no severity) - '48'→'048', '5'→'005'
|
|
1275
|
+
Base-only is intentional; passes _RE_VALID_APR_DRG.
|
|
1276
|
+
8. Bare 3-digit fallback (no severity) - '139 Unspecified' → '139'
|
|
1277
|
+
|
|
1278
|
+
If none match, return the original value unchanged.
|
|
1279
|
+
"""
|
|
1280
|
+
if not code:
|
|
1281
|
+
return code
|
|
1282
|
+
|
|
1283
|
+
# 1. Canonical NNN-S (with optional qualifier / spacing).
|
|
1284
|
+
m = _RE_APR_DRG_BASE.match(code)
|
|
1285
|
+
if m:
|
|
1286
|
+
return f"{m.group(1)}-{m.group(2)}"
|
|
1287
|
+
|
|
1288
|
+
# 2. Period-separated: NNN.S (MUST be before bare-3-digit fallback).
|
|
1289
|
+
m = _RE_APR_DRG_PERIOD.match(code)
|
|
1290
|
+
if m:
|
|
1291
|
+
return f"{_pad3(m.group(1))}-{m.group(2)}"
|
|
1292
|
+
|
|
1293
|
+
# 3. Verbose 'APRnnn SOI s'.
|
|
1294
|
+
m = _RE_APR_DRG_VERBOSE.match(code)
|
|
1295
|
+
if m:
|
|
1296
|
+
return f"{m.group(1)}-{m.group(2)}"
|
|
1297
|
+
|
|
1298
|
+
# 4. Compact 'APRnnns' - DARK until precheck confirms non-enumeration source.
|
|
1299
|
+
if _APR_COMPACT_ENABLED:
|
|
1300
|
+
m = _RE_APR_DRG_COMPACT.match(code)
|
|
1301
|
+
if m:
|
|
1302
|
+
return f"{m.group(1)}-{m.group(2)}"
|
|
1303
|
+
|
|
1304
|
+
# 5. Legacy bare 4-digit NNNS form: '0011' → '001-1'.
|
|
1305
|
+
m = _RE_APR_DRG_NNNS.match(code)
|
|
1306
|
+
if m:
|
|
1307
|
+
return f"{m.group(1)}-{m.group(2)}"
|
|
1308
|
+
|
|
1309
|
+
# 6. Short leading-zero hyphen: '48-3' → '048-3', '5-2' → '005-2'.
|
|
1310
|
+
m = _RE_APR_DRG_SHORT_HYPHEN.match(code)
|
|
1311
|
+
if m:
|
|
1312
|
+
return f"{_pad3(m.group(1))}-{m.group(2)}"
|
|
1313
|
+
|
|
1314
|
+
# 7. Short bare base (no severity): '48' → '048', '5' → '005'.
|
|
1315
|
+
m = _RE_APR_DRG_SHORT_BASE.match(code)
|
|
1316
|
+
if m:
|
|
1317
|
+
return _pad3(m.group(1))
|
|
1318
|
+
|
|
1319
|
+
# 8. Last resort: extract leading 3-digit base (partial/unknown qualifier text).
|
|
1320
|
+
m = re.match(r'^\s*(\d{3})', code)
|
|
1321
|
+
if m:
|
|
1322
|
+
return m.group(1)
|
|
1323
|
+
|
|
1324
|
+
return code
|
|
1325
|
+
|
|
1326
|
+
|
|
1327
|
+
# ============================================================================
|
|
1328
|
+
# CODE REJECTION: detect corrupt rows that should be skipped
|
|
1329
|
+
# ============================================================================
|
|
1330
|
+
# After normalize_code() has done its best, some rows are still irreparably
|
|
1331
|
+
# corrupt and should be dropped. Examples seen in real files:
|
|
1332
|
+
# - code_type is a bare numeric literal like '1' or '2' (column mis-parse)
|
|
1333
|
+
# - MS-DRG rows with 'N.NNNN' code values (decimal DRG, not a real format)
|
|
1334
|
+
# Return a short machine-readable reason string, or None if the row is OK.
|
|
1335
|
+
|
|
1336
|
+
|
|
1337
|
+
_RE_NUMERIC_ONLY = re.compile(r'^\d+$')
|
|
1338
|
+
_RE_MS_DRG_DECIMAL = re.compile(r'^\d+\.\d+$')
|
|
1339
|
+
# Structurally valid code_types are short uppercase tokens composed of
|
|
1340
|
+
# letters, digits, '-' and '/' only. Real-world examples span CPT, HCPCS,
|
|
1341
|
+
# MS-DRG, APR-DRG, TRIS-DRG, CPT/HCPCS, CDM, NDC, ICD, APC, EAPG, RC, CDT,
|
|
1342
|
+
# CMG, R-DRG, LOCAL, HIPPS. We allow unknown-but-plausible tokens through
|
|
1343
|
+
# (e.g. a future 'XYZ-DRG') while rejecting obvious column-shift garbage
|
|
1344
|
+
# like stray descriptions, decimals, or sentence fragments.
|
|
1345
|
+
_RE_VALID_CODE_TYPE_SHAPE = re.compile(r'^[A-Z][A-Z0-9]*(?:[-/][A-Z0-9]+)*$')
|
|
1346
|
+
_MAX_CODE_TYPE_LEN = 16
|
|
1347
|
+
# When code_type itself matches a HCPCS/CDT Level II pattern (letter +
|
|
1348
|
+
# 4 digits), the columns have almost certainly shifted: what looks like
|
|
1349
|
+
# the 'type' is actually a code. 5-digit numeric values (e.g. a CPT
|
|
1350
|
+
# leaking into the type column) are caught separately by the numeric-only
|
|
1351
|
+
# rule. Seen in the wild: code='CDM', code_type='C1887' - where CDM is
|
|
1352
|
+
# the real type and C1887 is the real code.
|
|
1353
|
+
_RE_CODE_TYPE_LOOKS_LIKE_HCPCS = re.compile(r'^[A-Z]\d{4}$')
|
|
1354
|
+
|
|
1355
|
+
|
|
1356
|
+
def is_rejected_code(
|
|
1357
|
+
code: Optional[str],
|
|
1358
|
+
code_type: Optional[str],
|
|
1359
|
+
*,
|
|
1360
|
+
stats: Optional[ParseStats] = None,
|
|
1361
|
+
) -> Optional[str]:
|
|
1362
|
+
"""Decide whether a (code, code_type) pair is too corrupt to keep.
|
|
1363
|
+
|
|
1364
|
+
Returns a short reason string (e.g. 'numeric_code_type',
|
|
1365
|
+
'malformed_code_type', 'ms_drg_decimal') when the row must be
|
|
1366
|
+
rejected, or None when the row is acceptable. Callers should drop
|
|
1367
|
+
rejected rows rather than emit them.
|
|
1368
|
+
|
|
1369
|
+
This is intentionally conservative: only clearly broken inputs are
|
|
1370
|
+
rejected; everything borderline is accepted and classified as LOCAL by
|
|
1371
|
+
normalize_code() so a human can still review it downstream.
|
|
1372
|
+
|
|
1373
|
+
``stats``, when given, counts each rejection.
|
|
1374
|
+
"""
|
|
1375
|
+
reason = _rejection_reason(code, code_type)
|
|
1376
|
+
if reason and stats is not None:
|
|
1377
|
+
stats.record_rejected_code(code, code_type)
|
|
1378
|
+
return reason
|
|
1379
|
+
|
|
1380
|
+
|
|
1381
|
+
def _rejection_reason(
|
|
1382
|
+
code: Optional[str],
|
|
1383
|
+
code_type: Optional[str],
|
|
1384
|
+
) -> Optional[str]:
|
|
1385
|
+
# Empty rows are not "corrupt", just uninteresting - let callers decide.
|
|
1386
|
+
if not code and not code_type:
|
|
1387
|
+
return None
|
|
1388
|
+
|
|
1389
|
+
# code_type is purely numeric (e.g. '1', '2', '12') - almost always a
|
|
1390
|
+
# column mis-parse where a price or count leaked into the type column.
|
|
1391
|
+
if code_type:
|
|
1392
|
+
ct_stripped = code_type.strip()
|
|
1393
|
+
if ct_stripped and _RE_NUMERIC_ONLY.fullmatch(ct_stripped):
|
|
1394
|
+
return 'numeric_code_type'
|
|
1395
|
+
|
|
1396
|
+
# HCPCS/CDT Level II lookalike: a single uppercase letter followed by
|
|
1397
|
+
# exactly 4 digits (e.g. 'C1887', 'J0591', 'D9999') is a HCPCS code,
|
|
1398
|
+
# not a code_type. Seeing this in the code_type column means the
|
|
1399
|
+
# columns have shifted and the real type sits elsewhere.
|
|
1400
|
+
if ct_stripped and _RE_CODE_TYPE_LOOKS_LIKE_HCPCS.fullmatch(ct_stripped.upper()):
|
|
1401
|
+
return 'malformed_code_type'
|
|
1402
|
+
|
|
1403
|
+
# Structural sanity: real code_types are short token-like values.
|
|
1404
|
+
# Anything containing whitespace, decimals, or punctuation other
|
|
1405
|
+
# than '-' / '/' is almost certainly a CSV column-shift artefact
|
|
1406
|
+
# (e.g. an unescaped inch-mark in a description causes
|
|
1407
|
+
# csv.DictReader to swallow the separator and shift every
|
|
1408
|
+
# subsequent column by one). We validate the uppercased form, so
|
|
1409
|
+
# lowercase-only tokens are allowed if uppercasing yields a valid
|
|
1410
|
+
# shape, while prose like 'as remittances do not itemize...' is
|
|
1411
|
+
# rejected.
|
|
1412
|
+
if ct_stripped and len(ct_stripped) <= _MAX_CODE_TYPE_LEN:
|
|
1413
|
+
if not _RE_VALID_CODE_TYPE_SHAPE.fullmatch(ct_stripped.upper()):
|
|
1414
|
+
return 'malformed_code_type'
|
|
1415
|
+
elif ct_stripped:
|
|
1416
|
+
return 'malformed_code_type'
|
|
1417
|
+
|
|
1418
|
+
# MS-DRG with a decimal value (e.g. '0.1234') is not a valid DRG format.
|
|
1419
|
+
# Real MS-DRGs are 3-digit integers. These rows came from weird
|
|
1420
|
+
# column-swap bugs in some chargemasters.
|
|
1421
|
+
if code and code_type and code_type.upper() == 'MS-DRG':
|
|
1422
|
+
if _RE_MS_DRG_DECIMAL.fullmatch(code.strip()):
|
|
1423
|
+
return 'ms_drg_decimal'
|
|
1424
|
+
|
|
1425
|
+
return None
|
|
1426
|
+
|
|
1427
|
+
|
|
1428
|
+
def _extract_code_from_garbage(raw: str) -> Tuple[Optional[str], Optional[str]]:
|
|
1429
|
+
"""
|
|
1430
|
+
Try to extract a valid CPT/HCPCS/CDT code from a malformed code string.
|
|
1431
|
+
|
|
1432
|
+
Handles patterns like:
|
|
1433
|
+
- XJ0591X → J0591 (HCPCS wrapped in garbage)
|
|
1434
|
+
- AD9999 → D9999 (CDT dental code with A-prefix)
|
|
1435
|
+
- A29999.41 → 29999 (A-prefix CPT with decimal sub-code)
|
|
1436
|
+
- A0237T → 0237T (A-prefix Category III)
|
|
1437
|
+
|
|
1438
|
+
Returns: (extracted_code, code_type) or (None, None) if nothing found.
|
|
1439
|
+
"""
|
|
1440
|
+
# Pattern 1: A-prefix + Category III code (A0237T → 0237T)
|
|
1441
|
+
# Must check before HCPCS extraction to avoid matching the A as HCPCS prefix
|
|
1442
|
+
m = re.fullmatch(r'A(\d{4}T)', raw)
|
|
1443
|
+
if m:
|
|
1444
|
+
return m.group(1), 'HCPCS'
|
|
1445
|
+
|
|
1446
|
+
# Pattern 2: A-prefix + 5-digit CPT with optional decimal (A29999.41 → 29999)
|
|
1447
|
+
# Stanford convention: A = facility component prefix on CPT codes
|
|
1448
|
+
m = re.fullmatch(r'A(\d{5})(?:\.\d+)?', raw)
|
|
1449
|
+
if m:
|
|
1450
|
+
return m.group(1), 'CPT'
|
|
1451
|
+
|
|
1452
|
+
# Pattern 3: A-prefix + HCPCS with optional decimal (A4649.0099 → A4649)
|
|
1453
|
+
# Real HCPCS code A4649 with sub-code suffix
|
|
1454
|
+
m = re.fullmatch(r'([A-Z]\d{4})\.\d+', raw)
|
|
1455
|
+
if m:
|
|
1456
|
+
return m.group(1), 'HCPCS'
|
|
1457
|
+
|
|
1458
|
+
# Pattern 4: Embedded HCPCS/CDT code in garbage (XJ0591X → J0591, AD9999 → D9999)
|
|
1459
|
+
m = re.search(r'(?<!\d)([A-Z]\d{4})(?!\d)', raw)
|
|
1460
|
+
if m:
|
|
1461
|
+
extracted = m.group(1)
|
|
1462
|
+
# D-prefix codes are CDT (dental), all others are HCPCS Level II
|
|
1463
|
+
ext_type = 'CDT' if extracted[0] == 'D' else 'HCPCS'
|
|
1464
|
+
return extracted, ext_type
|
|
1465
|
+
|
|
1466
|
+
# Pattern 5: Embedded CPT code in garbage
|
|
1467
|
+
m = re.search(r'(?<![A-Z\d])(\d{5})(?!\d)', raw)
|
|
1468
|
+
if m:
|
|
1469
|
+
return m.group(1), 'CPT'
|
|
1470
|
+
|
|
1471
|
+
# Pattern 6: Embedded PLA code in garbage (digits + letter suffix)
|
|
1472
|
+
m = re.search(r'(?<!\d)(\d{4}[A-Z])(?![A-Z\d])', raw)
|
|
1473
|
+
if m:
|
|
1474
|
+
return m.group(1), 'CPT'
|
|
1475
|
+
|
|
1476
|
+
return None, None
|
|
1477
|
+
|
|
1478
|
+
|
|
1479
|
+
def apply_code_extraction(
|
|
1480
|
+
code: Optional[str],
|
|
1481
|
+
code_type: Optional[str],
|
|
1482
|
+
config: Optional[Dict],
|
|
1483
|
+
) -> Tuple[Optional[str], Optional[str]]:
|
|
1484
|
+
"""
|
|
1485
|
+
Extract standard codes from composite code strings based on config rules.
|
|
1486
|
+
|
|
1487
|
+
Some hospitals (e.g., UCLA) encode multiple fields into a single composite
|
|
1488
|
+
"code" value like ``RRUCLA-7616710100-1000-67101-0761-8580-Y`` where a
|
|
1489
|
+
real CPT code (67101) is embedded at a known position.
|
|
1490
|
+
|
|
1491
|
+
This function splits such composite codes and extracts the embedded
|
|
1492
|
+
standard code. The caller should then pass the result through
|
|
1493
|
+
``normalize_code()`` for classification.
|
|
1494
|
+
|
|
1495
|
+
Config format (the per-source config dict)::
|
|
1496
|
+
|
|
1497
|
+
{
|
|
1498
|
+
"code_extraction": {
|
|
1499
|
+
"pattern": "^RRUCLA", # regex - only codes matching this are processed
|
|
1500
|
+
"separator": "-", # split character
|
|
1501
|
+
"code_position": 3, # 0-indexed segment with the standard code
|
|
1502
|
+
"fallback_position": 1 # segment to use as CDM code when code_position is empty
|
|
1503
|
+
}
|
|
1504
|
+
}
|
|
1505
|
+
|
|
1506
|
+
Returns: (extracted_code, extracted_code_type)
|
|
1507
|
+
- If extraction succeeds: the embedded code with code_type=None
|
|
1508
|
+
(caller should pass through ``normalize_code()`` for classification)
|
|
1509
|
+
- If code_position is empty: (parts[fallback_position], 'CDM')
|
|
1510
|
+
- If no config or no match: (code, code_type) unchanged
|
|
1511
|
+
"""
|
|
1512
|
+
if not config or not code:
|
|
1513
|
+
return code, code_type
|
|
1514
|
+
|
|
1515
|
+
extraction = config.get('code_extraction')
|
|
1516
|
+
if not extraction:
|
|
1517
|
+
return code, code_type
|
|
1518
|
+
|
|
1519
|
+
pattern = extraction.get('pattern')
|
|
1520
|
+
if not pattern:
|
|
1521
|
+
return code, code_type
|
|
1522
|
+
|
|
1523
|
+
code_stripped = code.strip()
|
|
1524
|
+
if not re.search(pattern, code_stripped):
|
|
1525
|
+
return code, code_type
|
|
1526
|
+
|
|
1527
|
+
separator = extraction.get('separator', '-')
|
|
1528
|
+
parts = code_stripped.split(separator)
|
|
1529
|
+
pos = extraction.get('code_position', 3)
|
|
1530
|
+
|
|
1531
|
+
# Extract the standard code from the expected position
|
|
1532
|
+
if pos < len(parts) and parts[pos]:
|
|
1533
|
+
return parts[pos], None
|
|
1534
|
+
|
|
1535
|
+
# Fallback: use CDM ID from another position
|
|
1536
|
+
fb_pos = extraction.get('fallback_position', 1)
|
|
1537
|
+
if fb_pos < len(parts) and parts[fb_pos]:
|
|
1538
|
+
return parts[fb_pos], 'CDM'
|
|
1539
|
+
|
|
1540
|
+
return code, code_type
|
|
1541
|
+
|
|
1542
|
+
|
|
1543
|
+
def infer_billing_class(
|
|
1544
|
+
code: Optional[str],
|
|
1545
|
+
code_type: Optional[str],
|
|
1546
|
+
description: Optional[str],
|
|
1547
|
+
billing_class: Optional[str] = None,
|
|
1548
|
+
) -> Tuple[Optional[str], Optional[str], Optional[str]]:
|
|
1549
|
+
"""
|
|
1550
|
+
Infer billing_class and normalize code when not explicitly provided.
|
|
1551
|
+
|
|
1552
|
+
Returns: (billing_class, normalized_code, code_type)
|
|
1553
|
+
|
|
1554
|
+
Heuristic rules (applied only when billing_class is None):
|
|
1555
|
+
1. A-prefix on CPT/HCPCS codes: 'A' + a full 5-digit CPT = facility
|
|
1556
|
+
component. Strip the 'A' prefix, set billing_class = 'facility', fix
|
|
1557
|
+
code_type. Does NOT fire on 'A' + 4 digits - that shape is a genuine
|
|
1558
|
+
HCPCS Level II code, not a prefixed CPT.
|
|
1559
|
+
2. Description prefix 'PR ': professional component.
|
|
1560
|
+
3. Description prefix 'HC ' on RC items: facility component.
|
|
1561
|
+
4. Revenue codes (code_type='RC'): facility by definition.
|
|
1562
|
+
"""
|
|
1563
|
+
normalized_code = code
|
|
1564
|
+
norm_type = code_type
|
|
1565
|
+
|
|
1566
|
+
# If billing_class already set (from source data), don't override
|
|
1567
|
+
if billing_class:
|
|
1568
|
+
return billing_class, normalized_code, norm_type
|
|
1569
|
+
|
|
1570
|
+
if not code and not description:
|
|
1571
|
+
return None, normalized_code, norm_type
|
|
1572
|
+
|
|
1573
|
+
# Rule 1: A-prefix on CPT/HCPCS codes → facility, strip prefix
|
|
1574
|
+
# e.g., A19325 (HCPCS) → code=19325, code_type=CPT, billing_class=facility
|
|
1575
|
+
# e.g., A0237T (HCPCS) → code=0237T, code_type=HCPCS, billing_class=facility
|
|
1576
|
+
# The SUFFIX, not the whole code, has to be a valid CPT. A genuine HCPCS
|
|
1577
|
+
# Level II code is exactly letter + 4 digits (length 5), so the old
|
|
1578
|
+
# `len(code) >= 5` guard swallowed every A-prefixed Level II supply code:
|
|
1579
|
+
# A4322 (irrigation syringe) became CPT '4322', which is not a valid CPT
|
|
1580
|
+
# at all. Nothing re-ran normalize_code() on the result, so the corrupt
|
|
1581
|
+
# pair went straight into the output.
|
|
1582
|
+
if (code and code_type in ('CPT', 'HCPCS', None)
|
|
1583
|
+
and len(code) >= 6 and code[0] == 'A'):
|
|
1584
|
+
suffix = code[1:]
|
|
1585
|
+
if re.fullmatch(r'\d{5}', suffix):
|
|
1586
|
+
# A + 5-digit CPT → CPT (strip A, reclassify)
|
|
1587
|
+
return 'facility', suffix, 'CPT'
|
|
1588
|
+
if re.fullmatch(r'\d{4}T', suffix):
|
|
1589
|
+
# A + Category III → HCPCS (strip A, keep HCPCS)
|
|
1590
|
+
return 'facility', suffix, 'HCPCS'
|
|
1591
|
+
|
|
1592
|
+
# Rule 2: Description starts with 'PR ' → professional
|
|
1593
|
+
# Common at O'Connor, John Muir: "PR CT Thorax Diag W Con"
|
|
1594
|
+
if description and description.startswith('PR '):
|
|
1595
|
+
return 'professional', normalized_code, norm_type
|
|
1596
|
+
|
|
1597
|
+
# Rule 3: Revenue codes → facility (they are inherently facility charges)
|
|
1598
|
+
if code_type == 'RC':
|
|
1599
|
+
return 'facility', normalized_code, norm_type
|
|
1600
|
+
|
|
1601
|
+
# Rule 4: Description starts with 'HC ' on non-RC items → facility
|
|
1602
|
+
# (RC items already caught by Rule 3)
|
|
1603
|
+
if description and description.startswith('HC '):
|
|
1604
|
+
return 'facility', normalized_code, norm_type
|
|
1605
|
+
|
|
1606
|
+
return None, normalized_code, norm_type
|
|
1607
|
+
|