mrfkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mrfkit/headers.py ADDED
@@ -0,0 +1,737 @@
1
+ """Map MRF column headers to canonical fields.
2
+
3
+ Hospitals name the same column a hundred ways ("Gross Charge", "gross_price",
4
+ "IVPRICE1"). Every header is normalized (lowercase, underscores, no
5
+ punctuation) and looked up in ``SYNONYM_MAP``. Wide CMS files encode the payer
6
+ and plan inside the header (``standard_charge|Aetna|PPO|negotiated_dollar``);
7
+ ``parse_wide_payer_header`` takes those apart. A few hospital families use
8
+ their own wide layouts (Hawaii, Atrium), described by the tables below.
9
+ """
10
+
11
+ from __future__ import annotations
12
+
13
+ import re
14
+ from typing import Dict, Optional, Tuple
15
+
16
+ CANONICAL_FIELDS = {
17
+ 'hospital_name',
18
+ 'hospital_location',
19
+ 'hospital_ein',
20
+ 'cms_certification_number',
21
+ 'code',
22
+ 'code_type',
23
+ 'description',
24
+ 'modifiers',
25
+ 'gross_charge',
26
+ 'discounted_cash_price',
27
+ 'min_negotiated_rate',
28
+ 'max_negotiated_rate',
29
+ 'payer_name',
30
+ 'plan_name',
31
+ 'negotiated_rate',
32
+ 'negotiated_percentage',
33
+ 'negotiated_algorithm',
34
+ 'methodology',
35
+ 'estimated_amount',
36
+ 'billing_class',
37
+ 'setting',
38
+ 'additional_notes',
39
+ 'footnote',
40
+ 'median_amount',
41
+ 'pct_10',
42
+ 'pct_90',
43
+ 'claim_count',
44
+ 'drug_unit_of_measurement',
45
+ 'drug_type_of_measurement',
46
+ 'additional_generic_notes',
47
+ }
48
+
49
+ # Fields that belong to payer-specific rates (not the item's standard charges)
50
+ PAYER_FIELDS = {
51
+ 'payer_name', 'plan_name', 'negotiated_rate', 'negotiated_percentage',
52
+ 'negotiated_algorithm', 'methodology', 'estimated_amount',
53
+ 'additional_notes', 'footnote', 'median_amount', 'pct_10', 'pct_90', 'claim_count',
54
+ }
55
+
56
+ # Fields stored on the charge item (also passed through to each payer rate)
57
+ ITEM_LEVEL_FIELDS = {
58
+ 'billing_class', 'setting',
59
+ }
60
+
61
+
62
+ # Synonym dictionary: normalized_header -> canonical_field
63
+ SYNONYM_MAP = {
64
+ # code
65
+ 'billing_code': 'code',
66
+ 'code': 'code',
67
+ 'cpt': 'code',
68
+ 'hcpcs': 'code', # Paris (Cornerstone Regional) format
69
+ 'hcpcs_code': 'code',
70
+ 'hcpcscpt_code': 'code',
71
+ 'procedure_code': 'code',
72
+ 'procedurecode': 'code', # Hawaii hospitals (CamelCase → no underscores)
73
+ 'service_code': 'code',
74
+ 'code_value': 'code', # Ascension hospitals (CMS 3.0 variant)
75
+ 'codes': 'code', # Atrium anesthesia/BH sections (Codes, Code(s))
76
+ 'procedure_external_id': 'code', # Atrium Prof Discounted Cash Price section
77
+ 'cpthcpcs_code': 'code', # Middlesex Hospital (CPT/HCPCS Code)
78
+ 'ivcptcd': 'code', # Bear Lake Memorial chargemaster (IV-prefixed legacy headers)
79
+
80
+ # code_type
81
+ 'billing_code_type': 'code_type',
82
+ 'code_type': 'code_type',
83
+ 'coding_type': 'code_type',
84
+
85
+ # description
86
+ 'description': 'description',
87
+ 'item_description': 'description',
88
+ 'procedure_description': 'description',
89
+ 'service_description': 'description',
90
+ 'customer_description': 'description', # Ascension hospitals
91
+ 'code_descriptions': 'description', # Ascension hospitals (CMS 3.0 variant)
92
+ 'charge_description': 'description', # William Bee Ririe Hospital
93
+ 'procedure_external_id_record_name': 'description', # Atrium Prof section
94
+ 'billing_code_description': 'description', # Middlesex Hospital
95
+ 'ivdesc': 'description', # Bear Lake Memorial chargemaster
96
+
97
+ # gross_charge
98
+ 'gross_charge': 'gross_charge',
99
+ 'grosscharge': 'gross_charge',
100
+ 'standard_charge_gross': 'gross_charge',
101
+ 'gross': 'gross_charge',
102
+ 'charge_gross': 'gross_charge',
103
+ 'gross_price': 'gross_charge',
104
+ 'standard_charge': 'gross_charge', # generic standard_charge maps to gross
105
+ 'price': 'gross_charge', # Atrium Prof Discounted Cash Price section
106
+ 'gross_charge_per_cdm': 'gross_charge', # Middlesex Hospital
107
+ 'ivprice1': 'gross_charge', # Bear Lake Memorial chargemaster
108
+ 'gross_charges_estimate': 'gross_charge', # Ascension estimate files
109
+ 'gross_outpatient': 'gross_charge', # Troy Regional CMS wide variant
110
+
111
+ # discounted_cash_price
112
+ 'discounted_cash_price': 'discounted_cash_price',
113
+ 'cash_price': 'discounted_cash_price',
114
+ 'self_pay': 'discounted_cash_price',
115
+ 'selfpay': 'discounted_cash_price',
116
+ 'cash': 'discounted_cash_price',
117
+ 'discounted_price': 'discounted_cash_price',
118
+ 'standard_charge_discounted_cash': 'discounted_cash_price',
119
+ 'discounted_cash': 'discounted_cash_price',
120
+ 'discounted_cash_price_gross_charges': 'discounted_cash_price',
121
+ 'estimated_discounted_cash': 'discounted_cash_price',
122
+ 'standard_charge_estimated_discounted_cash': 'discounted_cash_price',
123
+ 'cash_price_estimate': 'discounted_cash_price', # Ascension estimate files
124
+
125
+ # min_negotiated_rate
126
+ 'min_negotiated_rate': 'min_negotiated_rate',
127
+ 'minimum_negotiated_rate': 'min_negotiated_rate',
128
+ 'minimumnegotiatedcharge': 'min_negotiated_rate', # Hawaii hospitals (CamelCase)
129
+ 'min_rate': 'min_negotiated_rate',
130
+ 'minimum': 'min_negotiated_rate',
131
+ 'standard_charge_min': 'min_negotiated_rate',
132
+ 'min': 'min_negotiated_rate',
133
+ 'min_ip_reimb': 'min_negotiated_rate', # Partners Healthcare pipe-delimited
134
+ 'min_negotiated_charge': 'min_negotiated_rate', # Atrium anesthesia section
135
+ 'minimum_reimbursement': 'min_negotiated_rate', # Atrium BH section
136
+
137
+ # max_negotiated_rate
138
+ 'max_negotiated_rate': 'max_negotiated_rate',
139
+ 'maximum_negotiated_rate': 'max_negotiated_rate',
140
+ 'maximumnegotiatedcharge': 'max_negotiated_rate', # Hawaii hospitals (CamelCase)
141
+ 'max_rate': 'max_negotiated_rate',
142
+ 'maximum': 'max_negotiated_rate',
143
+ 'standard_charge_max': 'max_negotiated_rate',
144
+ 'max': 'max_negotiated_rate',
145
+ 'max_ip_reimb': 'max_negotiated_rate', # Partners Healthcare pipe-delimited
146
+ 'max_negotiated_charge': 'max_negotiated_rate', # Atrium anesthesia section
147
+ 'maximum_reimbursement': 'max_negotiated_rate', # Atrium BH section
148
+
149
+ # hospital_name
150
+ 'hospital_name': 'hospital_name',
151
+ 'facility_name': 'hospital_name',
152
+ 'facility': 'hospital_name',
153
+
154
+ # hospital_location
155
+ 'hospital_location': 'hospital_location',
156
+ 'facility_location': 'hospital_location',
157
+ 'address': 'hospital_location',
158
+
159
+ # hospital_ein
160
+ 'ein': 'hospital_ein',
161
+ 'hospital_ein': 'hospital_ein',
162
+ 'tax_id': 'hospital_ein',
163
+
164
+ # cms_certification_number
165
+ 'cc_number': 'cms_certification_number',
166
+ 'cms_certification_number': 'cms_certification_number',
167
+ 'cms_id': 'cms_certification_number',
168
+ 'medicare_provider_number': 'cms_certification_number',
169
+
170
+ # payer_name
171
+ 'payer_name': 'payer_name',
172
+ 'payer': 'payer_name',
173
+ 'insurance_name': 'payer_name',
174
+ 'insurance_carrier': 'payer_name', # Ascension hospitals
175
+
176
+ # plan_name
177
+ 'plan_name': 'plan_name',
178
+ 'plan': 'plan_name',
179
+ 'plans': 'plan_name', # Plan(s) header → plans after normalization
180
+ 'insurance_plan': 'plan_name',
181
+ 'product': 'plan_name', # Atrium Prof section uses Product for plan
182
+
183
+ # negotiated_rate (dollar amount)
184
+ 'standard_charge_negotiated_dollar': 'negotiated_rate',
185
+ 'negotiated_dollar': 'negotiated_rate',
186
+ 'negotiated_rate': 'negotiated_rate',
187
+ 'ip_price': 'negotiated_rate', # Partners Healthcare pipe-delimited format
188
+
189
+ # negotiated_percentage (CVHFullCDMPriceTransparency ships the singular
190
+ # 'percent' form with spaces around the pipe; both variants must route
191
+ # to the same canonical field).
192
+ 'standard_charge_negotiated_percentage': 'negotiated_percentage',
193
+ 'standard_charge_negotiated_percent': 'negotiated_percentage',
194
+ 'negotiated_percentage': 'negotiated_percentage',
195
+ 'negotiated_percent': 'negotiated_percentage',
196
+
197
+ # negotiated_algorithm
198
+ 'standard_charge_negotiated_algorithm': 'negotiated_algorithm',
199
+ 'negotiated_algorithm': 'negotiated_algorithm',
200
+
201
+ # methodology
202
+ 'standard_charge_methodology': 'methodology',
203
+ 'methodology': 'methodology',
204
+ 'rate_methodology': 'methodology', # Middlesex Hospital
205
+
206
+ # estimated_amount
207
+ 'estimated_amount': 'estimated_amount',
208
+ 'estimate_amount': 'estimated_amount',
209
+ 'insurance_estimate': 'estimated_amount',
210
+ 'ip_expected_reimbursement': 'estimated_amount', # Partners Healthcare pipe-delimited
211
+
212
+ # billing_class
213
+ 'billing_class': 'billing_class',
214
+
215
+ # setting
216
+ 'setting': 'setting',
217
+
218
+ # modifiers
219
+ 'modifiers': 'modifiers',
220
+ 'modifier': 'modifiers',
221
+ 'modifier_code': 'modifiers',
222
+ 'procedure_modifier': 'modifiers', # Atrium Prof Discounted Cash Price
223
+
224
+ # drug fields
225
+ 'drug_unit_of_measurement': 'drug_unit_of_measurement',
226
+ 'drug_unit': 'drug_unit_of_measurement',
227
+ 'drug_type_of_measurement': 'drug_type_of_measurement',
228
+ 'drug_type': 'drug_type_of_measurement',
229
+
230
+ # additional_notes (payer-level notes)
231
+ 'additional_notes': 'additional_notes',
232
+ 'additional_payer_notes': 'additional_notes', # University Hospitals
233
+
234
+ # additional_generic_notes (item-level notes, distinct from payer-level)
235
+ 'additional_generic_notes': 'additional_generic_notes',
236
+ 'additional_generic_notes_with_lpp': 'additional_generic_notes',
237
+
238
+ # footnote (pricing logic: bundling rules, lesser-of provisions, etc.)
239
+ 'footnote': 'footnote',
240
+ 'pricing_footnote': 'footnote',
241
+ 'rate_footnote': 'footnote',
242
+
243
+ # statistical fields (bare / tall-format variants)
244
+ 'median_amount': 'median_amount',
245
+ 'median_allowed_amount': 'median_amount', # Valley Medical Center
246
+ '10th_percentile': 'pct_10',
247
+ 'tenth_percentile_allowed_amount': 'pct_10', # Valley Medical Center
248
+ '90th_percentile': 'pct_90',
249
+ 'ninetieth_percentile_allowed_amount': 'pct_90', # Valley Medical Center
250
+ 'count': 'claim_count',
251
+ 'allowed_amount_count': 'claim_count', # Valley Medical Center
252
+ 'count_of_compared_rates': 'claim_count', # Cleveland Clinic + others
253
+ }
254
+
255
+
256
+ def normalize_header(header: str) -> str:
257
+ """
258
+ Tier 1 normalization:
259
+ - lowercase
260
+ - trim whitespace
261
+ - strip surrounding quote characters (single, double, smart quotes)
262
+ that some publishers leak into header rows
263
+ - replace spaces and dashes with underscores
264
+ - remove punctuation except underscores
265
+ - collapse multiple underscores
266
+ """
267
+ h = header.strip().lower()
268
+ # Strip surrounding quotes (straight + smart). A few MRF publishers emit
269
+ # CSVs with literal quotes baked into header names, e.g.
270
+ # `"standard_charge|gross"` instead of `standard_charge|gross`. Without
271
+ # this strip, the pipe-pattern matcher in map_header() sees parts[0] as
272
+ # `"standard_charge` and fails to match.
273
+ h = h.strip('"\'\u201c\u201d\u2018\u2019')
274
+ h = re.sub(r'[\s\-]+', '_', h)
275
+ h = re.sub(r'[^\w_]', '', h)
276
+ h = re.sub(r'_+', '_', h)
277
+ h = h.strip('_')
278
+ return h
279
+
280
+
281
+
282
+ def map_header(
283
+ source_header: str,
284
+ *,
285
+ extra_synonyms: Optional[Dict[str, str]] = None,
286
+ ) -> Tuple[str, Optional[str]]:
287
+ """
288
+ Map source_header to canonical field via Tier 1 synonym mapping.
289
+ Handles piped patterns like 'code|1', 'code|1|type', 'standard_charge|gross'.
290
+ Returns: (normalized_header, mapped_field or None)
291
+
292
+ ``extra_synonyms`` maps normalized headers to canonical fields. It is
293
+ consulted only when the built-in table finds nothing, and a target that
294
+ is not in ``CANONICAL_FIELDS`` is ignored.
295
+ """
296
+ normalized = normalize_header(source_header)
297
+
298
+ # First try exact match
299
+ mapped_field = SYNONYM_MAP.get(normalized)
300
+
301
+ # Strip surrounding quotes for the pipe-pattern matcher below - some MRFs
302
+ # ship headers like `"standard_charge|gross"` (literal quotes included).
303
+ # normalize_header() already strips them; do the same here so split()
304
+ # produces clean parts.
305
+ raw_for_pipes = source_header.strip().strip('"\'\u201c\u201d\u2018\u2019')
306
+
307
+ # If no match and contains pipes, try pattern matching
308
+ if not mapped_field and '|' in raw_for_pipes:
309
+ parts = [p.strip() for p in raw_for_pipes.split('|')]
310
+
311
+ # Pattern 1: code|1|type -> code_type
312
+ if len(parts) >= 3 and 'type' in parts[-1].lower() and parts[0].lower() == 'code':
313
+ combined = f"{parts[0]}_{parts[-1]}"
314
+ combined_normalized = normalize_header(combined)
315
+ mapped_field = SYNONYM_MAP.get(combined_normalized)
316
+
317
+ # Pattern 2: standard_charge|gross, standard_charge|discounted_cash, etc.
318
+ # Try combining first + last for 2-part patterns only
319
+ elif not mapped_field and len(parts) == 2 and parts[0].lower() == 'standard_charge':
320
+ # Try last part alone first
321
+ last_normalized = normalize_header(parts[-1])
322
+ mapped_field = SYNONYM_MAP.get(last_normalized)
323
+
324
+ # If not found, try combining
325
+ if not mapped_field:
326
+ combined = f"{parts[0]}_{parts[-1]}"
327
+ combined_normalized = normalize_header(combined)
328
+ mapped_field = SYNONYM_MAP.get(combined_normalized)
329
+
330
+ # Pattern 3: code|1 -> code (base pattern)
331
+ elif not mapped_field and parts[0].lower() in ('code', ):
332
+ base_normalized = normalize_header(parts[0])
333
+ mapped_field = SYNONYM_MAP.get(base_normalized)
334
+
335
+ if not mapped_field and extra_synonyms:
336
+ target = extra_synonyms.get(normalized)
337
+ if target in CANONICAL_FIELDS:
338
+ mapped_field = target
339
+
340
+ return normalized, mapped_field
341
+
342
+
343
+ # Rate-type suffixes in CMS wide CSV compound headers
344
+ _WIDE_RATE_SUFFIXES = {
345
+ 'negotiated_dollar': 'negotiated_rate',
346
+ 'negotiated_rate_dollar': 'negotiated_rate',
347
+ 'negotiated_percentage': 'negotiated_percentage',
348
+ # Singular variant some publishers ship (e.g. compound forms of
349
+ # `standard_charge|PAYER|negotiated_percent`). Routes to the same
350
+ # canonical as `negotiated_percentage` so wide-payer CSVs adopting the
351
+ # singular spelling don't silently drop their per-payer rate.
352
+ 'negotiated_percent': 'negotiated_percentage',
353
+ 'negotiated_rate_percentage': 'negotiated_percentage',
354
+ 'negotiated_rate_percent': 'negotiated_percentage',
355
+ 'negotiated_algorithm': 'negotiated_algorithm',
356
+ 'negotiated_rate_algorithm': 'negotiated_algorithm',
357
+ 'methodology': 'methodology',
358
+ }
359
+
360
+ # Prefix variants accepted by parse_wide_payer_header(); keep in sync with its
361
+ # docstring and golden tests in tests/golden/parse_wide_payer_header.json.
362
+ _WIDE_STANDARD_CHARGE_PREFIXES = {
363
+ 'standard_charge',
364
+ 'standard_charges',
365
+ 'standard_charged',
366
+ }
367
+
368
+ _WIDE_ESTIMATED_AMOUNT_PREFIXES = {
369
+ 'estimated_amount',
370
+ 'estimate_amount',
371
+ }
372
+
373
+
374
+ def _wide_payer_plan(parts, min_len: int = 2):
375
+ if len(parts) < min_len or not parts[1].strip():
376
+ return None
377
+ payer = parts[1].strip()
378
+ plan = ' - '.join(p.strip() for p in parts[2:] if p.strip())
379
+ return payer, plan
380
+
381
+
382
+ def parse_wide_payer_header(source_header: str):
383
+ """
384
+ Parse a CMS "wide" CSV compound header into (payer, plan, rate_field).
385
+
386
+ Patterns handled (N = number of pipe-separated parts):
387
+
388
+ standard_charge|PAYER|rate_suffix (3 parts, plan is empty string)
389
+ standard_charge|PAYER|PLAN|rate_suffix (4 parts)
390
+ standard_charge|PAYER|PLAN|PRODUCT|rate_suffix (5 parts, plan becomes "PLAN - PRODUCT")
391
+
392
+ estimated_amount|PAYER|PLAN (3 parts)
393
+ estimated_amount|PAYER|PLAN|PRODUCT (4 parts, plan becomes "PLAN - PRODUCT")
394
+
395
+ additional_payer_notes|PAYER|PLAN (3 parts)
396
+ additional_payer_notes|PAYER|PLAN|PRODUCT (4 parts, plan becomes "PLAN - PRODUCT")
397
+
398
+ metric|PAYER|PLAN (3 parts, metric = median_amount/10th_percentile/90th_percentile/count)
399
+ metric|PAYER|PLAN|PRODUCT (4 parts, plan becomes "PLAN - PRODUCT")
400
+
401
+ When an extra PRODUCT segment is present (e.g. "MEDICARE"), it is appended
402
+ to the plan name with " - " separator so that payer rates are grouped
403
+ correctly without losing the product-line information.
404
+
405
+ Returns (payer, plan, canonical_field) or None if not a compound payer header.
406
+ """
407
+ if '|' not in source_header:
408
+ return None
409
+
410
+ # Strip surrounding quote characters that some publishers leave embedded
411
+ # in CSV headers (e.g. `"standard_charge|PAYER|gross"`). Without this the
412
+ # split below produces a leading `"standard_charge` part that no prefix
413
+ # check matches.
414
+ cleaned = source_header.strip().strip('"\'\u201c\u201d\u2018\u2019')
415
+ parts = [p.strip() for p in cleaned.split('|')]
416
+
417
+ # Summit BHC West Virginia uses a leading empty segment after the prefix:
418
+ # standard_charge||PAYER|PLAN|PRODUCT|suffix (6 parts)
419
+ # additional_payer_notes||PAYER|PLAN (4 parts)
420
+ # estimated_amount||PAYER|PLAN|PRODUCT (5 parts)
421
+ # Collapse the leading empty slot so the rest of this function sees the
422
+ # canonical 3/4/5-part shape.
423
+ if len(parts) >= 3 and parts[1] == '':
424
+ parts = [parts[0]] + parts[2:]
425
+
426
+ prefix = normalize_header(parts[0])
427
+
428
+ # standard_charge|PAYER|PLAN|rate_suffix (4 parts)
429
+ # standard_charge|PAYER|PLAN|PRODUCT|rate_suffix (5 parts)
430
+ # standard_charge|PAYER|rate_suffix (3 parts, no plan - e.g. Triwest)
431
+ # Some publishers use standard_charges/standard_charged or split payer/plan
432
+ # names across extra pipe segments; preserve the first segment as payer and
433
+ # fold the remaining middle segments into the plan name.
434
+ if prefix in _WIDE_STANDARD_CHARGE_PREFIXES and len(parts) >= 3:
435
+ suffix = normalize_header(parts[-1])
436
+ canonical = _WIDE_RATE_SUFFIXES.get(suffix)
437
+ if canonical:
438
+ parsed = _wide_payer_plan(parts[:-1])
439
+ if parsed:
440
+ payer, plan = parsed
441
+ return (payer, plan, canonical)
442
+
443
+ # estimated_amount|PAYER (2 parts, no plan - e.g. Triwest)
444
+ # estimated_amount|PAYER|PLAN (3 parts)
445
+ # estimated_amount|PAYER|PLAN|PRODUCT (4 parts)
446
+ if prefix in _WIDE_ESTIMATED_AMOUNT_PREFIXES and len(parts) >= 2:
447
+ parsed = _wide_payer_plan(parts)
448
+ if parsed:
449
+ payer, plan = parsed
450
+ return (payer, plan, 'estimated_amount')
451
+
452
+ # additional_payer_notes|PAYER (2 parts, no plan - e.g. Triwest)
453
+ # additional_payer_notes|PAYER|PLAN (3 parts)
454
+ # additional_payer_notes|PAYER|PLAN|PRODUCT (4 parts)
455
+ if prefix == 'additional_payer_notes' and len(parts) >= 2:
456
+ parsed = _wide_payer_plan(parts)
457
+ if parsed:
458
+ payer, plan = parsed
459
+ return (payer, plan, 'additional_notes')
460
+
461
+ # CMS statistical fields: metric|PAYER (2 parts, no plan - e.g. Triwest)
462
+ # CMS statistical fields: metric|PAYER|PLAN (3 parts)
463
+ # CMS statistical fields: metric|PAYER|PLAN|PRODUCT (4 parts)
464
+ _WIDE_STAT_PREFIXES = {
465
+ 'median_amount': 'median_amount',
466
+ '10th_percentile': 'pct_10',
467
+ '90th_percentile': 'pct_90',
468
+ 'count': 'claim_count',
469
+ }
470
+ if prefix in _WIDE_STAT_PREFIXES and len(parts) >= 2:
471
+ parsed = _wide_payer_plan(parts)
472
+ if parsed:
473
+ payer, plan = parsed
474
+ return (payer, plan, _WIDE_STAT_PREFIXES[prefix])
475
+
476
+ return None
477
+
478
+
479
+ # ---- Hawaii wide format: setting-specific columns ----
480
+ # Maps normalized column names to (canonical_field, setting) pairs.
481
+ # Each row becomes up to 3 charge items (one per setting).
482
+ HAWAII_SETTING_COLUMNS: Dict[str, Tuple[str, str]] = {
483
+ 'inpatientgrosscharge': ('gross_charge', 'inpatient'),
484
+ 'outpatientgrosscharge': ('gross_charge', 'outpatient'),
485
+ 'emergencyroomgrosscharge': ('gross_charge', 'emergency'),
486
+ 'discountedcashpriceinpatient': ('discounted_cash_price', 'inpatient'),
487
+ 'discountedcashpriceoutpatient': ('discounted_cash_price', 'outpatient'),
488
+ 'discountedcashpriceemergencyroom': ('discounted_cash_price', 'emergency'),
489
+ }
490
+
491
+ # The three setting-specific gross charge columns that signal the Hawaii format
492
+ _HAWAII_DETECT_COLUMNS = frozenset({
493
+ 'inpatientgrosscharge',
494
+ 'outpatientgrosscharge',
495
+ 'emergencyroomgrosscharge',
496
+ })
497
+
498
+ # Setting tokens used in Hawaii payer column headers
499
+ _HAWAII_SETTINGS = {
500
+ 'Inpatient': 'inpatient',
501
+ 'Outpatient': 'outpatient',
502
+ 'EmergencyRoom': 'emergency',
503
+ }
504
+
505
+ # Methodology tokens used in Hawaii payer column headers
506
+ _HAWAII_METHODOLOGIES = {
507
+ 'PercentofCharges': 'percent_of_charge',
508
+ 'PerDiem': 'per_diem',
509
+ 'FeeSchedule': 'fee_schedule',
510
+ 'CaseRate': 'case_rate',
511
+ }
512
+
513
+
514
+ def parse_hawaii_payer_header(
515
+ header: str,
516
+ ) -> Optional[Tuple[str, str, str]]:
517
+ """
518
+ Parse a Hawaii-format payer column header into (payer_name, setting, methodology).
519
+
520
+ Hawaii hospitals encode payer/setting/methodology as a single CamelCase
521
+ column header with underscores separating the three parts:
522
+
523
+ {PayerPlan}_{Setting}_{Methodology}
524
+ {PayerPlan}_{Setting}_{Methodology}_{N}
525
+
526
+ Examples:
527
+ QuestIntegrationHMSA-ABDPlans_Outpatient_PercentofCharges
528
+ → ("QuestIntegrationHMSA-ABDPlans", "outpatient", "percent_of_charge")
529
+
530
+ OhanaCCSSMI-BehavioralHealthPlans_EmergencyRoom_FeeSchedule
531
+ → ("OhanaCCSSMI-BehavioralHealthPlans", "emergency", "fee_schedule")
532
+
533
+ QuestIntegrationAlohacare-ABD&NONABDPlansPlans_Inpatient_PerDiem_3
534
+ → ("QuestIntegrationAlohacare-ABD&NONABDPlansPlans", "inpatient", "per_diem")
535
+
536
+ Returns (payer_name, setting, methodology) or None if not a match.
537
+ """
538
+ parts = header.split('_')
539
+ if len(parts) < 3:
540
+ return None
541
+
542
+ # Payer names may contain underscores too (rare), but after an optional
543
+ # trailing numeric suffix we assume the final token is the methodology
544
+ # and the token immediately before it is the setting. Everything to the
545
+ # left is treated as the payer name.
546
+
547
+ # Strip trailing numeric suffix (e.g. _3) used for duplicate columns
548
+ # before identifying methodology/setting tokens.
549
+ if parts[-1].isdigit():
550
+ parts = parts[:-1]
551
+ if len(parts) < 3:
552
+ return None
553
+
554
+ methodology_token = parts[-1]
555
+ methodology = _HAWAII_METHODOLOGIES.get(methodology_token)
556
+ if not methodology:
557
+ return None
558
+
559
+ setting_token = parts[-2]
560
+ setting = _HAWAII_SETTINGS.get(setting_token)
561
+ if not setting:
562
+ return None
563
+
564
+ payer = '_'.join(parts[:-2])
565
+ if not payer:
566
+ return None
567
+
568
+ return (payer, setting, methodology)
569
+
570
+
571
+ # ---- Atrium wide format: setting-specific columns with underscore-separated names ----
572
+ # Atrium Health hospitals use a flat JSON/CSV format where each row has
573
+ # setting-specific charge columns:
574
+ # " Inpatient Gross Charge " / " Outpatient Gross Charge "
575
+ # " Inpatient Negotiated Charge " / " Outpatient Negotiated Charge "
576
+ # " Inpatient Discounted Charge " / " Outpatient Discounted Charge "
577
+ # " Gross Charge - Facility " / " Gross Charge - Non-Facility "
578
+ # " Negotiated Charge - Facility " / " Negotiated Charge - Non-Facility "
579
+ # A "Min /Max" column indicates whether negotiated charges are min or max.
580
+ # Payer/Plan are in separate tall-format columns per row.
581
+ # Each row becomes up to 2 charge items (one per setting).
582
+ #
583
+ # This differs from Hawaii because:
584
+ # 1. Column names normalize with underscores (inpatient_gross_charge)
585
+ # 2. Payer/plan are in row-level columns (tall format), not in column headers
586
+ # 3. A Min/Max column determines which rate field gets the negotiated charge
587
+ # 4. Professional sections use Facility/Non-Facility instead of Inpatient/Outpatient
588
+
589
+ # Maps normalized column name → (canonical_field, setting)
590
+ ATRIUM_SETTING_COLUMNS: Dict[str, Tuple[str, str]] = {
591
+ # Hospital sections: Inpatient/Outpatient
592
+ 'inpatient_gross_charge': ('gross_charge', 'inpatient'),
593
+ 'outpatient_gross_charge': ('gross_charge', 'outpatient'),
594
+ 'inpatient_negotiated_charge': ('negotiated_charge', 'inpatient'),
595
+ 'outpatient_negotiated_charge': ('negotiated_charge', 'outpatient'),
596
+ 'inpatient_discounted_charge': ('discounted_cash_price', 'inpatient'),
597
+ 'outpatient_discounted_charge': ('discounted_cash_price', 'outpatient'),
598
+ # Professional sections: Facility/Non-Facility
599
+ # CMS MPFS defines facility fees as rendered in a hospital/ASC (≈inpatient)
600
+ # and non-facility fees as rendered in an office (≈outpatient), so they
601
+ # map onto the standard inpatient/outpatient settings.
602
+ 'gross_charge_facility': ('gross_charge', 'inpatient'),
603
+ 'gross_charge_non_facility': ('gross_charge', 'outpatient'),
604
+ 'negotiated_charge_facility': ('negotiated_charge', 'inpatient'),
605
+ 'negotiated_charge_non_facility': ('negotiated_charge', 'outpatient'),
606
+ }
607
+
608
+ # Minimum columns that signal the Atrium format - the two hospital gross
609
+ # charge columns are always present in Atrium files.
610
+ _ATRIUM_DETECT_COLUMNS = frozenset({
611
+ 'inpatient_gross_charge',
612
+ 'outpatient_gross_charge',
613
+ })
614
+
615
+
616
+ # ---------------------------------------------------------------------------
617
+ # Low-value unmapped header suppression (shared between CSV and JSON paths)
618
+ # ---------------------------------------------------------------------------
619
+ # Headers that appear in source files but have no canonical target and
620
+ # produce large volumes of unmapped cells without analytical value.
621
+ # Keeping these in one place ensures CSV and JSON readers stay in sync.
622
+ _SUPPRESS_UNMAPPED_HEADERS = {
623
+ # outpatient pricing handled by fallback columns
624
+ 'op_price', 'min_op_reimb', 'max_op_reimb', 'op_expected_reimbursement',
625
+ # pricing detail / cross-reference blobs
626
+ 'ip_pricing_detail', 'op_pricing_detail', 'ip_xr_detail', 'op_xr_detail',
627
+ # Partners-specific row metadata
628
+ 'facility_id', 'contract', 'procedure', 'quantity',
629
+ # code identifiers stored elsewhere or not yet canonical
630
+ 'rev_code', 'rev', 'ndc',
631
+ # internal hospital IDs / mnemonic codes (no analytical value)
632
+ 'cdm', 'procedure_id', 'mnemonic', 'neumonic',
633
+ # department / cost-center labels (no canonical field)
634
+ 'department',
635
+ # garbage / separator columns (including headers that normalize to empty)
636
+ 'd', '',
637
+ # section / category labels (no analytical value)
638
+ 'tabname',
639
+ # KU Great Bend billing-specific min/max columns (nearly all NULL,
640
+ # duplicates data already captured in standard_charge|min/max)
641
+ 'hospitalbilling_inpatient_min_price',
642
+ 'hospitalbilling_inpatient_max_price',
643
+ 'hospitalbilling_outpatient_min_price',
644
+ 'hospitalbilling_outpatient_max_price',
645
+ 'professionalbilling_inpatient_min_price',
646
+ 'professionalbilling_inpatient_max_price',
647
+ 'professionalbilling_outpatient_min_price',
648
+ 'professionalbilling_outpatient_max_price',
649
+ # Middlesex internal classification / fiscal columns
650
+ 'revenue_code_description', 'fscrptcat3name', 'fscname',
651
+ 'fsccategory', 'script_type', 'service_area',
652
+ # count / statistics metadata (mapped to claim_count in SYNONYM_MAP,
653
+ # kept here as safety net for edge cases)
654
+ 'count_of_compared_rates',
655
+ # Kennedy Krieger numbered notes (internal charge IDs / dept names)
656
+ 'additional_generic_notes1', 'additional_generic_notes2',
657
+ # Alameda / Philip Health internal pricing/metadata columns
658
+ # (kept individually for clarity; *_internal suffix regex below also
659
+ # catches these and future variants)
660
+ 'lpp_id_internal', 'algorithm_factors_internal', 'contract_internal',
661
+ 'count_internal', 'location_internal', 'pricing_detail_internal',
662
+ 'xr_detail_internal',
663
+ # revenue_code variant (rev_code already suppressed above)
664
+ 'revenue_code',
665
+ # CMS JSON metadata fields (allowed_amounts is a nested structure
666
+ # with no single canonical target; code type version is a string)
667
+ 'allowed_amounts', 'billing_code_type_version',
668
+ # timestamp / audit metadata (no canonical field, no analytical value)
669
+ 'last_updated',
670
+ # Secondary code columns. A charge item carries one code; a second code
671
+ # (typically a rev code paired with a CPT) is duplicated in the row's
672
+ # primary code/code_type columns or appears on a sibling row. Suppress
673
+ # to avoid millions of unmapped cells from Ascension files.
674
+ 'code_2', 'code_2_type', 'code_3', 'code_3_type',
675
+ 'code_4', 'code_4_type',
676
+ # Hospital-specific bare-token header columns that publishers leave in
677
+ # CSVs (e.g. SURGCTR, WESTLAKE - surgery-center / facility tags). They
678
+ # have no canonical target and produce huge volumes of unmapped rows.
679
+ 'surgctr', 'westlake',
680
+ # Per-payer free-text note columns shipped by publishers as a single
681
+ # rolled-up column. There is no per-payer notes field, and mapping to
682
+ # additional_notes would conflict with the canonical generic-notes
683
+ # column. Suppress.
684
+ 'additional_payer_specific_notes',
685
+ # Source-local identifiers and ancillary timing fields without a canonical
686
+ # price target.
687
+ 'charge_number', 'ivnum', 'mins_per_unit', 'financial_aid_policy',
688
+ # Internal item-id columns published as `Item#`, `Item #`,
689
+ # `Item No`, `Item No.` - they all normalize to `item` / `item_no`
690
+ # (Prowers Medical Center ships ~27K cells per ingest).
691
+ 'item', 'item_no',
692
+ # Rolled-up per-payer text notes column. Same family as
693
+ # `additional_payer_specific_notes` already suppressed above; some
694
+ # publishers spell the suffix `payor_reimbursement` instead.
695
+ 'additional_generic_notes_payor_reimbursement',
696
+ # Internal item-id / department metadata columns with no canonical target.
697
+ # `Item ID` (normalizes to `item_id`, a sibling of `item`/`item_no` above),
698
+ # `Dept #` (`dept`) and `Dept Name` (`dept_name`) - department cost-center
699
+ # labels, same family as `department` suppressed above. Observed shipping
700
+ # tens of thousands of unmapped cells each (Item ID ~35K, Dept #/Dept Name
701
+ # ~24K each). Value/price columns from the same files (Average Charge Per
702
+ # Unit, Amount, possible_amount) are deliberately NOT suppressed here - they
703
+ # carry real prices and are tracked as synonym candidates instead.
704
+ 'item_id', 'dept', 'dept_name',
705
+ }
706
+
707
+ # Regex patterns for low-value columns that match a family of header names.
708
+ # Used in addition to the exact-match set above. Each pattern is matched
709
+ # against the normalized header (lowercase, underscores only).
710
+ _SUPPRESS_UNMAPPED_PATTERNS = [
711
+ # Alameda / Philip Health internal columns: anything_internal
712
+ re.compile(r'^[a-z0-9_]+_internal$'),
713
+ # Generic unnamed CSV columns: column1 .. column999 (Watsonville, etc.)
714
+ re.compile(r'^column\d+$'),
715
+ # Bare numeric headers - data row mis-read as header (WakeMed Cary,
716
+ # Evanston, Swedish Covenant). Suppressing avoids millions of
717
+ # unmapped cells per file.
718
+ re.compile(r'^\d+$'),
719
+ # Bracket-form secondary code columns. `code[2]` etc
720
+ # normalize to `code2` (the brackets are stripped, NO underscore is
721
+ # inserted), so the existing exact-match entries for `code_2..code_4`
722
+ # / `code_2_type..code_4_type` miss the bracket variants. Cover
723
+ # N >= 2 only - `code[1]` / `code1` could be a primary code column
724
+ # and a publisher labelling it that way would be a real mapping
725
+ # problem we want to see in the unmapped cells, not silence here.
726
+ re.compile(r'^code(?:[2-9]|[1-9]\d)(type)?$'),
727
+ ]
728
+
729
+
730
+ def _is_unmapped_header_suppressed(normalized: str) -> bool:
731
+ """Return True if a normalized header should be excluded from
732
+ unmapped-cell tracking. Used by both CSV and JSON readers.
733
+ """
734
+ if normalized in _SUPPRESS_UNMAPPED_HEADERS:
735
+ return True
736
+ return any(p.match(normalized) for p in _SUPPRESS_UNMAPPED_PATTERNS)
737
+