mrfkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mrfkit/json_reader.py ADDED
@@ -0,0 +1,603 @@
1
+ """Read hospital MRF JSON files into records.
2
+
3
+ Two shapes exist. The CMS template nests everything: an item has
4
+ ``code_information``, ``standard_charges`` and, inside those,
5
+ ``payers_information``. Older and vendor files are flat: each item is a row
6
+ of key/value pairs, read exactly like a CSV row by
7
+ :class:`mrfkit.tabular.TableLayout`.
8
+ """
9
+
10
+ from __future__ import annotations
11
+
12
+ import itertools
13
+ import logging
14
+ from pathlib import Path
15
+ from typing import Any, Dict, Iterable, Iterator, List, Mapping, Optional, Tuple, Union
16
+
17
+ import ijson
18
+
19
+ from .codes import merge_modifier_into_field, normalize_code
20
+ from .files import detect_file_format, open_json
21
+ from .payers import is_self_pay_payer, normalize_payer_name
22
+ from .records import FileMetadata, HeaderMapping, ModifierInfo, StandardCharge
23
+ from .reference import ParseStats, ReferenceData
24
+ from .tabular import PEEK_ROWS, RecordBuilder, TableLayout
25
+ from .values import _STATE_NAME_TO_ABBR, _US_STATES, _safe_count, _safe_float, _sanitize_npi
26
+
27
+ log = logging.getLogger(__name__)
28
+
29
+ # Root keys that hold the item array, most likely first.
30
+ DATA_KEYS = ('standard_charge_information', 'items', 'data')
31
+
32
+ # Root arrays that are metadata, never charge data.
33
+ METADATA_ARRAY_KEYS = frozenset({
34
+ 'hospital_location', 'hospital_address', 'financial_aid_policy',
35
+ 'license_information', 'affirmation', 'modifier_information',
36
+ 'general_contract_provisions', 'location_name', 'type_2_npi',
37
+ })
38
+
39
+ _PREFIX_SNIFF_BYTES = 65536
40
+
41
+ # How the CMS nested fields are read, for HeaderMapping records.
42
+ _CMS_HEADER_MAPPINGS = [
43
+ ('description', 'description', 'description'),
44
+ ('code_information.code', 'code', 'code'),
45
+ ('code_information.type', 'code_type', 'code_type'),
46
+ ('standard_charges.gross_charge', 'gross_charge', 'gross_charge'),
47
+ ('standard_charges.discounted_cash', 'discounted_cash_price', 'discounted_cash_price'),
48
+ ('standard_charges.minimum', 'min_negotiated_rate', 'min_negotiated_rate'),
49
+ ('standard_charges.maximum', 'max_negotiated_rate', 'max_negotiated_rate'),
50
+ ('standard_charges.billing_class', 'billing_class', 'billing_class'),
51
+ ('standard_charges.setting', 'setting', 'setting'),
52
+ ('standard_charges.modifiers', 'modifiers', 'modifiers'),
53
+ ('drug_information.unit', 'drug_unit_of_measurement', 'drug_unit_of_measurement'),
54
+ ('drug_information.type', 'drug_type_of_measurement', 'drug_type_of_measurement'),
55
+ ('standard_charges.payers_information', 'payers_information', 'wide_payer:cms_nested'),
56
+ ]
57
+
58
+ # CMS JSON field names for each canonical field. Several spellings exist
59
+ # across template versions; the first non-null one wins.
60
+ _JSON_FIELD_MAP = {
61
+ # standard_charges level
62
+ 'gross_charge': 'gross_charge',
63
+ 'discounted_cash': 'discounted_cash_price',
64
+ 'discounted_cash_price': 'discounted_cash_price',
65
+ 'minimum': 'min_negotiated_rate',
66
+ 'min': 'min_negotiated_rate',
67
+ 'maximum': 'max_negotiated_rate',
68
+ 'max': 'max_negotiated_rate',
69
+ 'billing_class': 'billing_class',
70
+ 'setting': 'setting',
71
+ 'modifiers': 'modifiers',
72
+ 'modifier_code': 'modifiers', # CMS 3.0
73
+ 'additional_generic_notes': 'additional_generic_notes',
74
+ 'payers_information': 'payers_information',
75
+ # payers_information entries
76
+ 'payer_name': 'payer_name',
77
+ 'plan_name': 'plan_name',
78
+ 'standard_charge_dollar': 'negotiated_rate',
79
+ 'negotiated_dollar': 'negotiated_rate',
80
+ 'standard_charge_percentage': 'negotiated_percentage',
81
+ 'standard_charge_percent': 'negotiated_percentage',
82
+ 'negotiated_percentage': 'negotiated_percentage',
83
+ 'negotiated_percent': 'negotiated_percentage',
84
+ 'standard_charge_algorithm': 'negotiated_algorithm',
85
+ 'negotiated_algorithm': 'negotiated_algorithm',
86
+ 'methodology': 'methodology',
87
+ 'estimated_amount': 'estimated_amount',
88
+ 'additional_payer_notes': 'additional_notes',
89
+ 'footnote': 'footnote',
90
+ 'median_amount': 'median_amount',
91
+ 'median_allowed_amount': 'median_amount',
92
+ '10th_percentile': 'pct_10',
93
+ '90th_percentile': 'pct_90',
94
+ 'count': 'claim_count',
95
+ }
96
+
97
+ _JSON_CANONICAL_TO_KEYS: Dict[str, List[str]] = {}
98
+ for _key, _field in _JSON_FIELD_MAP.items():
99
+ _JSON_CANONICAL_TO_KEYS.setdefault(_field, []).append(_key)
100
+
101
+ # Preferred code when an item lists several: CPT, then HCPCS, RC, CDM, NDC.
102
+ _CODE_PRIORITY = {'CPT': 0, 'HCPCS': 1, 'RC': 2, 'CDM': 3, 'NDC': 4}
103
+
104
+
105
+ def iter_json(
106
+ path: Union[str, Path],
107
+ compression: Optional[str] = None,
108
+ *,
109
+ extra_synonyms: Optional[Mapping[str, str]] = None,
110
+ header_overrides: Optional[Mapping[str, str]] = None,
111
+ code_extraction: Optional[Mapping[str, Any]] = None,
112
+ ref: Optional[ReferenceData] = None,
113
+ stats: Optional[ParseStats] = None,
114
+ ) -> Iterator[Any]:
115
+ """Yield the records in a hospital MRF JSON file.
116
+
117
+ Order: :class:`FileMetadata` (when the root has any),
118
+ :class:`ModifierInfo` records, :class:`HeaderMapping` records, then the
119
+ data records. The file is opened several times (metadata, modifier
120
+ information, items), so *path* must be a file, not a stream.
121
+ """
122
+ path = Path(path)
123
+ stats = stats if stats is not None else ParseStats()
124
+ if compression is None:
125
+ _, compression = detect_file_format(path)
126
+
127
+ metadata, warning = _extract_json_root_metadata(path, compression)
128
+ if warning:
129
+ stats.warn(warning)
130
+ record = _file_metadata(metadata)
131
+ if record is not None:
132
+ yield record
133
+
134
+ # ponytail: modifier_information usually follows the big item array, so
135
+ # this is a second full read of the file. One pass with a push parser
136
+ # would save it if read time ever matters.
137
+ for entry in _extract_json_modifier_information(path, compression):
138
+ yield from _modifier_records(entry, ref)
139
+
140
+ prefix = _detect_json_array_prefix(path, compression)
141
+ if prefix is None:
142
+ key = _find_data_array_key(path, compression)
143
+ if key is None:
144
+ stats.warn("No data array found in the JSON root object")
145
+ return
146
+ log.warning("No known data array; reading the first root array %r", key)
147
+ prefix = f'{key}.item'
148
+
149
+ yield from records_from_items(
150
+ _iter_items(path, compression, prefix, stats),
151
+ extra_synonyms=extra_synonyms, header_overrides=header_overrides,
152
+ code_extraction=code_extraction, ref=ref, stats=stats,
153
+ )
154
+
155
+
156
+ def records_from_items(
157
+ items: Iterable[Any],
158
+ *,
159
+ extra_synonyms: Optional[Mapping[str, str]] = None,
160
+ header_overrides: Optional[Mapping[str, str]] = None,
161
+ code_extraction: Optional[Mapping[str, Any]] = None,
162
+ ref: Optional[ReferenceData] = None,
163
+ stats: Optional[ParseStats] = None,
164
+ ) -> Iterator[Any]:
165
+ """Turn parsed JSON items (dicts) into records.
166
+
167
+ The first item decides the shape: CMS nested, or flat rows.
168
+ """
169
+ stats = stats if stats is not None else ParseStats()
170
+ dicts = (obj for obj in items if isinstance(obj, dict))
171
+ first = next(dicts, None)
172
+ if first is None:
173
+ return
174
+
175
+ if _is_cms_format(first):
176
+ log.info("Detected CMS nested format")
177
+ for source, normalized, mapped in _CMS_HEADER_MAPPINGS:
178
+ yield HeaderMapping(source, normalized, mapped)
179
+ builder = RecordBuilder(code_extraction=code_extraction, ref=ref, stats=stats)
180
+ for obj in itertools.chain([first], dicts):
181
+ for row in _flatten_cms_item(obj):
182
+ stats.rows_read += 1
183
+ yield from _cms_row_records(row, builder)
184
+ return
185
+
186
+ peek = [first] + list(itertools.islice(dicts, PEEK_ROWS - 1))
187
+ layout = TableLayout(list(first.keys()), peek, extra_synonyms=extra_synonyms,
188
+ header_overrides=header_overrides, code_extraction=code_extraction,
189
+ ref=ref, stats=stats)
190
+ yield from layout.header_mappings()
191
+ for obj in itertools.chain(peek, dicts):
192
+ stats.rows_read += 1
193
+ yield from layout.row_to_records(obj)
194
+
195
+
196
+ # ---------------------------------------------------------------------------
197
+ # Streaming items
198
+ # ---------------------------------------------------------------------------
199
+
200
+ def _iter_items(path: Path, compression: Optional[str], prefix: str,
201
+ stats: ParseStats) -> Iterator[Any]:
202
+ """Stream the items under *prefix*, repairing the JSON only if needed.
203
+
204
+ The repair pass (control characters, double and trailing commas) is slow
205
+ pure Python, so the file is first read as is. On a syntax error it is
206
+ reopened with repair on, and the items already yielded are skipped:
207
+ repair only touches damaged bytes, and every item before the damage
208
+ parsed cleanly.
209
+ """
210
+ done = 0
211
+ try:
212
+ with open_json(path, compression) as fh:
213
+ for obj in ijson.items(fh, prefix):
214
+ yield obj
215
+ done += 1
216
+ return
217
+ except ijson.JSONError as exc:
218
+ stats.warn(f"JSON syntax error after {done} items ({exc}); retrying with repair")
219
+ with open_json(path, compression, sanitize=True) as fh:
220
+ for obj in itertools.islice(ijson.items(fh, prefix), done, None):
221
+ yield obj
222
+
223
+
224
+ def _detect_json_array_prefix(path: Path, compression: Optional[str]) -> Optional[str]:
225
+ """Return the ijson prefix of the item array, from the first 64 KB.
226
+
227
+ ``item`` for a root array, ``<key>.item`` when a known data key shows up
228
+ early, or None when the root is an object without one.
229
+ """
230
+ with open_json(path, compression) as fh:
231
+ head = fh.read(_PREFIX_SNIFF_BYTES).decode('utf-8', errors='replace').strip()
232
+ if head.startswith('['):
233
+ return 'item'
234
+ for key in DATA_KEYS:
235
+ if f'"{key}"' in head:
236
+ return f'{key}.item'
237
+ return None
238
+
239
+
240
+ def _find_data_array_key(path: Path, compression: Optional[str]) -> Optional[str]:
241
+ """Scan the root object for the array that holds the items.
242
+
243
+ A known data key wins wherever it appears; otherwise the first root
244
+ array that is not metadata. Used when the item array starts later than
245
+ the first 64 KB or has an unusual name.
246
+ """
247
+ first_candidate = None
248
+ with open_json(path, compression) as fh:
249
+ for prefix, event, _value in ijson.parse(fh):
250
+ if event != 'start_array' or not prefix or '.' in prefix:
251
+ continue
252
+ if prefix in METADATA_ARRAY_KEYS:
253
+ continue
254
+ if prefix in DATA_KEYS:
255
+ return prefix
256
+ if first_candidate is None:
257
+ first_candidate = prefix
258
+ return first_candidate
259
+
260
+
261
+ def _is_cms_format(obj: Mapping[str, Any]) -> bool:
262
+ """True for the CMS nested item shape."""
263
+ return 'code_information' in obj or 'standard_charges' in obj
264
+
265
+
266
+ # ---------------------------------------------------------------------------
267
+ # Root metadata and modifier information
268
+ # ---------------------------------------------------------------------------
269
+
270
+ def _extract_json_root_metadata(path: Path, compression: Optional[str]) -> Tuple[Dict[str, Any], Optional[str]]:
271
+ """Read the root-level metadata, stopping at the item array.
272
+
273
+ Returns ``(metadata, warning)``. Keys: hospital_name, hospital_location
274
+ and hospital_address (lists), license_number, license_state,
275
+ last_updated_on, version, attestation, attester_name,
276
+ confirm_attestation, type_2_npi (list). CMS 3.0 ``location_name`` is
277
+ reported as hospital_location.
278
+ """
279
+ scalar_keys = {'hospital_name', 'last_updated_on', 'version',
280
+ 'attestation', 'attester_name', 'confirm_attestation'}
281
+ array_keys = {'hospital_location', 'hospital_address', 'type_2_npi', 'location_name'}
282
+ metadata: Dict[str, Any] = {}
283
+ warning = None
284
+ try:
285
+ with open_json(path, compression) as fh:
286
+ current = None
287
+ array_key, array_values = None, []
288
+ in_license, license_info, license_key = False, {}, None
289
+ in_attestation, attestation_info, attestation_key = False, {}, None
290
+ for prefix, event, value in ijson.parse(fh):
291
+ if prefix == '' and event == 'map_key':
292
+ # A new root key: finish whatever was being collected.
293
+ if array_key and array_values:
294
+ metadata[array_key] = array_values
295
+ array_key, array_values = None, []
296
+ if in_license and license_info:
297
+ metadata['license_number'] = license_info.get('license_number')
298
+ metadata['license_state'] = license_info.get('state')
299
+ in_license, license_info = False, {}
300
+ if in_attestation and attestation_info:
301
+ metadata.update(attestation_info)
302
+ in_attestation, attestation_info = False, {}
303
+ current = value
304
+ if current in DATA_KEYS:
305
+ break # do not read the item array
306
+ elif prefix in scalar_keys and event in ('string', 'boolean', 'number'):
307
+ metadata[prefix] = value
308
+ elif current in array_keys:
309
+ if event == 'start_array':
310
+ array_key, array_values = current, []
311
+ elif event in ('string', 'number') and array_key:
312
+ array_values.append(str(value))
313
+ elif event == 'end_array' and array_key:
314
+ metadata[array_key] = array_values
315
+ array_key, array_values = None, []
316
+ elif current == 'license_information':
317
+ # An object, or an array of them.
318
+ if event == 'start_map':
319
+ in_license, license_info = True, {}
320
+ elif event == 'map_key' and in_license:
321
+ license_key = value
322
+ elif event in ('string', 'number') and in_license and license_key:
323
+ license_info[license_key] = value
324
+ license_key = None
325
+ elif event == 'end_map' and in_license:
326
+ metadata['license_number'] = license_info.get('license_number')
327
+ metadata['license_state'] = license_info.get('state')
328
+ in_license = False
329
+ elif current == 'attestation' and not in_attestation:
330
+ # CMS 3.0 nests attestation in an object; a CMS 2.0
331
+ # scalar is caught by the scalar branch above.
332
+ if event == 'start_map':
333
+ in_attestation, attestation_info = True, {}
334
+ elif in_attestation:
335
+ if event == 'map_key':
336
+ attestation_key = value
337
+ elif event in ('string', 'boolean', 'number') and attestation_key:
338
+ attestation_info[attestation_key] = value
339
+ attestation_key = None
340
+ elif event == 'end_map':
341
+ metadata.update(attestation_info)
342
+ in_attestation = False
343
+ except Exception as exc: # metadata is optional: never fail the file on it
344
+ warning = f"Could not extract JSON root metadata: {exc}"
345
+ if 'location_name' in metadata:
346
+ metadata['hospital_location'] = metadata.pop('location_name')
347
+ return metadata, warning
348
+
349
+
350
+ def _extract_json_modifier_information(path: Path, compression: Optional[str]) -> List[Dict[str, Any]]:
351
+ """Read the CMS 3.0 root ``modifier_information`` array.
352
+
353
+ Returns dicts with code, description, setting and
354
+ modifier_payer_information (payer_name, plan_name, description).
355
+ Entries without a code are dropped. Errors give an empty list.
356
+ """
357
+ results = []
358
+ try:
359
+ with open_json(path, compression) as fh:
360
+ for obj in ijson.items(fh, 'modifier_information.item'):
361
+ if not isinstance(obj, dict) or not obj.get('code', ''):
362
+ continue
363
+ results.append({
364
+ 'code': obj.get('code', ''),
365
+ 'description': obj.get('description'),
366
+ 'setting': obj.get('setting'),
367
+ 'modifier_payer_information': [
368
+ {'payer_name': p.get('payer_name', ''),
369
+ 'plan_name': p.get('plan_name', ''),
370
+ 'description': p.get('description')}
371
+ for p in (obj.get('modifier_payer_information') or [])
372
+ if isinstance(p, dict)
373
+ ],
374
+ })
375
+ except Exception as exc: # optional metadata
376
+ log.warning("Could not extract modifier_information: %s", exc)
377
+ return results
378
+
379
+
380
+ def _file_metadata(meta: Dict[str, Any]) -> Optional[FileMetadata]:
381
+ if not meta:
382
+ return None
383
+
384
+ def joined(key):
385
+ values = [v for v in (meta.get(key) or []) if v]
386
+ return '|'.join(values) or None
387
+
388
+ def text(key):
389
+ value = meta.get(key)
390
+ return str(value) if value is not None and str(value) != '' else None
391
+
392
+ license_number = text('license_number')
393
+ license_state = meta.get('license_state')
394
+ if license_state:
395
+ license_state = str(license_state).strip().upper()
396
+ if len(license_state) > 2:
397
+ license_state = _STATE_NAME_TO_ABBR.get(license_state)
398
+ if license_state not in _US_STATES:
399
+ license_state = None
400
+ # Older files write "12345|CA" in the license number itself.
401
+ if not license_state and license_number and '|' in license_number:
402
+ number, _, state = license_number.partition('|')
403
+ license_number = number.strip()
404
+ state = state.strip().upper()[:2]
405
+ license_state = state if state in _US_STATES else None
406
+
407
+ npis = [n for n in (_sanitize_npi(str(v)) for v in (meta.get('type_2_npi') or [])) if n]
408
+ confirm = meta.get('confirm_attestation')
409
+ return FileMetadata(
410
+ hospital_name=text('hospital_name'),
411
+ hospital_location=joined('hospital_location'),
412
+ hospital_address=joined('hospital_address'),
413
+ license_number=license_number,
414
+ license_state=license_state,
415
+ type_2_npi='|'.join(npis) or None,
416
+ last_updated_on=text('last_updated_on'),
417
+ version=text('version'),
418
+ attestation=text('attestation'),
419
+ attester_name=text('attester_name'),
420
+ confirm_attestation=bool(confirm) if confirm is not None else None,
421
+ )
422
+
423
+
424
+ def _modifier_records(entry: Dict[str, Any], ref: Optional[ReferenceData]) -> Iterator[ModifierInfo]:
425
+ """One general record per modifier, then one per payer that defines it."""
426
+ yield ModifierInfo(entry['code'], entry.get('description'), entry.get('setting'))
427
+ for info in entry['modifier_payer_information']:
428
+ payer_name, _plan = normalize_payer_name(info.get('payer_name', ''), ref=ref)
429
+ if not payer_name:
430
+ continue
431
+ yield ModifierInfo(
432
+ entry['code'], entry.get('description'), entry.get('setting'),
433
+ payer_name=payer_name, raw_payer_name=info.get('payer_name') or None,
434
+ plan_name=info.get('plan_name') or None, payer_description=info.get('description'),
435
+ )
436
+
437
+
438
+ # ---------------------------------------------------------------------------
439
+ # CMS nested items
440
+ # ---------------------------------------------------------------------------
441
+
442
+ def _json_get(obj: Mapping[str, Any], canonical: str, default=None):
443
+ """Return the first non-null value among the JSON spellings of *canonical*."""
444
+ for key in _JSON_CANONICAL_TO_KEYS.get(canonical, ()):
445
+ value = obj.get(key)
446
+ if value is not None:
447
+ return value
448
+ return default
449
+
450
+
451
+ def _flatten_cms_item(obj: Mapping[str, Any]) -> List[Dict[str, Any]]:
452
+ """Flatten one CMS item into one row per (billing_class, setting, modifiers).
453
+
454
+ The item's best code is used (CPT, then HCPCS, RC, CDM, NDC). Charges
455
+ sharing a group are merged, and their payers collected under 'payers'.
456
+ CMS 3.0 ``billing_class: both`` becomes a facility row and a
457
+ professional row with the same prices.
458
+ """
459
+ description = obj.get('description', '')
460
+ code_info = obj.get('code_information', [])
461
+ charges = obj.get('standard_charges', [])
462
+ drug_info = obj.get('drug_information', {})
463
+ if not code_info and not charges:
464
+ return []
465
+
466
+ drug_unit = str(drug_info.get('unit', '')) if drug_info else None
467
+ drug_type = str(drug_info.get('type', '')) if drug_info else None
468
+ drug_unit = drug_unit if drug_unit and drug_unit.strip() else None
469
+ drug_type = drug_type if drug_type and drug_type.strip() else None
470
+
471
+ best_code = best_type = None
472
+ best_priority = 99
473
+ for ci in code_info:
474
+ ctype = str(ci.get('type', '')).upper()
475
+ priority = _CODE_PRIORITY.get(ctype, 10)
476
+ if priority < best_priority:
477
+ best_code, best_type, best_priority = str(ci.get('code', '')), ctype, priority
478
+ best_code, best_type = normalize_code(best_code, best_type)
479
+
480
+ def row(bc, setting, modifiers, pricing):
481
+ def text(v):
482
+ return str(v) if v is not None else None
483
+ return {
484
+ 'description': description,
485
+ 'code': best_code,
486
+ 'code_type': best_type,
487
+ 'gross_charge': text(pricing['gross']),
488
+ 'discounted_cash_price': text(pricing['cash']),
489
+ 'min_negotiated_rate': text(pricing['min_rate']),
490
+ 'max_negotiated_rate': text(pricing['max_rate']),
491
+ 'billing_class': bc,
492
+ 'setting': setting,
493
+ 'modifiers': modifiers,
494
+ 'drug_unit_of_measurement': drug_unit,
495
+ 'drug_type_of_measurement': drug_type,
496
+ 'additional_generic_notes': pricing.get('additional_generic_notes'),
497
+ 'payers': pricing['payers'],
498
+ }
499
+
500
+ if not charges:
501
+ empty = {'gross': None, 'cash': None, 'min_rate': None, 'max_rate': None,
502
+ 'additional_generic_notes': None, 'payers': []}
503
+ return [row('', '', None, empty)]
504
+
505
+ groups: Dict[Tuple[str, str, Optional[str]], Dict[str, Any]] = {}
506
+ for c in charges:
507
+ bc = (_json_get(c, 'billing_class') or '').strip().lower()
508
+ bc = bc if bc in ('facility', 'professional', 'both') else ''
509
+ setting = (_json_get(c, 'setting') or '').strip().lower()
510
+ modifiers = _json_get(c, 'modifiers')
511
+ if isinstance(modifiers, list):
512
+ modifiers = ','.join(s for s in (str(m).strip() for m in modifiers) if s) or None
513
+ elif modifiers:
514
+ modifiers = str(modifiers).strip() or None
515
+ else:
516
+ modifiers = None
517
+ notes = _json_get(c, 'additional_generic_notes')
518
+ notes = notes.strip() or None if isinstance(notes, str) else None
519
+ payers = _json_get(c, 'payers_information') or []
520
+
521
+ key = (bc, setting, modifiers)
522
+ if key not in groups:
523
+ groups[key] = {
524
+ 'gross': _json_get(c, 'gross_charge'),
525
+ 'cash': _json_get(c, 'discounted_cash_price'),
526
+ 'min_rate': _json_get(c, 'min_negotiated_rate'),
527
+ 'max_rate': _json_get(c, 'max_negotiated_rate'),
528
+ 'additional_generic_notes': notes,
529
+ 'payers': list(payers),
530
+ }
531
+ else:
532
+ g = groups[key]
533
+ g['gross'] = g['gross'] or _json_get(c, 'gross_charge')
534
+ g['cash'] = g['cash'] or _json_get(c, 'discounted_cash_price')
535
+ g['min_rate'] = g['min_rate'] or _json_get(c, 'min_negotiated_rate')
536
+ g['max_rate'] = g['max_rate'] or _json_get(c, 'max_negotiated_rate')
537
+ g['additional_generic_notes'] = g['additional_generic_notes'] or notes
538
+ g['payers'].extend(payers)
539
+
540
+ rows = []
541
+ for (bc, setting, modifiers), pricing in groups.items():
542
+ for expanded in (('facility', 'professional') if bc == 'both' else (bc,)):
543
+ rows.append(row(expanded, setting, modifiers, pricing))
544
+ return rows
545
+
546
+
547
+ def _cms_row_records(row: Dict[str, Any], builder: RecordBuilder) -> Iterator[Any]:
548
+ """Records for one flattened CMS row."""
549
+ code, code_type, description = row.get('code'), row.get('code_type'), row.get('description')
550
+ if not any([code, code_type, description]):
551
+ return
552
+ cleaned = builder.clean_code(code, code_type, description)
553
+ if cleaned is None:
554
+ return
555
+ code, code_type, baked = cleaned
556
+ billing_class = row.get('billing_class')
557
+ setting = row.get('setting')
558
+ modifiers = merge_modifier_into_field(row.get('modifiers'), baked)
559
+ code, code_type, billing_class = builder.effective(code, code_type, description, billing_class)
560
+
561
+ yield from builder.item(code, code_type, description, billing_class, setting, modifiers, row)
562
+
563
+ gross = _safe_float(row.get('gross_charge'))
564
+ cash = _safe_float(row.get('discounted_cash_price'))
565
+ min_rate = _safe_float(row.get('min_negotiated_rate'))
566
+ max_rate = _safe_float(row.get('max_negotiated_rate'))
567
+ payers = row.get('payers', [])
568
+ # Self-pay entries are the cash price: keep the lowest.
569
+ for entry in payers:
570
+ raw = (_json_get(entry, 'payer_name') or '').strip()
571
+ if raw and is_self_pay_payer(raw):
572
+ rate = _safe_float(_json_get(entry, 'negotiated_rate') or _json_get(entry, 'estimated_amount'))
573
+ if rate is not None and (cash is None or rate < cash):
574
+ cash = rate
575
+ if any(v is not None for v in (gross, cash, min_rate, max_rate)):
576
+ yield StandardCharge(code, code_type, description, billing_class or '', setting or '',
577
+ modifiers, gross, cash, min_rate, max_rate)
578
+
579
+ for entry in payers:
580
+ raw = (_json_get(entry, 'payer_name') or '').strip()
581
+ if not raw or is_self_pay_payer(raw):
582
+ continue
583
+ values = dict(
584
+ negotiated_rate=_safe_float(_json_get(entry, 'negotiated_rate')),
585
+ negotiated_percentage=_safe_float(_json_get(entry, 'negotiated_percentage')),
586
+ negotiated_algorithm=_json_get(entry, 'negotiated_algorithm') or None,
587
+ methodology=_json_get(entry, 'methodology') or None,
588
+ estimated_amount=_safe_float(_json_get(entry, 'estimated_amount')),
589
+ )
590
+ if not any(values.values()):
591
+ continue
592
+ yield from builder.rate(
593
+ code, code_type, description, billing_class, setting, modifiers,
594
+ payer=raw, plan=(_json_get(entry, 'plan_name') or '').strip() or None,
595
+ rate_billing_class=billing_class, rate_setting=setting,
596
+ additional_notes=_json_get(entry, 'additional_notes') or None,
597
+ footnote=_json_get(entry, 'footnote') or None,
598
+ median_amount=_safe_float(_json_get(entry, 'median_amount')),
599
+ pct_10=_safe_float(_json_get(entry, 'pct_10')),
600
+ pct_90=_safe_float(_json_get(entry, 'pct_90')),
601
+ claim_count=_safe_count(_json_get(entry, 'claim_count')),
602
+ **values,
603
+ )