mrfkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mrfkit/__init__.py +68 -0
- mrfkit/__main__.py +3 -0
- mrfkit/cli.py +79 -0
- mrfkit/codes.py +1607 -0
- mrfkit/csv_reader.py +324 -0
- mrfkit/files.py +651 -0
- mrfkit/headers.py +737 -0
- mrfkit/json_reader.py +603 -0
- mrfkit/payers.py +2755 -0
- mrfkit/records.py +260 -0
- mrfkit/reference.py +136 -0
- mrfkit/sinks.py +126 -0
- mrfkit/tabular.py +651 -0
- mrfkit/tic.py +660 -0
- mrfkit/values.py +627 -0
- mrfkit-0.1.0.dist-info/METADATA +136 -0
- mrfkit-0.1.0.dist-info/RECORD +21 -0
- mrfkit-0.1.0.dist-info/WHEEL +4 -0
- mrfkit-0.1.0.dist-info/entry_points.txt +2 -0
- mrfkit-0.1.0.dist-info/licenses/LICENSE +201 -0
- mrfkit-0.1.0.dist-info/licenses/NOTICE +4 -0
mrfkit/json_reader.py
ADDED
|
@@ -0,0 +1,603 @@
|
|
|
1
|
+
"""Read hospital MRF JSON files into records.
|
|
2
|
+
|
|
3
|
+
Two shapes exist. The CMS template nests everything: an item has
|
|
4
|
+
``code_information``, ``standard_charges`` and, inside those,
|
|
5
|
+
``payers_information``. Older and vendor files are flat: each item is a row
|
|
6
|
+
of key/value pairs, read exactly like a CSV row by
|
|
7
|
+
:class:`mrfkit.tabular.TableLayout`.
|
|
8
|
+
"""
|
|
9
|
+
|
|
10
|
+
from __future__ import annotations
|
|
11
|
+
|
|
12
|
+
import itertools
|
|
13
|
+
import logging
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import Any, Dict, Iterable, Iterator, List, Mapping, Optional, Tuple, Union
|
|
16
|
+
|
|
17
|
+
import ijson
|
|
18
|
+
|
|
19
|
+
from .codes import merge_modifier_into_field, normalize_code
|
|
20
|
+
from .files import detect_file_format, open_json
|
|
21
|
+
from .payers import is_self_pay_payer, normalize_payer_name
|
|
22
|
+
from .records import FileMetadata, HeaderMapping, ModifierInfo, StandardCharge
|
|
23
|
+
from .reference import ParseStats, ReferenceData
|
|
24
|
+
from .tabular import PEEK_ROWS, RecordBuilder, TableLayout
|
|
25
|
+
from .values import _STATE_NAME_TO_ABBR, _US_STATES, _safe_count, _safe_float, _sanitize_npi
|
|
26
|
+
|
|
27
|
+
log = logging.getLogger(__name__)
|
|
28
|
+
|
|
29
|
+
# Root keys that hold the item array, most likely first.
|
|
30
|
+
DATA_KEYS = ('standard_charge_information', 'items', 'data')
|
|
31
|
+
|
|
32
|
+
# Root arrays that are metadata, never charge data.
|
|
33
|
+
METADATA_ARRAY_KEYS = frozenset({
|
|
34
|
+
'hospital_location', 'hospital_address', 'financial_aid_policy',
|
|
35
|
+
'license_information', 'affirmation', 'modifier_information',
|
|
36
|
+
'general_contract_provisions', 'location_name', 'type_2_npi',
|
|
37
|
+
})
|
|
38
|
+
|
|
39
|
+
_PREFIX_SNIFF_BYTES = 65536
|
|
40
|
+
|
|
41
|
+
# How the CMS nested fields are read, for HeaderMapping records.
|
|
42
|
+
_CMS_HEADER_MAPPINGS = [
|
|
43
|
+
('description', 'description', 'description'),
|
|
44
|
+
('code_information.code', 'code', 'code'),
|
|
45
|
+
('code_information.type', 'code_type', 'code_type'),
|
|
46
|
+
('standard_charges.gross_charge', 'gross_charge', 'gross_charge'),
|
|
47
|
+
('standard_charges.discounted_cash', 'discounted_cash_price', 'discounted_cash_price'),
|
|
48
|
+
('standard_charges.minimum', 'min_negotiated_rate', 'min_negotiated_rate'),
|
|
49
|
+
('standard_charges.maximum', 'max_negotiated_rate', 'max_negotiated_rate'),
|
|
50
|
+
('standard_charges.billing_class', 'billing_class', 'billing_class'),
|
|
51
|
+
('standard_charges.setting', 'setting', 'setting'),
|
|
52
|
+
('standard_charges.modifiers', 'modifiers', 'modifiers'),
|
|
53
|
+
('drug_information.unit', 'drug_unit_of_measurement', 'drug_unit_of_measurement'),
|
|
54
|
+
('drug_information.type', 'drug_type_of_measurement', 'drug_type_of_measurement'),
|
|
55
|
+
('standard_charges.payers_information', 'payers_information', 'wide_payer:cms_nested'),
|
|
56
|
+
]
|
|
57
|
+
|
|
58
|
+
# CMS JSON field names for each canonical field. Several spellings exist
|
|
59
|
+
# across template versions; the first non-null one wins.
|
|
60
|
+
_JSON_FIELD_MAP = {
|
|
61
|
+
# standard_charges level
|
|
62
|
+
'gross_charge': 'gross_charge',
|
|
63
|
+
'discounted_cash': 'discounted_cash_price',
|
|
64
|
+
'discounted_cash_price': 'discounted_cash_price',
|
|
65
|
+
'minimum': 'min_negotiated_rate',
|
|
66
|
+
'min': 'min_negotiated_rate',
|
|
67
|
+
'maximum': 'max_negotiated_rate',
|
|
68
|
+
'max': 'max_negotiated_rate',
|
|
69
|
+
'billing_class': 'billing_class',
|
|
70
|
+
'setting': 'setting',
|
|
71
|
+
'modifiers': 'modifiers',
|
|
72
|
+
'modifier_code': 'modifiers', # CMS 3.0
|
|
73
|
+
'additional_generic_notes': 'additional_generic_notes',
|
|
74
|
+
'payers_information': 'payers_information',
|
|
75
|
+
# payers_information entries
|
|
76
|
+
'payer_name': 'payer_name',
|
|
77
|
+
'plan_name': 'plan_name',
|
|
78
|
+
'standard_charge_dollar': 'negotiated_rate',
|
|
79
|
+
'negotiated_dollar': 'negotiated_rate',
|
|
80
|
+
'standard_charge_percentage': 'negotiated_percentage',
|
|
81
|
+
'standard_charge_percent': 'negotiated_percentage',
|
|
82
|
+
'negotiated_percentage': 'negotiated_percentage',
|
|
83
|
+
'negotiated_percent': 'negotiated_percentage',
|
|
84
|
+
'standard_charge_algorithm': 'negotiated_algorithm',
|
|
85
|
+
'negotiated_algorithm': 'negotiated_algorithm',
|
|
86
|
+
'methodology': 'methodology',
|
|
87
|
+
'estimated_amount': 'estimated_amount',
|
|
88
|
+
'additional_payer_notes': 'additional_notes',
|
|
89
|
+
'footnote': 'footnote',
|
|
90
|
+
'median_amount': 'median_amount',
|
|
91
|
+
'median_allowed_amount': 'median_amount',
|
|
92
|
+
'10th_percentile': 'pct_10',
|
|
93
|
+
'90th_percentile': 'pct_90',
|
|
94
|
+
'count': 'claim_count',
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
_JSON_CANONICAL_TO_KEYS: Dict[str, List[str]] = {}
|
|
98
|
+
for _key, _field in _JSON_FIELD_MAP.items():
|
|
99
|
+
_JSON_CANONICAL_TO_KEYS.setdefault(_field, []).append(_key)
|
|
100
|
+
|
|
101
|
+
# Preferred code when an item lists several: CPT, then HCPCS, RC, CDM, NDC.
|
|
102
|
+
_CODE_PRIORITY = {'CPT': 0, 'HCPCS': 1, 'RC': 2, 'CDM': 3, 'NDC': 4}
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def iter_json(
|
|
106
|
+
path: Union[str, Path],
|
|
107
|
+
compression: Optional[str] = None,
|
|
108
|
+
*,
|
|
109
|
+
extra_synonyms: Optional[Mapping[str, str]] = None,
|
|
110
|
+
header_overrides: Optional[Mapping[str, str]] = None,
|
|
111
|
+
code_extraction: Optional[Mapping[str, Any]] = None,
|
|
112
|
+
ref: Optional[ReferenceData] = None,
|
|
113
|
+
stats: Optional[ParseStats] = None,
|
|
114
|
+
) -> Iterator[Any]:
|
|
115
|
+
"""Yield the records in a hospital MRF JSON file.
|
|
116
|
+
|
|
117
|
+
Order: :class:`FileMetadata` (when the root has any),
|
|
118
|
+
:class:`ModifierInfo` records, :class:`HeaderMapping` records, then the
|
|
119
|
+
data records. The file is opened several times (metadata, modifier
|
|
120
|
+
information, items), so *path* must be a file, not a stream.
|
|
121
|
+
"""
|
|
122
|
+
path = Path(path)
|
|
123
|
+
stats = stats if stats is not None else ParseStats()
|
|
124
|
+
if compression is None:
|
|
125
|
+
_, compression = detect_file_format(path)
|
|
126
|
+
|
|
127
|
+
metadata, warning = _extract_json_root_metadata(path, compression)
|
|
128
|
+
if warning:
|
|
129
|
+
stats.warn(warning)
|
|
130
|
+
record = _file_metadata(metadata)
|
|
131
|
+
if record is not None:
|
|
132
|
+
yield record
|
|
133
|
+
|
|
134
|
+
# ponytail: modifier_information usually follows the big item array, so
|
|
135
|
+
# this is a second full read of the file. One pass with a push parser
|
|
136
|
+
# would save it if read time ever matters.
|
|
137
|
+
for entry in _extract_json_modifier_information(path, compression):
|
|
138
|
+
yield from _modifier_records(entry, ref)
|
|
139
|
+
|
|
140
|
+
prefix = _detect_json_array_prefix(path, compression)
|
|
141
|
+
if prefix is None:
|
|
142
|
+
key = _find_data_array_key(path, compression)
|
|
143
|
+
if key is None:
|
|
144
|
+
stats.warn("No data array found in the JSON root object")
|
|
145
|
+
return
|
|
146
|
+
log.warning("No known data array; reading the first root array %r", key)
|
|
147
|
+
prefix = f'{key}.item'
|
|
148
|
+
|
|
149
|
+
yield from records_from_items(
|
|
150
|
+
_iter_items(path, compression, prefix, stats),
|
|
151
|
+
extra_synonyms=extra_synonyms, header_overrides=header_overrides,
|
|
152
|
+
code_extraction=code_extraction, ref=ref, stats=stats,
|
|
153
|
+
)
|
|
154
|
+
|
|
155
|
+
|
|
156
|
+
def records_from_items(
|
|
157
|
+
items: Iterable[Any],
|
|
158
|
+
*,
|
|
159
|
+
extra_synonyms: Optional[Mapping[str, str]] = None,
|
|
160
|
+
header_overrides: Optional[Mapping[str, str]] = None,
|
|
161
|
+
code_extraction: Optional[Mapping[str, Any]] = None,
|
|
162
|
+
ref: Optional[ReferenceData] = None,
|
|
163
|
+
stats: Optional[ParseStats] = None,
|
|
164
|
+
) -> Iterator[Any]:
|
|
165
|
+
"""Turn parsed JSON items (dicts) into records.
|
|
166
|
+
|
|
167
|
+
The first item decides the shape: CMS nested, or flat rows.
|
|
168
|
+
"""
|
|
169
|
+
stats = stats if stats is not None else ParseStats()
|
|
170
|
+
dicts = (obj for obj in items if isinstance(obj, dict))
|
|
171
|
+
first = next(dicts, None)
|
|
172
|
+
if first is None:
|
|
173
|
+
return
|
|
174
|
+
|
|
175
|
+
if _is_cms_format(first):
|
|
176
|
+
log.info("Detected CMS nested format")
|
|
177
|
+
for source, normalized, mapped in _CMS_HEADER_MAPPINGS:
|
|
178
|
+
yield HeaderMapping(source, normalized, mapped)
|
|
179
|
+
builder = RecordBuilder(code_extraction=code_extraction, ref=ref, stats=stats)
|
|
180
|
+
for obj in itertools.chain([first], dicts):
|
|
181
|
+
for row in _flatten_cms_item(obj):
|
|
182
|
+
stats.rows_read += 1
|
|
183
|
+
yield from _cms_row_records(row, builder)
|
|
184
|
+
return
|
|
185
|
+
|
|
186
|
+
peek = [first] + list(itertools.islice(dicts, PEEK_ROWS - 1))
|
|
187
|
+
layout = TableLayout(list(first.keys()), peek, extra_synonyms=extra_synonyms,
|
|
188
|
+
header_overrides=header_overrides, code_extraction=code_extraction,
|
|
189
|
+
ref=ref, stats=stats)
|
|
190
|
+
yield from layout.header_mappings()
|
|
191
|
+
for obj in itertools.chain(peek, dicts):
|
|
192
|
+
stats.rows_read += 1
|
|
193
|
+
yield from layout.row_to_records(obj)
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
# ---------------------------------------------------------------------------
|
|
197
|
+
# Streaming items
|
|
198
|
+
# ---------------------------------------------------------------------------
|
|
199
|
+
|
|
200
|
+
def _iter_items(path: Path, compression: Optional[str], prefix: str,
|
|
201
|
+
stats: ParseStats) -> Iterator[Any]:
|
|
202
|
+
"""Stream the items under *prefix*, repairing the JSON only if needed.
|
|
203
|
+
|
|
204
|
+
The repair pass (control characters, double and trailing commas) is slow
|
|
205
|
+
pure Python, so the file is first read as is. On a syntax error it is
|
|
206
|
+
reopened with repair on, and the items already yielded are skipped:
|
|
207
|
+
repair only touches damaged bytes, and every item before the damage
|
|
208
|
+
parsed cleanly.
|
|
209
|
+
"""
|
|
210
|
+
done = 0
|
|
211
|
+
try:
|
|
212
|
+
with open_json(path, compression) as fh:
|
|
213
|
+
for obj in ijson.items(fh, prefix):
|
|
214
|
+
yield obj
|
|
215
|
+
done += 1
|
|
216
|
+
return
|
|
217
|
+
except ijson.JSONError as exc:
|
|
218
|
+
stats.warn(f"JSON syntax error after {done} items ({exc}); retrying with repair")
|
|
219
|
+
with open_json(path, compression, sanitize=True) as fh:
|
|
220
|
+
for obj in itertools.islice(ijson.items(fh, prefix), done, None):
|
|
221
|
+
yield obj
|
|
222
|
+
|
|
223
|
+
|
|
224
|
+
def _detect_json_array_prefix(path: Path, compression: Optional[str]) -> Optional[str]:
|
|
225
|
+
"""Return the ijson prefix of the item array, from the first 64 KB.
|
|
226
|
+
|
|
227
|
+
``item`` for a root array, ``<key>.item`` when a known data key shows up
|
|
228
|
+
early, or None when the root is an object without one.
|
|
229
|
+
"""
|
|
230
|
+
with open_json(path, compression) as fh:
|
|
231
|
+
head = fh.read(_PREFIX_SNIFF_BYTES).decode('utf-8', errors='replace').strip()
|
|
232
|
+
if head.startswith('['):
|
|
233
|
+
return 'item'
|
|
234
|
+
for key in DATA_KEYS:
|
|
235
|
+
if f'"{key}"' in head:
|
|
236
|
+
return f'{key}.item'
|
|
237
|
+
return None
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _find_data_array_key(path: Path, compression: Optional[str]) -> Optional[str]:
|
|
241
|
+
"""Scan the root object for the array that holds the items.
|
|
242
|
+
|
|
243
|
+
A known data key wins wherever it appears; otherwise the first root
|
|
244
|
+
array that is not metadata. Used when the item array starts later than
|
|
245
|
+
the first 64 KB or has an unusual name.
|
|
246
|
+
"""
|
|
247
|
+
first_candidate = None
|
|
248
|
+
with open_json(path, compression) as fh:
|
|
249
|
+
for prefix, event, _value in ijson.parse(fh):
|
|
250
|
+
if event != 'start_array' or not prefix or '.' in prefix:
|
|
251
|
+
continue
|
|
252
|
+
if prefix in METADATA_ARRAY_KEYS:
|
|
253
|
+
continue
|
|
254
|
+
if prefix in DATA_KEYS:
|
|
255
|
+
return prefix
|
|
256
|
+
if first_candidate is None:
|
|
257
|
+
first_candidate = prefix
|
|
258
|
+
return first_candidate
|
|
259
|
+
|
|
260
|
+
|
|
261
|
+
def _is_cms_format(obj: Mapping[str, Any]) -> bool:
|
|
262
|
+
"""True for the CMS nested item shape."""
|
|
263
|
+
return 'code_information' in obj or 'standard_charges' in obj
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
# ---------------------------------------------------------------------------
|
|
267
|
+
# Root metadata and modifier information
|
|
268
|
+
# ---------------------------------------------------------------------------
|
|
269
|
+
|
|
270
|
+
def _extract_json_root_metadata(path: Path, compression: Optional[str]) -> Tuple[Dict[str, Any], Optional[str]]:
|
|
271
|
+
"""Read the root-level metadata, stopping at the item array.
|
|
272
|
+
|
|
273
|
+
Returns ``(metadata, warning)``. Keys: hospital_name, hospital_location
|
|
274
|
+
and hospital_address (lists), license_number, license_state,
|
|
275
|
+
last_updated_on, version, attestation, attester_name,
|
|
276
|
+
confirm_attestation, type_2_npi (list). CMS 3.0 ``location_name`` is
|
|
277
|
+
reported as hospital_location.
|
|
278
|
+
"""
|
|
279
|
+
scalar_keys = {'hospital_name', 'last_updated_on', 'version',
|
|
280
|
+
'attestation', 'attester_name', 'confirm_attestation'}
|
|
281
|
+
array_keys = {'hospital_location', 'hospital_address', 'type_2_npi', 'location_name'}
|
|
282
|
+
metadata: Dict[str, Any] = {}
|
|
283
|
+
warning = None
|
|
284
|
+
try:
|
|
285
|
+
with open_json(path, compression) as fh:
|
|
286
|
+
current = None
|
|
287
|
+
array_key, array_values = None, []
|
|
288
|
+
in_license, license_info, license_key = False, {}, None
|
|
289
|
+
in_attestation, attestation_info, attestation_key = False, {}, None
|
|
290
|
+
for prefix, event, value in ijson.parse(fh):
|
|
291
|
+
if prefix == '' and event == 'map_key':
|
|
292
|
+
# A new root key: finish whatever was being collected.
|
|
293
|
+
if array_key and array_values:
|
|
294
|
+
metadata[array_key] = array_values
|
|
295
|
+
array_key, array_values = None, []
|
|
296
|
+
if in_license and license_info:
|
|
297
|
+
metadata['license_number'] = license_info.get('license_number')
|
|
298
|
+
metadata['license_state'] = license_info.get('state')
|
|
299
|
+
in_license, license_info = False, {}
|
|
300
|
+
if in_attestation and attestation_info:
|
|
301
|
+
metadata.update(attestation_info)
|
|
302
|
+
in_attestation, attestation_info = False, {}
|
|
303
|
+
current = value
|
|
304
|
+
if current in DATA_KEYS:
|
|
305
|
+
break # do not read the item array
|
|
306
|
+
elif prefix in scalar_keys and event in ('string', 'boolean', 'number'):
|
|
307
|
+
metadata[prefix] = value
|
|
308
|
+
elif current in array_keys:
|
|
309
|
+
if event == 'start_array':
|
|
310
|
+
array_key, array_values = current, []
|
|
311
|
+
elif event in ('string', 'number') and array_key:
|
|
312
|
+
array_values.append(str(value))
|
|
313
|
+
elif event == 'end_array' and array_key:
|
|
314
|
+
metadata[array_key] = array_values
|
|
315
|
+
array_key, array_values = None, []
|
|
316
|
+
elif current == 'license_information':
|
|
317
|
+
# An object, or an array of them.
|
|
318
|
+
if event == 'start_map':
|
|
319
|
+
in_license, license_info = True, {}
|
|
320
|
+
elif event == 'map_key' and in_license:
|
|
321
|
+
license_key = value
|
|
322
|
+
elif event in ('string', 'number') and in_license and license_key:
|
|
323
|
+
license_info[license_key] = value
|
|
324
|
+
license_key = None
|
|
325
|
+
elif event == 'end_map' and in_license:
|
|
326
|
+
metadata['license_number'] = license_info.get('license_number')
|
|
327
|
+
metadata['license_state'] = license_info.get('state')
|
|
328
|
+
in_license = False
|
|
329
|
+
elif current == 'attestation' and not in_attestation:
|
|
330
|
+
# CMS 3.0 nests attestation in an object; a CMS 2.0
|
|
331
|
+
# scalar is caught by the scalar branch above.
|
|
332
|
+
if event == 'start_map':
|
|
333
|
+
in_attestation, attestation_info = True, {}
|
|
334
|
+
elif in_attestation:
|
|
335
|
+
if event == 'map_key':
|
|
336
|
+
attestation_key = value
|
|
337
|
+
elif event in ('string', 'boolean', 'number') and attestation_key:
|
|
338
|
+
attestation_info[attestation_key] = value
|
|
339
|
+
attestation_key = None
|
|
340
|
+
elif event == 'end_map':
|
|
341
|
+
metadata.update(attestation_info)
|
|
342
|
+
in_attestation = False
|
|
343
|
+
except Exception as exc: # metadata is optional: never fail the file on it
|
|
344
|
+
warning = f"Could not extract JSON root metadata: {exc}"
|
|
345
|
+
if 'location_name' in metadata:
|
|
346
|
+
metadata['hospital_location'] = metadata.pop('location_name')
|
|
347
|
+
return metadata, warning
|
|
348
|
+
|
|
349
|
+
|
|
350
|
+
def _extract_json_modifier_information(path: Path, compression: Optional[str]) -> List[Dict[str, Any]]:
|
|
351
|
+
"""Read the CMS 3.0 root ``modifier_information`` array.
|
|
352
|
+
|
|
353
|
+
Returns dicts with code, description, setting and
|
|
354
|
+
modifier_payer_information (payer_name, plan_name, description).
|
|
355
|
+
Entries without a code are dropped. Errors give an empty list.
|
|
356
|
+
"""
|
|
357
|
+
results = []
|
|
358
|
+
try:
|
|
359
|
+
with open_json(path, compression) as fh:
|
|
360
|
+
for obj in ijson.items(fh, 'modifier_information.item'):
|
|
361
|
+
if not isinstance(obj, dict) or not obj.get('code', ''):
|
|
362
|
+
continue
|
|
363
|
+
results.append({
|
|
364
|
+
'code': obj.get('code', ''),
|
|
365
|
+
'description': obj.get('description'),
|
|
366
|
+
'setting': obj.get('setting'),
|
|
367
|
+
'modifier_payer_information': [
|
|
368
|
+
{'payer_name': p.get('payer_name', ''),
|
|
369
|
+
'plan_name': p.get('plan_name', ''),
|
|
370
|
+
'description': p.get('description')}
|
|
371
|
+
for p in (obj.get('modifier_payer_information') or [])
|
|
372
|
+
if isinstance(p, dict)
|
|
373
|
+
],
|
|
374
|
+
})
|
|
375
|
+
except Exception as exc: # optional metadata
|
|
376
|
+
log.warning("Could not extract modifier_information: %s", exc)
|
|
377
|
+
return results
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def _file_metadata(meta: Dict[str, Any]) -> Optional[FileMetadata]:
|
|
381
|
+
if not meta:
|
|
382
|
+
return None
|
|
383
|
+
|
|
384
|
+
def joined(key):
|
|
385
|
+
values = [v for v in (meta.get(key) or []) if v]
|
|
386
|
+
return '|'.join(values) or None
|
|
387
|
+
|
|
388
|
+
def text(key):
|
|
389
|
+
value = meta.get(key)
|
|
390
|
+
return str(value) if value is not None and str(value) != '' else None
|
|
391
|
+
|
|
392
|
+
license_number = text('license_number')
|
|
393
|
+
license_state = meta.get('license_state')
|
|
394
|
+
if license_state:
|
|
395
|
+
license_state = str(license_state).strip().upper()
|
|
396
|
+
if len(license_state) > 2:
|
|
397
|
+
license_state = _STATE_NAME_TO_ABBR.get(license_state)
|
|
398
|
+
if license_state not in _US_STATES:
|
|
399
|
+
license_state = None
|
|
400
|
+
# Older files write "12345|CA" in the license number itself.
|
|
401
|
+
if not license_state and license_number and '|' in license_number:
|
|
402
|
+
number, _, state = license_number.partition('|')
|
|
403
|
+
license_number = number.strip()
|
|
404
|
+
state = state.strip().upper()[:2]
|
|
405
|
+
license_state = state if state in _US_STATES else None
|
|
406
|
+
|
|
407
|
+
npis = [n for n in (_sanitize_npi(str(v)) for v in (meta.get('type_2_npi') or [])) if n]
|
|
408
|
+
confirm = meta.get('confirm_attestation')
|
|
409
|
+
return FileMetadata(
|
|
410
|
+
hospital_name=text('hospital_name'),
|
|
411
|
+
hospital_location=joined('hospital_location'),
|
|
412
|
+
hospital_address=joined('hospital_address'),
|
|
413
|
+
license_number=license_number,
|
|
414
|
+
license_state=license_state,
|
|
415
|
+
type_2_npi='|'.join(npis) or None,
|
|
416
|
+
last_updated_on=text('last_updated_on'),
|
|
417
|
+
version=text('version'),
|
|
418
|
+
attestation=text('attestation'),
|
|
419
|
+
attester_name=text('attester_name'),
|
|
420
|
+
confirm_attestation=bool(confirm) if confirm is not None else None,
|
|
421
|
+
)
|
|
422
|
+
|
|
423
|
+
|
|
424
|
+
def _modifier_records(entry: Dict[str, Any], ref: Optional[ReferenceData]) -> Iterator[ModifierInfo]:
|
|
425
|
+
"""One general record per modifier, then one per payer that defines it."""
|
|
426
|
+
yield ModifierInfo(entry['code'], entry.get('description'), entry.get('setting'))
|
|
427
|
+
for info in entry['modifier_payer_information']:
|
|
428
|
+
payer_name, _plan = normalize_payer_name(info.get('payer_name', ''), ref=ref)
|
|
429
|
+
if not payer_name:
|
|
430
|
+
continue
|
|
431
|
+
yield ModifierInfo(
|
|
432
|
+
entry['code'], entry.get('description'), entry.get('setting'),
|
|
433
|
+
payer_name=payer_name, raw_payer_name=info.get('payer_name') or None,
|
|
434
|
+
plan_name=info.get('plan_name') or None, payer_description=info.get('description'),
|
|
435
|
+
)
|
|
436
|
+
|
|
437
|
+
|
|
438
|
+
# ---------------------------------------------------------------------------
|
|
439
|
+
# CMS nested items
|
|
440
|
+
# ---------------------------------------------------------------------------
|
|
441
|
+
|
|
442
|
+
def _json_get(obj: Mapping[str, Any], canonical: str, default=None):
|
|
443
|
+
"""Return the first non-null value among the JSON spellings of *canonical*."""
|
|
444
|
+
for key in _JSON_CANONICAL_TO_KEYS.get(canonical, ()):
|
|
445
|
+
value = obj.get(key)
|
|
446
|
+
if value is not None:
|
|
447
|
+
return value
|
|
448
|
+
return default
|
|
449
|
+
|
|
450
|
+
|
|
451
|
+
def _flatten_cms_item(obj: Mapping[str, Any]) -> List[Dict[str, Any]]:
|
|
452
|
+
"""Flatten one CMS item into one row per (billing_class, setting, modifiers).
|
|
453
|
+
|
|
454
|
+
The item's best code is used (CPT, then HCPCS, RC, CDM, NDC). Charges
|
|
455
|
+
sharing a group are merged, and their payers collected under 'payers'.
|
|
456
|
+
CMS 3.0 ``billing_class: both`` becomes a facility row and a
|
|
457
|
+
professional row with the same prices.
|
|
458
|
+
"""
|
|
459
|
+
description = obj.get('description', '')
|
|
460
|
+
code_info = obj.get('code_information', [])
|
|
461
|
+
charges = obj.get('standard_charges', [])
|
|
462
|
+
drug_info = obj.get('drug_information', {})
|
|
463
|
+
if not code_info and not charges:
|
|
464
|
+
return []
|
|
465
|
+
|
|
466
|
+
drug_unit = str(drug_info.get('unit', '')) if drug_info else None
|
|
467
|
+
drug_type = str(drug_info.get('type', '')) if drug_info else None
|
|
468
|
+
drug_unit = drug_unit if drug_unit and drug_unit.strip() else None
|
|
469
|
+
drug_type = drug_type if drug_type and drug_type.strip() else None
|
|
470
|
+
|
|
471
|
+
best_code = best_type = None
|
|
472
|
+
best_priority = 99
|
|
473
|
+
for ci in code_info:
|
|
474
|
+
ctype = str(ci.get('type', '')).upper()
|
|
475
|
+
priority = _CODE_PRIORITY.get(ctype, 10)
|
|
476
|
+
if priority < best_priority:
|
|
477
|
+
best_code, best_type, best_priority = str(ci.get('code', '')), ctype, priority
|
|
478
|
+
best_code, best_type = normalize_code(best_code, best_type)
|
|
479
|
+
|
|
480
|
+
def row(bc, setting, modifiers, pricing):
|
|
481
|
+
def text(v):
|
|
482
|
+
return str(v) if v is not None else None
|
|
483
|
+
return {
|
|
484
|
+
'description': description,
|
|
485
|
+
'code': best_code,
|
|
486
|
+
'code_type': best_type,
|
|
487
|
+
'gross_charge': text(pricing['gross']),
|
|
488
|
+
'discounted_cash_price': text(pricing['cash']),
|
|
489
|
+
'min_negotiated_rate': text(pricing['min_rate']),
|
|
490
|
+
'max_negotiated_rate': text(pricing['max_rate']),
|
|
491
|
+
'billing_class': bc,
|
|
492
|
+
'setting': setting,
|
|
493
|
+
'modifiers': modifiers,
|
|
494
|
+
'drug_unit_of_measurement': drug_unit,
|
|
495
|
+
'drug_type_of_measurement': drug_type,
|
|
496
|
+
'additional_generic_notes': pricing.get('additional_generic_notes'),
|
|
497
|
+
'payers': pricing['payers'],
|
|
498
|
+
}
|
|
499
|
+
|
|
500
|
+
if not charges:
|
|
501
|
+
empty = {'gross': None, 'cash': None, 'min_rate': None, 'max_rate': None,
|
|
502
|
+
'additional_generic_notes': None, 'payers': []}
|
|
503
|
+
return [row('', '', None, empty)]
|
|
504
|
+
|
|
505
|
+
groups: Dict[Tuple[str, str, Optional[str]], Dict[str, Any]] = {}
|
|
506
|
+
for c in charges:
|
|
507
|
+
bc = (_json_get(c, 'billing_class') or '').strip().lower()
|
|
508
|
+
bc = bc if bc in ('facility', 'professional', 'both') else ''
|
|
509
|
+
setting = (_json_get(c, 'setting') or '').strip().lower()
|
|
510
|
+
modifiers = _json_get(c, 'modifiers')
|
|
511
|
+
if isinstance(modifiers, list):
|
|
512
|
+
modifiers = ','.join(s for s in (str(m).strip() for m in modifiers) if s) or None
|
|
513
|
+
elif modifiers:
|
|
514
|
+
modifiers = str(modifiers).strip() or None
|
|
515
|
+
else:
|
|
516
|
+
modifiers = None
|
|
517
|
+
notes = _json_get(c, 'additional_generic_notes')
|
|
518
|
+
notes = notes.strip() or None if isinstance(notes, str) else None
|
|
519
|
+
payers = _json_get(c, 'payers_information') or []
|
|
520
|
+
|
|
521
|
+
key = (bc, setting, modifiers)
|
|
522
|
+
if key not in groups:
|
|
523
|
+
groups[key] = {
|
|
524
|
+
'gross': _json_get(c, 'gross_charge'),
|
|
525
|
+
'cash': _json_get(c, 'discounted_cash_price'),
|
|
526
|
+
'min_rate': _json_get(c, 'min_negotiated_rate'),
|
|
527
|
+
'max_rate': _json_get(c, 'max_negotiated_rate'),
|
|
528
|
+
'additional_generic_notes': notes,
|
|
529
|
+
'payers': list(payers),
|
|
530
|
+
}
|
|
531
|
+
else:
|
|
532
|
+
g = groups[key]
|
|
533
|
+
g['gross'] = g['gross'] or _json_get(c, 'gross_charge')
|
|
534
|
+
g['cash'] = g['cash'] or _json_get(c, 'discounted_cash_price')
|
|
535
|
+
g['min_rate'] = g['min_rate'] or _json_get(c, 'min_negotiated_rate')
|
|
536
|
+
g['max_rate'] = g['max_rate'] or _json_get(c, 'max_negotiated_rate')
|
|
537
|
+
g['additional_generic_notes'] = g['additional_generic_notes'] or notes
|
|
538
|
+
g['payers'].extend(payers)
|
|
539
|
+
|
|
540
|
+
rows = []
|
|
541
|
+
for (bc, setting, modifiers), pricing in groups.items():
|
|
542
|
+
for expanded in (('facility', 'professional') if bc == 'both' else (bc,)):
|
|
543
|
+
rows.append(row(expanded, setting, modifiers, pricing))
|
|
544
|
+
return rows
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
def _cms_row_records(row: Dict[str, Any], builder: RecordBuilder) -> Iterator[Any]:
|
|
548
|
+
"""Records for one flattened CMS row."""
|
|
549
|
+
code, code_type, description = row.get('code'), row.get('code_type'), row.get('description')
|
|
550
|
+
if not any([code, code_type, description]):
|
|
551
|
+
return
|
|
552
|
+
cleaned = builder.clean_code(code, code_type, description)
|
|
553
|
+
if cleaned is None:
|
|
554
|
+
return
|
|
555
|
+
code, code_type, baked = cleaned
|
|
556
|
+
billing_class = row.get('billing_class')
|
|
557
|
+
setting = row.get('setting')
|
|
558
|
+
modifiers = merge_modifier_into_field(row.get('modifiers'), baked)
|
|
559
|
+
code, code_type, billing_class = builder.effective(code, code_type, description, billing_class)
|
|
560
|
+
|
|
561
|
+
yield from builder.item(code, code_type, description, billing_class, setting, modifiers, row)
|
|
562
|
+
|
|
563
|
+
gross = _safe_float(row.get('gross_charge'))
|
|
564
|
+
cash = _safe_float(row.get('discounted_cash_price'))
|
|
565
|
+
min_rate = _safe_float(row.get('min_negotiated_rate'))
|
|
566
|
+
max_rate = _safe_float(row.get('max_negotiated_rate'))
|
|
567
|
+
payers = row.get('payers', [])
|
|
568
|
+
# Self-pay entries are the cash price: keep the lowest.
|
|
569
|
+
for entry in payers:
|
|
570
|
+
raw = (_json_get(entry, 'payer_name') or '').strip()
|
|
571
|
+
if raw and is_self_pay_payer(raw):
|
|
572
|
+
rate = _safe_float(_json_get(entry, 'negotiated_rate') or _json_get(entry, 'estimated_amount'))
|
|
573
|
+
if rate is not None and (cash is None or rate < cash):
|
|
574
|
+
cash = rate
|
|
575
|
+
if any(v is not None for v in (gross, cash, min_rate, max_rate)):
|
|
576
|
+
yield StandardCharge(code, code_type, description, billing_class or '', setting or '',
|
|
577
|
+
modifiers, gross, cash, min_rate, max_rate)
|
|
578
|
+
|
|
579
|
+
for entry in payers:
|
|
580
|
+
raw = (_json_get(entry, 'payer_name') or '').strip()
|
|
581
|
+
if not raw or is_self_pay_payer(raw):
|
|
582
|
+
continue
|
|
583
|
+
values = dict(
|
|
584
|
+
negotiated_rate=_safe_float(_json_get(entry, 'negotiated_rate')),
|
|
585
|
+
negotiated_percentage=_safe_float(_json_get(entry, 'negotiated_percentage')),
|
|
586
|
+
negotiated_algorithm=_json_get(entry, 'negotiated_algorithm') or None,
|
|
587
|
+
methodology=_json_get(entry, 'methodology') or None,
|
|
588
|
+
estimated_amount=_safe_float(_json_get(entry, 'estimated_amount')),
|
|
589
|
+
)
|
|
590
|
+
if not any(values.values()):
|
|
591
|
+
continue
|
|
592
|
+
yield from builder.rate(
|
|
593
|
+
code, code_type, description, billing_class, setting, modifiers,
|
|
594
|
+
payer=raw, plan=(_json_get(entry, 'plan_name') or '').strip() or None,
|
|
595
|
+
rate_billing_class=billing_class, rate_setting=setting,
|
|
596
|
+
additional_notes=_json_get(entry, 'additional_notes') or None,
|
|
597
|
+
footnote=_json_get(entry, 'footnote') or None,
|
|
598
|
+
median_amount=_safe_float(_json_get(entry, 'median_amount')),
|
|
599
|
+
pct_10=_safe_float(_json_get(entry, 'pct_10')),
|
|
600
|
+
pct_90=_safe_float(_json_get(entry, 'pct_90')),
|
|
601
|
+
claim_count=_safe_count(_json_get(entry, 'claim_count')),
|
|
602
|
+
**values,
|
|
603
|
+
)
|