mrfkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mrfkit/__init__.py +68 -0
- mrfkit/__main__.py +3 -0
- mrfkit/cli.py +79 -0
- mrfkit/codes.py +1607 -0
- mrfkit/csv_reader.py +324 -0
- mrfkit/files.py +651 -0
- mrfkit/headers.py +737 -0
- mrfkit/json_reader.py +603 -0
- mrfkit/payers.py +2755 -0
- mrfkit/records.py +260 -0
- mrfkit/reference.py +136 -0
- mrfkit/sinks.py +126 -0
- mrfkit/tabular.py +651 -0
- mrfkit/tic.py +660 -0
- mrfkit/values.py +627 -0
- mrfkit-0.1.0.dist-info/METADATA +136 -0
- mrfkit-0.1.0.dist-info/RECORD +21 -0
- mrfkit-0.1.0.dist-info/WHEEL +4 -0
- mrfkit-0.1.0.dist-info/entry_points.txt +2 -0
- mrfkit-0.1.0.dist-info/licenses/LICENSE +201 -0
- mrfkit-0.1.0.dist-info/licenses/NOTICE +4 -0
mrfkit/csv_reader.py
ADDED
|
@@ -0,0 +1,324 @@
|
|
|
1
|
+
"""Read hospital MRF CSV files into records."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import io
|
|
7
|
+
import itertools
|
|
8
|
+
import logging
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
from typing import Any, Dict, Iterator, List, Mapping, Optional, TextIO, Union
|
|
11
|
+
|
|
12
|
+
from .codes import _MAX_CODE_TYPE_LEN, _RE_VALID_CODE_TYPE_SHAPE
|
|
13
|
+
from .files import detect_file_format, open_text
|
|
14
|
+
from .headers import map_header
|
|
15
|
+
from .records import FileMetadata
|
|
16
|
+
from .reference import ParseStats, ReferenceData
|
|
17
|
+
from .tabular import PEEK_ROWS, SectionBoundary, TableLayout
|
|
18
|
+
from .values import _STATE_NAME_TO_ABBR, _US_STATES, _sanitize_npi
|
|
19
|
+
|
|
20
|
+
log = logging.getLogger(__name__)
|
|
21
|
+
|
|
22
|
+
# Real MRF cells (long notes, stuffed code lists) exceed the csv default of
|
|
23
|
+
# 128 KB. Raised once at import, so a caller can still lower it.
|
|
24
|
+
FIELD_SIZE_LIMIT = 10 * 1024 * 1024
|
|
25
|
+
csv.field_size_limit(max(csv.field_size_limit(), FIELD_SIZE_LIMIT))
|
|
26
|
+
|
|
27
|
+
PREAMBLE_LINES = 10
|
|
28
|
+
QUOTING_SAMPLE_ROWS = 200
|
|
29
|
+
_WARN_SKIPPED_ROWS = 5
|
|
30
|
+
|
|
31
|
+
# "Key: Value" preamble lines, as some non-CMS files write their metadata.
|
|
32
|
+
_KV_PREAMBLE_KEYS = {
|
|
33
|
+
'hospital name': 'hospital_name',
|
|
34
|
+
'hospital location': 'hospital_location',
|
|
35
|
+
'hospital address': 'hospital_address',
|
|
36
|
+
'price effective date': 'last_updated_on',
|
|
37
|
+
'last updated on': 'last_updated_on',
|
|
38
|
+
'cms certification number': 'CMS Certification Number',
|
|
39
|
+
'version': 'version',
|
|
40
|
+
'npi': 'type_2_npi',
|
|
41
|
+
'type 2 npi': 'type_2_npi',
|
|
42
|
+
'ein': 'ein',
|
|
43
|
+
'attestation': 'attestation',
|
|
44
|
+
'attester name': 'attester_name',
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def iter_csv(
|
|
49
|
+
source: Union[str, Path, TextIO],
|
|
50
|
+
compression: Optional[str] = None,
|
|
51
|
+
*,
|
|
52
|
+
extra_synonyms: Optional[Mapping[str, str]] = None,
|
|
53
|
+
header_overrides: Optional[Mapping[str, str]] = None,
|
|
54
|
+
code_extraction: Optional[Mapping[str, Any]] = None,
|
|
55
|
+
ref: Optional[ReferenceData] = None,
|
|
56
|
+
stats: Optional[ParseStats] = None,
|
|
57
|
+
) -> Iterator[Any]:
|
|
58
|
+
"""Yield the records in a hospital MRF CSV file.
|
|
59
|
+
|
|
60
|
+
Order: one :class:`FileMetadata` (when the file has a metadata preamble),
|
|
61
|
+
one :class:`HeaderMapping` per column, then charge items, standard
|
|
62
|
+
charges, payer rates and unmapped cells as rows are read.
|
|
63
|
+
|
|
64
|
+
*source* is a path (compression detected unless given) or an open text
|
|
65
|
+
stream. Pass a :class:`ParseStats` to collect counts and warnings.
|
|
66
|
+
"""
|
|
67
|
+
options = dict(extra_synonyms=extra_synonyms, header_overrides=header_overrides,
|
|
68
|
+
code_extraction=code_extraction, ref=ref,
|
|
69
|
+
stats=stats if stats is not None else ParseStats())
|
|
70
|
+
if hasattr(source, 'read'):
|
|
71
|
+
yield from _read(source, **options)
|
|
72
|
+
return
|
|
73
|
+
path = Path(source)
|
|
74
|
+
if compression is None:
|
|
75
|
+
_, compression = detect_file_format(path)
|
|
76
|
+
with open_text(path, compression) as fh:
|
|
77
|
+
yield from _read(fh, **options)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def _read(fh: TextIO, *, stats: ParseStats, **layout_options) -> Iterator[Any]:
|
|
81
|
+
preamble = list(itertools.islice(fh, PREAMBLE_LINES))
|
|
82
|
+
if not preamble:
|
|
83
|
+
raise ValueError("CSV file is empty")
|
|
84
|
+
|
|
85
|
+
header_idx = _find_header_row(preamble)
|
|
86
|
+
delimiter = _sniff_delimiter(preamble[header_idx])
|
|
87
|
+
if delimiter != ',':
|
|
88
|
+
log.info("Detected %s-delimited format", {'|': 'pipe', '\t': 'tab'}[delimiter])
|
|
89
|
+
|
|
90
|
+
metadata = _read_metadata(preamble, header_idx, delimiter)
|
|
91
|
+
if metadata is not None:
|
|
92
|
+
yield metadata
|
|
93
|
+
|
|
94
|
+
sample = list(itertools.islice(fh, QUOTING_SAMPLE_ROWS))
|
|
95
|
+
quoting = _detect_csv_quoting(preamble[header_idx:], sample, delimiter)
|
|
96
|
+
|
|
97
|
+
lines = itertools.chain(preamble[header_idx:], sample, fh)
|
|
98
|
+
reader = csv.DictReader(lines, delimiter=delimiter, quoting=quoting)
|
|
99
|
+
if not reader.fieldnames:
|
|
100
|
+
raise ValueError("CSV has no headers")
|
|
101
|
+
|
|
102
|
+
peeked = []
|
|
103
|
+
for _ in range(PEEK_ROWS):
|
|
104
|
+
try:
|
|
105
|
+
peeked.append(next(reader))
|
|
106
|
+
except StopIteration:
|
|
107
|
+
break
|
|
108
|
+
except csv.Error as exc:
|
|
109
|
+
_skip_row(stats, len(peeked) + 1, exc)
|
|
110
|
+
break
|
|
111
|
+
|
|
112
|
+
layout = TableLayout(reader.fieldnames, peeked, stats=stats, **layout_options)
|
|
113
|
+
yield from layout.header_mappings()
|
|
114
|
+
|
|
115
|
+
rows = itertools.chain(peeked, reader)
|
|
116
|
+
row_number = 0
|
|
117
|
+
while True:
|
|
118
|
+
try:
|
|
119
|
+
row = next(rows)
|
|
120
|
+
except StopIteration:
|
|
121
|
+
break
|
|
122
|
+
except csv.Error as exc:
|
|
123
|
+
row_number += 1
|
|
124
|
+
_skip_row(stats, row_number, exc)
|
|
125
|
+
continue
|
|
126
|
+
row_number += 1
|
|
127
|
+
stats.rows_read += 1
|
|
128
|
+
try:
|
|
129
|
+
yield from layout.row_to_records(row, first_row=row_number == 1, check_sections=True)
|
|
130
|
+
except SectionBoundary:
|
|
131
|
+
# A stacked multi-table file: later sections repeat the data under
|
|
132
|
+
# narrower per-payer headers and cannot be read against this one.
|
|
133
|
+
dropped = sum(1 for _ in _drain(rows))
|
|
134
|
+
stats.rows_skipped += 1
|
|
135
|
+
stats.warn(f"Multi-section CSV: stopped at row ~{row_number} on a later section "
|
|
136
|
+
f"header; dropped ~{dropped} trailing rows")
|
|
137
|
+
break
|
|
138
|
+
|
|
139
|
+
if stats.rows_skipped:
|
|
140
|
+
stats.warnings.insert(0, f"Skipped {stats.rows_skipped} CSV rows due to parse errors "
|
|
141
|
+
f"(e.g. oversized fields)")
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def _drain(rows):
|
|
145
|
+
while True:
|
|
146
|
+
try:
|
|
147
|
+
yield next(rows)
|
|
148
|
+
except StopIteration:
|
|
149
|
+
return
|
|
150
|
+
except csv.Error:
|
|
151
|
+
continue
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
def _skip_row(stats: ParseStats, row_number: int, exc: Exception) -> None:
|
|
155
|
+
stats.rows_skipped += 1
|
|
156
|
+
if stats.rows_skipped <= _WARN_SKIPPED_ROWS:
|
|
157
|
+
stats.warn(f"Skipped CSV row ~{row_number}: {exc}")
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
# ---------------------------------------------------------------------------
|
|
161
|
+
# Header row, delimiter, quoting
|
|
162
|
+
# ---------------------------------------------------------------------------
|
|
163
|
+
|
|
164
|
+
def _find_header_row(preamble: List[str]) -> int:
|
|
165
|
+
"""Return the index of the data header among the first lines.
|
|
166
|
+
|
|
167
|
+
Candidates mention ``description`` or ``code|``. Some files glue the data
|
|
168
|
+
header onto the end of the metadata value row, so among candidates the
|
|
169
|
+
one whose fields map best wins, then the shorter line (the clean header
|
|
170
|
+
is a suffix of the glued one), then the earliest.
|
|
171
|
+
"""
|
|
172
|
+
def score(line: str):
|
|
173
|
+
try:
|
|
174
|
+
fields = next(csv.reader(io.StringIO(line)))
|
|
175
|
+
except (csv.Error, StopIteration):
|
|
176
|
+
return (-1, 0)
|
|
177
|
+
mapped = sum(1 for f in fields if f.strip() and map_header(f)[1])
|
|
178
|
+
return (mapped, -len(fields))
|
|
179
|
+
|
|
180
|
+
candidates = [i for i, line in enumerate(preamble)
|
|
181
|
+
if 'description' in line.lower() or 'code|' in line.lower()]
|
|
182
|
+
if not candidates:
|
|
183
|
+
log.warning("Could not detect the data header row, using the first row")
|
|
184
|
+
return 0
|
|
185
|
+
return max(candidates, key=lambda i: score(preamble[i]))
|
|
186
|
+
|
|
187
|
+
|
|
188
|
+
def _sniff_delimiter(header_line: str) -> str:
|
|
189
|
+
"""Comma unless the header splits into fewer than 3 fields on commas.
|
|
190
|
+
|
|
191
|
+
CMS wide headers (``standard_charge|AETNA|PPO|negotiated_dollar``) are
|
|
192
|
+
full of pipes, so counting pipes against commas would pick the pipe.
|
|
193
|
+
"""
|
|
194
|
+
if len(next(csv.reader(io.StringIO(header_line)))) >= 3:
|
|
195
|
+
return ','
|
|
196
|
+
pipes, tabs = header_line.count('|'), header_line.count('\t')
|
|
197
|
+
if pipes > tabs and pipes > 0:
|
|
198
|
+
return '|'
|
|
199
|
+
if tabs > 0:
|
|
200
|
+
return '\t'
|
|
201
|
+
return ','
|
|
202
|
+
|
|
203
|
+
|
|
204
|
+
def _score_csv_quoting_mode(header_lines, sample_lines, delimiter, quoting) -> Dict[str, int]:
|
|
205
|
+
"""Parse a sample under *quoting* and count the symptoms of a bad choice.
|
|
206
|
+
|
|
207
|
+
Lower is better: rows whose width does not match the header, fields that
|
|
208
|
+
swallowed a newline (quote fusion), and code_type values that are not a
|
|
209
|
+
code-type shape (the classic sign of a column shift).
|
|
210
|
+
"""
|
|
211
|
+
buf = io.StringIO(''.join(header_lines) + ''.join(sample_lines))
|
|
212
|
+
try:
|
|
213
|
+
reader = csv.DictReader(buf, delimiter=delimiter, quoting=quoting)
|
|
214
|
+
fieldnames = reader.fieldnames or []
|
|
215
|
+
except csv.Error:
|
|
216
|
+
return {'length_mismatches': 10**9, 'malformed_code_types': 10**9,
|
|
217
|
+
'embedded_newlines': 0, 'rows_seen': 0, 'expected_cols': 0}
|
|
218
|
+
ct_col = next((h for h in fieldnames if map_header(h)[1] == 'code_type'), None)
|
|
219
|
+
scores = {'length_mismatches': 0, 'malformed_code_types': 0, 'embedded_newlines': 0,
|
|
220
|
+
'rows_seen': 0, 'expected_cols': len(fieldnames)}
|
|
221
|
+
try:
|
|
222
|
+
for row in reader:
|
|
223
|
+
scores['rows_seen'] += 1
|
|
224
|
+
# DictReader puts overflow under a None key and fills underflow with None.
|
|
225
|
+
if None in row or any(row.get(h) is None for h in fieldnames):
|
|
226
|
+
scores['length_mismatches'] += 1
|
|
227
|
+
# MRF fields never legitimately span lines.
|
|
228
|
+
if any(isinstance(v, str) and '\n' in v for v in row.values()):
|
|
229
|
+
scores['embedded_newlines'] += 1
|
|
230
|
+
if ct_col:
|
|
231
|
+
ct = (row.get(ct_col) or '').strip()
|
|
232
|
+
if ct and (len(ct) > _MAX_CODE_TYPE_LEN
|
|
233
|
+
or not _RE_VALID_CODE_TYPE_SHAPE.fullmatch(ct.upper())):
|
|
234
|
+
scores['malformed_code_types'] += 1
|
|
235
|
+
except csv.Error:
|
|
236
|
+
scores['length_mismatches'] += 10**6
|
|
237
|
+
return scores
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _detect_csv_quoting(header_lines, sample_lines, delimiter) -> int:
|
|
241
|
+
"""Pick QUOTE_MINIMAL or QUOTE_NONE for this file.
|
|
242
|
+
|
|
243
|
+
A stray ``"`` at the start of a field (``|"needle 6 inch``) makes
|
|
244
|
+
QUOTE_MINIMAL swallow delimiters and newlines until the next quote,
|
|
245
|
+
fusing rows. Parse a sample both ways and keep the cleaner one; ties go
|
|
246
|
+
to QUOTE_MINIMAL.
|
|
247
|
+
"""
|
|
248
|
+
if not sample_lines:
|
|
249
|
+
return csv.QUOTE_MINIMAL
|
|
250
|
+
|
|
251
|
+
def total(s):
|
|
252
|
+
return s['length_mismatches'] * 10 + s['embedded_newlines'] * 10 + s['malformed_code_types']
|
|
253
|
+
|
|
254
|
+
minimal = total(_score_csv_quoting_mode(header_lines, sample_lines, delimiter, csv.QUOTE_MINIMAL))
|
|
255
|
+
none = total(_score_csv_quoting_mode(header_lines, sample_lines, delimiter, csv.QUOTE_NONE))
|
|
256
|
+
if none < minimal:
|
|
257
|
+
log.info("CSV quoting: QUOTE_NONE (minimal=%d, none=%d)", minimal, none)
|
|
258
|
+
return csv.QUOTE_NONE
|
|
259
|
+
return csv.QUOTE_MINIMAL
|
|
260
|
+
|
|
261
|
+
|
|
262
|
+
# ---------------------------------------------------------------------------
|
|
263
|
+
# Metadata preamble
|
|
264
|
+
# ---------------------------------------------------------------------------
|
|
265
|
+
|
|
266
|
+
def _read_metadata(preamble: List[str], header_idx: int, delimiter: str) -> Optional[FileMetadata]:
|
|
267
|
+
meta = _try_parse_kv_preamble(preamble, header_idx)
|
|
268
|
+
if meta is None and header_idx >= 2:
|
|
269
|
+
rows = list(csv.reader(io.StringIO(''.join(preamble[:header_idx])), delimiter=delimiter))
|
|
270
|
+
if len(rows) >= 2:
|
|
271
|
+
keys, values = rows[0], rows[1]
|
|
272
|
+
meta = {keys[i].strip().lstrip(''): values[i].strip()
|
|
273
|
+
for i in range(min(len(keys), len(values))) if keys[i].strip()}
|
|
274
|
+
if meta is None:
|
|
275
|
+
return None
|
|
276
|
+
|
|
277
|
+
def get(key):
|
|
278
|
+
value = (meta.get(key) or '').strip()
|
|
279
|
+
if value.startswith('"') and value.endswith('"'):
|
|
280
|
+
value = value[1:-1].strip()
|
|
281
|
+
return value or None
|
|
282
|
+
|
|
283
|
+
license_number, license_state = None, None
|
|
284
|
+
for key, value in meta.items():
|
|
285
|
+
if key.lower().strip().startswith('license_number'):
|
|
286
|
+
# The header carries the state: license_number|CA
|
|
287
|
+
if '|' in key:
|
|
288
|
+
state = key.split('|', 1)[1].strip().upper()
|
|
289
|
+
if len(state) > 2:
|
|
290
|
+
state = _STATE_NAME_TO_ABBR.get(state, state[:2])
|
|
291
|
+
license_state = state if state in _US_STATES else None
|
|
292
|
+
license_number = value.split('|')[0].strip().strip('"').strip() or None
|
|
293
|
+
break
|
|
294
|
+
|
|
295
|
+
return FileMetadata(
|
|
296
|
+
hospital_name=get('hospital_name'),
|
|
297
|
+
hospital_location=get('hospital_location'),
|
|
298
|
+
hospital_address=get('hospital_address'),
|
|
299
|
+
cms_certification_number=get('CMS Certification Number'),
|
|
300
|
+
license_number=license_number,
|
|
301
|
+
license_state=license_state,
|
|
302
|
+
type_2_npi=_sanitize_npi((meta.get('type_2_npi') or '').strip()),
|
|
303
|
+
ein=get('ein'),
|
|
304
|
+
last_updated_on=get('last_updated_on'),
|
|
305
|
+
version=get('version'),
|
|
306
|
+
attestation=get('attestation'),
|
|
307
|
+
attester_name=get('attester_name'),
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
|
|
311
|
+
def _try_parse_kv_preamble(preamble_lines: List[str], header_row_idx: int) -> Optional[Dict[str, str]]:
|
|
312
|
+
"""Parse "Hospital Name: Acme,,,," style lines, or return None."""
|
|
313
|
+
if header_row_idx < 1:
|
|
314
|
+
return None
|
|
315
|
+
meta = {}
|
|
316
|
+
for line in preamble_lines[:header_row_idx]:
|
|
317
|
+
stripped = line.strip().rstrip(',').strip()
|
|
318
|
+
if ':' not in stripped:
|
|
319
|
+
continue
|
|
320
|
+
key, _, value = stripped.partition(':')
|
|
321
|
+
canonical = _KV_PREAMBLE_KEYS.get(key.strip().lower().lstrip(''))
|
|
322
|
+
if canonical:
|
|
323
|
+
meta[canonical] = value.strip().rstrip(',').strip()
|
|
324
|
+
return meta or None
|