mrfkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mrfkit/csv_reader.py ADDED
@@ -0,0 +1,324 @@
1
+ """Read hospital MRF CSV files into records."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ import io
7
+ import itertools
8
+ import logging
9
+ from pathlib import Path
10
+ from typing import Any, Dict, Iterator, List, Mapping, Optional, TextIO, Union
11
+
12
+ from .codes import _MAX_CODE_TYPE_LEN, _RE_VALID_CODE_TYPE_SHAPE
13
+ from .files import detect_file_format, open_text
14
+ from .headers import map_header
15
+ from .records import FileMetadata
16
+ from .reference import ParseStats, ReferenceData
17
+ from .tabular import PEEK_ROWS, SectionBoundary, TableLayout
18
+ from .values import _STATE_NAME_TO_ABBR, _US_STATES, _sanitize_npi
19
+
20
+ log = logging.getLogger(__name__)
21
+
22
+ # Real MRF cells (long notes, stuffed code lists) exceed the csv default of
23
+ # 128 KB. Raised once at import, so a caller can still lower it.
24
+ FIELD_SIZE_LIMIT = 10 * 1024 * 1024
25
+ csv.field_size_limit(max(csv.field_size_limit(), FIELD_SIZE_LIMIT))
26
+
27
+ PREAMBLE_LINES = 10
28
+ QUOTING_SAMPLE_ROWS = 200
29
+ _WARN_SKIPPED_ROWS = 5
30
+
31
+ # "Key: Value" preamble lines, as some non-CMS files write their metadata.
32
+ _KV_PREAMBLE_KEYS = {
33
+ 'hospital name': 'hospital_name',
34
+ 'hospital location': 'hospital_location',
35
+ 'hospital address': 'hospital_address',
36
+ 'price effective date': 'last_updated_on',
37
+ 'last updated on': 'last_updated_on',
38
+ 'cms certification number': 'CMS Certification Number',
39
+ 'version': 'version',
40
+ 'npi': 'type_2_npi',
41
+ 'type 2 npi': 'type_2_npi',
42
+ 'ein': 'ein',
43
+ 'attestation': 'attestation',
44
+ 'attester name': 'attester_name',
45
+ }
46
+
47
+
48
+ def iter_csv(
49
+ source: Union[str, Path, TextIO],
50
+ compression: Optional[str] = None,
51
+ *,
52
+ extra_synonyms: Optional[Mapping[str, str]] = None,
53
+ header_overrides: Optional[Mapping[str, str]] = None,
54
+ code_extraction: Optional[Mapping[str, Any]] = None,
55
+ ref: Optional[ReferenceData] = None,
56
+ stats: Optional[ParseStats] = None,
57
+ ) -> Iterator[Any]:
58
+ """Yield the records in a hospital MRF CSV file.
59
+
60
+ Order: one :class:`FileMetadata` (when the file has a metadata preamble),
61
+ one :class:`HeaderMapping` per column, then charge items, standard
62
+ charges, payer rates and unmapped cells as rows are read.
63
+
64
+ *source* is a path (compression detected unless given) or an open text
65
+ stream. Pass a :class:`ParseStats` to collect counts and warnings.
66
+ """
67
+ options = dict(extra_synonyms=extra_synonyms, header_overrides=header_overrides,
68
+ code_extraction=code_extraction, ref=ref,
69
+ stats=stats if stats is not None else ParseStats())
70
+ if hasattr(source, 'read'):
71
+ yield from _read(source, **options)
72
+ return
73
+ path = Path(source)
74
+ if compression is None:
75
+ _, compression = detect_file_format(path)
76
+ with open_text(path, compression) as fh:
77
+ yield from _read(fh, **options)
78
+
79
+
80
+ def _read(fh: TextIO, *, stats: ParseStats, **layout_options) -> Iterator[Any]:
81
+ preamble = list(itertools.islice(fh, PREAMBLE_LINES))
82
+ if not preamble:
83
+ raise ValueError("CSV file is empty")
84
+
85
+ header_idx = _find_header_row(preamble)
86
+ delimiter = _sniff_delimiter(preamble[header_idx])
87
+ if delimiter != ',':
88
+ log.info("Detected %s-delimited format", {'|': 'pipe', '\t': 'tab'}[delimiter])
89
+
90
+ metadata = _read_metadata(preamble, header_idx, delimiter)
91
+ if metadata is not None:
92
+ yield metadata
93
+
94
+ sample = list(itertools.islice(fh, QUOTING_SAMPLE_ROWS))
95
+ quoting = _detect_csv_quoting(preamble[header_idx:], sample, delimiter)
96
+
97
+ lines = itertools.chain(preamble[header_idx:], sample, fh)
98
+ reader = csv.DictReader(lines, delimiter=delimiter, quoting=quoting)
99
+ if not reader.fieldnames:
100
+ raise ValueError("CSV has no headers")
101
+
102
+ peeked = []
103
+ for _ in range(PEEK_ROWS):
104
+ try:
105
+ peeked.append(next(reader))
106
+ except StopIteration:
107
+ break
108
+ except csv.Error as exc:
109
+ _skip_row(stats, len(peeked) + 1, exc)
110
+ break
111
+
112
+ layout = TableLayout(reader.fieldnames, peeked, stats=stats, **layout_options)
113
+ yield from layout.header_mappings()
114
+
115
+ rows = itertools.chain(peeked, reader)
116
+ row_number = 0
117
+ while True:
118
+ try:
119
+ row = next(rows)
120
+ except StopIteration:
121
+ break
122
+ except csv.Error as exc:
123
+ row_number += 1
124
+ _skip_row(stats, row_number, exc)
125
+ continue
126
+ row_number += 1
127
+ stats.rows_read += 1
128
+ try:
129
+ yield from layout.row_to_records(row, first_row=row_number == 1, check_sections=True)
130
+ except SectionBoundary:
131
+ # A stacked multi-table file: later sections repeat the data under
132
+ # narrower per-payer headers and cannot be read against this one.
133
+ dropped = sum(1 for _ in _drain(rows))
134
+ stats.rows_skipped += 1
135
+ stats.warn(f"Multi-section CSV: stopped at row ~{row_number} on a later section "
136
+ f"header; dropped ~{dropped} trailing rows")
137
+ break
138
+
139
+ if stats.rows_skipped:
140
+ stats.warnings.insert(0, f"Skipped {stats.rows_skipped} CSV rows due to parse errors "
141
+ f"(e.g. oversized fields)")
142
+
143
+
144
+ def _drain(rows):
145
+ while True:
146
+ try:
147
+ yield next(rows)
148
+ except StopIteration:
149
+ return
150
+ except csv.Error:
151
+ continue
152
+
153
+
154
+ def _skip_row(stats: ParseStats, row_number: int, exc: Exception) -> None:
155
+ stats.rows_skipped += 1
156
+ if stats.rows_skipped <= _WARN_SKIPPED_ROWS:
157
+ stats.warn(f"Skipped CSV row ~{row_number}: {exc}")
158
+
159
+
160
+ # ---------------------------------------------------------------------------
161
+ # Header row, delimiter, quoting
162
+ # ---------------------------------------------------------------------------
163
+
164
+ def _find_header_row(preamble: List[str]) -> int:
165
+ """Return the index of the data header among the first lines.
166
+
167
+ Candidates mention ``description`` or ``code|``. Some files glue the data
168
+ header onto the end of the metadata value row, so among candidates the
169
+ one whose fields map best wins, then the shorter line (the clean header
170
+ is a suffix of the glued one), then the earliest.
171
+ """
172
+ def score(line: str):
173
+ try:
174
+ fields = next(csv.reader(io.StringIO(line)))
175
+ except (csv.Error, StopIteration):
176
+ return (-1, 0)
177
+ mapped = sum(1 for f in fields if f.strip() and map_header(f)[1])
178
+ return (mapped, -len(fields))
179
+
180
+ candidates = [i for i, line in enumerate(preamble)
181
+ if 'description' in line.lower() or 'code|' in line.lower()]
182
+ if not candidates:
183
+ log.warning("Could not detect the data header row, using the first row")
184
+ return 0
185
+ return max(candidates, key=lambda i: score(preamble[i]))
186
+
187
+
188
+ def _sniff_delimiter(header_line: str) -> str:
189
+ """Comma unless the header splits into fewer than 3 fields on commas.
190
+
191
+ CMS wide headers (``standard_charge|AETNA|PPO|negotiated_dollar``) are
192
+ full of pipes, so counting pipes against commas would pick the pipe.
193
+ """
194
+ if len(next(csv.reader(io.StringIO(header_line)))) >= 3:
195
+ return ','
196
+ pipes, tabs = header_line.count('|'), header_line.count('\t')
197
+ if pipes > tabs and pipes > 0:
198
+ return '|'
199
+ if tabs > 0:
200
+ return '\t'
201
+ return ','
202
+
203
+
204
+ def _score_csv_quoting_mode(header_lines, sample_lines, delimiter, quoting) -> Dict[str, int]:
205
+ """Parse a sample under *quoting* and count the symptoms of a bad choice.
206
+
207
+ Lower is better: rows whose width does not match the header, fields that
208
+ swallowed a newline (quote fusion), and code_type values that are not a
209
+ code-type shape (the classic sign of a column shift).
210
+ """
211
+ buf = io.StringIO(''.join(header_lines) + ''.join(sample_lines))
212
+ try:
213
+ reader = csv.DictReader(buf, delimiter=delimiter, quoting=quoting)
214
+ fieldnames = reader.fieldnames or []
215
+ except csv.Error:
216
+ return {'length_mismatches': 10**9, 'malformed_code_types': 10**9,
217
+ 'embedded_newlines': 0, 'rows_seen': 0, 'expected_cols': 0}
218
+ ct_col = next((h for h in fieldnames if map_header(h)[1] == 'code_type'), None)
219
+ scores = {'length_mismatches': 0, 'malformed_code_types': 0, 'embedded_newlines': 0,
220
+ 'rows_seen': 0, 'expected_cols': len(fieldnames)}
221
+ try:
222
+ for row in reader:
223
+ scores['rows_seen'] += 1
224
+ # DictReader puts overflow under a None key and fills underflow with None.
225
+ if None in row or any(row.get(h) is None for h in fieldnames):
226
+ scores['length_mismatches'] += 1
227
+ # MRF fields never legitimately span lines.
228
+ if any(isinstance(v, str) and '\n' in v for v in row.values()):
229
+ scores['embedded_newlines'] += 1
230
+ if ct_col:
231
+ ct = (row.get(ct_col) or '').strip()
232
+ if ct and (len(ct) > _MAX_CODE_TYPE_LEN
233
+ or not _RE_VALID_CODE_TYPE_SHAPE.fullmatch(ct.upper())):
234
+ scores['malformed_code_types'] += 1
235
+ except csv.Error:
236
+ scores['length_mismatches'] += 10**6
237
+ return scores
238
+
239
+
240
+ def _detect_csv_quoting(header_lines, sample_lines, delimiter) -> int:
241
+ """Pick QUOTE_MINIMAL or QUOTE_NONE for this file.
242
+
243
+ A stray ``"`` at the start of a field (``|"needle 6 inch``) makes
244
+ QUOTE_MINIMAL swallow delimiters and newlines until the next quote,
245
+ fusing rows. Parse a sample both ways and keep the cleaner one; ties go
246
+ to QUOTE_MINIMAL.
247
+ """
248
+ if not sample_lines:
249
+ return csv.QUOTE_MINIMAL
250
+
251
+ def total(s):
252
+ return s['length_mismatches'] * 10 + s['embedded_newlines'] * 10 + s['malformed_code_types']
253
+
254
+ minimal = total(_score_csv_quoting_mode(header_lines, sample_lines, delimiter, csv.QUOTE_MINIMAL))
255
+ none = total(_score_csv_quoting_mode(header_lines, sample_lines, delimiter, csv.QUOTE_NONE))
256
+ if none < minimal:
257
+ log.info("CSV quoting: QUOTE_NONE (minimal=%d, none=%d)", minimal, none)
258
+ return csv.QUOTE_NONE
259
+ return csv.QUOTE_MINIMAL
260
+
261
+
262
+ # ---------------------------------------------------------------------------
263
+ # Metadata preamble
264
+ # ---------------------------------------------------------------------------
265
+
266
+ def _read_metadata(preamble: List[str], header_idx: int, delimiter: str) -> Optional[FileMetadata]:
267
+ meta = _try_parse_kv_preamble(preamble, header_idx)
268
+ if meta is None and header_idx >= 2:
269
+ rows = list(csv.reader(io.StringIO(''.join(preamble[:header_idx])), delimiter=delimiter))
270
+ if len(rows) >= 2:
271
+ keys, values = rows[0], rows[1]
272
+ meta = {keys[i].strip().lstrip(''): values[i].strip()
273
+ for i in range(min(len(keys), len(values))) if keys[i].strip()}
274
+ if meta is None:
275
+ return None
276
+
277
+ def get(key):
278
+ value = (meta.get(key) or '').strip()
279
+ if value.startswith('"') and value.endswith('"'):
280
+ value = value[1:-1].strip()
281
+ return value or None
282
+
283
+ license_number, license_state = None, None
284
+ for key, value in meta.items():
285
+ if key.lower().strip().startswith('license_number'):
286
+ # The header carries the state: license_number|CA
287
+ if '|' in key:
288
+ state = key.split('|', 1)[1].strip().upper()
289
+ if len(state) > 2:
290
+ state = _STATE_NAME_TO_ABBR.get(state, state[:2])
291
+ license_state = state if state in _US_STATES else None
292
+ license_number = value.split('|')[0].strip().strip('"').strip() or None
293
+ break
294
+
295
+ return FileMetadata(
296
+ hospital_name=get('hospital_name'),
297
+ hospital_location=get('hospital_location'),
298
+ hospital_address=get('hospital_address'),
299
+ cms_certification_number=get('CMS Certification Number'),
300
+ license_number=license_number,
301
+ license_state=license_state,
302
+ type_2_npi=_sanitize_npi((meta.get('type_2_npi') or '').strip()),
303
+ ein=get('ein'),
304
+ last_updated_on=get('last_updated_on'),
305
+ version=get('version'),
306
+ attestation=get('attestation'),
307
+ attester_name=get('attester_name'),
308
+ )
309
+
310
+
311
+ def _try_parse_kv_preamble(preamble_lines: List[str], header_row_idx: int) -> Optional[Dict[str, str]]:
312
+ """Parse "Hospital Name: Acme,,,," style lines, or return None."""
313
+ if header_row_idx < 1:
314
+ return None
315
+ meta = {}
316
+ for line in preamble_lines[:header_row_idx]:
317
+ stripped = line.strip().rstrip(',').strip()
318
+ if ':' not in stripped:
319
+ continue
320
+ key, _, value = stripped.partition(':')
321
+ canonical = _KV_PREAMBLE_KEYS.get(key.strip().lower().lstrip(''))
322
+ if canonical:
323
+ meta[canonical] = value.strip().rstrip(',').strip()
324
+ return meta or None