mrfkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mrfkit/__init__.py +68 -0
- mrfkit/__main__.py +3 -0
- mrfkit/cli.py +79 -0
- mrfkit/codes.py +1607 -0
- mrfkit/csv_reader.py +324 -0
- mrfkit/files.py +651 -0
- mrfkit/headers.py +737 -0
- mrfkit/json_reader.py +603 -0
- mrfkit/payers.py +2755 -0
- mrfkit/records.py +260 -0
- mrfkit/reference.py +136 -0
- mrfkit/sinks.py +126 -0
- mrfkit/tabular.py +651 -0
- mrfkit/tic.py +660 -0
- mrfkit/values.py +627 -0
- mrfkit-0.1.0.dist-info/METADATA +136 -0
- mrfkit-0.1.0.dist-info/RECORD +21 -0
- mrfkit-0.1.0.dist-info/WHEEL +4 -0
- mrfkit-0.1.0.dist-info/entry_points.txt +2 -0
- mrfkit-0.1.0.dist-info/licenses/LICENSE +201 -0
- mrfkit-0.1.0.dist-info/licenses/NOTICE +4 -0
mrfkit/files.py
ADDED
|
@@ -0,0 +1,651 @@
|
|
|
1
|
+
"""Open MRF files: detect the format, decompress, and repair broken text.
|
|
2
|
+
|
|
3
|
+
Hospital files arrive as plain, gzip or zip, sometimes with a wrong or missing
|
|
4
|
+
extension, sometimes in UTF-16, often with stray Windows-1252 bytes, and the
|
|
5
|
+
JSON ones regularly carry double or trailing commas. Everything here streams:
|
|
6
|
+
memory stays flat no matter how big the file is.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
import codecs
|
|
12
|
+
import copy
|
|
13
|
+
import gzip
|
|
14
|
+
import io
|
|
15
|
+
import logging
|
|
16
|
+
import zipfile
|
|
17
|
+
import zlib
|
|
18
|
+
from contextlib import contextmanager
|
|
19
|
+
from pathlib import Path
|
|
20
|
+
from typing import BinaryIO, Iterator, List, Optional, TextIO, Tuple
|
|
21
|
+
|
|
22
|
+
import inflate64
|
|
23
|
+
|
|
24
|
+
log = logging.getLogger(__name__)
|
|
25
|
+
|
|
26
|
+
# Zip method 9. The stdlib zipfile cannot read it, and Windows uses it for
|
|
27
|
+
# large archives, so real MRFs show up compressed this way.
|
|
28
|
+
ZIP_DEFLATE64 = 9
|
|
29
|
+
|
|
30
|
+
_UTF8_BOM = b'\xef\xbb\xbf'
|
|
31
|
+
_UTF16_BOMS = (b'\xff\xfe', b'\xfe\xff')
|
|
32
|
+
_SNIFF_BYTES = 4096
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
# ---------------------------------------------------------------------------
|
|
36
|
+
# Format detection
|
|
37
|
+
# ---------------------------------------------------------------------------
|
|
38
|
+
|
|
39
|
+
def detect_file_format(path: Path) -> Tuple[str, Optional[str]]:
|
|
40
|
+
"""Return ``(file_format, compression)`` for *path*.
|
|
41
|
+
|
|
42
|
+
*file_format* is ``'csv'`` or ``'json'``; *compression* is ``'gz'``,
|
|
43
|
+
``'zip'`` or ``None``.
|
|
44
|
+
|
|
45
|
+
The extension decides first. Plain ``.csv`` and ``.json`` names are
|
|
46
|
+
checked against the content, because misnamed files are common. Anything
|
|
47
|
+
else (API downloads often have no extension) is sniffed.
|
|
48
|
+
|
|
49
|
+
Raises ``ValueError`` for Excel files and ambiguous zips, and
|
|
50
|
+
``zipfile.BadZipFile`` for corrupt zips.
|
|
51
|
+
"""
|
|
52
|
+
path = Path(path)
|
|
53
|
+
name = path.name.lower()
|
|
54
|
+
|
|
55
|
+
if looks_like_html(path):
|
|
56
|
+
raise ValueError(
|
|
57
|
+
f"{path.name} is an HTML page, not an MRF: the download probably hit "
|
|
58
|
+
f"an error page or a bot challenge.")
|
|
59
|
+
if name.endswith(('.xlsx', '.xls')):
|
|
60
|
+
raise ValueError(
|
|
61
|
+
f"Excel/XLSX files are not supported: {path.name}. "
|
|
62
|
+
f"Please provide the CSV or JSON version of this MRF."
|
|
63
|
+
)
|
|
64
|
+
if name.endswith('.zip'):
|
|
65
|
+
_reject_if_xlsx(path)
|
|
66
|
+
return detect_format_from_zip(path), 'zip'
|
|
67
|
+
if name.endswith('.csv.gz'):
|
|
68
|
+
return 'csv', 'gz'
|
|
69
|
+
if name.endswith('.json.gz'):
|
|
70
|
+
return 'json', 'gz'
|
|
71
|
+
if name.endswith(('.csv', '.json')):
|
|
72
|
+
ext_format = 'csv' if name.endswith('.csv') else 'json'
|
|
73
|
+
try:
|
|
74
|
+
content_format, content_compression = _detect_format_from_content(path)
|
|
75
|
+
except ValueError:
|
|
76
|
+
# Empty file: nothing to contradict the extension.
|
|
77
|
+
return ext_format, None
|
|
78
|
+
if content_compression is not None:
|
|
79
|
+
# Compressed despite a plain extension: trust the content.
|
|
80
|
+
return content_format, content_compression
|
|
81
|
+
if content_format != ext_format:
|
|
82
|
+
log.warning(
|
|
83
|
+
"Extension says %s but content looks like %s, trusting content: %s",
|
|
84
|
+
ext_format, content_format, path.name,
|
|
85
|
+
)
|
|
86
|
+
return content_format, None
|
|
87
|
+
return ext_format, None
|
|
88
|
+
return _detect_format_from_content(path)
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
# Signs of an HTML error page saved under an MRF name: ASP.NET download
|
|
92
|
+
# stubs, CDN 403/404 pages, Cloudflare challenges.
|
|
93
|
+
_HTML_MARKERS = (b"<!doctype html", b"<html", b"<head", b"attention required", b"__viewstate")
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def looks_like_html(path: Path) -> bool:
|
|
97
|
+
"""True when the first 512 bytes of *path* look like an HTML page.
|
|
98
|
+
|
|
99
|
+
Compressed files and anything starting with ``{`` or ``[`` never match,
|
|
100
|
+
so a real MRF is not flagged.
|
|
101
|
+
"""
|
|
102
|
+
try:
|
|
103
|
+
with open(path, 'rb') as fh:
|
|
104
|
+
head = fh.read(512)
|
|
105
|
+
except OSError:
|
|
106
|
+
return False
|
|
107
|
+
if not head or head[:2] in (b'\x1f\x8b', b'PK'):
|
|
108
|
+
return False
|
|
109
|
+
stripped = head.removeprefix(_UTF8_BOM).lstrip()
|
|
110
|
+
if stripped.startswith((b'{', b'[')):
|
|
111
|
+
return False
|
|
112
|
+
lowered = stripped.lower()
|
|
113
|
+
return any(marker in lowered for marker in _HTML_MARKERS)
|
|
114
|
+
|
|
115
|
+
|
|
116
|
+
def _detect_format_from_content(path: Path) -> Tuple[str, Optional[str]]:
|
|
117
|
+
"""Detect the format from the first bytes of *path*.
|
|
118
|
+
|
|
119
|
+
Gzip and zip are recognized by magic bytes. Text whose first
|
|
120
|
+
non-whitespace byte is ``{`` or ``[`` is JSON, anything else is CSV.
|
|
121
|
+
UTF-16 text is transcoded before sniffing. Raises ``ValueError`` for an
|
|
122
|
+
empty file.
|
|
123
|
+
"""
|
|
124
|
+
with open(path, 'rb') as f:
|
|
125
|
+
head = f.read(_SNIFF_BYTES)
|
|
126
|
+
|
|
127
|
+
if not head:
|
|
128
|
+
raise ValueError(f"Cannot detect format: file is empty: {path.name}")
|
|
129
|
+
|
|
130
|
+
if head[:4] == b'PK\x03\x04':
|
|
131
|
+
_reject_if_xlsx(path)
|
|
132
|
+
return detect_format_from_zip(path), 'zip'
|
|
133
|
+
|
|
134
|
+
if head[:2] == b'\x1f\x8b':
|
|
135
|
+
with gzip.open(path, 'rb') as gz:
|
|
136
|
+
inner_head = gz.read(_SNIFF_BYTES)
|
|
137
|
+
return _sniff_text(_transcode_to_utf8(inner_head)), 'gz'
|
|
138
|
+
|
|
139
|
+
return _sniff_text(_transcode_to_utf8(head)), None
|
|
140
|
+
|
|
141
|
+
|
|
142
|
+
def _sniff_text(head: bytes) -> str:
|
|
143
|
+
stripped = head.lstrip(b' \t\r\n' + _UTF8_BOM)
|
|
144
|
+
return 'json' if stripped[:1] in (b'{', b'[') else 'csv'
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
def validate_zip_integrity(zip_path: Path) -> None:
|
|
148
|
+
"""Raise ``zipfile.BadZipFile`` if the zip is corrupt or truncated.
|
|
149
|
+
|
|
150
|
+
Reads every data member in full and checks its CRC, so a damaged archive
|
|
151
|
+
fails here with a clear error instead of halfway through a parse.
|
|
152
|
+
"""
|
|
153
|
+
try:
|
|
154
|
+
with zipfile.ZipFile(zip_path, 'r') as zf:
|
|
155
|
+
for name in _zip_data_members(zf):
|
|
156
|
+
with _open_zip_member(zf, name) as member:
|
|
157
|
+
while member.read(1 << 20):
|
|
158
|
+
pass
|
|
159
|
+
except (zlib.error, EOFError) as exc:
|
|
160
|
+
raise zipfile.BadZipFile(
|
|
161
|
+
f"ZIP decompression error ({Path(zip_path).name}): {exc}") from exc
|
|
162
|
+
|
|
163
|
+
|
|
164
|
+
def _reject_if_xlsx(zip_path: Path) -> None:
|
|
165
|
+
"""Raise ``ValueError`` if the zip is really an XLSX workbook."""
|
|
166
|
+
try:
|
|
167
|
+
with zipfile.ZipFile(zip_path, 'r') as zf:
|
|
168
|
+
lower_names = [n.lower() for n in zf.namelist()]
|
|
169
|
+
except zipfile.BadZipFile:
|
|
170
|
+
return # Not a valid zip: let the caller report that.
|
|
171
|
+
if '[content_types].xml' in lower_names or any(n.startswith('xl/') for n in lower_names):
|
|
172
|
+
raise ValueError(
|
|
173
|
+
f"Excel/XLSX files are not supported: {Path(zip_path).name}. "
|
|
174
|
+
f"Please provide the CSV or JSON version of this MRF."
|
|
175
|
+
)
|
|
176
|
+
|
|
177
|
+
|
|
178
|
+
def _zip_data_members(zf: zipfile.ZipFile) -> List[str]:
|
|
179
|
+
"""Return the real data members, skipping directories and macOS sidecars.
|
|
180
|
+
|
|
181
|
+
Finder-made zips add ``__MACOSX/._name`` resource forks and sometimes a
|
|
182
|
+
``.DS_Store``. Counting those would make a one-file archive look like
|
|
183
|
+
three.
|
|
184
|
+
"""
|
|
185
|
+
members = []
|
|
186
|
+
for info in zf.infolist():
|
|
187
|
+
if info.is_dir():
|
|
188
|
+
continue
|
|
189
|
+
base = info.filename.rsplit('/', 1)[-1]
|
|
190
|
+
if info.filename.startswith('__MACOSX/') or base == '.DS_Store' or base.startswith('._'):
|
|
191
|
+
continue
|
|
192
|
+
members.append(info.filename)
|
|
193
|
+
return members
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
def _zip_single_member(zf: zipfile.ZipFile, zip_name: str = "") -> str:
|
|
197
|
+
"""Return the one data member of *zf*, or raise ``ValueError``."""
|
|
198
|
+
members = _zip_data_members(zf)
|
|
199
|
+
if len(members) != 1:
|
|
200
|
+
suffix = f" in {zip_name}" if zip_name else ""
|
|
201
|
+
detail = f": {members}" if members else ""
|
|
202
|
+
raise ValueError(
|
|
203
|
+
f"ZIP must contain exactly one file, found {len(members)}{suffix}{detail}")
|
|
204
|
+
return members[0]
|
|
205
|
+
|
|
206
|
+
|
|
207
|
+
def detect_format_from_zip(zip_path: Path) -> str:
|
|
208
|
+
"""Return ``'csv'`` or ``'json'`` for the single data file inside a zip.
|
|
209
|
+
|
|
210
|
+
Uses the inner extension when it is ``.csv`` or ``.json`` and sniffs the
|
|
211
|
+
content otherwise. Rejects nested gzip or zip. Validates the archive first.
|
|
212
|
+
"""
|
|
213
|
+
validate_zip_integrity(zip_path)
|
|
214
|
+
|
|
215
|
+
with zipfile.ZipFile(zip_path, 'r') as zf:
|
|
216
|
+
inner = _zip_single_member(zf, Path(zip_path).name)
|
|
217
|
+
inner_name = inner.lower()
|
|
218
|
+
if inner_name.endswith('.csv'):
|
|
219
|
+
return 'csv'
|
|
220
|
+
if inner_name.endswith('.json'):
|
|
221
|
+
return 'json'
|
|
222
|
+
with _open_zip_member(zf, inner) as f:
|
|
223
|
+
head = f.read(_SNIFF_BYTES)
|
|
224
|
+
if not head:
|
|
225
|
+
raise ValueError(f"ZIP inner file is empty: {inner_name}")
|
|
226
|
+
if head[:2] == b'\x1f\x8b':
|
|
227
|
+
raise ValueError(f"ZIP inner file appears to be gzip-compressed: {inner_name}")
|
|
228
|
+
if head[:4] == b'PK\x03\x04':
|
|
229
|
+
raise ValueError(f"ZIP inner file appears to be another ZIP archive: {inner_name}")
|
|
230
|
+
return _sniff_text(_transcode_to_utf8(head))
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
# ---------------------------------------------------------------------------
|
|
234
|
+
# Opening files
|
|
235
|
+
# ---------------------------------------------------------------------------
|
|
236
|
+
|
|
237
|
+
@contextmanager
|
|
238
|
+
def open_binary(path: Path, compression: Optional[str]) -> Iterator[BinaryIO]:
|
|
239
|
+
"""Yield the decompressed bytes of *path* as a readable stream."""
|
|
240
|
+
path = Path(path)
|
|
241
|
+
if compression == 'gz':
|
|
242
|
+
with gzip.open(path, 'rb') as fh:
|
|
243
|
+
yield fh
|
|
244
|
+
elif compression == 'zip':
|
|
245
|
+
with zipfile.ZipFile(path, 'r') as zf:
|
|
246
|
+
with _open_zip_member(zf, _zip_single_member(zf, path.name)) as fh:
|
|
247
|
+
yield fh
|
|
248
|
+
else:
|
|
249
|
+
with open(path, 'rb') as fh:
|
|
250
|
+
yield fh
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
@contextmanager
|
|
254
|
+
def open_text(path: Path, compression: Optional[str]) -> Iterator[TextIO]:
|
|
255
|
+
"""Yield the decompressed content of *path* as text.
|
|
256
|
+
|
|
257
|
+
UTF-16 (with or without a BOM) is detected and decoded. Everything else
|
|
258
|
+
is read as UTF-8, with undecodable bytes replaced by U+FFFD. A BOM, when
|
|
259
|
+
present, stays in the text as U+FEFF; header normalization removes it.
|
|
260
|
+
"""
|
|
261
|
+
with open_binary(path, compression) as raw:
|
|
262
|
+
buffered = raw if hasattr(raw, 'peek') else io.BufferedReader(raw)
|
|
263
|
+
encoding = _detect_utf16_encoding(buffered.peek(32)[:32]) or 'utf-8'
|
|
264
|
+
text = io.TextIOWrapper(buffered, encoding=encoding, errors='replace')
|
|
265
|
+
try:
|
|
266
|
+
yield text
|
|
267
|
+
finally:
|
|
268
|
+
if not text.closed:
|
|
269
|
+
text.detach() # leave closing to open_binary
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
@contextmanager
|
|
273
|
+
def open_json(path: Path, compression: Optional[str],
|
|
274
|
+
sanitize: bool = False) -> Iterator[BinaryIO]:
|
|
275
|
+
"""Yield the decompressed content of *path* as UTF-8 bytes for ijson.
|
|
276
|
+
|
|
277
|
+
Encoding problems are always repaired: BOMs are stripped, UTF-16 is
|
|
278
|
+
transcoded and invalid UTF-8 becomes U+FFFD.
|
|
279
|
+
|
|
280
|
+
With ``sanitize=True`` the stream also drops control characters and fixes
|
|
281
|
+
double and trailing commas. That pass runs in pure Python, byte by byte,
|
|
282
|
+
so parse with ``sanitize=False`` first and retry with ``True`` only when
|
|
283
|
+
ijson reports a syntax error.
|
|
284
|
+
"""
|
|
285
|
+
with open_binary(path, compression) as raw:
|
|
286
|
+
reader = Utf8SanitizingReader(raw)
|
|
287
|
+
yield JsonSanitizingReader(reader) if sanitize else reader
|
|
288
|
+
|
|
289
|
+
|
|
290
|
+
# ---------------------------------------------------------------------------
|
|
291
|
+
# Deflate64 zip members
|
|
292
|
+
# ---------------------------------------------------------------------------
|
|
293
|
+
|
|
294
|
+
def _open_zip_member(zf: zipfile.ZipFile, name: str) -> BinaryIO:
|
|
295
|
+
"""Open a zip member for reading, including Deflate64 members."""
|
|
296
|
+
info = zf.getinfo(name)
|
|
297
|
+
if info.compress_type != ZIP_DEFLATE64:
|
|
298
|
+
return zf.open(info)
|
|
299
|
+
# Open the member as STORED so zipfile hands back the compressed bytes
|
|
300
|
+
# untouched, then inflate them ourselves. ZipInfo has __slots__, so copy.
|
|
301
|
+
raw_info = copy.copy(info)
|
|
302
|
+
raw_info.compress_type = zipfile.ZIP_STORED
|
|
303
|
+
raw_info.file_size = info.compress_size
|
|
304
|
+
raw_info.CRC = None # skip zipfile's check, _Deflate64Reader does its own
|
|
305
|
+
return io.BufferedReader(_Deflate64Reader(zf.open(raw_info), info))
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
class _Deflate64Reader(io.RawIOBase):
|
|
309
|
+
"""Inflate a Deflate64 stream and verify its CRC and size at the end."""
|
|
310
|
+
|
|
311
|
+
def __init__(self, raw: BinaryIO, info: zipfile.ZipInfo, chunk_size: int = 1 << 16):
|
|
312
|
+
self._raw = raw
|
|
313
|
+
self._info = info
|
|
314
|
+
self._chunk_size = chunk_size
|
|
315
|
+
self._inflater = inflate64.Inflater()
|
|
316
|
+
self._out = b''
|
|
317
|
+
self._pos = 0
|
|
318
|
+
self._crc = 0
|
|
319
|
+
self._size = 0
|
|
320
|
+
self._eof = False
|
|
321
|
+
|
|
322
|
+
def readable(self) -> bool:
|
|
323
|
+
return True
|
|
324
|
+
|
|
325
|
+
def readinto(self, b) -> int:
|
|
326
|
+
while self._pos >= len(self._out):
|
|
327
|
+
if self._eof:
|
|
328
|
+
return 0
|
|
329
|
+
chunk = self._raw.read(self._chunk_size)
|
|
330
|
+
if chunk:
|
|
331
|
+
try:
|
|
332
|
+
self._out, self._pos = self._inflater.inflate(chunk), 0
|
|
333
|
+
except ValueError as exc:
|
|
334
|
+
# inflate64 reports bad data as ValueError; callers treat
|
|
335
|
+
# ValueError as "unsupported file", so name it properly.
|
|
336
|
+
raise zipfile.BadZipFile(
|
|
337
|
+
f"Corrupt Deflate64 data in {self._info.filename}: {exc}") from exc
|
|
338
|
+
else:
|
|
339
|
+
self._eof = True
|
|
340
|
+
self._verify()
|
|
341
|
+
n = min(len(b), len(self._out) - self._pos)
|
|
342
|
+
data = self._out[self._pos:self._pos + n]
|
|
343
|
+
b[:n] = data
|
|
344
|
+
self._pos += n
|
|
345
|
+
self._crc = zlib.crc32(data, self._crc)
|
|
346
|
+
self._size += n
|
|
347
|
+
return n
|
|
348
|
+
|
|
349
|
+
def _verify(self) -> None:
|
|
350
|
+
if self._size != self._info.file_size or self._crc != self._info.CRC:
|
|
351
|
+
raise zipfile.BadZipFile(
|
|
352
|
+
f"Bad CRC-32 or size for Deflate64 member {self._info.filename}")
|
|
353
|
+
|
|
354
|
+
def close(self) -> None:
|
|
355
|
+
if not self.closed:
|
|
356
|
+
self._raw.close()
|
|
357
|
+
super().close()
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
# ---------------------------------------------------------------------------
|
|
361
|
+
# Encoding repair
|
|
362
|
+
# ---------------------------------------------------------------------------
|
|
363
|
+
|
|
364
|
+
def _detect_utf16_encoding(probe: bytes) -> Optional[str]:
|
|
365
|
+
"""Return ``'utf-16-le'``, ``'utf-16-be'`` or ``None`` for *probe*.
|
|
366
|
+
|
|
367
|
+
A BOM decides when present. Without one, four or more NUL bytes in the
|
|
368
|
+
first 32 bytes mean UTF-16, and which positions hold the NULs give the
|
|
369
|
+
byte order.
|
|
370
|
+
"""
|
|
371
|
+
if len(probe) < 2:
|
|
372
|
+
return None
|
|
373
|
+
if probe[:2] == b'\xff\xfe':
|
|
374
|
+
return 'utf-16-le'
|
|
375
|
+
if probe[:2] == b'\xfe\xff':
|
|
376
|
+
return 'utf-16-be'
|
|
377
|
+
snippet = probe[:32]
|
|
378
|
+
if snippet.count(b'\x00') >= 4:
|
|
379
|
+
le_score = sum(1 for i in range(1, len(snippet), 2) if snippet[i] == 0)
|
|
380
|
+
be_score = sum(1 for i in range(0, len(snippet), 2) if snippet[i] == 0)
|
|
381
|
+
if le_score > be_score:
|
|
382
|
+
return 'utf-16-le'
|
|
383
|
+
if be_score > le_score:
|
|
384
|
+
return 'utf-16-be'
|
|
385
|
+
return None
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
def _transcode_to_utf8(data: bytes) -> bytes:
|
|
389
|
+
"""Return *data* as UTF-8 if it is UTF-16, unchanged otherwise. Drops a UTF-16 BOM."""
|
|
390
|
+
enc = _detect_utf16_encoding(data)
|
|
391
|
+
if enc is None:
|
|
392
|
+
return data
|
|
393
|
+
if data[:2] in _UTF16_BOMS:
|
|
394
|
+
data = data[2:]
|
|
395
|
+
return data.decode(enc, errors='replace').encode('utf-8')
|
|
396
|
+
|
|
397
|
+
|
|
398
|
+
class Utf8SanitizingReader:
|
|
399
|
+
"""Wrap a binary stream so everything read from it is valid UTF-8.
|
|
400
|
+
|
|
401
|
+
Strips a leading UTF-8 BOM, transcodes UTF-16 (detected on the first
|
|
402
|
+
read) and replaces invalid bytes, such as stray Windows-1252, with
|
|
403
|
+
U+FFFD. An incremental decoder carries multi-byte sequences split across
|
|
404
|
+
chunks. ``read(n)`` returns at most *n* bytes even when a replacement
|
|
405
|
+
grows one input byte into three.
|
|
406
|
+
"""
|
|
407
|
+
|
|
408
|
+
def __init__(self, fh: BinaryIO, chunk_size: int = 256 * 1024):
|
|
409
|
+
self._fh = fh
|
|
410
|
+
self._chunk_size = chunk_size
|
|
411
|
+
self._decoder = codecs.getincrementaldecoder('utf-8')('replace')
|
|
412
|
+
self._buf = b''
|
|
413
|
+
self._eof = False
|
|
414
|
+
self._first_read = True
|
|
415
|
+
|
|
416
|
+
def _strip_bom(self, raw: bytes) -> bytes:
|
|
417
|
+
if not self._first_read:
|
|
418
|
+
return raw
|
|
419
|
+
self._first_read = False
|
|
420
|
+
if raw[:3] == _UTF8_BOM:
|
|
421
|
+
raw = raw[3:]
|
|
422
|
+
enc = _detect_utf16_encoding(raw)
|
|
423
|
+
if enc is not None:
|
|
424
|
+
self._decoder = codecs.getincrementaldecoder(enc)('replace')
|
|
425
|
+
if raw[:2] in _UTF16_BOMS:
|
|
426
|
+
raw = raw[2:]
|
|
427
|
+
return raw
|
|
428
|
+
|
|
429
|
+
def _pull(self) -> None:
|
|
430
|
+
"""Decode one more chunk into the buffer, or mark EOF."""
|
|
431
|
+
raw = self._fh.read(self._chunk_size)
|
|
432
|
+
if raw:
|
|
433
|
+
raw = self._strip_bom(raw) # may swap in a UTF-16 decoder, so call it first
|
|
434
|
+
text = self._decoder.decode(raw, final=False)
|
|
435
|
+
else:
|
|
436
|
+
self._eof = True
|
|
437
|
+
text = self._decoder.decode(b'', final=True)
|
|
438
|
+
if text:
|
|
439
|
+
self._buf += text.encode('utf-8')
|
|
440
|
+
|
|
441
|
+
def read(self, n: int = -1) -> bytes:
|
|
442
|
+
if n is None or n < 0:
|
|
443
|
+
while not self._eof:
|
|
444
|
+
self._pull()
|
|
445
|
+
result, self._buf = self._buf, b''
|
|
446
|
+
return result
|
|
447
|
+
while len(self._buf) < n and not self._eof:
|
|
448
|
+
self._pull()
|
|
449
|
+
result, self._buf = self._buf[:n], self._buf[n:]
|
|
450
|
+
return result
|
|
451
|
+
|
|
452
|
+
def readline(self) -> bytes:
|
|
453
|
+
"""Return bytes up to and including the next newline, or to EOF."""
|
|
454
|
+
while True:
|
|
455
|
+
nl = self._buf.find(b'\n')
|
|
456
|
+
if nl != -1:
|
|
457
|
+
line, self._buf = self._buf[:nl + 1], self._buf[nl + 1:]
|
|
458
|
+
return line
|
|
459
|
+
if self._eof:
|
|
460
|
+
line, self._buf = self._buf, b''
|
|
461
|
+
return line
|
|
462
|
+
self._pull()
|
|
463
|
+
|
|
464
|
+
def __iter__(self):
|
|
465
|
+
return self
|
|
466
|
+
|
|
467
|
+
def __next__(self) -> bytes:
|
|
468
|
+
line = self.readline()
|
|
469
|
+
if not line:
|
|
470
|
+
raise StopIteration
|
|
471
|
+
return line
|
|
472
|
+
|
|
473
|
+
def seekable(self) -> bool:
|
|
474
|
+
return False
|
|
475
|
+
|
|
476
|
+
def readable(self) -> bool:
|
|
477
|
+
return True
|
|
478
|
+
|
|
479
|
+
def close(self) -> None:
|
|
480
|
+
self._fh.close()
|
|
481
|
+
|
|
482
|
+
|
|
483
|
+
# Bytes JsonSanitizingReader treats as control characters and whitespace.
|
|
484
|
+
_CTRL_CHAR_BYTES = frozenset(range(0x00, 0x09)) | frozenset(range(0x0E, 0x20)) | {0x0B, 0x0C}
|
|
485
|
+
_WS_BYTES = frozenset(b' \t\r\n')
|
|
486
|
+
|
|
487
|
+
|
|
488
|
+
class JsonSanitizingReader:
|
|
489
|
+
"""Fix common JSON damage in a stream, without loading the whole file.
|
|
490
|
+
|
|
491
|
+
Applied to a UTF-8 stream (normally a :class:`Utf8SanitizingReader`):
|
|
492
|
+
|
|
493
|
+
1. strip control characters ``[\\x00-\\x08\\x0b\\x0c\\x0e-\\x1f]``
|
|
494
|
+
2. collapse double commas ``,,`` into one
|
|
495
|
+
3. drop trailing commas before ``]`` or ``}``
|
|
496
|
+
|
|
497
|
+
Comma fixes only apply outside string literals. Quote and escape state is
|
|
498
|
+
carried across chunks, and a comma whose meaning depends on the next
|
|
499
|
+
chunk is held back until that chunk arrives. Memory stays at about two
|
|
500
|
+
chunks when the caller reads in bounded sizes, as ijson does.
|
|
501
|
+
"""
|
|
502
|
+
|
|
503
|
+
def __init__(self, fh, chunk_size: int = 256 * 1024):
|
|
504
|
+
self._fh = fh
|
|
505
|
+
self._chunk_size = chunk_size
|
|
506
|
+
self._out_buf = b''
|
|
507
|
+
self._eof = False
|
|
508
|
+
self._in_string = False
|
|
509
|
+
self._escaped = False
|
|
510
|
+
# A comma (plus any whitespace) at the end of a chunk, waiting for
|
|
511
|
+
# lookahead from the next one.
|
|
512
|
+
self._pending = b''
|
|
513
|
+
|
|
514
|
+
def _process_chunk(self, raw: bytes) -> bytes:
|
|
515
|
+
"""Return the sanitized form of *raw*, holding back an undecided comma."""
|
|
516
|
+
if self._pending:
|
|
517
|
+
raw = self._pending + raw
|
|
518
|
+
self._pending = b''
|
|
519
|
+
|
|
520
|
+
out = bytearray()
|
|
521
|
+
i = 0
|
|
522
|
+
n = len(raw)
|
|
523
|
+
|
|
524
|
+
while i < n:
|
|
525
|
+
bch = raw[i]
|
|
526
|
+
|
|
527
|
+
if self._in_string:
|
|
528
|
+
if bch in _CTRL_CHAR_BYTES:
|
|
529
|
+
i += 1
|
|
530
|
+
continue
|
|
531
|
+
out.append(bch)
|
|
532
|
+
if self._escaped:
|
|
533
|
+
self._escaped = False
|
|
534
|
+
elif bch == 0x5C: # backslash
|
|
535
|
+
self._escaped = True
|
|
536
|
+
elif bch == 0x22: # closing quote
|
|
537
|
+
self._in_string = False
|
|
538
|
+
i += 1
|
|
539
|
+
continue
|
|
540
|
+
|
|
541
|
+
if bch in _CTRL_CHAR_BYTES:
|
|
542
|
+
i += 1
|
|
543
|
+
continue
|
|
544
|
+
|
|
545
|
+
if bch == 0x22: # opening quote
|
|
546
|
+
self._in_string = True
|
|
547
|
+
out.append(bch)
|
|
548
|
+
i += 1
|
|
549
|
+
continue
|
|
550
|
+
|
|
551
|
+
if bch == 0x2C: # comma
|
|
552
|
+
j = i + 1
|
|
553
|
+
while j < n and raw[j] in _WS_BYTES:
|
|
554
|
+
j += 1
|
|
555
|
+
|
|
556
|
+
if j >= n:
|
|
557
|
+
# Need the next chunk to decide.
|
|
558
|
+
self._pending = bytes(raw[i:])
|
|
559
|
+
return bytes(out)
|
|
560
|
+
|
|
561
|
+
next_byte = raw[j]
|
|
562
|
+
|
|
563
|
+
if next_byte == 0x2C:
|
|
564
|
+
# Double comma: swallow the run of extra commas.
|
|
565
|
+
while j < n and raw[j] == 0x2C:
|
|
566
|
+
j += 1
|
|
567
|
+
while j < n and raw[j] in _WS_BYTES:
|
|
568
|
+
j += 1
|
|
569
|
+
if j >= n:
|
|
570
|
+
# Still undecided: keep one comma for the next chunk.
|
|
571
|
+
self._pending = bytes(raw[i:i + 1])
|
|
572
|
+
i = j
|
|
573
|
+
continue
|
|
574
|
+
if raw[j] in (0x5D, 0x7D): # ] or }
|
|
575
|
+
i = j # trailing comma: drop it
|
|
576
|
+
continue
|
|
577
|
+
out.append(0x2C)
|
|
578
|
+
k = i + 1
|
|
579
|
+
while k < n and raw[k] in _WS_BYTES:
|
|
580
|
+
out.append(raw[k])
|
|
581
|
+
k += 1
|
|
582
|
+
i = j
|
|
583
|
+
continue
|
|
584
|
+
|
|
585
|
+
if next_byte in (0x5D, 0x7D):
|
|
586
|
+
# Trailing comma: drop it, keep the whitespace.
|
|
587
|
+
out.extend(raw[i + 1:j])
|
|
588
|
+
i = j
|
|
589
|
+
continue
|
|
590
|
+
|
|
591
|
+
out.append(bch)
|
|
592
|
+
i += 1
|
|
593
|
+
continue
|
|
594
|
+
|
|
595
|
+
out.append(bch)
|
|
596
|
+
i += 1
|
|
597
|
+
|
|
598
|
+
return bytes(out)
|
|
599
|
+
|
|
600
|
+
def _pull(self) -> None:
|
|
601
|
+
"""Sanitize one more chunk into the buffer, or mark EOF."""
|
|
602
|
+
raw = self._fh.read(self._chunk_size)
|
|
603
|
+
if raw:
|
|
604
|
+
self._out_buf += self._process_chunk(raw)
|
|
605
|
+
else:
|
|
606
|
+
# A comma still pending at EOF means the file ends mid-token.
|
|
607
|
+
# Emit it rather than lose data.
|
|
608
|
+
self._eof = True
|
|
609
|
+
self._out_buf += self._pending
|
|
610
|
+
self._pending = b''
|
|
611
|
+
|
|
612
|
+
def read(self, n: int = -1) -> bytes:
|
|
613
|
+
if n is None or n < 0:
|
|
614
|
+
while not self._eof:
|
|
615
|
+
self._pull()
|
|
616
|
+
result, self._out_buf = self._out_buf, b''
|
|
617
|
+
return result
|
|
618
|
+
while len(self._out_buf) < n and not self._eof:
|
|
619
|
+
self._pull()
|
|
620
|
+
result, self._out_buf = self._out_buf[:n], self._out_buf[n:]
|
|
621
|
+
return result
|
|
622
|
+
|
|
623
|
+
def readline(self) -> bytes:
|
|
624
|
+
"""Return bytes up to and including the next newline, or to EOF."""
|
|
625
|
+
while True:
|
|
626
|
+
nl = self._out_buf.find(b'\n')
|
|
627
|
+
if nl != -1:
|
|
628
|
+
line, self._out_buf = self._out_buf[:nl + 1], self._out_buf[nl + 1:]
|
|
629
|
+
return line
|
|
630
|
+
if self._eof:
|
|
631
|
+
line, self._out_buf = self._out_buf, b''
|
|
632
|
+
return line
|
|
633
|
+
self._pull()
|
|
634
|
+
|
|
635
|
+
def __iter__(self):
|
|
636
|
+
return self
|
|
637
|
+
|
|
638
|
+
def __next__(self) -> bytes:
|
|
639
|
+
line = self.readline()
|
|
640
|
+
if not line:
|
|
641
|
+
raise StopIteration
|
|
642
|
+
return line
|
|
643
|
+
|
|
644
|
+
def seekable(self) -> bool:
|
|
645
|
+
return False
|
|
646
|
+
|
|
647
|
+
def readable(self) -> bool:
|
|
648
|
+
return True
|
|
649
|
+
|
|
650
|
+
def close(self) -> None:
|
|
651
|
+
self._fh.close()
|