mrfkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mrfkit/files.py ADDED
@@ -0,0 +1,651 @@
1
+ """Open MRF files: detect the format, decompress, and repair broken text.
2
+
3
+ Hospital files arrive as plain, gzip or zip, sometimes with a wrong or missing
4
+ extension, sometimes in UTF-16, often with stray Windows-1252 bytes, and the
5
+ JSON ones regularly carry double or trailing commas. Everything here streams:
6
+ memory stays flat no matter how big the file is.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ import codecs
12
+ import copy
13
+ import gzip
14
+ import io
15
+ import logging
16
+ import zipfile
17
+ import zlib
18
+ from contextlib import contextmanager
19
+ from pathlib import Path
20
+ from typing import BinaryIO, Iterator, List, Optional, TextIO, Tuple
21
+
22
+ import inflate64
23
+
24
+ log = logging.getLogger(__name__)
25
+
26
+ # Zip method 9. The stdlib zipfile cannot read it, and Windows uses it for
27
+ # large archives, so real MRFs show up compressed this way.
28
+ ZIP_DEFLATE64 = 9
29
+
30
+ _UTF8_BOM = b'\xef\xbb\xbf'
31
+ _UTF16_BOMS = (b'\xff\xfe', b'\xfe\xff')
32
+ _SNIFF_BYTES = 4096
33
+
34
+
35
+ # ---------------------------------------------------------------------------
36
+ # Format detection
37
+ # ---------------------------------------------------------------------------
38
+
39
+ def detect_file_format(path: Path) -> Tuple[str, Optional[str]]:
40
+ """Return ``(file_format, compression)`` for *path*.
41
+
42
+ *file_format* is ``'csv'`` or ``'json'``; *compression* is ``'gz'``,
43
+ ``'zip'`` or ``None``.
44
+
45
+ The extension decides first. Plain ``.csv`` and ``.json`` names are
46
+ checked against the content, because misnamed files are common. Anything
47
+ else (API downloads often have no extension) is sniffed.
48
+
49
+ Raises ``ValueError`` for Excel files and ambiguous zips, and
50
+ ``zipfile.BadZipFile`` for corrupt zips.
51
+ """
52
+ path = Path(path)
53
+ name = path.name.lower()
54
+
55
+ if looks_like_html(path):
56
+ raise ValueError(
57
+ f"{path.name} is an HTML page, not an MRF: the download probably hit "
58
+ f"an error page or a bot challenge.")
59
+ if name.endswith(('.xlsx', '.xls')):
60
+ raise ValueError(
61
+ f"Excel/XLSX files are not supported: {path.name}. "
62
+ f"Please provide the CSV or JSON version of this MRF."
63
+ )
64
+ if name.endswith('.zip'):
65
+ _reject_if_xlsx(path)
66
+ return detect_format_from_zip(path), 'zip'
67
+ if name.endswith('.csv.gz'):
68
+ return 'csv', 'gz'
69
+ if name.endswith('.json.gz'):
70
+ return 'json', 'gz'
71
+ if name.endswith(('.csv', '.json')):
72
+ ext_format = 'csv' if name.endswith('.csv') else 'json'
73
+ try:
74
+ content_format, content_compression = _detect_format_from_content(path)
75
+ except ValueError:
76
+ # Empty file: nothing to contradict the extension.
77
+ return ext_format, None
78
+ if content_compression is not None:
79
+ # Compressed despite a plain extension: trust the content.
80
+ return content_format, content_compression
81
+ if content_format != ext_format:
82
+ log.warning(
83
+ "Extension says %s but content looks like %s, trusting content: %s",
84
+ ext_format, content_format, path.name,
85
+ )
86
+ return content_format, None
87
+ return ext_format, None
88
+ return _detect_format_from_content(path)
89
+
90
+
91
+ # Signs of an HTML error page saved under an MRF name: ASP.NET download
92
+ # stubs, CDN 403/404 pages, Cloudflare challenges.
93
+ _HTML_MARKERS = (b"<!doctype html", b"<html", b"<head", b"attention required", b"__viewstate")
94
+
95
+
96
+ def looks_like_html(path: Path) -> bool:
97
+ """True when the first 512 bytes of *path* look like an HTML page.
98
+
99
+ Compressed files and anything starting with ``{`` or ``[`` never match,
100
+ so a real MRF is not flagged.
101
+ """
102
+ try:
103
+ with open(path, 'rb') as fh:
104
+ head = fh.read(512)
105
+ except OSError:
106
+ return False
107
+ if not head or head[:2] in (b'\x1f\x8b', b'PK'):
108
+ return False
109
+ stripped = head.removeprefix(_UTF8_BOM).lstrip()
110
+ if stripped.startswith((b'{', b'[')):
111
+ return False
112
+ lowered = stripped.lower()
113
+ return any(marker in lowered for marker in _HTML_MARKERS)
114
+
115
+
116
+ def _detect_format_from_content(path: Path) -> Tuple[str, Optional[str]]:
117
+ """Detect the format from the first bytes of *path*.
118
+
119
+ Gzip and zip are recognized by magic bytes. Text whose first
120
+ non-whitespace byte is ``{`` or ``[`` is JSON, anything else is CSV.
121
+ UTF-16 text is transcoded before sniffing. Raises ``ValueError`` for an
122
+ empty file.
123
+ """
124
+ with open(path, 'rb') as f:
125
+ head = f.read(_SNIFF_BYTES)
126
+
127
+ if not head:
128
+ raise ValueError(f"Cannot detect format: file is empty: {path.name}")
129
+
130
+ if head[:4] == b'PK\x03\x04':
131
+ _reject_if_xlsx(path)
132
+ return detect_format_from_zip(path), 'zip'
133
+
134
+ if head[:2] == b'\x1f\x8b':
135
+ with gzip.open(path, 'rb') as gz:
136
+ inner_head = gz.read(_SNIFF_BYTES)
137
+ return _sniff_text(_transcode_to_utf8(inner_head)), 'gz'
138
+
139
+ return _sniff_text(_transcode_to_utf8(head)), None
140
+
141
+
142
+ def _sniff_text(head: bytes) -> str:
143
+ stripped = head.lstrip(b' \t\r\n' + _UTF8_BOM)
144
+ return 'json' if stripped[:1] in (b'{', b'[') else 'csv'
145
+
146
+
147
+ def validate_zip_integrity(zip_path: Path) -> None:
148
+ """Raise ``zipfile.BadZipFile`` if the zip is corrupt or truncated.
149
+
150
+ Reads every data member in full and checks its CRC, so a damaged archive
151
+ fails here with a clear error instead of halfway through a parse.
152
+ """
153
+ try:
154
+ with zipfile.ZipFile(zip_path, 'r') as zf:
155
+ for name in _zip_data_members(zf):
156
+ with _open_zip_member(zf, name) as member:
157
+ while member.read(1 << 20):
158
+ pass
159
+ except (zlib.error, EOFError) as exc:
160
+ raise zipfile.BadZipFile(
161
+ f"ZIP decompression error ({Path(zip_path).name}): {exc}") from exc
162
+
163
+
164
+ def _reject_if_xlsx(zip_path: Path) -> None:
165
+ """Raise ``ValueError`` if the zip is really an XLSX workbook."""
166
+ try:
167
+ with zipfile.ZipFile(zip_path, 'r') as zf:
168
+ lower_names = [n.lower() for n in zf.namelist()]
169
+ except zipfile.BadZipFile:
170
+ return # Not a valid zip: let the caller report that.
171
+ if '[content_types].xml' in lower_names or any(n.startswith('xl/') for n in lower_names):
172
+ raise ValueError(
173
+ f"Excel/XLSX files are not supported: {Path(zip_path).name}. "
174
+ f"Please provide the CSV or JSON version of this MRF."
175
+ )
176
+
177
+
178
+ def _zip_data_members(zf: zipfile.ZipFile) -> List[str]:
179
+ """Return the real data members, skipping directories and macOS sidecars.
180
+
181
+ Finder-made zips add ``__MACOSX/._name`` resource forks and sometimes a
182
+ ``.DS_Store``. Counting those would make a one-file archive look like
183
+ three.
184
+ """
185
+ members = []
186
+ for info in zf.infolist():
187
+ if info.is_dir():
188
+ continue
189
+ base = info.filename.rsplit('/', 1)[-1]
190
+ if info.filename.startswith('__MACOSX/') or base == '.DS_Store' or base.startswith('._'):
191
+ continue
192
+ members.append(info.filename)
193
+ return members
194
+
195
+
196
+ def _zip_single_member(zf: zipfile.ZipFile, zip_name: str = "") -> str:
197
+ """Return the one data member of *zf*, or raise ``ValueError``."""
198
+ members = _zip_data_members(zf)
199
+ if len(members) != 1:
200
+ suffix = f" in {zip_name}" if zip_name else ""
201
+ detail = f": {members}" if members else ""
202
+ raise ValueError(
203
+ f"ZIP must contain exactly one file, found {len(members)}{suffix}{detail}")
204
+ return members[0]
205
+
206
+
207
+ def detect_format_from_zip(zip_path: Path) -> str:
208
+ """Return ``'csv'`` or ``'json'`` for the single data file inside a zip.
209
+
210
+ Uses the inner extension when it is ``.csv`` or ``.json`` and sniffs the
211
+ content otherwise. Rejects nested gzip or zip. Validates the archive first.
212
+ """
213
+ validate_zip_integrity(zip_path)
214
+
215
+ with zipfile.ZipFile(zip_path, 'r') as zf:
216
+ inner = _zip_single_member(zf, Path(zip_path).name)
217
+ inner_name = inner.lower()
218
+ if inner_name.endswith('.csv'):
219
+ return 'csv'
220
+ if inner_name.endswith('.json'):
221
+ return 'json'
222
+ with _open_zip_member(zf, inner) as f:
223
+ head = f.read(_SNIFF_BYTES)
224
+ if not head:
225
+ raise ValueError(f"ZIP inner file is empty: {inner_name}")
226
+ if head[:2] == b'\x1f\x8b':
227
+ raise ValueError(f"ZIP inner file appears to be gzip-compressed: {inner_name}")
228
+ if head[:4] == b'PK\x03\x04':
229
+ raise ValueError(f"ZIP inner file appears to be another ZIP archive: {inner_name}")
230
+ return _sniff_text(_transcode_to_utf8(head))
231
+
232
+
233
+ # ---------------------------------------------------------------------------
234
+ # Opening files
235
+ # ---------------------------------------------------------------------------
236
+
237
+ @contextmanager
238
+ def open_binary(path: Path, compression: Optional[str]) -> Iterator[BinaryIO]:
239
+ """Yield the decompressed bytes of *path* as a readable stream."""
240
+ path = Path(path)
241
+ if compression == 'gz':
242
+ with gzip.open(path, 'rb') as fh:
243
+ yield fh
244
+ elif compression == 'zip':
245
+ with zipfile.ZipFile(path, 'r') as zf:
246
+ with _open_zip_member(zf, _zip_single_member(zf, path.name)) as fh:
247
+ yield fh
248
+ else:
249
+ with open(path, 'rb') as fh:
250
+ yield fh
251
+
252
+
253
+ @contextmanager
254
+ def open_text(path: Path, compression: Optional[str]) -> Iterator[TextIO]:
255
+ """Yield the decompressed content of *path* as text.
256
+
257
+ UTF-16 (with or without a BOM) is detected and decoded. Everything else
258
+ is read as UTF-8, with undecodable bytes replaced by U+FFFD. A BOM, when
259
+ present, stays in the text as U+FEFF; header normalization removes it.
260
+ """
261
+ with open_binary(path, compression) as raw:
262
+ buffered = raw if hasattr(raw, 'peek') else io.BufferedReader(raw)
263
+ encoding = _detect_utf16_encoding(buffered.peek(32)[:32]) or 'utf-8'
264
+ text = io.TextIOWrapper(buffered, encoding=encoding, errors='replace')
265
+ try:
266
+ yield text
267
+ finally:
268
+ if not text.closed:
269
+ text.detach() # leave closing to open_binary
270
+
271
+
272
+ @contextmanager
273
+ def open_json(path: Path, compression: Optional[str],
274
+ sanitize: bool = False) -> Iterator[BinaryIO]:
275
+ """Yield the decompressed content of *path* as UTF-8 bytes for ijson.
276
+
277
+ Encoding problems are always repaired: BOMs are stripped, UTF-16 is
278
+ transcoded and invalid UTF-8 becomes U+FFFD.
279
+
280
+ With ``sanitize=True`` the stream also drops control characters and fixes
281
+ double and trailing commas. That pass runs in pure Python, byte by byte,
282
+ so parse with ``sanitize=False`` first and retry with ``True`` only when
283
+ ijson reports a syntax error.
284
+ """
285
+ with open_binary(path, compression) as raw:
286
+ reader = Utf8SanitizingReader(raw)
287
+ yield JsonSanitizingReader(reader) if sanitize else reader
288
+
289
+
290
+ # ---------------------------------------------------------------------------
291
+ # Deflate64 zip members
292
+ # ---------------------------------------------------------------------------
293
+
294
+ def _open_zip_member(zf: zipfile.ZipFile, name: str) -> BinaryIO:
295
+ """Open a zip member for reading, including Deflate64 members."""
296
+ info = zf.getinfo(name)
297
+ if info.compress_type != ZIP_DEFLATE64:
298
+ return zf.open(info)
299
+ # Open the member as STORED so zipfile hands back the compressed bytes
300
+ # untouched, then inflate them ourselves. ZipInfo has __slots__, so copy.
301
+ raw_info = copy.copy(info)
302
+ raw_info.compress_type = zipfile.ZIP_STORED
303
+ raw_info.file_size = info.compress_size
304
+ raw_info.CRC = None # skip zipfile's check, _Deflate64Reader does its own
305
+ return io.BufferedReader(_Deflate64Reader(zf.open(raw_info), info))
306
+
307
+
308
+ class _Deflate64Reader(io.RawIOBase):
309
+ """Inflate a Deflate64 stream and verify its CRC and size at the end."""
310
+
311
+ def __init__(self, raw: BinaryIO, info: zipfile.ZipInfo, chunk_size: int = 1 << 16):
312
+ self._raw = raw
313
+ self._info = info
314
+ self._chunk_size = chunk_size
315
+ self._inflater = inflate64.Inflater()
316
+ self._out = b''
317
+ self._pos = 0
318
+ self._crc = 0
319
+ self._size = 0
320
+ self._eof = False
321
+
322
+ def readable(self) -> bool:
323
+ return True
324
+
325
+ def readinto(self, b) -> int:
326
+ while self._pos >= len(self._out):
327
+ if self._eof:
328
+ return 0
329
+ chunk = self._raw.read(self._chunk_size)
330
+ if chunk:
331
+ try:
332
+ self._out, self._pos = self._inflater.inflate(chunk), 0
333
+ except ValueError as exc:
334
+ # inflate64 reports bad data as ValueError; callers treat
335
+ # ValueError as "unsupported file", so name it properly.
336
+ raise zipfile.BadZipFile(
337
+ f"Corrupt Deflate64 data in {self._info.filename}: {exc}") from exc
338
+ else:
339
+ self._eof = True
340
+ self._verify()
341
+ n = min(len(b), len(self._out) - self._pos)
342
+ data = self._out[self._pos:self._pos + n]
343
+ b[:n] = data
344
+ self._pos += n
345
+ self._crc = zlib.crc32(data, self._crc)
346
+ self._size += n
347
+ return n
348
+
349
+ def _verify(self) -> None:
350
+ if self._size != self._info.file_size or self._crc != self._info.CRC:
351
+ raise zipfile.BadZipFile(
352
+ f"Bad CRC-32 or size for Deflate64 member {self._info.filename}")
353
+
354
+ def close(self) -> None:
355
+ if not self.closed:
356
+ self._raw.close()
357
+ super().close()
358
+
359
+
360
+ # ---------------------------------------------------------------------------
361
+ # Encoding repair
362
+ # ---------------------------------------------------------------------------
363
+
364
+ def _detect_utf16_encoding(probe: bytes) -> Optional[str]:
365
+ """Return ``'utf-16-le'``, ``'utf-16-be'`` or ``None`` for *probe*.
366
+
367
+ A BOM decides when present. Without one, four or more NUL bytes in the
368
+ first 32 bytes mean UTF-16, and which positions hold the NULs give the
369
+ byte order.
370
+ """
371
+ if len(probe) < 2:
372
+ return None
373
+ if probe[:2] == b'\xff\xfe':
374
+ return 'utf-16-le'
375
+ if probe[:2] == b'\xfe\xff':
376
+ return 'utf-16-be'
377
+ snippet = probe[:32]
378
+ if snippet.count(b'\x00') >= 4:
379
+ le_score = sum(1 for i in range(1, len(snippet), 2) if snippet[i] == 0)
380
+ be_score = sum(1 for i in range(0, len(snippet), 2) if snippet[i] == 0)
381
+ if le_score > be_score:
382
+ return 'utf-16-le'
383
+ if be_score > le_score:
384
+ return 'utf-16-be'
385
+ return None
386
+
387
+
388
+ def _transcode_to_utf8(data: bytes) -> bytes:
389
+ """Return *data* as UTF-8 if it is UTF-16, unchanged otherwise. Drops a UTF-16 BOM."""
390
+ enc = _detect_utf16_encoding(data)
391
+ if enc is None:
392
+ return data
393
+ if data[:2] in _UTF16_BOMS:
394
+ data = data[2:]
395
+ return data.decode(enc, errors='replace').encode('utf-8')
396
+
397
+
398
+ class Utf8SanitizingReader:
399
+ """Wrap a binary stream so everything read from it is valid UTF-8.
400
+
401
+ Strips a leading UTF-8 BOM, transcodes UTF-16 (detected on the first
402
+ read) and replaces invalid bytes, such as stray Windows-1252, with
403
+ U+FFFD. An incremental decoder carries multi-byte sequences split across
404
+ chunks. ``read(n)`` returns at most *n* bytes even when a replacement
405
+ grows one input byte into three.
406
+ """
407
+
408
+ def __init__(self, fh: BinaryIO, chunk_size: int = 256 * 1024):
409
+ self._fh = fh
410
+ self._chunk_size = chunk_size
411
+ self._decoder = codecs.getincrementaldecoder('utf-8')('replace')
412
+ self._buf = b''
413
+ self._eof = False
414
+ self._first_read = True
415
+
416
+ def _strip_bom(self, raw: bytes) -> bytes:
417
+ if not self._first_read:
418
+ return raw
419
+ self._first_read = False
420
+ if raw[:3] == _UTF8_BOM:
421
+ raw = raw[3:]
422
+ enc = _detect_utf16_encoding(raw)
423
+ if enc is not None:
424
+ self._decoder = codecs.getincrementaldecoder(enc)('replace')
425
+ if raw[:2] in _UTF16_BOMS:
426
+ raw = raw[2:]
427
+ return raw
428
+
429
+ def _pull(self) -> None:
430
+ """Decode one more chunk into the buffer, or mark EOF."""
431
+ raw = self._fh.read(self._chunk_size)
432
+ if raw:
433
+ raw = self._strip_bom(raw) # may swap in a UTF-16 decoder, so call it first
434
+ text = self._decoder.decode(raw, final=False)
435
+ else:
436
+ self._eof = True
437
+ text = self._decoder.decode(b'', final=True)
438
+ if text:
439
+ self._buf += text.encode('utf-8')
440
+
441
+ def read(self, n: int = -1) -> bytes:
442
+ if n is None or n < 0:
443
+ while not self._eof:
444
+ self._pull()
445
+ result, self._buf = self._buf, b''
446
+ return result
447
+ while len(self._buf) < n and not self._eof:
448
+ self._pull()
449
+ result, self._buf = self._buf[:n], self._buf[n:]
450
+ return result
451
+
452
+ def readline(self) -> bytes:
453
+ """Return bytes up to and including the next newline, or to EOF."""
454
+ while True:
455
+ nl = self._buf.find(b'\n')
456
+ if nl != -1:
457
+ line, self._buf = self._buf[:nl + 1], self._buf[nl + 1:]
458
+ return line
459
+ if self._eof:
460
+ line, self._buf = self._buf, b''
461
+ return line
462
+ self._pull()
463
+
464
+ def __iter__(self):
465
+ return self
466
+
467
+ def __next__(self) -> bytes:
468
+ line = self.readline()
469
+ if not line:
470
+ raise StopIteration
471
+ return line
472
+
473
+ def seekable(self) -> bool:
474
+ return False
475
+
476
+ def readable(self) -> bool:
477
+ return True
478
+
479
+ def close(self) -> None:
480
+ self._fh.close()
481
+
482
+
483
+ # Bytes JsonSanitizingReader treats as control characters and whitespace.
484
+ _CTRL_CHAR_BYTES = frozenset(range(0x00, 0x09)) | frozenset(range(0x0E, 0x20)) | {0x0B, 0x0C}
485
+ _WS_BYTES = frozenset(b' \t\r\n')
486
+
487
+
488
+ class JsonSanitizingReader:
489
+ """Fix common JSON damage in a stream, without loading the whole file.
490
+
491
+ Applied to a UTF-8 stream (normally a :class:`Utf8SanitizingReader`):
492
+
493
+ 1. strip control characters ``[\\x00-\\x08\\x0b\\x0c\\x0e-\\x1f]``
494
+ 2. collapse double commas ``,,`` into one
495
+ 3. drop trailing commas before ``]`` or ``}``
496
+
497
+ Comma fixes only apply outside string literals. Quote and escape state is
498
+ carried across chunks, and a comma whose meaning depends on the next
499
+ chunk is held back until that chunk arrives. Memory stays at about two
500
+ chunks when the caller reads in bounded sizes, as ijson does.
501
+ """
502
+
503
+ def __init__(self, fh, chunk_size: int = 256 * 1024):
504
+ self._fh = fh
505
+ self._chunk_size = chunk_size
506
+ self._out_buf = b''
507
+ self._eof = False
508
+ self._in_string = False
509
+ self._escaped = False
510
+ # A comma (plus any whitespace) at the end of a chunk, waiting for
511
+ # lookahead from the next one.
512
+ self._pending = b''
513
+
514
+ def _process_chunk(self, raw: bytes) -> bytes:
515
+ """Return the sanitized form of *raw*, holding back an undecided comma."""
516
+ if self._pending:
517
+ raw = self._pending + raw
518
+ self._pending = b''
519
+
520
+ out = bytearray()
521
+ i = 0
522
+ n = len(raw)
523
+
524
+ while i < n:
525
+ bch = raw[i]
526
+
527
+ if self._in_string:
528
+ if bch in _CTRL_CHAR_BYTES:
529
+ i += 1
530
+ continue
531
+ out.append(bch)
532
+ if self._escaped:
533
+ self._escaped = False
534
+ elif bch == 0x5C: # backslash
535
+ self._escaped = True
536
+ elif bch == 0x22: # closing quote
537
+ self._in_string = False
538
+ i += 1
539
+ continue
540
+
541
+ if bch in _CTRL_CHAR_BYTES:
542
+ i += 1
543
+ continue
544
+
545
+ if bch == 0x22: # opening quote
546
+ self._in_string = True
547
+ out.append(bch)
548
+ i += 1
549
+ continue
550
+
551
+ if bch == 0x2C: # comma
552
+ j = i + 1
553
+ while j < n and raw[j] in _WS_BYTES:
554
+ j += 1
555
+
556
+ if j >= n:
557
+ # Need the next chunk to decide.
558
+ self._pending = bytes(raw[i:])
559
+ return bytes(out)
560
+
561
+ next_byte = raw[j]
562
+
563
+ if next_byte == 0x2C:
564
+ # Double comma: swallow the run of extra commas.
565
+ while j < n and raw[j] == 0x2C:
566
+ j += 1
567
+ while j < n and raw[j] in _WS_BYTES:
568
+ j += 1
569
+ if j >= n:
570
+ # Still undecided: keep one comma for the next chunk.
571
+ self._pending = bytes(raw[i:i + 1])
572
+ i = j
573
+ continue
574
+ if raw[j] in (0x5D, 0x7D): # ] or }
575
+ i = j # trailing comma: drop it
576
+ continue
577
+ out.append(0x2C)
578
+ k = i + 1
579
+ while k < n and raw[k] in _WS_BYTES:
580
+ out.append(raw[k])
581
+ k += 1
582
+ i = j
583
+ continue
584
+
585
+ if next_byte in (0x5D, 0x7D):
586
+ # Trailing comma: drop it, keep the whitespace.
587
+ out.extend(raw[i + 1:j])
588
+ i = j
589
+ continue
590
+
591
+ out.append(bch)
592
+ i += 1
593
+ continue
594
+
595
+ out.append(bch)
596
+ i += 1
597
+
598
+ return bytes(out)
599
+
600
+ def _pull(self) -> None:
601
+ """Sanitize one more chunk into the buffer, or mark EOF."""
602
+ raw = self._fh.read(self._chunk_size)
603
+ if raw:
604
+ self._out_buf += self._process_chunk(raw)
605
+ else:
606
+ # A comma still pending at EOF means the file ends mid-token.
607
+ # Emit it rather than lose data.
608
+ self._eof = True
609
+ self._out_buf += self._pending
610
+ self._pending = b''
611
+
612
+ def read(self, n: int = -1) -> bytes:
613
+ if n is None or n < 0:
614
+ while not self._eof:
615
+ self._pull()
616
+ result, self._out_buf = self._out_buf, b''
617
+ return result
618
+ while len(self._out_buf) < n and not self._eof:
619
+ self._pull()
620
+ result, self._out_buf = self._out_buf[:n], self._out_buf[n:]
621
+ return result
622
+
623
+ def readline(self) -> bytes:
624
+ """Return bytes up to and including the next newline, or to EOF."""
625
+ while True:
626
+ nl = self._out_buf.find(b'\n')
627
+ if nl != -1:
628
+ line, self._out_buf = self._out_buf[:nl + 1], self._out_buf[nl + 1:]
629
+ return line
630
+ if self._eof:
631
+ line, self._out_buf = self._out_buf, b''
632
+ return line
633
+ self._pull()
634
+
635
+ def __iter__(self):
636
+ return self
637
+
638
+ def __next__(self) -> bytes:
639
+ line = self.readline()
640
+ if not line:
641
+ raise StopIteration
642
+ return line
643
+
644
+ def seekable(self) -> bool:
645
+ return False
646
+
647
+ def readable(self) -> bool:
648
+ return True
649
+
650
+ def close(self) -> None:
651
+ self._fh.close()