mrfkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
mrfkit/__init__.py ADDED
@@ -0,0 +1,68 @@
1
+ """Parse US price transparency files (MRFs) into clean, normalized rows.
2
+
3
+ import mrfkit
4
+
5
+ for record in mrfkit.iter_records("hospital_standardcharges.csv"):
6
+ ...
7
+
8
+ Hospital files (CSV or JSON) and insurer Transparency in Coverage files
9
+ (in-network rates and tables of contents) are both read. See
10
+ :mod:`mrfkit.records` for what comes out.
11
+ """
12
+
13
+ from __future__ import annotations
14
+
15
+ from pathlib import Path
16
+ from typing import Any, Iterator, Optional, Union
17
+
18
+ from .csv_reader import iter_csv
19
+ from .files import detect_file_format
20
+ from .json_reader import iter_json
21
+ from .records import (
22
+ ChargeItem,
23
+ FileMetadata,
24
+ HeaderMapping,
25
+ ModifierInfo,
26
+ PayerRate,
27
+ StandardCharge,
28
+ TicFileMetadata,
29
+ TicIndexEntry,
30
+ TicProviderGroup,
31
+ TicRate,
32
+ UnmappedCell,
33
+ )
34
+ from .reference import ParseStats, ReferenceData
35
+ from .sinks import CsvSink, ParquetSink, open_sink
36
+ from .tic import detect_tic_file, group_tic_networks, iter_tic_in_network, iter_tic_index, tic_network_key
37
+
38
+ __version__ = "0.1.0"
39
+
40
+ __all__ = [
41
+ "ChargeItem", "CsvSink", "FileMetadata", "HeaderMapping", "ModifierInfo", "ParquetSink",
42
+ "ParseStats", "PayerRate", "ReferenceData", "StandardCharge", "TicFileMetadata",
43
+ "TicIndexEntry", "TicProviderGroup", "TicRate", "UnmappedCell", "detect_file_format",
44
+ "group_tic_networks", "iter_csv", "iter_json", "iter_records", "iter_tic_in_network",
45
+ "iter_tic_index", "open_sink", "tic_network_key",
46
+ ]
47
+
48
+
49
+ def iter_records(path: Union[str, Path], compression: Optional[str] = None, **options: Any) -> Iterator[Any]:
50
+ """Yield every record in an MRF, CSV or JSON, plain or compressed.
51
+
52
+ A JSON file whose root has ``in_network`` or ``provider_references`` is
53
+ read by :func:`iter_tic_in_network`, one with ``reporting_structure`` by
54
+ :func:`iter_tic_index`, and every other file as a hospital MRF.
55
+
56
+ *options* go to :func:`iter_csv` / :func:`iter_json`: ``stats``,
57
+ ``extra_synonyms``, ``header_overrides``, ``code_extraction``, ``ref``.
58
+ The TiC readers take only ``stats``.
59
+ """
60
+ fmt, detected = detect_file_format(Path(path))
61
+ compression = compression or detected
62
+ if fmt == "json":
63
+ kind = detect_tic_file(path, compression)
64
+ if kind is not None:
65
+ reader = iter_tic_in_network if kind == "in_network" else iter_tic_index
66
+ return reader(path, compression, stats=options.get("stats"))
67
+ reader = iter_csv if fmt == "csv" else iter_json
68
+ return reader(path, compression, **options)
mrfkit/__main__.py ADDED
@@ -0,0 +1,3 @@
1
+ from .cli import main
2
+
3
+ raise SystemExit(main())
mrfkit/cli.py ADDED
@@ -0,0 +1,79 @@
1
+ """``mrfkit FILE...``: turn hospital and insurer MRFs into CSV or Parquet tables."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import argparse
6
+ import json
7
+ import logging
8
+ import sys
9
+ from collections import Counter
10
+ from pathlib import Path
11
+ from typing import List, Optional
12
+
13
+ from . import __version__, iter_records
14
+ from .reference import ParseStats
15
+ from .sinks import open_sink
16
+
17
+
18
+ def _parser() -> argparse.ArgumentParser:
19
+ p = argparse.ArgumentParser(
20
+ prog="mrfkit",
21
+ description="Turn US price transparency files (hospital CSV or JSON, insurer "
22
+ "Transparency in Coverage JSON; plain, gzip or zip) into clean tables: "
23
+ "one file per record type.")
24
+ p.add_argument("files", nargs="+", type=Path, help="MRF files to read")
25
+ p.add_argument("-o", "--out", type=Path, default=Path("mrfkit-out"),
26
+ help="output directory (default: mrfkit-out). With several input "
27
+ "files, each gets its own subdirectory")
28
+ p.add_argument("-f", "--format", choices=["csv", "parquet"], default="csv",
29
+ help="output format (parquet needs: pip install \"mrfkit[parquet]\")")
30
+ p.add_argument("--synonyms", type=Path,
31
+ help="JSON file mapping extra header names to fields, "
32
+ "e.g. {\"charge_amt\": \"gross_charge\"}")
33
+ p.add_argument("-v", "--verbose", action="store_true",
34
+ help="log layout detection and list unmapped JSON paths")
35
+ p.add_argument("--version", action="version", version=f"mrfkit {__version__}")
36
+ return p
37
+
38
+
39
+ def main(argv: Optional[List[str]] = None) -> int:
40
+ parser = _parser()
41
+ args = parser.parse_args(argv)
42
+ if args.format == "parquet":
43
+ try:
44
+ import pyarrow # noqa: F401
45
+ except ImportError:
46
+ parser.error('Parquet output needs pyarrow: pip install "mrfkit[parquet]"')
47
+ logging.basicConfig(level=logging.INFO if args.verbose else logging.WARNING,
48
+ format="%(levelname)s %(message)s")
49
+ synonyms = json.loads(args.synonyms.read_text()) if args.synonyms else None
50
+
51
+ failed = 0
52
+ for path in args.files:
53
+ out_dir = args.out if len(args.files) == 1 else args.out / path.name.split(".")[0]
54
+ stats = ParseStats()
55
+ counts: Counter = Counter()
56
+ try:
57
+ with open_sink(out_dir, args.format) as sink:
58
+ for record in iter_records(path, stats=stats, extra_synonyms=synonyms):
59
+ sink.write(record)
60
+ counts[record.TABLE] += 1
61
+ except (OSError, ValueError) as exc:
62
+ failed += 1
63
+ print(f"{path}: error: {exc}", file=sys.stderr)
64
+ continue
65
+ summary = ", ".join(f"{n:,} {table}" for table, n in sorted(counts.items()))
66
+ print(f"{path} -> {out_dir}: {stats.rows_read:,} rows read; {summary or 'no records'}",
67
+ file=sys.stderr)
68
+ if stats.rejected_codes:
69
+ print(f" {stats.rejected_codes:,} rows dropped for corrupt codes", file=sys.stderr)
70
+ for warning in stats.warnings:
71
+ print(f" warning: {warning}", file=sys.stderr)
72
+ if stats.unmapped_paths:
73
+ n_paths = len(stats.unmapped_paths)
74
+ print(f" {n_paths:,} unmapped JSON path{'' if n_paths == 1 else 's'}"
75
+ + ("" if args.verbose else " (-v lists them)"), file=sys.stderr)
76
+ if args.verbose:
77
+ for json_path, n in sorted(stats.unmapped_paths.items()):
78
+ print(f" {json_path}: {n:,}", file=sys.stderr)
79
+ return 1 if failed else 0