mrfkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mrfkit/__init__.py +68 -0
- mrfkit/__main__.py +3 -0
- mrfkit/cli.py +79 -0
- mrfkit/codes.py +1607 -0
- mrfkit/csv_reader.py +324 -0
- mrfkit/files.py +651 -0
- mrfkit/headers.py +737 -0
- mrfkit/json_reader.py +603 -0
- mrfkit/payers.py +2755 -0
- mrfkit/records.py +260 -0
- mrfkit/reference.py +136 -0
- mrfkit/sinks.py +126 -0
- mrfkit/tabular.py +651 -0
- mrfkit/tic.py +660 -0
- mrfkit/values.py +627 -0
- mrfkit-0.1.0.dist-info/METADATA +136 -0
- mrfkit-0.1.0.dist-info/RECORD +21 -0
- mrfkit-0.1.0.dist-info/WHEEL +4 -0
- mrfkit-0.1.0.dist-info/entry_points.txt +2 -0
- mrfkit-0.1.0.dist-info/licenses/LICENSE +201 -0
- mrfkit-0.1.0.dist-info/licenses/NOTICE +4 -0
mrfkit/__init__.py
ADDED
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
"""Parse US price transparency files (MRFs) into clean, normalized rows.
|
|
2
|
+
|
|
3
|
+
import mrfkit
|
|
4
|
+
|
|
5
|
+
for record in mrfkit.iter_records("hospital_standardcharges.csv"):
|
|
6
|
+
...
|
|
7
|
+
|
|
8
|
+
Hospital files (CSV or JSON) and insurer Transparency in Coverage files
|
|
9
|
+
(in-network rates and tables of contents) are both read. See
|
|
10
|
+
:mod:`mrfkit.records` for what comes out.
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
from __future__ import annotations
|
|
14
|
+
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
from typing import Any, Iterator, Optional, Union
|
|
17
|
+
|
|
18
|
+
from .csv_reader import iter_csv
|
|
19
|
+
from .files import detect_file_format
|
|
20
|
+
from .json_reader import iter_json
|
|
21
|
+
from .records import (
|
|
22
|
+
ChargeItem,
|
|
23
|
+
FileMetadata,
|
|
24
|
+
HeaderMapping,
|
|
25
|
+
ModifierInfo,
|
|
26
|
+
PayerRate,
|
|
27
|
+
StandardCharge,
|
|
28
|
+
TicFileMetadata,
|
|
29
|
+
TicIndexEntry,
|
|
30
|
+
TicProviderGroup,
|
|
31
|
+
TicRate,
|
|
32
|
+
UnmappedCell,
|
|
33
|
+
)
|
|
34
|
+
from .reference import ParseStats, ReferenceData
|
|
35
|
+
from .sinks import CsvSink, ParquetSink, open_sink
|
|
36
|
+
from .tic import detect_tic_file, group_tic_networks, iter_tic_in_network, iter_tic_index, tic_network_key
|
|
37
|
+
|
|
38
|
+
__version__ = "0.1.0"
|
|
39
|
+
|
|
40
|
+
__all__ = [
|
|
41
|
+
"ChargeItem", "CsvSink", "FileMetadata", "HeaderMapping", "ModifierInfo", "ParquetSink",
|
|
42
|
+
"ParseStats", "PayerRate", "ReferenceData", "StandardCharge", "TicFileMetadata",
|
|
43
|
+
"TicIndexEntry", "TicProviderGroup", "TicRate", "UnmappedCell", "detect_file_format",
|
|
44
|
+
"group_tic_networks", "iter_csv", "iter_json", "iter_records", "iter_tic_in_network",
|
|
45
|
+
"iter_tic_index", "open_sink", "tic_network_key",
|
|
46
|
+
]
|
|
47
|
+
|
|
48
|
+
|
|
49
|
+
def iter_records(path: Union[str, Path], compression: Optional[str] = None, **options: Any) -> Iterator[Any]:
|
|
50
|
+
"""Yield every record in an MRF, CSV or JSON, plain or compressed.
|
|
51
|
+
|
|
52
|
+
A JSON file whose root has ``in_network`` or ``provider_references`` is
|
|
53
|
+
read by :func:`iter_tic_in_network`, one with ``reporting_structure`` by
|
|
54
|
+
:func:`iter_tic_index`, and every other file as a hospital MRF.
|
|
55
|
+
|
|
56
|
+
*options* go to :func:`iter_csv` / :func:`iter_json`: ``stats``,
|
|
57
|
+
``extra_synonyms``, ``header_overrides``, ``code_extraction``, ``ref``.
|
|
58
|
+
The TiC readers take only ``stats``.
|
|
59
|
+
"""
|
|
60
|
+
fmt, detected = detect_file_format(Path(path))
|
|
61
|
+
compression = compression or detected
|
|
62
|
+
if fmt == "json":
|
|
63
|
+
kind = detect_tic_file(path, compression)
|
|
64
|
+
if kind is not None:
|
|
65
|
+
reader = iter_tic_in_network if kind == "in_network" else iter_tic_index
|
|
66
|
+
return reader(path, compression, stats=options.get("stats"))
|
|
67
|
+
reader = iter_csv if fmt == "csv" else iter_json
|
|
68
|
+
return reader(path, compression, **options)
|
mrfkit/__main__.py
ADDED
mrfkit/cli.py
ADDED
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
"""``mrfkit FILE...``: turn hospital and insurer MRFs into CSV or Parquet tables."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import argparse
|
|
6
|
+
import json
|
|
7
|
+
import logging
|
|
8
|
+
import sys
|
|
9
|
+
from collections import Counter
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import List, Optional
|
|
12
|
+
|
|
13
|
+
from . import __version__, iter_records
|
|
14
|
+
from .reference import ParseStats
|
|
15
|
+
from .sinks import open_sink
|
|
16
|
+
|
|
17
|
+
|
|
18
|
+
def _parser() -> argparse.ArgumentParser:
|
|
19
|
+
p = argparse.ArgumentParser(
|
|
20
|
+
prog="mrfkit",
|
|
21
|
+
description="Turn US price transparency files (hospital CSV or JSON, insurer "
|
|
22
|
+
"Transparency in Coverage JSON; plain, gzip or zip) into clean tables: "
|
|
23
|
+
"one file per record type.")
|
|
24
|
+
p.add_argument("files", nargs="+", type=Path, help="MRF files to read")
|
|
25
|
+
p.add_argument("-o", "--out", type=Path, default=Path("mrfkit-out"),
|
|
26
|
+
help="output directory (default: mrfkit-out). With several input "
|
|
27
|
+
"files, each gets its own subdirectory")
|
|
28
|
+
p.add_argument("-f", "--format", choices=["csv", "parquet"], default="csv",
|
|
29
|
+
help="output format (parquet needs: pip install \"mrfkit[parquet]\")")
|
|
30
|
+
p.add_argument("--synonyms", type=Path,
|
|
31
|
+
help="JSON file mapping extra header names to fields, "
|
|
32
|
+
"e.g. {\"charge_amt\": \"gross_charge\"}")
|
|
33
|
+
p.add_argument("-v", "--verbose", action="store_true",
|
|
34
|
+
help="log layout detection and list unmapped JSON paths")
|
|
35
|
+
p.add_argument("--version", action="version", version=f"mrfkit {__version__}")
|
|
36
|
+
return p
|
|
37
|
+
|
|
38
|
+
|
|
39
|
+
def main(argv: Optional[List[str]] = None) -> int:
|
|
40
|
+
parser = _parser()
|
|
41
|
+
args = parser.parse_args(argv)
|
|
42
|
+
if args.format == "parquet":
|
|
43
|
+
try:
|
|
44
|
+
import pyarrow # noqa: F401
|
|
45
|
+
except ImportError:
|
|
46
|
+
parser.error('Parquet output needs pyarrow: pip install "mrfkit[parquet]"')
|
|
47
|
+
logging.basicConfig(level=logging.INFO if args.verbose else logging.WARNING,
|
|
48
|
+
format="%(levelname)s %(message)s")
|
|
49
|
+
synonyms = json.loads(args.synonyms.read_text()) if args.synonyms else None
|
|
50
|
+
|
|
51
|
+
failed = 0
|
|
52
|
+
for path in args.files:
|
|
53
|
+
out_dir = args.out if len(args.files) == 1 else args.out / path.name.split(".")[0]
|
|
54
|
+
stats = ParseStats()
|
|
55
|
+
counts: Counter = Counter()
|
|
56
|
+
try:
|
|
57
|
+
with open_sink(out_dir, args.format) as sink:
|
|
58
|
+
for record in iter_records(path, stats=stats, extra_synonyms=synonyms):
|
|
59
|
+
sink.write(record)
|
|
60
|
+
counts[record.TABLE] += 1
|
|
61
|
+
except (OSError, ValueError) as exc:
|
|
62
|
+
failed += 1
|
|
63
|
+
print(f"{path}: error: {exc}", file=sys.stderr)
|
|
64
|
+
continue
|
|
65
|
+
summary = ", ".join(f"{n:,} {table}" for table, n in sorted(counts.items()))
|
|
66
|
+
print(f"{path} -> {out_dir}: {stats.rows_read:,} rows read; {summary or 'no records'}",
|
|
67
|
+
file=sys.stderr)
|
|
68
|
+
if stats.rejected_codes:
|
|
69
|
+
print(f" {stats.rejected_codes:,} rows dropped for corrupt codes", file=sys.stderr)
|
|
70
|
+
for warning in stats.warnings:
|
|
71
|
+
print(f" warning: {warning}", file=sys.stderr)
|
|
72
|
+
if stats.unmapped_paths:
|
|
73
|
+
n_paths = len(stats.unmapped_paths)
|
|
74
|
+
print(f" {n_paths:,} unmapped JSON path{'' if n_paths == 1 else 's'}"
|
|
75
|
+
+ ("" if args.verbose else " (-v lists them)"), file=sys.stderr)
|
|
76
|
+
if args.verbose:
|
|
77
|
+
for json_path, n in sorted(stats.unmapped_paths.items()):
|
|
78
|
+
print(f" {json_path}: {n:,}", file=sys.stderr)
|
|
79
|
+
return 1 if failed else 0
|