pharmadata 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (166) hide show
  1. pharmadata/__init__.py +41 -0
  2. pharmadata/_core/__init__.py +1 -0
  3. pharmadata/_core/collection.py +119 -0
  4. pharmadata/_core/data.py +295 -0
  5. pharmadata/_core/meta.py +159 -0
  6. pharmadata/_core/output_format.py +139 -0
  7. pharmadata/_data/cdiscpilotadam/_collection.json +6 -0
  8. pharmadata/_data/cdiscpilotadam/_meta.json +3140 -0
  9. pharmadata/_data/cdiscpilotadam/adadas.parquet +0 -0
  10. pharmadata/_data/cdiscpilotadam/adae.parquet +0 -0
  11. pharmadata/_data/cdiscpilotadam/adcibc.parquet +0 -0
  12. pharmadata/_data/cdiscpilotadam/adlbc.parquet +0 -0
  13. pharmadata/_data/cdiscpilotadam/adlbcpv.parquet +0 -0
  14. pharmadata/_data/cdiscpilotadam/adlbh.parquet +0 -0
  15. pharmadata/_data/cdiscpilotadam/adlbhpv.parquet +0 -0
  16. pharmadata/_data/cdiscpilotadam/adlbhy.parquet +0 -0
  17. pharmadata/_data/cdiscpilotadam/adnpix.parquet +0 -0
  18. pharmadata/_data/cdiscpilotadam/adsl.parquet +0 -0
  19. pharmadata/_data/cdiscpilotadam/adtte.parquet +0 -0
  20. pharmadata/_data/cdiscpilotadam/advs.parquet +0 -0
  21. pharmadata/_data/cdiscpilotsdtm/_collection.json +6 -0
  22. pharmadata/_data/cdiscpilotsdtm/_meta.json +3321 -0
  23. pharmadata/_data/cdiscpilotsdtm/ae.parquet +0 -0
  24. pharmadata/_data/cdiscpilotsdtm/cm.parquet +0 -0
  25. pharmadata/_data/cdiscpilotsdtm/dm.parquet +0 -0
  26. pharmadata/_data/cdiscpilotsdtm/ds.parquet +0 -0
  27. pharmadata/_data/cdiscpilotsdtm/ex.parquet +0 -0
  28. pharmadata/_data/cdiscpilotsdtm/lbch.parquet +0 -0
  29. pharmadata/_data/cdiscpilotsdtm/lbhe.parquet +0 -0
  30. pharmadata/_data/cdiscpilotsdtm/lbur.parquet +0 -0
  31. pharmadata/_data/cdiscpilotsdtm/mh.parquet +0 -0
  32. pharmadata/_data/cdiscpilotsdtm/qsco.parquet +0 -0
  33. pharmadata/_data/cdiscpilotsdtm/qsda.parquet +0 -0
  34. pharmadata/_data/cdiscpilotsdtm/qsgi.parquet +0 -0
  35. pharmadata/_data/cdiscpilotsdtm/qshi.parquet +0 -0
  36. pharmadata/_data/cdiscpilotsdtm/qsmm.parquet +0 -0
  37. pharmadata/_data/cdiscpilotsdtm/qsni.parquet +0 -0
  38. pharmadata/_data/cdiscpilotsdtm/relrec.parquet +0 -0
  39. pharmadata/_data/cdiscpilotsdtm/sc.parquet +0 -0
  40. pharmadata/_data/cdiscpilotsdtm/se.parquet +0 -0
  41. pharmadata/_data/cdiscpilotsdtm/suppae.parquet +0 -0
  42. pharmadata/_data/cdiscpilotsdtm/suppdm.parquet +0 -0
  43. pharmadata/_data/cdiscpilotsdtm/suppds.parquet +0 -0
  44. pharmadata/_data/cdiscpilotsdtm/supplbch.parquet +0 -0
  45. pharmadata/_data/cdiscpilotsdtm/supplbhe.parquet +0 -0
  46. pharmadata/_data/cdiscpilotsdtm/supplbur.parquet +0 -0
  47. pharmadata/_data/cdiscpilotsdtm/sv.parquet +0 -0
  48. pharmadata/_data/cdiscpilotsdtm/ta.parquet +0 -0
  49. pharmadata/_data/cdiscpilotsdtm/te.parquet +0 -0
  50. pharmadata/_data/cdiscpilotsdtm/ti.parquet +0 -0
  51. pharmadata/_data/cdiscpilotsdtm/ts.parquet +0 -0
  52. pharmadata/_data/cdiscpilotsdtm/tv.parquet +0 -0
  53. pharmadata/_data/cdiscpilotsdtm/vs.parquet +0 -0
  54. pharmadata/_data/pharmaverseadam/_collection.json +6 -0
  55. pharmadata/_data/pharmaverseadam/_meta.json +14421 -0
  56. pharmadata/_data/pharmaverseadam/adab.parquet +0 -0
  57. pharmadata/_data/pharmaverseadam/adae.parquet +0 -0
  58. pharmadata/_data/pharmaverseadam/adapet_neuro.parquet +0 -0
  59. pharmadata/_data/pharmaverseadam/adbcva_ophtha.parquet +0 -0
  60. pharmadata/_data/pharmaverseadam/adce_vaccine.parquet +0 -0
  61. pharmadata/_data/pharmaverseadam/adcm.parquet +0 -0
  62. pharmadata/_data/pharmaverseadam/adcoeq_metabolic.parquet +0 -0
  63. pharmadata/_data/pharmaverseadam/adeg.parquet +0 -0
  64. pharmadata/_data/pharmaverseadam/adex.parquet +0 -0
  65. pharmadata/_data/pharmaverseadam/adface_vaccine.parquet +0 -0
  66. pharmadata/_data/pharmaverseadam/adis_vaccine.parquet +0 -0
  67. pharmadata/_data/pharmaverseadam/adlb.parquet +0 -0
  68. pharmadata/_data/pharmaverseadam/adlb_metabolic.parquet +0 -0
  69. pharmadata/_data/pharmaverseadam/adlb_neuro.parquet +0 -0
  70. pharmadata/_data/pharmaverseadam/adlbhy.parquet +0 -0
  71. pharmadata/_data/pharmaverseadam/admh.parquet +0 -0
  72. pharmadata/_data/pharmaverseadam/adnv_neuro.parquet +0 -0
  73. pharmadata/_data/pharmaverseadam/adoe_ophtha.parquet +0 -0
  74. pharmadata/_data/pharmaverseadam/adpc.parquet +0 -0
  75. pharmadata/_data/pharmaverseadam/adpp.parquet +0 -0
  76. pharmadata/_data/pharmaverseadam/adppk.parquet +0 -0
  77. pharmadata/_data/pharmaverseadam/adrs_onco.parquet +0 -0
  78. pharmadata/_data/pharmaverseadam/adsl.parquet +0 -0
  79. pharmadata/_data/pharmaverseadam/adsl_vaccine.parquet +0 -0
  80. pharmadata/_data/pharmaverseadam/adtpet_neuro.parquet +0 -0
  81. pharmadata/_data/pharmaverseadam/adtr_onco.parquet +0 -0
  82. pharmadata/_data/pharmaverseadam/adtte_onco.parquet +0 -0
  83. pharmadata/_data/pharmaverseadam/advfq_ophtha.parquet +0 -0
  84. pharmadata/_data/pharmaverseadam/advs.parquet +0 -0
  85. pharmadata/_data/pharmaverseadam/advs_metabolic.parquet +0 -0
  86. pharmadata/_data/pharmaverseadam/advs_peds.parquet +0 -0
  87. pharmadata/_data/pharmaversesdtm/_collection.json +6 -0
  88. pharmadata/_data/pharmaversesdtm/_meta.json +7518 -0
  89. pharmadata/_data/pharmaversesdtm/ae.parquet +0 -0
  90. pharmadata/_data/pharmaversesdtm/ae_ophtha.parquet +0 -0
  91. pharmadata/_data/pharmaversesdtm/ag_neuro.parquet +0 -0
  92. pharmadata/_data/pharmaversesdtm/be.parquet +0 -0
  93. pharmadata/_data/pharmaversesdtm/ce_vaccine.parquet +0 -0
  94. pharmadata/_data/pharmaversesdtm/cm.parquet +0 -0
  95. pharmadata/_data/pharmaversesdtm/dm.parquet +0 -0
  96. pharmadata/_data/pharmaversesdtm/dm_metabolic.parquet +0 -0
  97. pharmadata/_data/pharmaversesdtm/dm_neuro.parquet +0 -0
  98. pharmadata/_data/pharmaversesdtm/dm_peds.parquet +0 -0
  99. pharmadata/_data/pharmaversesdtm/dm_vaccine.parquet +0 -0
  100. pharmadata/_data/pharmaversesdtm/ds.parquet +0 -0
  101. pharmadata/_data/pharmaversesdtm/eg.parquet +0 -0
  102. pharmadata/_data/pharmaversesdtm/ex.parquet +0 -0
  103. pharmadata/_data/pharmaversesdtm/ex_ophtha.parquet +0 -0
  104. pharmadata/_data/pharmaversesdtm/ex_vaccine.parquet +0 -0
  105. pharmadata/_data/pharmaversesdtm/face_vaccine.parquet +0 -0
  106. pharmadata/_data/pharmaversesdtm/is_ada.parquet +0 -0
  107. pharmadata/_data/pharmaversesdtm/is_vaccine.parquet +0 -0
  108. pharmadata/_data/pharmaversesdtm/lb.parquet +0 -0
  109. pharmadata/_data/pharmaversesdtm/lb_metabolic.parquet +0 -0
  110. pharmadata/_data/pharmaversesdtm/lb_neuro.parquet +0 -0
  111. pharmadata/_data/pharmaversesdtm/lb_onco_pcwg3.parquet +0 -0
  112. pharmadata/_data/pharmaversesdtm/mb.parquet +0 -0
  113. pharmadata/_data/pharmaversesdtm/mh.parquet +0 -0
  114. pharmadata/_data/pharmaversesdtm/ms.parquet +0 -0
  115. pharmadata/_data/pharmaversesdtm/nv_neuro.parquet +0 -0
  116. pharmadata/_data/pharmaversesdtm/oe_ophtha.parquet +0 -0
  117. pharmadata/_data/pharmaversesdtm/pc.parquet +0 -0
  118. pharmadata/_data/pharmaversesdtm/pp.parquet +0 -0
  119. pharmadata/_data/pharmaversesdtm/qs_metabolic.parquet +0 -0
  120. pharmadata/_data/pharmaversesdtm/qs_ophtha.parquet +0 -0
  121. pharmadata/_data/pharmaversesdtm/rs_onco.parquet +0 -0
  122. pharmadata/_data/pharmaversesdtm/rs_onco_ca125.parquet +0 -0
  123. pharmadata/_data/pharmaversesdtm/rs_onco_imwg.parquet +0 -0
  124. pharmadata/_data/pharmaversesdtm/rs_onco_irecist.parquet +0 -0
  125. pharmadata/_data/pharmaversesdtm/rs_onco_lymphoma.parquet +0 -0
  126. pharmadata/_data/pharmaversesdtm/rs_onco_pcwg3.parquet +0 -0
  127. pharmadata/_data/pharmaversesdtm/rs_onco_recist.parquet +0 -0
  128. pharmadata/_data/pharmaversesdtm/sc_ophtha.parquet +0 -0
  129. pharmadata/_data/pharmaversesdtm/sdg_db.parquet +0 -0
  130. pharmadata/_data/pharmaversesdtm/smq_db.parquet +0 -0
  131. pharmadata/_data/pharmaversesdtm/suppae.parquet +0 -0
  132. pharmadata/_data/pharmaversesdtm/suppce_vaccine.parquet +0 -0
  133. pharmadata/_data/pharmaversesdtm/suppdm.parquet +0 -0
  134. pharmadata/_data/pharmaversesdtm/suppdm_vaccine.parquet +0 -0
  135. pharmadata/_data/pharmaversesdtm/suppds.parquet +0 -0
  136. pharmadata/_data/pharmaversesdtm/suppex_vaccine.parquet +0 -0
  137. pharmadata/_data/pharmaversesdtm/suppface_vaccine.parquet +0 -0
  138. pharmadata/_data/pharmaversesdtm/suppis_vaccine.parquet +0 -0
  139. pharmadata/_data/pharmaversesdtm/suppnv_neuro.parquet +0 -0
  140. pharmadata/_data/pharmaversesdtm/supprs_onco_ca125.parquet +0 -0
  141. pharmadata/_data/pharmaversesdtm/supprs_onco_imwg.parquet +0 -0
  142. pharmadata/_data/pharmaversesdtm/supptr_onco.parquet +0 -0
  143. pharmadata/_data/pharmaversesdtm/sv.parquet +0 -0
  144. pharmadata/_data/pharmaversesdtm/tr_onco.parquet +0 -0
  145. pharmadata/_data/pharmaversesdtm/tr_onco_recist.parquet +0 -0
  146. pharmadata/_data/pharmaversesdtm/ts.parquet +0 -0
  147. pharmadata/_data/pharmaversesdtm/tu_onco.parquet +0 -0
  148. pharmadata/_data/pharmaversesdtm/tu_onco_recist.parquet +0 -0
  149. pharmadata/_data/pharmaversesdtm/vs.parquet +0 -0
  150. pharmadata/_data/pharmaversesdtm/vs_metabolic.parquet +0 -0
  151. pharmadata/_data/pharmaversesdtm/vs_peds.parquet +0 -0
  152. pharmadata/_data/pharmaversesdtm/vs_vaccine.parquet +0 -0
  153. pharmadata/cdiscpilotadam/__init__.py +136 -0
  154. pharmadata/cdiscpilotadam/__init__.pyi +647 -0
  155. pharmadata/cdiscpilotsdtm/__init__.py +136 -0
  156. pharmadata/cdiscpilotsdtm/__init__.pyi +845 -0
  157. pharmadata/pharmaverseadam/__init__.py +136 -0
  158. pharmadata/pharmaverseadam/__init__.pyi +2690 -0
  159. pharmadata/pharmaversesdtm/__init__.py +136 -0
  160. pharmadata/pharmaversesdtm/__init__.pyi +1772 -0
  161. pharmadata/py.typed +0 -0
  162. pharmadata-0.1.0.dist-info/METADATA +71 -0
  163. pharmadata-0.1.0.dist-info/RECORD +166 -0
  164. pharmadata-0.1.0.dist-info/WHEEL +4 -0
  165. pharmadata-0.1.0.dist-info/licenses/LICENSE +177 -0
  166. pharmadata-0.1.0.dist-info/licenses/NOTICE +139 -0
pharmadata/__init__.py ADDED
@@ -0,0 +1,41 @@
1
+ """Access CDISC ADaM and SDTM test datasets in Python for clinical programming."""
2
+
3
+ import importlib
4
+ import importlib.metadata as _metadata
5
+
6
+ from pharmadata._core.data import COLLECTIONS as _COLLECTIONS
7
+ from pharmadata._core.meta import DatasetMeta
8
+ from pharmadata._core.output_format import (
9
+ get_output_format,
10
+ set_output_arrow,
11
+ set_output_pandas,
12
+ set_output_polars,
13
+ )
14
+ from pharmadata._core.output_format import (
15
+ use_output_format as output_format,
16
+ )
17
+
18
+ # The version lives in pyproject.toml
19
+ try:
20
+ __version__ = _metadata.version("pharmadata")
21
+ except (
22
+ _metadata.PackageNotFoundError
23
+ ): # pragma: no cover - only in an unpacked source tree
24
+ __version__ = "0.0.0"
25
+
26
+ # Import every discovered collection so ``pharmadata.<name>`` resolves. The set
27
+ # comes from ``_data``, not a hardcoded list, so a new collection appears here
28
+ # without an edit.
29
+ for _name in _COLLECTIONS:
30
+ globals()[_name] = importlib.import_module(f"pharmadata.{_name}")
31
+
32
+ __all__ = [
33
+ "DatasetMeta",
34
+ "__version__",
35
+ "get_output_format",
36
+ "output_format",
37
+ "set_output_arrow",
38
+ "set_output_pandas",
39
+ "set_output_polars",
40
+ *sorted(_COLLECTIONS),
41
+ ]
@@ -0,0 +1 @@
1
+ """Core infrastructure: registry, metadata, output format, and collection wiring."""
@@ -0,0 +1,119 @@
1
+ """Wire a collection onto its module: lazy dataset attributes, ``__all__``, ``__dir__``."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import sys
6
+ from functools import cache
7
+ from typing import TYPE_CHECKING, cast
8
+
9
+ import polars as pl
10
+
11
+ from pharmadata._core import data, meta, output_format
12
+
13
+ if TYPE_CHECKING:
14
+ # pandas is an optional extra; pyarrow is only named by the annotations
15
+ # (both live behind ``data``, which reads and converts the arrow base).
16
+ import pandas as pd
17
+ import pyarrow as pa
18
+
19
+ # The functions every collection module defines (and that __all__ lists).
20
+ HELPER_FUNCTIONS = (
21
+ "list_datasets",
22
+ "meta",
23
+ "meta_tables",
24
+ "meta_columns",
25
+ "meta_specs",
26
+ "to_dict",
27
+ )
28
+
29
+
30
+ @cache
31
+ def _load_polars(collection: str, name: str) -> pl.DataFrame:
32
+ """The process-wide cached polars frame; callers get a clone via ``load``."""
33
+ # from_arrow of a Table is a DataFrame; the stub returns the wider union.
34
+ return cast("pl.DataFrame", pl.from_arrow(data.load_dataset(collection, name)))
35
+
36
+
37
+ @cache
38
+ def _load_pandas(collection: str, name: str) -> pd.DataFrame:
39
+ """The process-wide cached pandas frame; callers get a copy via ``load``."""
40
+ # pyarrow's parquet reader initialises the pandas shim, which crashes when
41
+ # pandas is absent; ``load`` calls ``data.require_pandas`` before this runs.
42
+ return cast(
43
+ "pd.DataFrame", data.to_output(data.load_dataset(collection, name), "pandas")
44
+ )
45
+
46
+
47
+ def load(collection: str, name: str) -> pa.Table | pl.DataFrame | pd.DataFrame:
48
+ """One dataset on the active output format.
49
+
50
+ The arrow base is immutable, so ``"arrow"`` shares the cached table; polars
51
+ returns a fresh clone and pandas a deep copy, so a mutation never poisons
52
+ the cache another caller reads.
53
+ """
54
+ fmt = output_format.get_output_format()
55
+ if fmt == "arrow":
56
+ return data.load_dataset(collection, name)
57
+ if fmt == "pandas":
58
+ # before load_dataset: pyarrow would initialise the pandas shim first
59
+ data.require_pandas()
60
+ frame: pa.Table | pl.DataFrame | pd.DataFrame = _load_pandas(
61
+ collection, name
62
+ ).copy()
63
+ else:
64
+ frame = _load_polars(collection, name).clone()
65
+ # a clone/copy does not carry the cached frame's instance docstring
66
+ frame.__doc__ = data.dataset_doc(collection, name)
67
+ return frame
68
+
69
+
70
+ def load_all(
71
+ collection: str,
72
+ ) -> dict[str, pa.Table | pl.DataFrame | pd.DataFrame]:
73
+ """Every dataset of a collection keyed by name, honoring the active output format.
74
+
75
+ The values are polars frames by default, pyarrow Tables when
76
+ ``output_format="arrow"`` is active, and pandas frames when
77
+ ``output_format="pandas"`` is active.
78
+ """
79
+ return {name: load(collection, name) for name in data.list_datasets(collection)}
80
+
81
+
82
+ def install(module_name: str, spec: data.Collection) -> None:
83
+ """Expose *spec*'s datasets as lazy attributes on the module *module_name*."""
84
+ module = sys.modules.get(module_name)
85
+ if module is None: # pragma: no cover - only under an unusual loader
86
+ msg = f"{module_name} is not in sys.modules; cannot install collection API"
87
+ raise RuntimeError(msg)
88
+
89
+ names = data.dataset_names(spec.name)
90
+ sorted_names = data.list_datasets(spec.name)
91
+ module.__dict__["__all__"] = [
92
+ *HELPER_FUNCTIONS,
93
+ *sorted_names,
94
+ *(f"{name}_meta" for name in sorted_names),
95
+ ]
96
+
97
+ def __getattr__(
98
+ name: str,
99
+ ) -> pa.Table | pl.DataFrame | pd.DataFrame | meta.DatasetMeta:
100
+ """The dataset named *name*, or its metadata as ``<name>_meta``."""
101
+ if name.startswith("__") and name.endswith("__"):
102
+ # keep introspection (copy, pickle, IPython) on its normal path
103
+ raise AttributeError(name)
104
+ if name in names:
105
+ return load(spec.name, name)
106
+ stem = name.removesuffix("_meta")
107
+ if name.endswith("_meta") and stem in names:
108
+ return meta.build(spec.name, stem)
109
+ msg = (
110
+ f"module {module_name!r} has no dataset {name!r}; "
111
+ f"available: {', '.join(sorted_names)}"
112
+ )
113
+ raise AttributeError(msg)
114
+
115
+ def __dir__() -> list[str]:
116
+ return sorted(module.__dict__["__all__"])
117
+
118
+ module.__dict__["__getattr__"] = __getattr__
119
+ module.__dict__["__dir__"] = __dir__
@@ -0,0 +1,295 @@
1
+ """Dataset registry and loading: which datasets exist, their metadata, and I/O."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ from dataclasses import dataclass
7
+ from functools import cache
8
+ from pathlib import Path
9
+ from typing import TYPE_CHECKING, TypedDict, cast
10
+
11
+ import polars as pl
12
+ import pyarrow as pa
13
+ import pyarrow.parquet as pq
14
+
15
+ from pharmadata._core.output_format import OutputFormat, get_output_format
16
+
17
+ if TYPE_CHECKING:
18
+ # pandas is an optional extra; only named by the annotations below.
19
+ import pandas as pd
20
+
21
+
22
+ @dataclass(frozen=True, slots=True)
23
+ class Collection:
24
+ """One named group of datasets: module, title, and where it comes from.
25
+
26
+ ``url`` and ``example`` are optional: a bare new collection needs only a
27
+ ``title`` and ``source``. ``example`` falls back to the first dataset.
28
+ """
29
+
30
+ name: str # the submodule and the _data directory name
31
+ title: str # for the generated docs and the nav
32
+ source: str # the upstream source, named on its index page
33
+ url: str = "" # where that source is published
34
+ example: str = "" # a representative dataset, used in docstring examples
35
+
36
+
37
+ # The _meta.json contract: one entry per dataset, one spec per column.
38
+ class ColumnSpec(TypedDict):
39
+ """One column's entry in a collection's ``_meta.json``."""
40
+
41
+ label: str | None
42
+ type: str
43
+ mandatory: str | None
44
+ role: str | None
45
+
46
+
47
+ class DatasetMetaEntry(TypedDict):
48
+ """One dataset's entry in a collection's ``_meta.json``."""
49
+
50
+ label: str | None
51
+ structure: str | None
52
+ n_rows: int
53
+ columns: dict[str, ColumnSpec]
54
+
55
+
56
+ # The directory the shipped data lives in, a sibling of this ``_core`` package.
57
+ _DATA_ROOT = Path(__file__).resolve().parent.parent / "_data"
58
+
59
+
60
+ def _resolve_example(child: Path, example: str) -> str:
61
+ """Return *example*, else the first dataset declared in ``_meta.json``."""
62
+ if example:
63
+ return example
64
+ meta_path = child / "_meta.json"
65
+ if not meta_path.is_file():
66
+ return ""
67
+ datasets = json.loads(meta_path.read_text(encoding="utf-8"))
68
+ return next(iter(sorted(datasets)), "")
69
+
70
+
71
+ def _discover_collections() -> dict[str, Collection]:
72
+ """Every collection declared by a ``_collection.json`` under ``_data``.
73
+
74
+ A directory becomes a collection by dropping a ``_collection.json`` beside
75
+ its ``_meta.json``. Only ``title`` and ``source`` are required; ``url`` and
76
+ ``example`` are optional, and an omitted ``example`` falls back to the first
77
+ dataset. Nothing is hardcoded, so a new dataset group is picked up without
78
+ editing Python.
79
+ """
80
+ found: dict[str, Collection] = {}
81
+ if not _DATA_ROOT.is_dir(): # pragma: no cover - the data ships with the package
82
+ return found
83
+ for child in sorted(_DATA_ROOT.iterdir()):
84
+ config = child / "_collection.json"
85
+ if child.is_dir() and config.is_file():
86
+ raw = json.loads(config.read_text(encoding="utf-8"))
87
+ found[child.name] = Collection(
88
+ name=child.name,
89
+ title=str(raw["title"]),
90
+ source=str(raw["source"]),
91
+ url=str(raw.get("url", "")),
92
+ example=_resolve_example(child, str(raw.get("example", ""))),
93
+ )
94
+ return found
95
+
96
+
97
+ # The single source of truth for which collections exist: read from the data.
98
+ COLLECTIONS: dict[str, Collection] = _discover_collections()
99
+
100
+
101
+ @cache
102
+ def data_dir(collection: str) -> Path:
103
+ """Directory holding the parquet files and metadata of a collection."""
104
+ if collection not in COLLECTIONS:
105
+ msg = (
106
+ f"unknown collection {collection!r}; expected one of {sorted(COLLECTIONS)}"
107
+ )
108
+ raise ValueError(msg)
109
+ return _DATA_ROOT / collection
110
+
111
+
112
+ @cache
113
+ def raw_meta(collection: str) -> dict[str, DatasetMetaEntry]:
114
+ """Raw metadata of a collection: dataset -> label/structure/columns."""
115
+ path = data_dir(collection) / "_meta.json"
116
+ return json.loads(path.read_text(encoding="utf-8"))
117
+
118
+
119
+ @cache
120
+ def dataset_names(collection: str) -> frozenset[str]:
121
+ """Names of the datasets available in a collection."""
122
+ return frozenset(raw_meta(collection))
123
+
124
+
125
+ def check_dataset(collection: str, name: str) -> None:
126
+ """Raise a helpful error when *name* is not a dataset in *collection*."""
127
+ if name not in dataset_names(collection):
128
+ msg = (
129
+ f"no dataset {name!r} in collection {collection!r}; "
130
+ f"available: {', '.join(list_datasets(collection))}"
131
+ )
132
+ raise ValueError(msg)
133
+
134
+
135
+ def _meta_entry(collection: str, name: str) -> DatasetMetaEntry:
136
+ """The raw metadata dict of one dataset, after validating it exists."""
137
+ check_dataset(collection, name)
138
+ return raw_meta(collection)[name]
139
+
140
+
141
+ def list_datasets(collection: str) -> list[str]:
142
+ """Names of the datasets available in a collection, sorted alphabetically."""
143
+ return sorted(dataset_names(collection))
144
+
145
+
146
+ @cache
147
+ def load_dataset(collection: str, name: str) -> pa.Table:
148
+ """Load a dataset as an immutable pyarrow Table, the internal base.
149
+
150
+ pyarrow reads the parquet on every platform, including the browser, where
151
+ polars' own parquet reader is unavailable; the served frame is the arrow
152
+ table converted on demand by :func:`to_output`.
153
+ """
154
+ check_dataset(collection, name)
155
+ return pq.read_table(data_dir(collection) / f"{name}.parquet")
156
+
157
+
158
+ # The one pandas-missing message, raised wherever "pandas" output is asked for.
159
+ PANDAS_REQUIRED_MSG = (
160
+ 'pandas is required for output_format="pandas"; '
161
+ 'install with: pip install "pharmadata[pandas]"'
162
+ )
163
+
164
+
165
+ def require_pandas() -> None:
166
+ """Raise ImportError unless pandas can be imported.
167
+
168
+ ``sys.modules["pandas"] = None`` (the test fixture for a missing pandas)
169
+ makes ``import pandas`` return None without raising, so check the bound
170
+ name. Readers of parquet must call this first: pyarrow's reader
171
+ initialises the pandas shim and crashes on it when pandas is absent.
172
+ """
173
+ try:
174
+ import pandas as pd
175
+ except ImportError:
176
+ pd = None # type: ignore[assignment]
177
+ if pd is None:
178
+ msg = PANDAS_REQUIRED_MSG
179
+ raise ImportError(msg)
180
+
181
+
182
+ def to_output(
183
+ table: pa.Table, fmt: OutputFormat
184
+ ) -> pa.Table | pl.DataFrame | pd.DataFrame:
185
+ """Convert the arrow base into the requested output format.
186
+
187
+ ``"arrow"`` hands the table back untouched; polars crosses the Arrow C data
188
+ interface with ``pl.from_arrow``; pandas converts through ``to_pandas``,
189
+ naming the extra to install when pandas is absent.
190
+ """
191
+ if fmt == "arrow":
192
+ return table
193
+ if fmt == "pandas":
194
+ require_pandas()
195
+ return table.to_pandas()
196
+ if fmt == "polars":
197
+ # a Table converts to a DataFrame; from_arrow's stub returns the wider union.
198
+ return cast("pl.DataFrame", pl.from_arrow(table))
199
+ msg = f"unknown output format: {fmt!r}" # pragma: no cover
200
+ raise ValueError(msg) # pragma: no cover
201
+
202
+
203
+ @cache
204
+ def dataset_doc(collection: str, name: str) -> str:
205
+ """The docstring of a dataset, generated from its CDISC metadata."""
206
+ entry = _meta_entry(collection, name)
207
+ spec = COLLECTIONS[collection]
208
+ rows, cols = shape(collection, name)
209
+ lines = [
210
+ f"{name} - {entry['label'] or name}",
211
+ "",
212
+ f"{spec.title}, from {spec.source} (<{spec.url}>).",
213
+ f"{rows} rows x {cols} columns.",
214
+ ]
215
+ if entry["structure"]:
216
+ lines.append(f"Structure: {entry['structure']}.")
217
+ lines += ["", "Variables", "---------"]
218
+ lines += [
219
+ f"{column} ({info['type']}): {info['label'] or ''}".rstrip()
220
+ for column, info in entry["columns"].items()
221
+ ]
222
+ return "\n".join(lines) + "\n"
223
+
224
+
225
+ @cache
226
+ def shape(collection: str, name: str) -> tuple[int, int]:
227
+ """Row and column counts of a dataset, without loading any data."""
228
+ entry = _meta_entry(collection, name)
229
+ return entry["n_rows"], len(entry["columns"])
230
+
231
+
232
+ @cache
233
+ def _meta_tables_table(collection: str) -> pa.Table:
234
+ """Arrow base of the dataset-level metadata: one row per dataset."""
235
+ meta = raw_meta(collection)
236
+ names = list_datasets(collection)
237
+ return pa.table(
238
+ {
239
+ "dataset": names,
240
+ "label": [meta[n]["label"] for n in names],
241
+ "n_rows": [meta[n]["n_rows"] for n in names],
242
+ "n_cols": [len(meta[n]["columns"]) for n in names],
243
+ }
244
+ )
245
+
246
+
247
+ @cache
248
+ def _meta_specs_table(collection: str) -> pa.Table:
249
+ """Arrow base of the column-level specifications of every dataset."""
250
+ meta = raw_meta(collection)
251
+ dataset: list[str] = []
252
+ variable: list[str] = []
253
+ label: list[str | None] = []
254
+ column_type: list[str] = []
255
+ mandatory: list[str | None] = []
256
+ role: list[str | None] = []
257
+ for name in list_datasets(collection):
258
+ for col, info in meta[name]["columns"].items():
259
+ dataset.append(name)
260
+ variable.append(col)
261
+ label.append(info["label"])
262
+ column_type.append(info["type"])
263
+ mandatory.append(info.get("mandatory"))
264
+ role.append(info.get("role"))
265
+ return pa.table(
266
+ {
267
+ "dataset": dataset,
268
+ "variable": variable,
269
+ "label": label,
270
+ "type": column_type,
271
+ "mandatory": mandatory,
272
+ "role": role,
273
+ }
274
+ )
275
+
276
+
277
+ def meta_tables(collection: str) -> pa.Table | pl.DataFrame | pd.DataFrame:
278
+ """Dataset-level metadata: one row per dataset, in the active format."""
279
+ return to_output(_meta_tables_table(collection), get_output_format())
280
+
281
+
282
+ def meta_columns(collection: str) -> pa.Table | pl.DataFrame | pd.DataFrame:
283
+ """Column-level metadata of every dataset (long format), in the active format.
284
+
285
+ The specifications minus the ``mandatory`` and ``role`` columns.
286
+ """
287
+ table = _meta_specs_table(collection).select(
288
+ ["dataset", "variable", "label", "type"]
289
+ )
290
+ return to_output(table, get_output_format())
291
+
292
+
293
+ def meta_specs(collection: str) -> pa.Table | pl.DataFrame | pd.DataFrame:
294
+ """Column-level specifications of every dataset (long format), active format."""
295
+ return to_output(_meta_specs_table(collection), get_output_format())
@@ -0,0 +1,159 @@
1
+ """Metadata of one dataset: the ``DatasetMeta`` value object."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from functools import cache
7
+ from typing import TYPE_CHECKING, ClassVar, Literal
8
+
9
+ import pyarrow as pa
10
+
11
+ from pharmadata._core import data, output_format
12
+
13
+ if TYPE_CHECKING:
14
+ import pandas as pd
15
+ import polars as pl
16
+
17
+
18
+ @dataclass(frozen=True, slots=True)
19
+ class DatasetMeta:
20
+ """Metadata of one dataset, in the style of pyreadstat's metadata object.
21
+
22
+ A read-only view: the field names and parallel-tuple layout follow
23
+ pyreadstat, extended with the CDISC dimensions (``structure``,
24
+ ``mandatory``, ``role``). The data itself is reached through the module
25
+ attribute of the same name (e.g. ``pharmaverseadam.adsl``).
26
+
27
+ Attributes
28
+ ----------
29
+ dataset : str
30
+ The dataset name, e.g. ``"adsl"``.
31
+ collection : str
32
+ The collection it belongs to, e.g. ``"pharmaverseadam"``.
33
+ file_label : str or None
34
+ The dataset label, equivalent to a SAS dataset label.
35
+ structure : str or None
36
+ The CDISC structure, e.g. ``"One record per subject"``; None when the
37
+ source specifications do not provide one.
38
+ number_rows : int
39
+ Row count, read from the parquet footer without loading any data.
40
+ number_columns : int
41
+ Column count, i.e. ``len(column_names)``.
42
+ column_names : tuple of str
43
+ The column names, in column order.
44
+ column_labels : tuple of str or None
45
+ The column labels, parallel to ``column_names``.
46
+ column_types : tuple of str or None
47
+ The column types, parallel to ``column_names``.
48
+ column_mandatory : tuple of str or None
49
+ The CDISC mandatory flags, parallel to ``column_names``.
50
+ column_roles : tuple of str or None
51
+ The CDISC roles, parallel to ``column_names``.
52
+ file_format : str
53
+ Always ``"parquet"``, the storage format of every dataset.
54
+ file_encoding : str
55
+ Always ``"utf-8"``: the encoding of the shipped parquet data, not of
56
+ the original source file.
57
+
58
+ Examples
59
+ --------
60
+ >>> meta = pharmaverseadam.meta("adsl")
61
+ >>> meta.dataset
62
+ 'adsl'
63
+ >>> meta.file_label
64
+ 'Subject Level Analysis'
65
+ >>> meta.column_names_to_labels["STUDYID"]
66
+ 'Study Identifier'
67
+ >>> (meta.number_rows, meta.number_columns)
68
+ (306, 55)
69
+ """
70
+
71
+ dataset: str
72
+ collection: str
73
+ file_label: str | None
74
+ structure: str | None
75
+ number_rows: int
76
+ column_names: tuple[str, ...]
77
+ column_labels: tuple[str | None, ...]
78
+ column_types: tuple[str | None, ...]
79
+ column_mandatory: tuple[str | None, ...]
80
+ column_roles: tuple[str | None, ...]
81
+
82
+ file_format: ClassVar[Literal["parquet"]] = "parquet"
83
+ file_encoding: ClassVar[Literal["utf-8"]] = "utf-8"
84
+
85
+ def __repr__(self) -> str:
86
+ """One line: which dataset, and how big."""
87
+ return (
88
+ f"DatasetMeta({self.collection}.{self.dataset}, "
89
+ f"{self.number_rows} rows x {self.number_columns} columns)"
90
+ )
91
+
92
+ @property
93
+ def number_columns(self) -> int:
94
+ """Column count, i.e. ``len(column_names)``."""
95
+ return len(self.column_names)
96
+
97
+ @property
98
+ def column_names_to_labels(self) -> dict[str, str | None]:
99
+ """The column names mapped to their labels, in column order.
100
+
101
+ Returns
102
+ -------
103
+ dict[str, str or None]
104
+ A dict mapping each column name to its CDISC label.
105
+
106
+ Examples
107
+ --------
108
+ >>> pharmaverseadam.adsl_meta.column_names_to_labels["AGE"]
109
+ 'Age'
110
+ """
111
+ return dict(zip(self.column_names, self.column_labels, strict=True))
112
+
113
+ @property
114
+ def specs(self) -> pa.Table | pl.DataFrame | pd.DataFrame:
115
+ """Column-level specifications of the dataset, in the active format.
116
+
117
+ Returns
118
+ -------
119
+ pa.Table or pl.DataFrame or pd.DataFrame
120
+ One row per variable with the columns ``variable, label, type,
121
+ mandatory, role`` (mandatory/role are null when the source
122
+ specifications do not provide them). The frame is served in the
123
+ active output format, like the dataset attributes.
124
+
125
+ Examples
126
+ --------
127
+ >>> pharmaverseadam.adsl_meta.specs.columns
128
+ ['variable', 'label', 'type', 'mandatory', 'role']
129
+ """
130
+ table = pa.table(
131
+ {
132
+ "variable": self.column_names,
133
+ "label": self.column_labels,
134
+ "type": self.column_types,
135
+ "mandatory": self.column_mandatory,
136
+ "role": self.column_roles,
137
+ }
138
+ )
139
+ return data.to_output(table, output_format.get_output_format())
140
+
141
+
142
+ @cache
143
+ def build(collection: str, name: str) -> DatasetMeta:
144
+ """The DatasetMeta of one dataset, built once and shared process-wide."""
145
+ entry = data._meta_entry(collection, name)
146
+ n_rows, _ = data.shape(collection, name)
147
+ columns = entry["columns"]
148
+ return DatasetMeta(
149
+ dataset=name,
150
+ collection=collection,
151
+ file_label=entry.get("label"),
152
+ structure=entry.get("structure"),
153
+ number_rows=n_rows,
154
+ column_names=tuple(columns),
155
+ column_labels=tuple(info.get("label") for info in columns.values()),
156
+ column_types=tuple(info.get("type") for info in columns.values()),
157
+ column_mandatory=tuple(info.get("mandatory") for info in columns.values()),
158
+ column_roles=tuple(info.get("role") for info in columns.values()),
159
+ )