phactor 0.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
phactor/pandas.py ADDED
@@ -0,0 +1,589 @@
1
+ """DataFrame converters for Phactor cohort analysis results.
2
+
3
+ Install with::
4
+
5
+ pip install 'phactor[pandas]'
6
+ """
7
+ from __future__ import annotations
8
+
9
+ from typing import TYPE_CHECKING
10
+
11
+ if TYPE_CHECKING:
12
+ import pandas as pd
13
+
14
+ try:
15
+ import pandas as pd
16
+ except ImportError:
17
+ raise ImportError(
18
+ "Install phactor[pandas] for DataFrame support: "
19
+ "pip install 'phactor[pandas]'"
20
+ ) from None
21
+
22
+ from phactor.cohorts.models import (
23
+ AnalyzeCohortsPayload,
24
+ DemographicBreakdown,
25
+ )
26
+
27
+ __all__ = [
28
+ "analyses_to_dataframe",
29
+ "cohort_groups_to_dataframe",
30
+ "criteria_groups_to_dataframe",
31
+ "criterion_impacts_to_dataframe",
32
+ "demographics_to_dataframe",
33
+ "geographic_groups_to_dataframe",
34
+ "site_radius_to_dataframe",
35
+ "site_distance_bands_to_dataframe",
36
+ ]
37
+
38
+ _DEMOGRAPHIC_CATEGORIES = pd.CategoricalDtype(
39
+ categories=["age", "gender", "race", "geographic"], ordered=False
40
+ )
41
+
42
+ _ANALYSES_COLS: dict[str, str] = {
43
+ "analysis_id": "string",
44
+ "provider_id": "string",
45
+ "provider_name": "string",
46
+ "total_patients": "Int64",
47
+ }
48
+
49
+ _COHORT_GROUPS_COLS: dict[str, str] = {
50
+ "provider_id": "string",
51
+ "provider_name": "string",
52
+ "pathway_index": "int64",
53
+ "total_patients_in_group": "Int64",
54
+ "evaluated": "bool",
55
+ }
56
+
57
+ _CRITERIA_GROUPS_COLS: dict[str, str] = {
58
+ "provider_id": "string",
59
+ "provider_name": "string",
60
+ "pathway_index": "int64",
61
+ "group_kind": "string",
62
+ "group_name": "string",
63
+ "total_patients_in_group": "Int64",
64
+ "evaluated": "bool",
65
+ }
66
+
67
+ _CRITERION_IMPACTS_COLS: dict[str, str] = {
68
+ "provider_id": "string",
69
+ "provider_name": "string",
70
+ "pathway_index": "int64",
71
+ "group_kind": "string",
72
+ "group_name": "string",
73
+ "proposition_index": "int64",
74
+ "proposition_label": "string",
75
+ "matching_count": "int64",
76
+ "excluded_count": "int64",
77
+ "impact_percentage": "float64",
78
+ }
79
+
80
+ _DEMOGRAPHICS_COLS: dict[str, str] = {
81
+ "category": "string",
82
+ "label": "string",
83
+ "count": "int64",
84
+ "percentage": "float64",
85
+ }
86
+
87
+ _GEOGRAPHIC_GROUPS_COLS: dict[str, str] = {
88
+ "provider_id": "string",
89
+ "provider_name": "string",
90
+ "label": "string",
91
+ "count": "int64",
92
+ "percentage": "float64",
93
+ }
94
+
95
+ _SITE_RADIUS_COLS: dict[str, str] = {
96
+ "provider_id": "string",
97
+ "provider_name": "string",
98
+ "site_name": "string",
99
+ "latitude": "float64",
100
+ "longitude": "float64",
101
+ "radius_miles": "int64",
102
+ "count": "int64",
103
+ "percentage": "float64",
104
+ }
105
+
106
+ _SITE_DISTANCE_BANDS_COLS: dict[str, str] = {
107
+ "provider_id": "string",
108
+ "provider_name": "string",
109
+ "site_name": "string",
110
+ "latitude": "float64",
111
+ "longitude": "float64",
112
+ "band_label": "string",
113
+ "min_miles": "int64",
114
+ "max_miles": "Int64",
115
+ "count": "int64",
116
+ "percentage": "float64",
117
+ }
118
+
119
+
120
+ def _empty(columns: dict[str, str]) -> pd.DataFrame:
121
+ """Return an empty DataFrame with the given column names and dtypes."""
122
+ return pd.DataFrame({col: pd.Series(dtype=dt) for col, dt in columns.items()})
123
+
124
+
125
+ def _to_df(rows: list[dict[str, object]], columns: dict[str, str]) -> pd.DataFrame:
126
+ """Build a DataFrame from rows with explicit dtypes, or an empty schema."""
127
+ if not rows:
128
+ return _empty(columns)
129
+ df: pd.DataFrame = pd.DataFrame(rows).astype(columns)
130
+ return df
131
+
132
+
133
+ def analyses_to_dataframe(payload: AnalyzeCohortsPayload) -> pd.DataFrame:
134
+ """Convert an ``AnalyzeCohortsPayload`` to a flat DataFrame of provider analyses.
135
+
136
+ Each row represents one :class:`~phactor.cohorts.models.ProviderCohortAnalysis`.
137
+
138
+ Columns
139
+ -------
140
+ analysis_id : string
141
+ Unique identifier for the analysis.
142
+ provider_id : string
143
+ Unique identifier for the provider.
144
+ provider_name : string
145
+ Human-readable provider name.
146
+ total_patients : Int64
147
+ Total number of patients in the provider's dataset. Null means the
148
+ provider request was not evaluated.
149
+ """
150
+ rows: list[dict[str, object]] = [
151
+ {
152
+ "analysis_id": analysis.id,
153
+ "provider_id": analysis.provider.id,
154
+ "provider_name": analysis.provider.name,
155
+ "total_patients": analysis.total_patients,
156
+ }
157
+ for analysis in payload.analyses
158
+ ]
159
+ return _to_df(rows, _ANALYSES_COLS)
160
+
161
+
162
+ def cohort_groups_to_dataframe(
163
+ payload: AnalyzeCohortsPayload,
164
+ *,
165
+ provider_id: str | None = None,
166
+ ) -> pd.DataFrame:
167
+ """Convert pathway-level cohort group results to a flat DataFrame.
168
+
169
+ Each row represents one complete cohort pathway. Its
170
+ ``total_patients_in_group`` value is the net result after combining all
171
+ inclusion groups and subtracting all exclusion groups.
172
+
173
+ Parameters
174
+ ----------
175
+ payload:
176
+ The full cohort analysis payload.
177
+ provider_id:
178
+ When set, only rows belonging to this provider are included.
179
+
180
+ Columns
181
+ -------
182
+ provider_id : string
183
+ Unique identifier for the provider.
184
+ provider_name : string
185
+ Human-readable provider name.
186
+ pathway_index : int64
187
+ Zero-based position of the complete pathway in the request.
188
+ total_patients_in_group : Int64
189
+ Net matching patient count. Null means the pathway was not evaluated.
190
+ evaluated : bool
191
+ Whether the pathway was evaluated.
192
+ """
193
+ rows: list[dict[str, object]] = []
194
+ for analysis in payload.analyses:
195
+ if provider_id is not None and analysis.provider.id != provider_id:
196
+ continue
197
+ for pathway_index, pathway in enumerate(analysis.cohort_groups):
198
+ rows.append(
199
+ {
200
+ "provider_id": analysis.provider.id,
201
+ "provider_name": analysis.provider.name,
202
+ "pathway_index": pathway_index,
203
+ "total_patients_in_group": pathway.total_patients_in_group,
204
+ "evaluated": pathway.evaluated,
205
+ }
206
+ )
207
+ return _to_df(rows, _COHORT_GROUPS_COLS)
208
+
209
+
210
+ def criteria_groups_to_dataframe(
211
+ payload: AnalyzeCohortsPayload,
212
+ *,
213
+ provider_id: str | None = None,
214
+ ) -> pd.DataFrame:
215
+ """Convert nested inclusion and exclusion criteria groups to a flat DataFrame.
216
+
217
+ Each row represents one named criteria group inside a complete cohort
218
+ pathway. ``group_kind`` preserves whether that group includes or excludes
219
+ patients.
220
+
221
+ Criteria-group counts are diagnostic component counts and are not additive:
222
+ inclusion groups can overlap, and exclusion counts are subtracted from the
223
+ pathway result. Use :func:`cohort_groups_to_dataframe` for net pathway counts.
224
+
225
+ Parameters
226
+ ----------
227
+ payload:
228
+ The full cohort analysis payload.
229
+ provider_id:
230
+ When set, only rows belonging to this provider are included.
231
+
232
+ Columns
233
+ -------
234
+ provider_id : string
235
+ Unique identifier for the provider.
236
+ provider_name : string
237
+ Human-readable provider name.
238
+ pathway_index : int64
239
+ Zero-based position of the complete pathway in the request.
240
+ group_kind : string
241
+ ``inclusion`` or ``exclusion``.
242
+ group_name : string
243
+ Name of the nested criteria group.
244
+ total_patients_in_group : Int64
245
+ Matching patient count. Null means the group was not evaluated.
246
+ evaluated : bool
247
+ Whether the criteria group was evaluated.
248
+ """
249
+ rows: list[dict[str, object]] = []
250
+ for analysis in payload.analyses:
251
+ if provider_id is not None and analysis.provider.id != provider_id:
252
+ continue
253
+ for pathway_index, pathway in enumerate(analysis.cohort_groups):
254
+ for group_kind, criteria_groups in (
255
+ ("inclusion", pathway.inclusion_groups),
256
+ ("exclusion", pathway.exclusion_groups),
257
+ ):
258
+ rows.extend(
259
+ {
260
+ "provider_id": analysis.provider.id,
261
+ "provider_name": analysis.provider.name,
262
+ "pathway_index": pathway_index,
263
+ "group_kind": group_kind,
264
+ "group_name": criteria_group.name,
265
+ "total_patients_in_group": criteria_group.total_patients_in_group,
266
+ "evaluated": criteria_group.evaluated,
267
+ }
268
+ for criteria_group in criteria_groups
269
+ )
270
+ return _to_df(rows, _CRITERIA_GROUPS_COLS)
271
+
272
+
273
+ def criterion_impacts_to_dataframe(
274
+ payload: AnalyzeCohortsPayload,
275
+ *,
276
+ provider_id: str | None = None,
277
+ group_name: str | None = None,
278
+ ) -> pd.DataFrame:
279
+ """Convert criterion impact results to a flat DataFrame.
280
+
281
+ Each row represents one :class:`~phactor.cohorts.models.CriterionImpact`
282
+ nested within a cohort group, across all (or a filtered subset of) analyses.
283
+
284
+ Parameters
285
+ ----------
286
+ payload:
287
+ The full cohort analysis payload.
288
+ provider_id:
289
+ When set, only rows belonging to this provider are included.
290
+ group_name:
291
+ When set, only rows belonging to this cohort group name are included.
292
+
293
+ Columns
294
+ -------
295
+ provider_id : string
296
+ Unique identifier for the provider.
297
+ provider_name : string
298
+ Human-readable provider name.
299
+ group_name : string
300
+ Name of the cohort group containing this criterion.
301
+ proposition_index : int64
302
+ Zero-based index of the criterion proposition.
303
+ proposition_label : string
304
+ Human-readable label for the criterion proposition.
305
+ matching_count : int64
306
+ Number of patients matching this criterion.
307
+ excluded_count : int64
308
+ Number of patients excluded by this criterion.
309
+ impact_percentage : float64
310
+ Percentage impact of this criterion on the cohort.
311
+ """
312
+ rows: list[dict[str, object]] = []
313
+ for analysis in payload.analyses:
314
+ if provider_id is not None and analysis.provider.id != provider_id:
315
+ continue
316
+ for pathway_index, pathway in enumerate(analysis.cohort_groups):
317
+ for group_kind, criteria_groups in (
318
+ ("inclusion", pathway.inclusion_groups),
319
+ ("exclusion", pathway.exclusion_groups),
320
+ ):
321
+ for group in criteria_groups:
322
+ if group_name is not None and group.name != group_name:
323
+ continue
324
+ for impact in group.criterion_impacts or []:
325
+ rows.append(
326
+ {
327
+ "provider_id": analysis.provider.id,
328
+ "provider_name": analysis.provider.name,
329
+ "pathway_index": pathway_index,
330
+ "group_kind": group_kind,
331
+ "group_name": group.name,
332
+ "proposition_index": impact.proposition_index,
333
+ "proposition_label": impact.proposition_label,
334
+ "matching_count": impact.matching_count,
335
+ "excluded_count": impact.excluded_count,
336
+ "impact_percentage": impact.impact_percentage,
337
+ }
338
+ )
339
+ return _to_df(rows, _CRITERION_IMPACTS_COLS)
340
+
341
+
342
+ def demographics_to_dataframe(
343
+ breakdown: DemographicBreakdown,
344
+ *,
345
+ category: str | None = None,
346
+ ) -> pd.DataFrame:
347
+ """Convert a ``DemographicBreakdown`` to a flat DataFrame.
348
+
349
+ Works for both per-group breakdowns (``CohortGroupResult.demographic_breakdown``)
350
+ and combined breakdowns (``ProviderCohortAnalysis.combined_demographic_breakdown``).
351
+
352
+ Parameters
353
+ ----------
354
+ breakdown:
355
+ A single demographic breakdown object.
356
+ category:
357
+ When set, only rows for this demographic category are included.
358
+ Must be one of ``"age"``, ``"gender"``, ``"race"``, ``"geographic"``.
359
+
360
+ Columns
361
+ -------
362
+ category : CategoricalDtype(["age", "gender", "race", "geographic"])
363
+ The demographic category.
364
+ label : string
365
+ The demographic label (e.g. age range ``"18-24"``, gender ``"Male"``).
366
+ count : int64
367
+ Number of patients in this demographic bucket.
368
+ percentage : float64
369
+ Percentage of patients in this demographic bucket.
370
+ """
371
+ rows: list[dict[str, object]] = []
372
+
373
+ if category is None or category == "age":
374
+ for bucket in breakdown.age_distribution or []:
375
+ rows.append(
376
+ {
377
+ "category": "age",
378
+ "label": f"{bucket.range_from}-{bucket.range_to}",
379
+ "count": bucket.count,
380
+ "percentage": bucket.percentage,
381
+ }
382
+ )
383
+
384
+ if category is None or category == "gender":
385
+ for entry in breakdown.gender_distribution or []:
386
+ rows.append(
387
+ {
388
+ "category": "gender",
389
+ "label": entry.label,
390
+ "count": entry.count,
391
+ "percentage": entry.percentage,
392
+ }
393
+ )
394
+
395
+ if category is None or category == "race":
396
+ for entry in breakdown.race_distribution or []:
397
+ rows.append(
398
+ {
399
+ "category": "race",
400
+ "label": entry.label,
401
+ "count": entry.count,
402
+ "percentage": entry.percentage,
403
+ }
404
+ )
405
+
406
+ if category is None or category == "geographic":
407
+ for entry in breakdown.geographic_distribution or []:
408
+ rows.append(
409
+ {
410
+ "category": "geographic",
411
+ "label": entry.label,
412
+ "count": entry.count,
413
+ "percentage": entry.percentage,
414
+ }
415
+ )
416
+
417
+ df: pd.DataFrame = (
418
+ _empty(_DEMOGRAPHICS_COLS) if not rows else pd.DataFrame(rows).astype(_DEMOGRAPHICS_COLS)
419
+ )
420
+ df["category"] = df["category"].astype(_DEMOGRAPHIC_CATEGORIES)
421
+ return df
422
+
423
+
424
+ def geographic_groups_to_dataframe(
425
+ payload: AnalyzeCohortsPayload,
426
+ *,
427
+ provider_id: str | None = None,
428
+ ) -> pd.DataFrame:
429
+ """Convert geographic group results to a flat DataFrame.
430
+
431
+ Each row represents one :class:`~phactor.cohorts.models.GeographicGroup`
432
+ across all (or a filtered subset of) provider analyses.
433
+
434
+ Parameters
435
+ ----------
436
+ payload:
437
+ The full cohort analysis payload.
438
+ provider_id:
439
+ When set, only rows belonging to this provider are included.
440
+
441
+ Columns
442
+ -------
443
+ provider_id : string
444
+ Unique identifier for the provider.
445
+ provider_name : string
446
+ Human-readable provider name.
447
+ label : string
448
+ Geographic region label.
449
+ count : int64
450
+ Number of patients in this geographic region.
451
+ percentage : float64
452
+ Percentage of patients in this geographic region.
453
+ """
454
+ rows: list[dict[str, object]] = []
455
+ for analysis in payload.analyses:
456
+ if provider_id is not None and analysis.provider.id != provider_id:
457
+ continue
458
+ for geo_group in analysis.geographic_groups or []:
459
+ rows.append(
460
+ {
461
+ "provider_id": analysis.provider.id,
462
+ "provider_name": analysis.provider.name,
463
+ "label": geo_group.label,
464
+ "count": geo_group.count,
465
+ "percentage": geo_group.percentage,
466
+ }
467
+ )
468
+ return _to_df(rows, _GEOGRAPHIC_GROUPS_COLS)
469
+
470
+
471
+ def site_radius_to_dataframe(
472
+ payload: AnalyzeCohortsPayload,
473
+ *,
474
+ provider_id: str | None = None,
475
+ ) -> pd.DataFrame:
476
+ """Convert site radius results to a flat DataFrame.
477
+
478
+ Each row represents one :class:`~phactor.cohorts.models.SiteRadiusResult`
479
+ across all (or a filtered subset of) provider analyses.
480
+
481
+ Parameters
482
+ ----------
483
+ payload:
484
+ The full cohort analysis payload.
485
+ provider_id:
486
+ When set, only rows belonging to this provider are included.
487
+
488
+ Columns
489
+ -------
490
+ provider_id : string
491
+ Unique identifier for the provider.
492
+ provider_name : string
493
+ Human-readable provider name.
494
+ site_name : string
495
+ Name of the clinical site.
496
+ latitude : float64
497
+ Latitude coordinate of the site.
498
+ longitude : float64
499
+ Longitude coordinate of the site.
500
+ radius_miles : int64
501
+ Search radius in miles around the site.
502
+ count : int64
503
+ Number of patients within the radius.
504
+ percentage : float64
505
+ Percentage of patients within the radius.
506
+ """
507
+ rows: list[dict[str, object]] = []
508
+ for analysis in payload.analyses:
509
+ if provider_id is not None and analysis.provider.id != provider_id:
510
+ continue
511
+ for site in analysis.site_radius_results or []:
512
+ rows.append(
513
+ {
514
+ "provider_id": analysis.provider.id,
515
+ "provider_name": analysis.provider.name,
516
+ "site_name": site.site_name,
517
+ "latitude": site.latitude,
518
+ "longitude": site.longitude,
519
+ "radius_miles": site.radius_miles,
520
+ "count": site.count,
521
+ "percentage": site.percentage,
522
+ }
523
+ )
524
+ return _to_df(rows, _SITE_RADIUS_COLS)
525
+
526
+
527
+ def site_distance_bands_to_dataframe(
528
+ payload: AnalyzeCohortsPayload,
529
+ *,
530
+ provider_id: str | None = None,
531
+ ) -> pd.DataFrame:
532
+ """Convert site distance band results to a flat DataFrame.
533
+
534
+ Each row represents one :class:`~phactor.cohorts.models.DistanceBand` within
535
+ a :class:`~phactor.cohorts.models.SiteDistanceBandResult`, across all (or a
536
+ filtered subset of) provider analyses.
537
+
538
+ Parameters
539
+ ----------
540
+ payload:
541
+ The full cohort analysis payload.
542
+ provider_id:
543
+ When set, only rows belonging to this provider are included.
544
+
545
+ Columns
546
+ -------
547
+ provider_id : string
548
+ Unique identifier for the provider.
549
+ provider_name : string
550
+ Human-readable provider name.
551
+ site_name : string
552
+ Name of the clinical site.
553
+ latitude : float64
554
+ Latitude coordinate of the site.
555
+ longitude : float64
556
+ Longitude coordinate of the site.
557
+ band_label : string
558
+ Human-readable label for the distance band (e.g. ``"0-25 miles"``).
559
+ min_miles : int64
560
+ Lower bound of the distance band in miles (inclusive).
561
+ max_miles : Int64
562
+ Upper bound of the distance band in miles (inclusive). Nullable —
563
+ ``pd.NA`` for open-ended bands such as ``"100+ miles"``.
564
+ count : int64
565
+ Number of patients within this distance band.
566
+ percentage : float64
567
+ Percentage of patients within this distance band.
568
+ """
569
+ rows: list[dict[str, object]] = []
570
+ for analysis in payload.analyses:
571
+ if provider_id is not None and analysis.provider.id != provider_id:
572
+ continue
573
+ for site_result in analysis.site_distance_band_results or []:
574
+ for band in site_result.bands:
575
+ rows.append(
576
+ {
577
+ "provider_id": analysis.provider.id,
578
+ "provider_name": analysis.provider.name,
579
+ "site_name": site_result.site_name,
580
+ "latitude": site_result.latitude,
581
+ "longitude": site_result.longitude,
582
+ "band_label": band.label,
583
+ "min_miles": band.min_miles,
584
+ "max_miles": band.max_miles,
585
+ "count": band.count,
586
+ "percentage": band.percentage,
587
+ }
588
+ )
589
+ return _to_df(rows, _SITE_DISTANCE_BANDS_COLS)
phactor/py.typed ADDED
File without changes