open-pharma-plugins 2.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. mcp_framework.py +495 -0
  2. open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
  3. open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
  4. open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
  5. open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
  6. open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
  7. open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
  8. open_pharma_plugins_campaign_studio/__init__.py +14 -0
  9. open_pharma_plugins_campaign_studio/__main__.py +11 -0
  10. open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
  11. open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
  12. open_pharma_plugins_campaign_studio/_renderer.py +119 -0
  13. open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
  14. open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
  15. open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
  16. open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
  17. open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
  18. open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
  19. open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
  20. open_pharma_plugins_campaign_studio/models/_common.py +12 -0
  21. open_pharma_plugins_campaign_studio/models/brief.py +72 -0
  22. open_pharma_plugins_campaign_studio/models/claims.py +15 -0
  23. open_pharma_plugins_campaign_studio/models/copy.py +47 -0
  24. open_pharma_plugins_campaign_studio/models/journey.py +21 -0
  25. open_pharma_plugins_campaign_studio/models/message.py +25 -0
  26. open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
  27. open_pharma_plugins_campaign_studio/models/validation.py +32 -0
  28. open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
  29. open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
  30. open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
  31. open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
  32. open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
  33. open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
  34. open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
  35. open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
  36. open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
  37. open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
  38. open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
  39. open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
  40. open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
  41. open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
  42. open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
  43. open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
  44. open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
  45. open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
  46. open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
  47. open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
  48. open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
  49. open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
  50. open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
  51. open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
  52. open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
  53. open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
  54. open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
  55. open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
  56. open_pharma_plugins_competitive_intelligence/models.py +525 -0
  57. open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
  58. open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
  59. open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
  60. open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
  61. open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
  62. open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
  63. open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
  64. open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
  65. open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
  66. open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
  67. open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
  68. open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
  69. open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
  70. open_pharma_plugins_field_training/__init__.py +13 -0
  71. open_pharma_plugins_field_training/__main__.py +11 -0
  72. open_pharma_plugins_field_training/_content_store.py +131 -0
  73. open_pharma_plugins_field_training/_grounding.py +75 -0
  74. open_pharma_plugins_field_training/_html_renderers.py +546 -0
  75. open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
  76. open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
  77. open_pharma_plugins_field_training/models.py +265 -0
  78. open_pharma_plugins_field_training/tools/__init__.py +0 -0
  79. open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
  80. open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
  81. open_pharma_plugins_field_training/tools/list_documents.py +51 -0
  82. open_pharma_plugins_field_training/tools/render_output.py +118 -0
  83. open_pharma_plugins_field_training/tools/search_content.py +57 -0
  84. open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
  85. open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
  86. open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
  87. open_pharma_plugins_hcp_intelligence/batch.py +891 -0
  88. open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
  89. open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
  90. open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
  91. open_pharma_plugins_hcp_intelligence/models.py +356 -0
  92. open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
  93. open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
  94. open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
  95. open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
  96. open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
  97. open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
  98. open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
  99. open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
  100. open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
  101. open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
  102. open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
  103. open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
  104. open_pharma_plugins_next_best_engagement/__init__.py +14 -0
  105. open_pharma_plugins_next_best_engagement/__main__.py +11 -0
  106. open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
  107. open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
  108. open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
  109. open_pharma_plugins_next_best_engagement/_universe.py +145 -0
  110. open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
  111. open_pharma_plugins_next_best_engagement/models.py +135 -0
  112. open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
  113. open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
  114. open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
  115. open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
  116. open_pharma_plugins_territory_alignment/__init__.py +14 -0
  117. open_pharma_plugins_territory_alignment/__main__.py +11 -0
  118. open_pharma_plugins_territory_alignment/data.py +300 -0
  119. open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
  120. open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
  121. open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
  122. open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
  123. open_pharma_plugins_territory_alignment/geo.py +175 -0
  124. open_pharma_plugins_territory_alignment/models.py +201 -0
  125. open_pharma_plugins_territory_alignment/scoring.py +125 -0
  126. open_pharma_plugins_territory_alignment/solver.py +504 -0
  127. open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
  128. open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
  129. open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
  130. open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
  131. open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
  132. open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
  133. open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
  134. shared/__init__.py +11 -0
  135. shared/env.py +217 -0
  136. shared/filesystem.py +110 -0
@@ -0,0 +1,342 @@
1
+ """PubMed ESearch/EFetch adapter with nested XML preservation."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import json
6
+ import re
7
+ import xml.etree.ElementTree as ET
8
+ from datetime import datetime, timezone
9
+ from typing import Any, Mapping
10
+ from urllib.parse import urlencode
11
+
12
+ from pydantic import ValidationError
13
+
14
+ from shared.filesystem import sanitize_url
15
+
16
+ from ._cache import cache_lookup, cache_store
17
+ from ._transport import HttpRequest, HttpTransport, TransportError, UrllibTransport
18
+ from .models import (
19
+ CacheProvenance,
20
+ CoverageStatus,
21
+ Publication,
22
+ PublicationSearchRequest,
23
+ SourceError,
24
+ SourceName,
25
+ SourceRequestEvidence,
26
+ SourceResult,
27
+ aggregate_cache_status,
28
+ )
29
+
30
+ _EUTILS = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils"
31
+ DEFAULT_TRANSPORT: HttpTransport = UrllibTransport()
32
+
33
+
34
+ def build_search_url(request: PublicationSearchRequest, *, api_key: str) -> str:
35
+ params = {
36
+ "db": "pubmed",
37
+ "term": _exact_query(request),
38
+ "retmax": str(request.max_results),
39
+ "sort": "date",
40
+ "datetype": "pdat",
41
+ "reldate": str(request.days_back),
42
+ "retmode": "json",
43
+ }
44
+ if api_key:
45
+ params["api_key"] = api_key
46
+ return f"{_EUTILS}/esearch.fcgi?{urlencode(params)}"
47
+
48
+
49
+ def search_publications(
50
+ request: PublicationSearchRequest,
51
+ *,
52
+ transport: HttpTransport | None = None,
53
+ now: datetime | None = None,
54
+ ) -> SourceResult:
55
+ from shared.env import get_env
56
+
57
+ transport = transport or DEFAULT_TRANSPORT
58
+ retrieved_at = _utc(now)
59
+ api_key = get_env("NCBI_API_KEY", "")
60
+ search_wire_url = build_search_url(request, api_key=api_key)
61
+ search_url = sanitize_url(search_wire_url)
62
+ search_params = {
63
+ "query": _exact_query(request),
64
+ "days_back": request.days_back,
65
+ "max_results": request.max_results,
66
+ }
67
+ search_lookup = cache_lookup("pubmed:esearch", search_params)
68
+ try:
69
+ search_payload = search_lookup.payload
70
+ if search_payload is None:
71
+ response = transport.request(HttpRequest(method="GET", url=search_wire_url))
72
+ search_payload = _json_mapping(response.body)
73
+ ids, total = _parse_search(search_payload)
74
+ cache_store("pubmed:esearch", search_params, search_payload)
75
+ else:
76
+ ids, total = _parse_search(search_payload)
77
+ except TransportError as error:
78
+ return _failure(request, search_url, retrieved_at, search_lookup.status, error.code, str(error))
79
+ except (ValueError, json.JSONDecodeError, UnicodeDecodeError):
80
+ return _failure(
81
+ request,
82
+ search_url,
83
+ retrieved_at,
84
+ search_lookup.status,
85
+ "schema_mismatch",
86
+ "PubMed ESearch response shape was invalid",
87
+ )
88
+
89
+ search_evidence = SourceRequestEvidence(
90
+ query=_exact_query(request),
91
+ source_url=search_url,
92
+ retrieved_at=retrieved_at,
93
+ cache=CacheProvenance(status=search_lookup.status, cached_at=search_lookup.cached_at),
94
+ status=CoverageStatus.COMPLETE,
95
+ record_count=len(ids),
96
+ )
97
+ if not ids:
98
+ return SourceResult(
99
+ source=SourceName.PUBMED,
100
+ provider="pubmed",
101
+ status=CoverageStatus.COMPLETE,
102
+ query=_exact_query(request),
103
+ source_url=search_url,
104
+ retrieved_at=retrieved_at,
105
+ cache=CacheProvenance(status=search_lookup.status, cached_at=search_lookup.cached_at),
106
+ records=[],
107
+ total_available=total,
108
+ requests=[search_evidence],
109
+ )
110
+
111
+ fetch_wire_url = _build_fetch_url(ids, api_key=api_key)
112
+ fetch_url = sanitize_url(fetch_wire_url)
113
+ fetch_params = {"ids": ids}
114
+ fetch_lookup = cache_lookup("pubmed:efetch", fetch_params)
115
+ try:
116
+ xml_text = fetch_lookup.payload
117
+ if xml_text is None:
118
+ response = transport.request(HttpRequest(method="GET", url=fetch_wire_url, timeout_seconds=60))
119
+ xml_text = response.body.decode("utf-8")
120
+ records, dropped = _parse_xml(xml_text.encode("utf-8"))
121
+ cache_store("pubmed:efetch", fetch_params, xml_text)
122
+ else:
123
+ if not isinstance(xml_text, str):
124
+ raise ValueError("cached PubMed XML must be text")
125
+ records, dropped = _parse_xml(xml_text.encode("utf-8"))
126
+ except TransportError:
127
+ return _fetch_failure(request, fetch_url, retrieved_at, fetch_lookup.status, "fetch_failed")
128
+ except (ValueError, ValidationError, ET.ParseError, UnicodeDecodeError):
129
+ return _fetch_failure(request, fetch_url, retrieved_at, fetch_lookup.status, "schema_mismatch")
130
+
131
+ if dropped and not records:
132
+ return _fetch_failure(request, fetch_url, retrieved_at, fetch_lookup.status, "schema_mismatch")
133
+ status = CoverageStatus.PARTIAL if dropped else CoverageStatus.COMPLETE
134
+ error = SourceError(code="schema_mismatch", message="some PubMed records were invalid") if dropped else None
135
+ fetch_evidence = SourceRequestEvidence(
136
+ query=",".join(ids),
137
+ source_url=fetch_url,
138
+ retrieved_at=retrieved_at,
139
+ cache=CacheProvenance(status=fetch_lookup.status, cached_at=fetch_lookup.cached_at),
140
+ status=status,
141
+ record_count=len(records),
142
+ error=error,
143
+ )
144
+ limitations = [f"Dropped {dropped} malformed PubMed record(s)."] if dropped else []
145
+ return SourceResult(
146
+ source=SourceName.PUBMED,
147
+ provider="pubmed",
148
+ status=status,
149
+ query=_exact_query(request),
150
+ source_url=search_url,
151
+ retrieved_at=retrieved_at,
152
+ cache=CacheProvenance(status=aggregate_cache_status([search_lookup.status, fetch_lookup.status])),
153
+ records=records,
154
+ total_available=total,
155
+ requests=[search_evidence, fetch_evidence],
156
+ limitations=limitations,
157
+ error=error,
158
+ )
159
+
160
+
161
+ def _parse_search(payload: Mapping[str, Any]) -> tuple[list[str], int]:
162
+ result = payload.get("esearchresult")
163
+ if not isinstance(result, Mapping):
164
+ raise ValueError("esearchresult must be a mapping")
165
+ ids = result.get("idlist")
166
+ if not isinstance(ids, list) or not all(isinstance(value, str) for value in ids):
167
+ raise ValueError("idlist must be a string list")
168
+ total = int(result.get("count", 0))
169
+ if total < len(ids):
170
+ raise ValueError("PubMed count cannot be smaller than returned IDs")
171
+ return ids, total
172
+
173
+
174
+ def _parse_xml(xml_data: bytes) -> tuple[list[dict[str, Any]], int]:
175
+ root = ET.fromstring(xml_data)
176
+ articles = root.findall(".//PubmedArticle")
177
+ if not articles:
178
+ raise ValueError("PubMed EFetch contained no articles")
179
+ records = []
180
+ dropped = 0
181
+ for element in articles:
182
+ try:
183
+ records.append(_parse_article(element).model_dump(mode="json"))
184
+ except (ValueError, ValidationError, TypeError):
185
+ dropped += 1
186
+ if not records:
187
+ raise ValueError("no valid PubMed articles")
188
+ return records, dropped
189
+
190
+
191
+ def _parse_article(article: ET.Element) -> Publication:
192
+ medline = article.find("MedlineCitation")
193
+ if medline is None:
194
+ raise ValueError("MedlineCitation is required")
195
+ pmid = _text(medline.find("PMID"))
196
+ body = medline.find("Article")
197
+ if body is None:
198
+ raise ValueError("Article is required")
199
+ title = _element_text(body.find("ArticleTitle"))
200
+ abstract_parts = []
201
+ for block in body.findall(".//Abstract/AbstractText"):
202
+ text = _element_text(block)
203
+ label = block.attrib.get("Label", "").strip()
204
+ abstract_parts.append(f"{label}: {text}" if label else text)
205
+ authors = []
206
+ for author in body.findall(".//AuthorList/Author")[:5]:
207
+ last = author.findtext("LastName", "").strip()
208
+ fore = author.findtext("ForeName", "").strip()
209
+ if last:
210
+ authors.append(f"{last} {fore}".strip())
211
+ pub_types = [_element_text(value) for value in body.findall(".//PublicationTypeList/PublicationType")]
212
+ return Publication(
213
+ pmid=pmid,
214
+ title=title,
215
+ authors=authors,
216
+ journal=_element_text(body.find(".//Journal/Title"), required=False),
217
+ pub_date=_publication_date(body),
218
+ abstract_excerpt=" ".join(abstract_parts)[:1000],
219
+ pub_types=[value for value in pub_types if value],
220
+ source_url=f"https://pubmed.ncbi.nlm.nih.gov/{pmid}/",
221
+ )
222
+
223
+
224
+ def _publication_date(article: ET.Element) -> str:
225
+ article_date = article.find(".//ArticleDate")
226
+ if article_date is not None:
227
+ year = article_date.findtext("Year", "")
228
+ month = article_date.findtext("Month", "")
229
+ day = article_date.findtext("Day", "")
230
+ if year:
231
+ return _join_date(year, month, day)
232
+ pub_date = article.find(".//Journal/JournalIssue/PubDate")
233
+ if pub_date is None:
234
+ return ""
235
+ year = pub_date.findtext("Year", "")
236
+ if year:
237
+ return _join_date(year, pub_date.findtext("Month", ""), pub_date.findtext("Day", ""))
238
+ return pub_date.findtext("MedlineDate", "").strip()
239
+
240
+
241
+ def _join_date(year: str, month: str, day: str) -> str:
242
+ month_value = _month(month)
243
+ if day:
244
+ return f"{year}-{month_value}-{day.zfill(2)}"
245
+ return f"{year}-{month_value}" if month_value else year
246
+
247
+
248
+ def _month(value: str) -> str:
249
+ value = value.strip()
250
+ if value.isdigit():
251
+ return value.zfill(2)
252
+ months = {
253
+ "jan": "01",
254
+ "feb": "02",
255
+ "mar": "03",
256
+ "apr": "04",
257
+ "may": "05",
258
+ "jun": "06",
259
+ "jul": "07",
260
+ "aug": "08",
261
+ "sep": "09",
262
+ "oct": "10",
263
+ "nov": "11",
264
+ "dec": "12",
265
+ }
266
+ return months.get(value[:3].lower(), "")
267
+
268
+
269
+ def _element_text(element: ET.Element | None, *, required: bool = True) -> str:
270
+ if element is None:
271
+ if required:
272
+ raise ValueError("required XML element is missing")
273
+ return ""
274
+ value = re.sub(r"\s+", " ", "".join(element.itertext())).strip()
275
+ if required and not value:
276
+ raise ValueError("required XML text is missing")
277
+ return value
278
+
279
+
280
+ def _text(element: ET.Element | None) -> str:
281
+ return _element_text(element)
282
+
283
+
284
+ def _build_fetch_url(ids: list[str], *, api_key: str) -> str:
285
+ params = {"db": "pubmed", "id": ",".join(ids), "retmode": "xml"}
286
+ if api_key:
287
+ params["api_key"] = api_key
288
+ return f"{_EUTILS}/efetch.fcgi?{urlencode(params)}"
289
+
290
+
291
+ def _exact_query(request: PublicationSearchRequest) -> str:
292
+ return f"{request.query} AND clinical trial[pt]"
293
+
294
+
295
+ def _failure(request, url, retrieved_at, cache_status, code, message) -> SourceResult:
296
+ error = SourceError(code=code, message=message)
297
+ evidence = SourceRequestEvidence(
298
+ query=_exact_query(request),
299
+ source_url=url,
300
+ retrieved_at=retrieved_at,
301
+ cache=CacheProvenance(status=cache_status),
302
+ status=CoverageStatus.FAILED,
303
+ record_count=0,
304
+ error=error,
305
+ )
306
+ return SourceResult(
307
+ source=SourceName.PUBMED,
308
+ provider="pubmed",
309
+ status=CoverageStatus.FAILED,
310
+ query=_exact_query(request),
311
+ source_url=url,
312
+ retrieved_at=retrieved_at,
313
+ cache=CacheProvenance(status=cache_status),
314
+ requests=[evidence],
315
+ limitations=["No trustworthy PubMed response was obtained."],
316
+ error=error,
317
+ )
318
+
319
+
320
+ def _fetch_failure(request, url, retrieved_at, cache_status, code) -> SourceResult:
321
+ return _failure(
322
+ request,
323
+ url,
324
+ retrieved_at,
325
+ cache_status,
326
+ code,
327
+ "PubMed EFetch did not produce trustworthy publication records",
328
+ )
329
+
330
+
331
+ def _json_mapping(body: bytes) -> Mapping[str, Any]:
332
+ payload = json.loads(body.decode("utf-8"))
333
+ if not isinstance(payload, Mapping):
334
+ raise ValueError("provider response must be a mapping")
335
+ return payload
336
+
337
+
338
+ def _utc(value: datetime | None) -> datetime:
339
+ now = value or datetime.now(timezone.utc)
340
+ if now.tzinfo is None:
341
+ raise ValueError("now must be timezone-aware")
342
+ return now.astimezone(timezone.utc)
@@ -0,0 +1,140 @@
1
+ """One-pass regulatory collection across openFDA and DailyMed."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from dataclasses import dataclass
6
+ from datetime import datetime, timezone
7
+ from typing import Callable
8
+
9
+ from . import _dailymed, _fda
10
+ from .models import (
11
+ CacheProvenance,
12
+ CoverageStatus,
13
+ LabelHistoryEntry,
14
+ RegulatoryEvent,
15
+ RegulatorySearchRequest,
16
+ RegulatorySearchResult,
17
+ SourceError,
18
+ SourceName,
19
+ SourceResult,
20
+ aggregate_cache_status,
21
+ aggregate_coverage,
22
+ )
23
+
24
+
25
+ @dataclass(frozen=True)
26
+ class RegulatoryProviders:
27
+ openfda: Callable[[RegulatorySearchRequest, datetime], SourceResult]
28
+ dailymed_search: Callable[[RegulatorySearchRequest, datetime], SourceResult]
29
+ dailymed_history: Callable[[str, datetime], SourceResult]
30
+
31
+
32
+ DEFAULT_PROVIDERS = RegulatoryProviders(
33
+ openfda=lambda request, now: _fda.search_openfda(request, now=now),
34
+ dailymed_search=lambda request, now: _dailymed.search_dailymed(request, now=now),
35
+ dailymed_history=lambda set_id, now: _dailymed.get_dailymed_history(set_id, now=now),
36
+ )
37
+
38
+
39
+ def scan_regulatory(
40
+ request: RegulatorySearchRequest,
41
+ *,
42
+ providers: RegulatoryProviders | None = None,
43
+ now: datetime | None = None,
44
+ ) -> RegulatorySearchResult:
45
+ providers = providers or DEFAULT_PROVIDERS
46
+ collected_at = _utc(now)
47
+ openfda = providers.openfda(request, collected_at)
48
+ dailymed = None
49
+ if request.include_label_history:
50
+ search = providers.dailymed_search(request, collected_at)
51
+ histories = []
52
+ if search.status in {CoverageStatus.COMPLETE, CoverageStatus.PARTIAL}:
53
+ seen_set_ids: set[str] = set()
54
+ for record in search.records:
55
+ set_id = str(record.get("set_id", ""))
56
+ if set_id and set_id not in seen_set_ids:
57
+ seen_set_ids.add(set_id)
58
+ histories.append(providers.dailymed_history(set_id, collected_at))
59
+ dailymed = _combine_dailymed(request, search, histories)
60
+
61
+ events = _dedupe_events(openfda)
62
+ label_history = _dedupe_history(dailymed)
63
+ sources = [openfda] + ([dailymed] if dailymed is not None else [])
64
+ coverage = aggregate_coverage([source.status for source in sources])
65
+ limitations = [f"{source.source.value}: {limitation}" for source in sources for limitation in source.limitations]
66
+ return RegulatorySearchResult(
67
+ drug_name=request.drug_name,
68
+ coverage=coverage,
69
+ openfda=openfda,
70
+ dailymed=dailymed,
71
+ events=events,
72
+ label_history=label_history,
73
+ limitations=limitations,
74
+ )
75
+
76
+
77
+ def _combine_dailymed(
78
+ request: RegulatorySearchRequest,
79
+ search: SourceResult,
80
+ histories: list[SourceResult],
81
+ ) -> SourceResult:
82
+ if search.status in {CoverageStatus.FAILED, CoverageStatus.NOT_CONFIGURED}:
83
+ return search
84
+ sources = [search, *histories]
85
+ status = aggregate_coverage([source.status for source in sources])
86
+ records = [record for history in histories for record in history.records]
87
+ requests = [evidence for source in sources for evidence in source.requests]
88
+ limitations = [limitation for source in sources for limitation in source.limitations]
89
+ error = None
90
+ if status == CoverageStatus.PARTIAL:
91
+ error = SourceError(code="partial_coverage", message="one or more DailyMed requests failed")
92
+ elif status == CoverageStatus.FAILED:
93
+ error = next((source.error for source in sources if source.error is not None), None)
94
+ return SourceResult(
95
+ source=SourceName.DAILYMED,
96
+ provider="dailymed",
97
+ status=status,
98
+ query=request.drug_name,
99
+ source_url=search.source_url,
100
+ retrieved_at=search.retrieved_at,
101
+ cache=CacheProvenance(status=aggregate_cache_status([source.cache.status for source in sources])),
102
+ records=records,
103
+ total_available=len(records) if status in {CoverageStatus.COMPLETE, CoverageStatus.PARTIAL} else None,
104
+ requests=requests,
105
+ limitations=limitations,
106
+ error=error,
107
+ )
108
+
109
+
110
+ def _dedupe_events(source: SourceResult) -> list[RegulatoryEvent]:
111
+ seen: set[tuple[str, str, str]] = set()
112
+ events = []
113
+ for record in source.records:
114
+ event = RegulatoryEvent.model_validate(record)
115
+ key = (event.application_number, event.submission, event.date.isoformat() if event.date else "")
116
+ if key not in seen:
117
+ seen.add(key)
118
+ events.append(event)
119
+ return events
120
+
121
+
122
+ def _dedupe_history(source: SourceResult | None) -> list[LabelHistoryEntry]:
123
+ if source is None:
124
+ return []
125
+ seen: set[tuple[str, str, str]] = set()
126
+ entries = []
127
+ for record in source.records:
128
+ entry = LabelHistoryEntry.model_validate(record)
129
+ key = (entry.set_id, entry.spl_version, entry.published_date)
130
+ if key not in seen:
131
+ seen.add(key)
132
+ entries.append(entry)
133
+ return entries
134
+
135
+
136
+ def _utc(value: datetime | None) -> datetime:
137
+ now = value or datetime.now(timezone.utc)
138
+ if now.tzinfo is None:
139
+ raise ValueError("now must be timezone-aware")
140
+ return now.astimezone(timezone.utc)