open-pharma-plugins 2.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_framework.py +495 -0
- open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
- open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
- open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
- open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
- open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
- open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
- open_pharma_plugins_campaign_studio/__init__.py +14 -0
- open_pharma_plugins_campaign_studio/__main__.py +11 -0
- open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
- open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
- open_pharma_plugins_campaign_studio/_renderer.py +119 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
- open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
- open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
- open_pharma_plugins_campaign_studio/models/_common.py +12 -0
- open_pharma_plugins_campaign_studio/models/brief.py +72 -0
- open_pharma_plugins_campaign_studio/models/claims.py +15 -0
- open_pharma_plugins_campaign_studio/models/copy.py +47 -0
- open_pharma_plugins_campaign_studio/models/journey.py +21 -0
- open_pharma_plugins_campaign_studio/models/message.py +25 -0
- open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
- open_pharma_plugins_campaign_studio/models/validation.py +32 -0
- open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
- open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
- open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
- open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
- open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
- open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
- open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
- open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
- open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
- open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
- open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
- open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
- open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
- open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
- open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
- open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
- open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
- open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
- open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
- open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
- open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
- open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
- open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
- open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
- open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
- open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
- open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
- open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
- open_pharma_plugins_competitive_intelligence/models.py +525 -0
- open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
- open_pharma_plugins_field_training/__init__.py +13 -0
- open_pharma_plugins_field_training/__main__.py +11 -0
- open_pharma_plugins_field_training/_content_store.py +131 -0
- open_pharma_plugins_field_training/_grounding.py +75 -0
- open_pharma_plugins_field_training/_html_renderers.py +546 -0
- open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
- open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
- open_pharma_plugins_field_training/models.py +265 -0
- open_pharma_plugins_field_training/tools/__init__.py +0 -0
- open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
- open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
- open_pharma_plugins_field_training/tools/list_documents.py +51 -0
- open_pharma_plugins_field_training/tools/render_output.py +118 -0
- open_pharma_plugins_field_training/tools/search_content.py +57 -0
- open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
- open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
- open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
- open_pharma_plugins_hcp_intelligence/batch.py +891 -0
- open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
- open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
- open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
- open_pharma_plugins_hcp_intelligence/models.py +356 -0
- open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
- open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
- open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
- open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
- open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
- open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
- open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
- open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
- open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
- open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
- open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
- open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
- open_pharma_plugins_next_best_engagement/__init__.py +14 -0
- open_pharma_plugins_next_best_engagement/__main__.py +11 -0
- open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
- open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
- open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
- open_pharma_plugins_next_best_engagement/_universe.py +145 -0
- open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
- open_pharma_plugins_next_best_engagement/models.py +135 -0
- open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
- open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
- open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
- open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
- open_pharma_plugins_territory_alignment/__init__.py +14 -0
- open_pharma_plugins_territory_alignment/__main__.py +11 -0
- open_pharma_plugins_territory_alignment/data.py +300 -0
- open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
- open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
- open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
- open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
- open_pharma_plugins_territory_alignment/geo.py +175 -0
- open_pharma_plugins_territory_alignment/models.py +201 -0
- open_pharma_plugins_territory_alignment/scoring.py +125 -0
- open_pharma_plugins_territory_alignment/solver.py +504 -0
- open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
- open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
- open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
- open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
- open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
- open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
- open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
- shared/__init__.py +11 -0
- shared/env.py +217 -0
- shared/filesystem.py +110 -0
|
@@ -0,0 +1,569 @@
|
|
|
1
|
+
"""ClinicalTrials.gov API-v2 provider with explicit source evidence."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import json
|
|
6
|
+
from datetime import datetime, timezone
|
|
7
|
+
from typing import Any, Mapping
|
|
8
|
+
from urllib.parse import urlencode
|
|
9
|
+
|
|
10
|
+
from pydantic import ValidationError
|
|
11
|
+
|
|
12
|
+
from ._cache import cache_lookup, cache_store
|
|
13
|
+
from ._transport import HttpRequest, HttpTransport, TransportError, UrllibTransport
|
|
14
|
+
from .models import (
|
|
15
|
+
CacheProvenance,
|
|
16
|
+
CacheStatus,
|
|
17
|
+
CoverageStatus,
|
|
18
|
+
SourceError,
|
|
19
|
+
SourceName,
|
|
20
|
+
SourceRequestEvidence,
|
|
21
|
+
SourceResult,
|
|
22
|
+
Trial,
|
|
23
|
+
TrialArm,
|
|
24
|
+
TrialDetail,
|
|
25
|
+
TrialDetailRequest,
|
|
26
|
+
TrialIntervention,
|
|
27
|
+
TrialResultsSummary,
|
|
28
|
+
TrialSearchRequest,
|
|
29
|
+
aggregate_cache_status,
|
|
30
|
+
)
|
|
31
|
+
|
|
32
|
+
_BASE_URL = "https://clinicaltrials.gov/api/v2/studies"
|
|
33
|
+
_FIELDS = "|".join(
|
|
34
|
+
[
|
|
35
|
+
"NCTId",
|
|
36
|
+
"BriefTitle",
|
|
37
|
+
"OverallStatus",
|
|
38
|
+
"Phase",
|
|
39
|
+
"LeadSponsorName",
|
|
40
|
+
"CollaboratorName",
|
|
41
|
+
"Condition",
|
|
42
|
+
"InterventionName",
|
|
43
|
+
"InterventionType",
|
|
44
|
+
"EnrollmentCount",
|
|
45
|
+
"StartDate",
|
|
46
|
+
"PrimaryCompletionDate",
|
|
47
|
+
"CompletionDate",
|
|
48
|
+
"StudyType",
|
|
49
|
+
"PrimaryOutcomeMeasure",
|
|
50
|
+
"HasResults",
|
|
51
|
+
]
|
|
52
|
+
)
|
|
53
|
+
|
|
54
|
+
DEFAULT_TRANSPORT: HttpTransport = UrllibTransport()
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def build_search_url(request: TrialSearchRequest, *, page_token: str | None = None) -> str:
|
|
58
|
+
params: dict[str, str] = {
|
|
59
|
+
"query.term": request.query,
|
|
60
|
+
"fields": _FIELDS,
|
|
61
|
+
"pageSize": str(request.max_results),
|
|
62
|
+
"format": "json",
|
|
63
|
+
"countTotal": "true",
|
|
64
|
+
}
|
|
65
|
+
if request.phase:
|
|
66
|
+
params["filter.advanced"] = f"AREA[Phase]{request.phase}"
|
|
67
|
+
if request.status:
|
|
68
|
+
params["filter.overallStatus"] = request.status
|
|
69
|
+
if page_token:
|
|
70
|
+
params["pageToken"] = page_token
|
|
71
|
+
return f"{_BASE_URL}?{urlencode(params)}"
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def search_trials(
|
|
75
|
+
request: TrialSearchRequest,
|
|
76
|
+
*,
|
|
77
|
+
transport: HttpTransport | None = None,
|
|
78
|
+
now: datetime | None = None,
|
|
79
|
+
) -> SourceResult:
|
|
80
|
+
transport = transport or DEFAULT_TRANSPORT
|
|
81
|
+
retrieved_at = _utc(now)
|
|
82
|
+
records: list[dict[str, Any]] = []
|
|
83
|
+
requests: list[SourceRequestEvidence] = []
|
|
84
|
+
cache_states: list[CacheStatus] = []
|
|
85
|
+
limitations: list[str] = []
|
|
86
|
+
total_available: int | None = None
|
|
87
|
+
page_token: str | None = None
|
|
88
|
+
|
|
89
|
+
while len(records) < request.max_results:
|
|
90
|
+
url = build_search_url(request, page_token=page_token)
|
|
91
|
+
cache_params = _search_cache_params(request, page_token)
|
|
92
|
+
lookup = cache_lookup("clinicaltrials:search", cache_params)
|
|
93
|
+
cache_states.append(lookup.status)
|
|
94
|
+
try:
|
|
95
|
+
payload = lookup.payload
|
|
96
|
+
if payload is None:
|
|
97
|
+
response = transport.request(HttpRequest(method="GET", url=url))
|
|
98
|
+
payload = _json_mapping(response.body)
|
|
99
|
+
cache_store("clinicaltrials:search", cache_params, payload)
|
|
100
|
+
page_records, next_token, page_total, dropped = _parse_search_page(payload)
|
|
101
|
+
except TransportError as error:
|
|
102
|
+
return _search_failure(
|
|
103
|
+
request=request,
|
|
104
|
+
url=url,
|
|
105
|
+
retrieved_at=retrieved_at,
|
|
106
|
+
records=records,
|
|
107
|
+
requests=requests,
|
|
108
|
+
cache_states=cache_states,
|
|
109
|
+
total_available=total_available,
|
|
110
|
+
code="pagination_failed" if records else error.code,
|
|
111
|
+
message="later ClinicalTrials.gov page failed" if records else str(error),
|
|
112
|
+
)
|
|
113
|
+
except (ValueError, ValidationError, json.JSONDecodeError, UnicodeDecodeError):
|
|
114
|
+
return _search_failure(
|
|
115
|
+
request=request,
|
|
116
|
+
url=url,
|
|
117
|
+
retrieved_at=retrieved_at,
|
|
118
|
+
records=records,
|
|
119
|
+
requests=requests,
|
|
120
|
+
cache_states=cache_states,
|
|
121
|
+
total_available=total_available,
|
|
122
|
+
code="schema_mismatch",
|
|
123
|
+
message="ClinicalTrials.gov response shape was invalid",
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
if total_available is None:
|
|
127
|
+
if page_total is None:
|
|
128
|
+
return _search_failure(
|
|
129
|
+
request=request,
|
|
130
|
+
url=url,
|
|
131
|
+
retrieved_at=retrieved_at,
|
|
132
|
+
records=records,
|
|
133
|
+
requests=requests,
|
|
134
|
+
cache_states=cache_states,
|
|
135
|
+
total_available=None,
|
|
136
|
+
code="schema_mismatch",
|
|
137
|
+
message="ClinicalTrials.gov total count was missing",
|
|
138
|
+
)
|
|
139
|
+
total_available = page_total
|
|
140
|
+
|
|
141
|
+
if dropped and not page_records:
|
|
142
|
+
return _search_failure(
|
|
143
|
+
request=request,
|
|
144
|
+
url=url,
|
|
145
|
+
retrieved_at=retrieved_at,
|
|
146
|
+
records=records,
|
|
147
|
+
requests=requests,
|
|
148
|
+
cache_states=cache_states,
|
|
149
|
+
total_available=total_available,
|
|
150
|
+
code="schema_mismatch",
|
|
151
|
+
message="ClinicalTrials.gov page contained no usable records",
|
|
152
|
+
)
|
|
153
|
+
|
|
154
|
+
request_status = CoverageStatus.PARTIAL if dropped else CoverageStatus.COMPLETE
|
|
155
|
+
request_error = (
|
|
156
|
+
SourceError(code="schema_mismatch", message="some ClinicalTrials.gov records were invalid")
|
|
157
|
+
if dropped
|
|
158
|
+
else None
|
|
159
|
+
)
|
|
160
|
+
requests.append(
|
|
161
|
+
SourceRequestEvidence(
|
|
162
|
+
query=request.query,
|
|
163
|
+
source_url=url,
|
|
164
|
+
retrieved_at=retrieved_at,
|
|
165
|
+
cache=CacheProvenance(status=lookup.status, cached_at=lookup.cached_at),
|
|
166
|
+
status=request_status,
|
|
167
|
+
record_count=len(page_records),
|
|
168
|
+
error=request_error,
|
|
169
|
+
)
|
|
170
|
+
)
|
|
171
|
+
if dropped:
|
|
172
|
+
limitations.append(f"Dropped {dropped} malformed ClinicalTrials.gov record(s).")
|
|
173
|
+
remaining = request.max_results - len(records)
|
|
174
|
+
records.extend(page_records[:remaining])
|
|
175
|
+
page_token = next_token
|
|
176
|
+
if not page_token:
|
|
177
|
+
break
|
|
178
|
+
|
|
179
|
+
truncated = total_available is not None and total_available > len(records)
|
|
180
|
+
if truncated:
|
|
181
|
+
limitations.append(
|
|
182
|
+
f"Returned {len(records)} of {total_available} matching trials because max_results bounded collection."
|
|
183
|
+
)
|
|
184
|
+
status = (
|
|
185
|
+
CoverageStatus.PARTIAL
|
|
186
|
+
if truncated or any(r.status == CoverageStatus.PARTIAL for r in requests)
|
|
187
|
+
else CoverageStatus.COMPLETE
|
|
188
|
+
)
|
|
189
|
+
error = (
|
|
190
|
+
SourceError(code="truncated", message="ClinicalTrials.gov results were bounded or partially invalid")
|
|
191
|
+
if status == CoverageStatus.PARTIAL
|
|
192
|
+
else None
|
|
193
|
+
)
|
|
194
|
+
return SourceResult(
|
|
195
|
+
source=SourceName.CLINICAL_TRIALS,
|
|
196
|
+
provider="clinicaltrials.gov",
|
|
197
|
+
status=status,
|
|
198
|
+
query=request.query,
|
|
199
|
+
source_url=build_search_url(request),
|
|
200
|
+
retrieved_at=retrieved_at,
|
|
201
|
+
cache=CacheProvenance(status=aggregate_cache_status(cache_states)),
|
|
202
|
+
records=records,
|
|
203
|
+
total_available=total_available,
|
|
204
|
+
requests=requests,
|
|
205
|
+
limitations=limitations,
|
|
206
|
+
error=error,
|
|
207
|
+
)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
def get_trial_detail(
|
|
211
|
+
request: TrialDetailRequest,
|
|
212
|
+
*,
|
|
213
|
+
transport: HttpTransport | None = None,
|
|
214
|
+
now: datetime | None = None,
|
|
215
|
+
) -> SourceResult:
|
|
216
|
+
transport = transport or DEFAULT_TRANSPORT
|
|
217
|
+
retrieved_at = _utc(now)
|
|
218
|
+
url = f"{_BASE_URL}/{request.nct_id}"
|
|
219
|
+
lookup = cache_lookup("clinicaltrials:detail", {"nct_id": request.nct_id})
|
|
220
|
+
try:
|
|
221
|
+
payload = lookup.payload
|
|
222
|
+
if payload is None:
|
|
223
|
+
response = transport.request(HttpRequest(method="GET", url=url))
|
|
224
|
+
payload = _json_mapping(response.body)
|
|
225
|
+
cache_store("clinicaltrials:detail", {"nct_id": request.nct_id}, payload)
|
|
226
|
+
detail = _parse_trial_detail(payload)
|
|
227
|
+
except TransportError as error:
|
|
228
|
+
return _single_failure(
|
|
229
|
+
source_url=url,
|
|
230
|
+
query=request.nct_id,
|
|
231
|
+
retrieved_at=retrieved_at,
|
|
232
|
+
cache_status=lookup.status,
|
|
233
|
+
error=SourceError(code=error.code, message=str(error)),
|
|
234
|
+
)
|
|
235
|
+
except (ValueError, ValidationError, json.JSONDecodeError, UnicodeDecodeError):
|
|
236
|
+
return _single_failure(
|
|
237
|
+
source_url=url,
|
|
238
|
+
query=request.nct_id,
|
|
239
|
+
retrieved_at=retrieved_at,
|
|
240
|
+
cache_status=lookup.status,
|
|
241
|
+
error=SourceError(code="schema_mismatch", message="ClinicalTrials.gov detail shape was invalid"),
|
|
242
|
+
)
|
|
243
|
+
|
|
244
|
+
limitation = "The study endpoint exposes current status and milestone dates, not record-version history."
|
|
245
|
+
return SourceResult(
|
|
246
|
+
source=SourceName.CLINICAL_TRIALS,
|
|
247
|
+
provider="clinicaltrials.gov",
|
|
248
|
+
status=CoverageStatus.COMPLETE,
|
|
249
|
+
query=request.nct_id,
|
|
250
|
+
source_url=url,
|
|
251
|
+
retrieved_at=retrieved_at,
|
|
252
|
+
cache=CacheProvenance(status=lookup.status, cached_at=lookup.cached_at),
|
|
253
|
+
records=[detail.model_dump(mode="json")],
|
|
254
|
+
total_available=1,
|
|
255
|
+
requests=[
|
|
256
|
+
SourceRequestEvidence(
|
|
257
|
+
query=request.nct_id,
|
|
258
|
+
source_url=url,
|
|
259
|
+
retrieved_at=retrieved_at,
|
|
260
|
+
cache=CacheProvenance(status=lookup.status, cached_at=lookup.cached_at),
|
|
261
|
+
status=CoverageStatus.COMPLETE,
|
|
262
|
+
record_count=1,
|
|
263
|
+
)
|
|
264
|
+
],
|
|
265
|
+
limitations=[limitation],
|
|
266
|
+
)
|
|
267
|
+
|
|
268
|
+
|
|
269
|
+
def _parse_search_page(payload: Mapping[str, Any]) -> tuple[list[dict[str, Any]], str | None, int | None, int]:
|
|
270
|
+
studies = payload.get("studies")
|
|
271
|
+
if not isinstance(studies, list):
|
|
272
|
+
raise ValueError("studies must be a list")
|
|
273
|
+
total = payload.get("totalCount")
|
|
274
|
+
if total is not None and (not isinstance(total, int) or total < 0):
|
|
275
|
+
raise ValueError("totalCount must be a non-negative integer")
|
|
276
|
+
next_token = payload.get("nextPageToken")
|
|
277
|
+
if next_token is not None and not isinstance(next_token, str):
|
|
278
|
+
raise ValueError("nextPageToken must be a string")
|
|
279
|
+
records: list[dict[str, Any]] = []
|
|
280
|
+
dropped = 0
|
|
281
|
+
for study in studies:
|
|
282
|
+
try:
|
|
283
|
+
records.append(_parse_trial(study).model_dump(mode="json"))
|
|
284
|
+
except (ValueError, ValidationError, TypeError, AttributeError):
|
|
285
|
+
dropped += 1
|
|
286
|
+
return records, next_token, total, dropped
|
|
287
|
+
|
|
288
|
+
|
|
289
|
+
def _parse_trial(study: Any, *, detail: bool = False) -> Trial:
|
|
290
|
+
if not isinstance(study, Mapping) or not isinstance(study.get("hasResults"), bool):
|
|
291
|
+
raise ValueError("study and hasResults are required")
|
|
292
|
+
protocol = _mapping(study, "protocolSection")
|
|
293
|
+
identification = _mapping(protocol, "identificationModule")
|
|
294
|
+
status_module = _mapping(protocol, "statusModule")
|
|
295
|
+
sponsors = _mapping(protocol, "sponsorCollaboratorsModule")
|
|
296
|
+
design = _mapping(protocol, "designModule")
|
|
297
|
+
conditions = _mapping(protocol, "conditionsModule")
|
|
298
|
+
arms = _optional_mapping(protocol, "armsInterventionsModule")
|
|
299
|
+
outcomes = protocol.get("outcomesModule") or {}
|
|
300
|
+
if not isinstance(outcomes, Mapping):
|
|
301
|
+
raise ValueError("outcomesModule must be a mapping")
|
|
302
|
+
nct_id = _string(identification, "nctId")
|
|
303
|
+
title_key = "officialTitle" if detail and identification.get("officialTitle") else "briefTitle"
|
|
304
|
+
lead_sponsor = _mapping(sponsors, "leadSponsor")
|
|
305
|
+
collaborators = [
|
|
306
|
+
str(item.get("name", ""))
|
|
307
|
+
for item in _list(sponsors, "collaborators", default=[])
|
|
308
|
+
if isinstance(item, Mapping) and item.get("name")
|
|
309
|
+
]
|
|
310
|
+
interventions = [
|
|
311
|
+
TrialIntervention(
|
|
312
|
+
name=_string(item, "name"),
|
|
313
|
+
intervention_type=_string(item, "type"),
|
|
314
|
+
description=str(item.get("description", "")),
|
|
315
|
+
other_names=[str(name) for name in item.get("otherNames", []) if isinstance(name, str)],
|
|
316
|
+
)
|
|
317
|
+
for item in _list(arms, "interventions", default=[])
|
|
318
|
+
if isinstance(item, Mapping)
|
|
319
|
+
]
|
|
320
|
+
primary_outcomes = [
|
|
321
|
+
str(item.get("measure", ""))
|
|
322
|
+
for item in _list(outcomes, "primaryOutcomes", default=[])
|
|
323
|
+
if isinstance(item, Mapping) and item.get("measure")
|
|
324
|
+
]
|
|
325
|
+
phases = [str(value) for value in _list(design, "phases", default=[]) if isinstance(value, str)]
|
|
326
|
+
return Trial(
|
|
327
|
+
nct_id=nct_id,
|
|
328
|
+
title=_string(identification, title_key),
|
|
329
|
+
sponsor=_string(lead_sponsor, "name"),
|
|
330
|
+
collaborators=collaborators,
|
|
331
|
+
phase=", ".join(phases),
|
|
332
|
+
status=_string(status_module, "overallStatus"),
|
|
333
|
+
conditions=[str(value) for value in _list(conditions, "conditions", default=[]) if isinstance(value, str)],
|
|
334
|
+
interventions=interventions,
|
|
335
|
+
enrollment=_optional_int(_mapping(design, "enrollmentInfo").get("count")),
|
|
336
|
+
start_date=_date_value(status_module, "startDateStruct"),
|
|
337
|
+
primary_completion_date=_date_value(status_module, "primaryCompletionDateStruct"),
|
|
338
|
+
estimated_completion_date=_date_value(status_module, "completionDateStruct"),
|
|
339
|
+
study_type=str(design.get("studyType", "INTERVENTIONAL")),
|
|
340
|
+
primary_endpoints=primary_outcomes,
|
|
341
|
+
has_results=study["hasResults"],
|
|
342
|
+
source_url=f"https://clinicaltrials.gov/study/{nct_id}",
|
|
343
|
+
)
|
|
344
|
+
|
|
345
|
+
|
|
346
|
+
def _parse_trial_detail(payload: Mapping[str, Any]) -> TrialDetail:
|
|
347
|
+
trial = _parse_trial(payload, detail=True)
|
|
348
|
+
protocol = _mapping(payload, "protocolSection")
|
|
349
|
+
arms_module = _optional_mapping(protocol, "armsInterventionsModule")
|
|
350
|
+
outcomes = protocol.get("outcomesModule") or {}
|
|
351
|
+
if not isinstance(outcomes, Mapping):
|
|
352
|
+
raise ValueError("outcomesModule must be a mapping")
|
|
353
|
+
arms = [
|
|
354
|
+
TrialArm(
|
|
355
|
+
label=_string(item, "label"),
|
|
356
|
+
type=_string(item, "type"),
|
|
357
|
+
description=str(item.get("description", "")),
|
|
358
|
+
interventions=[str(name) for name in item.get("interventionNames", []) if isinstance(name, str)],
|
|
359
|
+
)
|
|
360
|
+
for item in _list(arms_module, "armGroups", default=[])
|
|
361
|
+
if isinstance(item, Mapping)
|
|
362
|
+
]
|
|
363
|
+
secondary = [
|
|
364
|
+
str(item.get("measure", ""))
|
|
365
|
+
for item in _list(outcomes, "secondaryOutcomes", default=[])
|
|
366
|
+
if isinstance(item, Mapping) and item.get("measure")
|
|
367
|
+
]
|
|
368
|
+
status_module = _mapping(protocol, "statusModule")
|
|
369
|
+
milestones = []
|
|
370
|
+
for label, key in (
|
|
371
|
+
("Study Start", "startDateStruct"),
|
|
372
|
+
("Primary Completion", "primaryCompletionDateStruct"),
|
|
373
|
+
("Study Completion", "completionDateStruct"),
|
|
374
|
+
):
|
|
375
|
+
value = status_module.get(key)
|
|
376
|
+
if isinstance(value, Mapping) and value.get("date"):
|
|
377
|
+
milestones.append({"milestone": label, "date": str(value["date"]), "type": str(value.get("type", ""))})
|
|
378
|
+
results = payload.get("resultsSection")
|
|
379
|
+
results_summary = None
|
|
380
|
+
if results is not None:
|
|
381
|
+
if not isinstance(results, Mapping):
|
|
382
|
+
raise ValueError("resultsSection must be a mapping")
|
|
383
|
+
measures = results.get("outcomeMeasuresModule") or {}
|
|
384
|
+
if not isinstance(measures, Mapping):
|
|
385
|
+
raise ValueError("outcomeMeasuresModule must be a mapping")
|
|
386
|
+
outcome_measures = measures.get("outcomeMeasures", [])
|
|
387
|
+
if not isinstance(outcome_measures, list):
|
|
388
|
+
raise ValueError("outcomeMeasures must be a list")
|
|
389
|
+
results_summary = TrialResultsSummary(
|
|
390
|
+
has_results=True,
|
|
391
|
+
primary_outcomes_count=len(outcome_measures),
|
|
392
|
+
adverse_events_reported=isinstance(results.get("adverseEventsModule"), Mapping),
|
|
393
|
+
)
|
|
394
|
+
references_module = protocol.get("referencesModule") or {}
|
|
395
|
+
if not isinstance(references_module, Mapping):
|
|
396
|
+
raise ValueError("referencesModule must be a mapping")
|
|
397
|
+
publications = [
|
|
398
|
+
str(item.get("pmid"))
|
|
399
|
+
for item in _list(references_module, "references", default=[])
|
|
400
|
+
if isinstance(item, Mapping) and item.get("pmid")
|
|
401
|
+
]
|
|
402
|
+
eligibility_module = protocol.get("eligibilityModule") or {}
|
|
403
|
+
if not isinstance(eligibility_module, Mapping):
|
|
404
|
+
raise ValueError("eligibilityModule must be a mapping")
|
|
405
|
+
eligibility = str(eligibility_module.get("eligibilityCriteria", "")) or None
|
|
406
|
+
if eligibility and len(eligibility) > 1000:
|
|
407
|
+
eligibility = eligibility[:1000] + "..."
|
|
408
|
+
return TrialDetail(
|
|
409
|
+
trial=trial,
|
|
410
|
+
arms=arms,
|
|
411
|
+
secondary_endpoints=secondary,
|
|
412
|
+
eligibility_criteria=eligibility,
|
|
413
|
+
status_history=[],
|
|
414
|
+
milestone_dates=milestones,
|
|
415
|
+
results_summary=results_summary,
|
|
416
|
+
publications=publications,
|
|
417
|
+
)
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
def _search_failure(
|
|
421
|
+
*,
|
|
422
|
+
request: TrialSearchRequest,
|
|
423
|
+
url: str,
|
|
424
|
+
retrieved_at: datetime,
|
|
425
|
+
records: list[dict[str, Any]],
|
|
426
|
+
requests: list[SourceRequestEvidence],
|
|
427
|
+
cache_states: list[CacheStatus],
|
|
428
|
+
total_available: int | None,
|
|
429
|
+
code: str,
|
|
430
|
+
message: str,
|
|
431
|
+
) -> SourceResult:
|
|
432
|
+
error = SourceError(code=code, message=message)
|
|
433
|
+
requests.append(
|
|
434
|
+
SourceRequestEvidence(
|
|
435
|
+
query=request.query,
|
|
436
|
+
source_url=url,
|
|
437
|
+
retrieved_at=retrieved_at,
|
|
438
|
+
cache=CacheProvenance(status=cache_states[-1]),
|
|
439
|
+
status=CoverageStatus.FAILED,
|
|
440
|
+
record_count=0,
|
|
441
|
+
error=error,
|
|
442
|
+
)
|
|
443
|
+
)
|
|
444
|
+
if records:
|
|
445
|
+
return SourceResult(
|
|
446
|
+
source=SourceName.CLINICAL_TRIALS,
|
|
447
|
+
provider="clinicaltrials.gov",
|
|
448
|
+
status=CoverageStatus.PARTIAL,
|
|
449
|
+
query=request.query,
|
|
450
|
+
source_url=build_search_url(request),
|
|
451
|
+
retrieved_at=retrieved_at,
|
|
452
|
+
cache=CacheProvenance(status=aggregate_cache_status(cache_states)),
|
|
453
|
+
records=records,
|
|
454
|
+
total_available=total_available,
|
|
455
|
+
requests=requests,
|
|
456
|
+
limitations=["ClinicalTrials.gov pagination stopped after a later request failed."],
|
|
457
|
+
error=SourceError(code="pagination_failed", message="later ClinicalTrials.gov page failed"),
|
|
458
|
+
)
|
|
459
|
+
return SourceResult(
|
|
460
|
+
source=SourceName.CLINICAL_TRIALS,
|
|
461
|
+
provider="clinicaltrials.gov",
|
|
462
|
+
status=CoverageStatus.FAILED,
|
|
463
|
+
query=request.query,
|
|
464
|
+
source_url=build_search_url(request),
|
|
465
|
+
retrieved_at=retrieved_at,
|
|
466
|
+
cache=CacheProvenance(status=aggregate_cache_status(cache_states)),
|
|
467
|
+
requests=requests,
|
|
468
|
+
limitations=["No trustworthy ClinicalTrials.gov response was obtained."],
|
|
469
|
+
error=error,
|
|
470
|
+
)
|
|
471
|
+
|
|
472
|
+
|
|
473
|
+
def _single_failure(
|
|
474
|
+
*, source_url: str, query: str, retrieved_at: datetime, cache_status: CacheStatus, error: SourceError
|
|
475
|
+
) -> SourceResult:
|
|
476
|
+
evidence = SourceRequestEvidence(
|
|
477
|
+
query=query,
|
|
478
|
+
source_url=source_url,
|
|
479
|
+
retrieved_at=retrieved_at,
|
|
480
|
+
cache=CacheProvenance(status=cache_status),
|
|
481
|
+
status=CoverageStatus.FAILED,
|
|
482
|
+
record_count=0,
|
|
483
|
+
error=error,
|
|
484
|
+
)
|
|
485
|
+
return SourceResult(
|
|
486
|
+
source=SourceName.CLINICAL_TRIALS,
|
|
487
|
+
provider="clinicaltrials.gov",
|
|
488
|
+
status=CoverageStatus.FAILED,
|
|
489
|
+
query=query,
|
|
490
|
+
source_url=source_url,
|
|
491
|
+
retrieved_at=retrieved_at,
|
|
492
|
+
cache=CacheProvenance(status=cache_status),
|
|
493
|
+
requests=[evidence],
|
|
494
|
+
limitations=["No trustworthy ClinicalTrials.gov detail response was obtained."],
|
|
495
|
+
error=error,
|
|
496
|
+
)
|
|
497
|
+
|
|
498
|
+
|
|
499
|
+
def _search_cache_params(request: TrialSearchRequest, page_token: str | None) -> dict[str, Any]:
|
|
500
|
+
return {
|
|
501
|
+
"query": request.query,
|
|
502
|
+
"phase": request.phase,
|
|
503
|
+
"status": request.status,
|
|
504
|
+
"page_size": request.max_results,
|
|
505
|
+
"page_token": page_token,
|
|
506
|
+
"count_total": True,
|
|
507
|
+
}
|
|
508
|
+
|
|
509
|
+
|
|
510
|
+
def _json_mapping(body: bytes) -> Mapping[str, Any]:
|
|
511
|
+
payload = json.loads(body.decode("utf-8"))
|
|
512
|
+
if not isinstance(payload, Mapping):
|
|
513
|
+
raise ValueError("provider response must be a mapping")
|
|
514
|
+
return payload
|
|
515
|
+
|
|
516
|
+
|
|
517
|
+
def _mapping(container: Mapping[str, Any], key: str) -> Mapping[str, Any]:
|
|
518
|
+
value = container.get(key)
|
|
519
|
+
if not isinstance(value, Mapping):
|
|
520
|
+
raise ValueError(f"{key} must be a mapping")
|
|
521
|
+
return value
|
|
522
|
+
|
|
523
|
+
|
|
524
|
+
def _optional_mapping(container: Mapping[str, Any], key: str) -> Mapping[str, Any]:
|
|
525
|
+
value = container.get(key)
|
|
526
|
+
if value is None:
|
|
527
|
+
return {}
|
|
528
|
+
if not isinstance(value, Mapping):
|
|
529
|
+
raise ValueError(f"{key} must be a mapping")
|
|
530
|
+
return value
|
|
531
|
+
|
|
532
|
+
|
|
533
|
+
def _list(container: Mapping[str, Any], key: str, *, default: list[Any] | None = None) -> list[Any]:
|
|
534
|
+
value = container.get(key, default)
|
|
535
|
+
if not isinstance(value, list):
|
|
536
|
+
raise ValueError(f"{key} must be a list")
|
|
537
|
+
return value
|
|
538
|
+
|
|
539
|
+
|
|
540
|
+
def _string(container: Mapping[str, Any], key: str) -> str:
|
|
541
|
+
value = container.get(key)
|
|
542
|
+
if not isinstance(value, str) or not value:
|
|
543
|
+
raise ValueError(f"{key} must be a non-empty string")
|
|
544
|
+
return value
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
def _date_value(status_module: Mapping[str, Any], key: str) -> str | None:
|
|
548
|
+
value = status_module.get(key)
|
|
549
|
+
if value is None:
|
|
550
|
+
return None
|
|
551
|
+
if not isinstance(value, Mapping):
|
|
552
|
+
raise ValueError(f"{key} must be a mapping")
|
|
553
|
+
date = value.get("date")
|
|
554
|
+
return str(date) if date else None
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
def _optional_int(value: Any) -> int | None:
|
|
558
|
+
if value is None:
|
|
559
|
+
return None
|
|
560
|
+
if not isinstance(value, int) or isinstance(value, bool):
|
|
561
|
+
raise ValueError("enrollment count must be an integer")
|
|
562
|
+
return value
|
|
563
|
+
|
|
564
|
+
|
|
565
|
+
def _utc(value: datetime | None) -> datetime:
|
|
566
|
+
now = value or datetime.now(timezone.utc)
|
|
567
|
+
if now.tzinfo is None:
|
|
568
|
+
raise ValueError("now must be timezone-aware")
|
|
569
|
+
return now.astimezone(timezone.utc)
|