open-pharma-plugins 2.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_framework.py +495 -0
- open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
- open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
- open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
- open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
- open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
- open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
- open_pharma_plugins_campaign_studio/__init__.py +14 -0
- open_pharma_plugins_campaign_studio/__main__.py +11 -0
- open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
- open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
- open_pharma_plugins_campaign_studio/_renderer.py +119 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
- open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
- open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
- open_pharma_plugins_campaign_studio/models/_common.py +12 -0
- open_pharma_plugins_campaign_studio/models/brief.py +72 -0
- open_pharma_plugins_campaign_studio/models/claims.py +15 -0
- open_pharma_plugins_campaign_studio/models/copy.py +47 -0
- open_pharma_plugins_campaign_studio/models/journey.py +21 -0
- open_pharma_plugins_campaign_studio/models/message.py +25 -0
- open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
- open_pharma_plugins_campaign_studio/models/validation.py +32 -0
- open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
- open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
- open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
- open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
- open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
- open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
- open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
- open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
- open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
- open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
- open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
- open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
- open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
- open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
- open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
- open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
- open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
- open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
- open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
- open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
- open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
- open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
- open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
- open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
- open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
- open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
- open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
- open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
- open_pharma_plugins_competitive_intelligence/models.py +525 -0
- open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
- open_pharma_plugins_field_training/__init__.py +13 -0
- open_pharma_plugins_field_training/__main__.py +11 -0
- open_pharma_plugins_field_training/_content_store.py +131 -0
- open_pharma_plugins_field_training/_grounding.py +75 -0
- open_pharma_plugins_field_training/_html_renderers.py +546 -0
- open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
- open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
- open_pharma_plugins_field_training/models.py +265 -0
- open_pharma_plugins_field_training/tools/__init__.py +0 -0
- open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
- open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
- open_pharma_plugins_field_training/tools/list_documents.py +51 -0
- open_pharma_plugins_field_training/tools/render_output.py +118 -0
- open_pharma_plugins_field_training/tools/search_content.py +57 -0
- open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
- open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
- open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
- open_pharma_plugins_hcp_intelligence/batch.py +891 -0
- open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
- open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
- open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
- open_pharma_plugins_hcp_intelligence/models.py +356 -0
- open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
- open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
- open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
- open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
- open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
- open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
- open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
- open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
- open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
- open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
- open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
- open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
- open_pharma_plugins_next_best_engagement/__init__.py +14 -0
- open_pharma_plugins_next_best_engagement/__main__.py +11 -0
- open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
- open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
- open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
- open_pharma_plugins_next_best_engagement/_universe.py +145 -0
- open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
- open_pharma_plugins_next_best_engagement/models.py +135 -0
- open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
- open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
- open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
- open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
- open_pharma_plugins_territory_alignment/__init__.py +14 -0
- open_pharma_plugins_territory_alignment/__main__.py +11 -0
- open_pharma_plugins_territory_alignment/data.py +300 -0
- open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
- open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
- open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
- open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
- open_pharma_plugins_territory_alignment/geo.py +175 -0
- open_pharma_plugins_territory_alignment/models.py +201 -0
- open_pharma_plugins_territory_alignment/scoring.py +125 -0
- open_pharma_plugins_territory_alignment/solver.py +504 -0
- open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
- open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
- open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
- open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
- open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
- open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
- open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
- shared/__init__.py +11 -0
- shared/env.py +217 -0
- shared/filesystem.py +110 -0
|
@@ -0,0 +1,199 @@
|
|
|
1
|
+
"""search_hcp_web — web search tailored for HCP profiling."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class SearchHcpWebArgs(BaseModel):
|
|
11
|
+
name: str = Field(description="Full name of the healthcare professional")
|
|
12
|
+
specialty: str | None = Field(default=None, description="Medical specialty (e.g. 'Cardiology')")
|
|
13
|
+
country: str | None = Field(default=None, description="Country of practice")
|
|
14
|
+
institution: str | None = Field(default=None, description="Known institution or affiliation")
|
|
15
|
+
query_focus: str | None = Field(
|
|
16
|
+
default=None,
|
|
17
|
+
description=(
|
|
18
|
+
"Optional focus to append to the search query, e.g. "
|
|
19
|
+
"'biography', 'society membership', 'education qualifications', "
|
|
20
|
+
"'advisory board committee'"
|
|
21
|
+
),
|
|
22
|
+
)
|
|
23
|
+
max_results: int = Field(
|
|
24
|
+
default=10,
|
|
25
|
+
ge=1,
|
|
26
|
+
le=20,
|
|
27
|
+
description="Maximum web results to return",
|
|
28
|
+
)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
TOOL: dict[str, Any] = {
|
|
32
|
+
"name": "search_hcp_web",
|
|
33
|
+
"description": (
|
|
34
|
+
"Search the web for information about a specific Healthcare Professional. "
|
|
35
|
+
"Constructs a targeted query from the HCP's name, specialty, country, and "
|
|
36
|
+
"institution to find institutional profiles, society memberships, conference "
|
|
37
|
+
"appearances, and biographical information. Returns URLs, titles, and snippets. "
|
|
38
|
+
"Use query_focus to steer toward specific profile sections (e.g. 'education', "
|
|
39
|
+
"'advisory board')."
|
|
40
|
+
),
|
|
41
|
+
"args": SearchHcpWebArgs,
|
|
42
|
+
}
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def handle(arguments: dict[str, Any]) -> list[dict[str, Any]]:
|
|
46
|
+
import json
|
|
47
|
+
from datetime import datetime, timezone
|
|
48
|
+
|
|
49
|
+
query_parts = [f'"{arguments["name"]}"']
|
|
50
|
+
if arguments.get("specialty"):
|
|
51
|
+
query_parts.append(arguments["specialty"])
|
|
52
|
+
if arguments.get("country"):
|
|
53
|
+
query_parts.append(arguments["country"])
|
|
54
|
+
if arguments.get("institution"):
|
|
55
|
+
query_parts.append(arguments["institution"])
|
|
56
|
+
if arguments.get("query_focus"):
|
|
57
|
+
query_parts.append(arguments["query_focus"])
|
|
58
|
+
else:
|
|
59
|
+
query_parts.append("doctor OR physician OR professor OR consultant")
|
|
60
|
+
|
|
61
|
+
query = " ".join(query_parts)
|
|
62
|
+
max_results = arguments.get("max_results", 10)
|
|
63
|
+
|
|
64
|
+
results = _web_search(query, max_results)
|
|
65
|
+
|
|
66
|
+
output = {
|
|
67
|
+
"query": query,
|
|
68
|
+
"results": results,
|
|
69
|
+
"search_backend": _backend_name(),
|
|
70
|
+
"searched_at": datetime.now(timezone.utc).isoformat(),
|
|
71
|
+
}
|
|
72
|
+
return [{"type": "text", "text": json.dumps(output, indent=2)}]
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _backend_name() -> str:
|
|
76
|
+
from shared.env import get_env
|
|
77
|
+
|
|
78
|
+
backend = get_env("OPEN_PHARMA_SEARCH_BACKEND", "auto").strip().lower()
|
|
79
|
+
if backend != "auto":
|
|
80
|
+
return backend
|
|
81
|
+
if get_env("SERPER_API_KEY", ""):
|
|
82
|
+
return "serper"
|
|
83
|
+
if get_env("TAVILY_API_KEY", ""):
|
|
84
|
+
return "tavily"
|
|
85
|
+
if get_env("EXA_API_KEY", ""):
|
|
86
|
+
return "exa"
|
|
87
|
+
return "none"
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def _web_search(query: str, max_results: int) -> list[dict]:
|
|
91
|
+
backend = _backend_name()
|
|
92
|
+
if backend == "serper":
|
|
93
|
+
return _serper_search(query, max_results)
|
|
94
|
+
if backend == "tavily":
|
|
95
|
+
return _tavily_search(query, max_results)
|
|
96
|
+
if backend == "exa":
|
|
97
|
+
return _exa_search(query, max_results)
|
|
98
|
+
raise RuntimeError("No web search backend configured. Set SERPER_API_KEY, TAVILY_API_KEY, or EXA_API_KEY.")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _serper_search(query: str, max_results: int) -> list[dict]:
|
|
102
|
+
import json
|
|
103
|
+
import urllib.request
|
|
104
|
+
|
|
105
|
+
from shared.env import get_env
|
|
106
|
+
|
|
107
|
+
body = json.dumps({"q": query, "num": max_results}).encode()
|
|
108
|
+
req = urllib.request.Request(
|
|
109
|
+
"https://google.serper.dev/search",
|
|
110
|
+
data=body,
|
|
111
|
+
headers={
|
|
112
|
+
"X-API-KEY": get_env("SERPER_API_KEY", ""),
|
|
113
|
+
"Content-Type": "application/json",
|
|
114
|
+
},
|
|
115
|
+
)
|
|
116
|
+
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
117
|
+
data = json.loads(resp.read())
|
|
118
|
+
|
|
119
|
+
return [
|
|
120
|
+
{
|
|
121
|
+
"url": r.get("link", ""),
|
|
122
|
+
"title": r.get("title", ""),
|
|
123
|
+
"snippet": r.get("snippet", ""),
|
|
124
|
+
"published_date": r.get("date"),
|
|
125
|
+
"domain": r.get("link", "").split("/")[2] if "/" in r.get("link", "") else None,
|
|
126
|
+
}
|
|
127
|
+
for r in data.get("organic", [])
|
|
128
|
+
]
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _tavily_search(query: str, max_results: int) -> list[dict]:
|
|
132
|
+
import json
|
|
133
|
+
import urllib.request
|
|
134
|
+
|
|
135
|
+
from shared.env import get_env
|
|
136
|
+
|
|
137
|
+
body = json.dumps(
|
|
138
|
+
{
|
|
139
|
+
"api_key": get_env("TAVILY_API_KEY", ""),
|
|
140
|
+
"query": query,
|
|
141
|
+
"max_results": max_results,
|
|
142
|
+
"search_depth": "advanced",
|
|
143
|
+
}
|
|
144
|
+
).encode()
|
|
145
|
+
req = urllib.request.Request(
|
|
146
|
+
"https://api.tavily.com/search",
|
|
147
|
+
data=body,
|
|
148
|
+
headers={"Content-Type": "application/json"},
|
|
149
|
+
)
|
|
150
|
+
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
151
|
+
data = json.loads(resp.read())
|
|
152
|
+
|
|
153
|
+
return [
|
|
154
|
+
{
|
|
155
|
+
"url": r.get("url", ""),
|
|
156
|
+
"title": r.get("title", ""),
|
|
157
|
+
"snippet": r.get("content", ""),
|
|
158
|
+
"published_date": r.get("published_date"),
|
|
159
|
+
"domain": r.get("url", "").split("/")[2] if "/" in r.get("url", "") else None,
|
|
160
|
+
}
|
|
161
|
+
for r in data.get("results", [])
|
|
162
|
+
]
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _exa_search(query: str, max_results: int) -> list[dict]:
|
|
166
|
+
import json
|
|
167
|
+
import urllib.request
|
|
168
|
+
|
|
169
|
+
from shared.env import get_env
|
|
170
|
+
|
|
171
|
+
body = json.dumps(
|
|
172
|
+
{
|
|
173
|
+
"query": query,
|
|
174
|
+
"numResults": max_results,
|
|
175
|
+
"useAutoprompt": True,
|
|
176
|
+
"contents": {"text": {"maxCharacters": 500}},
|
|
177
|
+
}
|
|
178
|
+
).encode()
|
|
179
|
+
req = urllib.request.Request(
|
|
180
|
+
"https://api.exa.ai/search",
|
|
181
|
+
data=body,
|
|
182
|
+
headers={
|
|
183
|
+
"x-api-key": get_env("EXA_API_KEY", ""),
|
|
184
|
+
"Content-Type": "application/json",
|
|
185
|
+
},
|
|
186
|
+
)
|
|
187
|
+
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
188
|
+
data = json.loads(resp.read())
|
|
189
|
+
|
|
190
|
+
return [
|
|
191
|
+
{
|
|
192
|
+
"url": r.get("url", ""),
|
|
193
|
+
"title": r.get("title", ""),
|
|
194
|
+
"snippet": r.get("text", ""),
|
|
195
|
+
"published_date": r.get("publishedDate"),
|
|
196
|
+
"domain": r.get("url", "").split("/")[2] if "/" in r.get("url", "") else None,
|
|
197
|
+
}
|
|
198
|
+
for r in data.get("results", [])
|
|
199
|
+
]
|
|
@@ -0,0 +1,214 @@
|
|
|
1
|
+
"""search_orcid — query the ORCID public API for researcher profiles."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class SearchOrcidArgs(BaseModel):
|
|
11
|
+
name: str = Field(
|
|
12
|
+
description="Full name of the researcher (e.g. 'Yvonne Lim')",
|
|
13
|
+
)
|
|
14
|
+
affiliation: str | None = Field(
|
|
15
|
+
default=None,
|
|
16
|
+
description="Known institution to help disambiguate (e.g. 'KK Women's and Children's Hospital')",
|
|
17
|
+
)
|
|
18
|
+
orcid_id: str | None = Field(
|
|
19
|
+
default=None,
|
|
20
|
+
description="If the ORCID ID is already known (e.g. '0000-0002-1234-5678'), fetch directly instead of searching",
|
|
21
|
+
)
|
|
22
|
+
max_results: int = Field(
|
|
23
|
+
default=5,
|
|
24
|
+
ge=1,
|
|
25
|
+
le=10,
|
|
26
|
+
description="Maximum candidate profiles to return when searching by name",
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
TOOL: dict[str, Any] = {
|
|
31
|
+
"name": "search_orcid",
|
|
32
|
+
"description": (
|
|
33
|
+
"Search the ORCID registry for a researcher's profile. ORCID provides "
|
|
34
|
+
"author-curated, globally unique researcher identifiers with verified "
|
|
35
|
+
"affiliations, education history, publication counts, and funding. Call "
|
|
36
|
+
"early in the HCP profiling workflow — it disambiguates common names and "
|
|
37
|
+
"fills education/affiliation gaps cheaply. Pass an orcid_id to fetch a "
|
|
38
|
+
"known profile directly, or search by name and optional affiliation."
|
|
39
|
+
),
|
|
40
|
+
"args": SearchOrcidArgs,
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
|
|
44
|
+
_BASE = "https://pub.orcid.org/v3.0"
|
|
45
|
+
_HEADERS = {"Accept": "application/json"}
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def handle(arguments: dict[str, Any]) -> list[dict[str, Any]]:
|
|
49
|
+
import json
|
|
50
|
+
from datetime import datetime, timezone
|
|
51
|
+
|
|
52
|
+
orcid_id = arguments.get("orcid_id")
|
|
53
|
+
if orcid_id:
|
|
54
|
+
profile = _fetch_profile(orcid_id)
|
|
55
|
+
output = {
|
|
56
|
+
"query": orcid_id,
|
|
57
|
+
"total_found": 1 if profile else 0,
|
|
58
|
+
"profiles": [profile] if profile else [],
|
|
59
|
+
"source": "orcid",
|
|
60
|
+
"searched_at": datetime.now(timezone.utc).isoformat(),
|
|
61
|
+
}
|
|
62
|
+
return [{"type": "text", "text": json.dumps(output, indent=2)}]
|
|
63
|
+
|
|
64
|
+
name = arguments["name"]
|
|
65
|
+
affiliation = arguments.get("affiliation")
|
|
66
|
+
max_results = arguments.get("max_results", 5)
|
|
67
|
+
|
|
68
|
+
query = _build_query(name, affiliation)
|
|
69
|
+
orcid_ids = _search_ids(query, max_results)
|
|
70
|
+
|
|
71
|
+
profiles = []
|
|
72
|
+
for oid in orcid_ids:
|
|
73
|
+
p = _fetch_profile(oid)
|
|
74
|
+
if p:
|
|
75
|
+
profiles.append(p)
|
|
76
|
+
|
|
77
|
+
output = {
|
|
78
|
+
"query": query,
|
|
79
|
+
"total_found": len(orcid_ids),
|
|
80
|
+
"profiles": profiles,
|
|
81
|
+
"source": "orcid",
|
|
82
|
+
"searched_at": datetime.now(timezone.utc).isoformat(),
|
|
83
|
+
}
|
|
84
|
+
return [{"type": "text", "text": json.dumps(output, indent=2)}]
|
|
85
|
+
|
|
86
|
+
|
|
87
|
+
def _build_query(name: str, affiliation: str | None) -> str:
|
|
88
|
+
parts = name.strip().split()
|
|
89
|
+
if len(parts) >= 2:
|
|
90
|
+
family = parts[-1]
|
|
91
|
+
given = " ".join(parts[:-1])
|
|
92
|
+
q = f"family-name:{family} AND given-names:{given}"
|
|
93
|
+
else:
|
|
94
|
+
q = f"family-name:{name}"
|
|
95
|
+
|
|
96
|
+
if affiliation:
|
|
97
|
+
q += f" AND affiliation-org-name:{affiliation}"
|
|
98
|
+
return q
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def _search_ids(query: str, max_results: int) -> list[str]:
|
|
102
|
+
import json
|
|
103
|
+
import urllib.parse
|
|
104
|
+
import urllib.request
|
|
105
|
+
|
|
106
|
+
url = f"{_BASE}/search/?q={urllib.parse.quote(query)}&rows={max_results}"
|
|
107
|
+
req = urllib.request.Request(url, headers=_HEADERS)
|
|
108
|
+
try:
|
|
109
|
+
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
110
|
+
data = json.loads(resp.read())
|
|
111
|
+
except Exception:
|
|
112
|
+
return []
|
|
113
|
+
|
|
114
|
+
results = data.get("result", []) or []
|
|
115
|
+
ids = []
|
|
116
|
+
for r in results:
|
|
117
|
+
oid = r.get("orcid-identifier", {}).get("path")
|
|
118
|
+
if oid:
|
|
119
|
+
ids.append(oid)
|
|
120
|
+
return ids
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def _fetch_profile(orcid_id: str) -> dict | None:
|
|
124
|
+
import json
|
|
125
|
+
import urllib.request
|
|
126
|
+
|
|
127
|
+
url = f"{_BASE}/{orcid_id}"
|
|
128
|
+
req = urllib.request.Request(url, headers=_HEADERS)
|
|
129
|
+
try:
|
|
130
|
+
with urllib.request.urlopen(req, timeout=15) as resp:
|
|
131
|
+
data = json.loads(resp.read())
|
|
132
|
+
except Exception:
|
|
133
|
+
return None
|
|
134
|
+
|
|
135
|
+
return _extract_profile(orcid_id, data)
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
def _extract_profile(orcid_id: str, data: dict) -> dict:
|
|
139
|
+
person = data.get("person", {}) or {}
|
|
140
|
+
activities = data.get("activities-summary", {}) or {}
|
|
141
|
+
|
|
142
|
+
name_obj = person.get("name", {}) or {}
|
|
143
|
+
given = (name_obj.get("given-names", {}) or {}).get("value", "")
|
|
144
|
+
family = (name_obj.get("family-name", {}) or {}).get("value", "")
|
|
145
|
+
|
|
146
|
+
bio_obj = person.get("biography", {}) or {}
|
|
147
|
+
biography = bio_obj.get("content", "") or ""
|
|
148
|
+
if len(biography) > 500:
|
|
149
|
+
biography = biography[:497] + "..."
|
|
150
|
+
|
|
151
|
+
education = _extract_affiliations(activities.get("educations", {}) or {}, "education-summary")
|
|
152
|
+
employment = _extract_affiliations(activities.get("employments", {}) or {}, "employment-summary")
|
|
153
|
+
|
|
154
|
+
work_groups = (activities.get("works", {}) or {}).get("group", []) or []
|
|
155
|
+
funding_groups = (activities.get("fundings", {}) or {}).get("group", []) or []
|
|
156
|
+
review_groups = (activities.get("peer-reviews", {}) or {}).get("group", []) or []
|
|
157
|
+
|
|
158
|
+
return {
|
|
159
|
+
"orcid_id": orcid_id,
|
|
160
|
+
"given_names": given,
|
|
161
|
+
"family_name": family,
|
|
162
|
+
"biography": biography or None,
|
|
163
|
+
"education": education,
|
|
164
|
+
"employment": employment,
|
|
165
|
+
"publication_count": len(work_groups),
|
|
166
|
+
"funding_count": len(funding_groups),
|
|
167
|
+
"peer_review_count": len(review_groups),
|
|
168
|
+
"profile_url": f"https://orcid.org/{orcid_id}",
|
|
169
|
+
}
|
|
170
|
+
|
|
171
|
+
|
|
172
|
+
def _extract_affiliations(section: dict, summary_key: str) -> list[dict]:
|
|
173
|
+
groups = section.get("affiliation-group", []) or []
|
|
174
|
+
results = []
|
|
175
|
+
for group in groups:
|
|
176
|
+
summaries = group.get("summaries", []) or []
|
|
177
|
+
for entry in summaries:
|
|
178
|
+
summary = entry.get(summary_key, {}) or {}
|
|
179
|
+
org = (summary.get("organization", {}) or {}).get("name", "")
|
|
180
|
+
role = (summary.get("role-title") or "") if summary_key == "employment-summary" else ""
|
|
181
|
+
degree = (summary.get("role-title") or "") if summary_key == "education-summary" else ""
|
|
182
|
+
dept = summary.get("department-name") or ""
|
|
183
|
+
|
|
184
|
+
start = summary.get("start-date") or {}
|
|
185
|
+
end = summary.get("end-date") or {}
|
|
186
|
+
start_year = _year_from(start)
|
|
187
|
+
end_year = _year_from(end)
|
|
188
|
+
|
|
189
|
+
entry_dict: dict[str, Any] = {"institution": org}
|
|
190
|
+
if summary_key == "education-summary":
|
|
191
|
+
entry_dict["degree"] = degree
|
|
192
|
+
else:
|
|
193
|
+
entry_dict["role"] = role
|
|
194
|
+
entry_dict["department"] = dept or None
|
|
195
|
+
entry_dict["start_year"] = start_year
|
|
196
|
+
entry_dict["end_year"] = end_year
|
|
197
|
+
results.append(entry_dict)
|
|
198
|
+
return results
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _year_from(date_obj: dict | None) -> int | None:
|
|
202
|
+
if not date_obj:
|
|
203
|
+
return None
|
|
204
|
+
year = date_obj.get("year", {})
|
|
205
|
+
if isinstance(year, dict):
|
|
206
|
+
val = year.get("value")
|
|
207
|
+
else:
|
|
208
|
+
val = year
|
|
209
|
+
if val:
|
|
210
|
+
try:
|
|
211
|
+
return int(val)
|
|
212
|
+
except (ValueError, TypeError):
|
|
213
|
+
pass
|
|
214
|
+
return None
|
|
@@ -0,0 +1,207 @@
|
|
|
1
|
+
"""search_publications — query PubMed via NCBI E-utilities."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
from typing import Any
|
|
6
|
+
|
|
7
|
+
from pydantic import BaseModel, Field
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class SearchPublicationsArgs(BaseModel):
|
|
11
|
+
author_name: str = Field(description="Full name of the author (e.g. 'Yvonne Lim')")
|
|
12
|
+
affiliation: str | None = Field(
|
|
13
|
+
default=None,
|
|
14
|
+
description="Institution or affiliation to narrow results (e.g. 'KK Women's and Children's Hospital')",
|
|
15
|
+
)
|
|
16
|
+
keywords: str | None = Field(
|
|
17
|
+
default=None,
|
|
18
|
+
description="Additional search terms: specialty, disease area, or MeSH terms",
|
|
19
|
+
)
|
|
20
|
+
year_from: int | None = Field(default=None, description="Earliest publication year (e.g. 2019)")
|
|
21
|
+
year_to: int | None = Field(default=None, description="Latest publication year (e.g. 2026)")
|
|
22
|
+
max_results: int = Field(
|
|
23
|
+
default=20,
|
|
24
|
+
ge=1,
|
|
25
|
+
le=100,
|
|
26
|
+
description="Maximum publications to return",
|
|
27
|
+
)
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
TOOL: dict[str, Any] = {
|
|
31
|
+
"name": "search_publications",
|
|
32
|
+
"description": (
|
|
33
|
+
"Search PubMed for publications by a specific author. Returns structured "
|
|
34
|
+
"records with PMID, title, authors, journal, year, abstract, MeSH terms, and "
|
|
35
|
+
"DOI. Use to identify an HCP's research output, publication themes, and "
|
|
36
|
+
"co-author network. Supports filtering by affiliation, keywords, and year range."
|
|
37
|
+
),
|
|
38
|
+
"args": SearchPublicationsArgs,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def handle(arguments: dict[str, Any]) -> list[dict[str, Any]]:
|
|
43
|
+
import json
|
|
44
|
+
import urllib.parse
|
|
45
|
+
import urllib.request
|
|
46
|
+
from datetime import datetime, timezone
|
|
47
|
+
|
|
48
|
+
from shared.env import get_env
|
|
49
|
+
|
|
50
|
+
api_key = get_env("NCBI_API_KEY", "")
|
|
51
|
+
base = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils"
|
|
52
|
+
|
|
53
|
+
author = arguments["author_name"]
|
|
54
|
+
parts = [f"{author}[Author]"]
|
|
55
|
+
if arguments.get("affiliation"):
|
|
56
|
+
parts.append(f"{arguments['affiliation']}[Affiliation]")
|
|
57
|
+
if arguments.get("keywords"):
|
|
58
|
+
parts.append(arguments["keywords"])
|
|
59
|
+
query = " AND ".join(parts)
|
|
60
|
+
|
|
61
|
+
if arguments.get("year_from") or arguments.get("year_to"):
|
|
62
|
+
mindate = str(arguments.get("year_from", 1900))
|
|
63
|
+
maxdate = str(arguments.get("year_to", 2099))
|
|
64
|
+
query += f" AND {mindate}:{maxdate}[dp]"
|
|
65
|
+
|
|
66
|
+
max_results = arguments.get("max_results", 20)
|
|
67
|
+
|
|
68
|
+
search_params = {
|
|
69
|
+
"db": "pubmed",
|
|
70
|
+
"term": query,
|
|
71
|
+
"retmax": str(max_results),
|
|
72
|
+
"retmode": "json",
|
|
73
|
+
"sort": "date",
|
|
74
|
+
}
|
|
75
|
+
if api_key:
|
|
76
|
+
search_params["api_key"] = api_key
|
|
77
|
+
|
|
78
|
+
search_url = f"{base}/esearch.fcgi?{urllib.parse.urlencode(search_params)}"
|
|
79
|
+
with urllib.request.urlopen(search_url, timeout=30) as resp:
|
|
80
|
+
search_data = json.loads(resp.read())
|
|
81
|
+
|
|
82
|
+
result = search_data.get("esearchresult", {})
|
|
83
|
+
total_count = int(result.get("count", 0))
|
|
84
|
+
id_list = result.get("idlist", [])
|
|
85
|
+
|
|
86
|
+
publications = []
|
|
87
|
+
if id_list:
|
|
88
|
+
fetch_params = {
|
|
89
|
+
"db": "pubmed",
|
|
90
|
+
"id": ",".join(id_list),
|
|
91
|
+
"retmode": "xml",
|
|
92
|
+
"rettype": "abstract",
|
|
93
|
+
}
|
|
94
|
+
if api_key:
|
|
95
|
+
fetch_params["api_key"] = api_key
|
|
96
|
+
|
|
97
|
+
fetch_url = f"{base}/efetch.fcgi?{urllib.parse.urlencode(fetch_params)}"
|
|
98
|
+
with urllib.request.urlopen(fetch_url, timeout=60) as resp:
|
|
99
|
+
xml_data = resp.read()
|
|
100
|
+
|
|
101
|
+
publications = _parse_pubmed_xml(xml_data)
|
|
102
|
+
|
|
103
|
+
output = {
|
|
104
|
+
"query": query,
|
|
105
|
+
"total_count": total_count,
|
|
106
|
+
"publications": publications,
|
|
107
|
+
"source": "pubmed",
|
|
108
|
+
"searched_at": datetime.now(timezone.utc).isoformat(),
|
|
109
|
+
}
|
|
110
|
+
return [{"type": "text", "text": json.dumps(output, indent=2)}]
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
def _parse_pubmed_xml(xml_bytes: bytes) -> list[dict]:
|
|
114
|
+
"""Parse PubMed efetch XML into structured publication records."""
|
|
115
|
+
import xml.etree.ElementTree as ET
|
|
116
|
+
|
|
117
|
+
root = ET.fromstring(xml_bytes)
|
|
118
|
+
pubs = []
|
|
119
|
+
|
|
120
|
+
for article in root.findall(".//PubmedArticle"):
|
|
121
|
+
medline = article.find("MedlineCitation")
|
|
122
|
+
if medline is None:
|
|
123
|
+
continue
|
|
124
|
+
|
|
125
|
+
pmid_el = medline.find("PMID")
|
|
126
|
+
pmid = pmid_el.text if pmid_el is not None else None
|
|
127
|
+
|
|
128
|
+
art = medline.find("Article")
|
|
129
|
+
if art is None:
|
|
130
|
+
continue
|
|
131
|
+
|
|
132
|
+
title_el = art.find("ArticleTitle")
|
|
133
|
+
title = "".join(title_el.itertext()) if title_el is not None else ""
|
|
134
|
+
|
|
135
|
+
journal_el = art.find("Journal/Title")
|
|
136
|
+
journal = journal_el.text if journal_el is not None else ""
|
|
137
|
+
|
|
138
|
+
year = None
|
|
139
|
+
for date_path in [
|
|
140
|
+
"Journal/JournalIssue/PubDate/Year",
|
|
141
|
+
"ArticleDate/Year",
|
|
142
|
+
]:
|
|
143
|
+
y_el = art.find(date_path)
|
|
144
|
+
if y_el is not None and y_el.text:
|
|
145
|
+
year = int(y_el.text)
|
|
146
|
+
break
|
|
147
|
+
if year is None:
|
|
148
|
+
medline_date = art.find("Journal/JournalIssue/PubDate/MedlineDate")
|
|
149
|
+
if medline_date is not None and medline_date.text:
|
|
150
|
+
import re
|
|
151
|
+
|
|
152
|
+
m = re.search(r"(\d{4})", medline_date.text)
|
|
153
|
+
if m:
|
|
154
|
+
year = int(m.group(1))
|
|
155
|
+
|
|
156
|
+
authors = []
|
|
157
|
+
for au in art.findall("AuthorList/Author"):
|
|
158
|
+
last = au.find("LastName")
|
|
159
|
+
fore = au.find("ForeName")
|
|
160
|
+
if last is not None:
|
|
161
|
+
name = last.text or ""
|
|
162
|
+
if fore is not None and fore.text:
|
|
163
|
+
name += f" {fore.text}"
|
|
164
|
+
authors.append(name)
|
|
165
|
+
|
|
166
|
+
abstract_parts = []
|
|
167
|
+
for abs_text in art.findall("Abstract/AbstractText"):
|
|
168
|
+
label = abs_text.get("Label", "")
|
|
169
|
+
text = "".join(abs_text.itertext())
|
|
170
|
+
if label:
|
|
171
|
+
abstract_parts.append(f"{label}: {text}")
|
|
172
|
+
else:
|
|
173
|
+
abstract_parts.append(text)
|
|
174
|
+
abstract = "\n".join(abstract_parts) if abstract_parts else None
|
|
175
|
+
|
|
176
|
+
doi = None
|
|
177
|
+
for eid in art.findall("ELocationID"):
|
|
178
|
+
if eid.get("EIdType") == "doi":
|
|
179
|
+
doi = eid.text
|
|
180
|
+
break
|
|
181
|
+
|
|
182
|
+
mesh_terms = []
|
|
183
|
+
for mh in medline.findall("MeshHeadingList/MeshHeading/DescriptorName"):
|
|
184
|
+
if mh.text:
|
|
185
|
+
mesh_terms.append(mh.text)
|
|
186
|
+
|
|
187
|
+
pub_types = []
|
|
188
|
+
for pt in art.findall("PublicationTypeList/PublicationType"):
|
|
189
|
+
if pt.text:
|
|
190
|
+
pub_types.append(pt.text)
|
|
191
|
+
|
|
192
|
+
pubs.append(
|
|
193
|
+
{
|
|
194
|
+
"pmid": pmid,
|
|
195
|
+
"title": title,
|
|
196
|
+
"authors": authors,
|
|
197
|
+
"journal": journal,
|
|
198
|
+
"year": year,
|
|
199
|
+
"doi": doi,
|
|
200
|
+
"abstract": abstract[:2000] if abstract and len(abstract) > 2000 else abstract,
|
|
201
|
+
"mesh_terms": mesh_terms,
|
|
202
|
+
"publication_types": pub_types,
|
|
203
|
+
"source_url": f"https://pubmed.ncbi.nlm.nih.gov/{pmid}/" if pmid else "",
|
|
204
|
+
}
|
|
205
|
+
)
|
|
206
|
+
|
|
207
|
+
return pubs
|