open-pharma-plugins 2.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. mcp_framework.py +495 -0
  2. open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
  3. open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
  4. open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
  5. open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
  6. open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
  7. open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
  8. open_pharma_plugins_campaign_studio/__init__.py +14 -0
  9. open_pharma_plugins_campaign_studio/__main__.py +11 -0
  10. open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
  11. open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
  12. open_pharma_plugins_campaign_studio/_renderer.py +119 -0
  13. open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
  14. open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
  15. open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
  16. open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
  17. open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
  18. open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
  19. open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
  20. open_pharma_plugins_campaign_studio/models/_common.py +12 -0
  21. open_pharma_plugins_campaign_studio/models/brief.py +72 -0
  22. open_pharma_plugins_campaign_studio/models/claims.py +15 -0
  23. open_pharma_plugins_campaign_studio/models/copy.py +47 -0
  24. open_pharma_plugins_campaign_studio/models/journey.py +21 -0
  25. open_pharma_plugins_campaign_studio/models/message.py +25 -0
  26. open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
  27. open_pharma_plugins_campaign_studio/models/validation.py +32 -0
  28. open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
  29. open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
  30. open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
  31. open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
  32. open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
  33. open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
  34. open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
  35. open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
  36. open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
  37. open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
  38. open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
  39. open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
  40. open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
  41. open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
  42. open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
  43. open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
  44. open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
  45. open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
  46. open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
  47. open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
  48. open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
  49. open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
  50. open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
  51. open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
  52. open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
  53. open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
  54. open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
  55. open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
  56. open_pharma_plugins_competitive_intelligence/models.py +525 -0
  57. open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
  58. open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
  59. open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
  60. open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
  61. open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
  62. open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
  63. open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
  64. open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
  65. open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
  66. open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
  67. open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
  68. open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
  69. open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
  70. open_pharma_plugins_field_training/__init__.py +13 -0
  71. open_pharma_plugins_field_training/__main__.py +11 -0
  72. open_pharma_plugins_field_training/_content_store.py +131 -0
  73. open_pharma_plugins_field_training/_grounding.py +75 -0
  74. open_pharma_plugins_field_training/_html_renderers.py +546 -0
  75. open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
  76. open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
  77. open_pharma_plugins_field_training/models.py +265 -0
  78. open_pharma_plugins_field_training/tools/__init__.py +0 -0
  79. open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
  80. open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
  81. open_pharma_plugins_field_training/tools/list_documents.py +51 -0
  82. open_pharma_plugins_field_training/tools/render_output.py +118 -0
  83. open_pharma_plugins_field_training/tools/search_content.py +57 -0
  84. open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
  85. open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
  86. open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
  87. open_pharma_plugins_hcp_intelligence/batch.py +891 -0
  88. open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
  89. open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
  90. open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
  91. open_pharma_plugins_hcp_intelligence/models.py +356 -0
  92. open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
  93. open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
  94. open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
  95. open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
  96. open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
  97. open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
  98. open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
  99. open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
  100. open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
  101. open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
  102. open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
  103. open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
  104. open_pharma_plugins_next_best_engagement/__init__.py +14 -0
  105. open_pharma_plugins_next_best_engagement/__main__.py +11 -0
  106. open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
  107. open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
  108. open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
  109. open_pharma_plugins_next_best_engagement/_universe.py +145 -0
  110. open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
  111. open_pharma_plugins_next_best_engagement/models.py +135 -0
  112. open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
  113. open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
  114. open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
  115. open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
  116. open_pharma_plugins_territory_alignment/__init__.py +14 -0
  117. open_pharma_plugins_territory_alignment/__main__.py +11 -0
  118. open_pharma_plugins_territory_alignment/data.py +300 -0
  119. open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
  120. open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
  121. open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
  122. open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
  123. open_pharma_plugins_territory_alignment/geo.py +175 -0
  124. open_pharma_plugins_territory_alignment/models.py +201 -0
  125. open_pharma_plugins_territory_alignment/scoring.py +125 -0
  126. open_pharma_plugins_territory_alignment/solver.py +504 -0
  127. open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
  128. open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
  129. open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
  130. open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
  131. open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
  132. open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
  133. open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
  134. shared/__init__.py +11 -0
  135. shared/env.py +217 -0
  136. shared/filesystem.py +110 -0
@@ -0,0 +1,199 @@
1
+ """search_hcp_web — web search tailored for HCP profiling."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from pydantic import BaseModel, Field
8
+
9
+
10
+ class SearchHcpWebArgs(BaseModel):
11
+ name: str = Field(description="Full name of the healthcare professional")
12
+ specialty: str | None = Field(default=None, description="Medical specialty (e.g. 'Cardiology')")
13
+ country: str | None = Field(default=None, description="Country of practice")
14
+ institution: str | None = Field(default=None, description="Known institution or affiliation")
15
+ query_focus: str | None = Field(
16
+ default=None,
17
+ description=(
18
+ "Optional focus to append to the search query, e.g. "
19
+ "'biography', 'society membership', 'education qualifications', "
20
+ "'advisory board committee'"
21
+ ),
22
+ )
23
+ max_results: int = Field(
24
+ default=10,
25
+ ge=1,
26
+ le=20,
27
+ description="Maximum web results to return",
28
+ )
29
+
30
+
31
+ TOOL: dict[str, Any] = {
32
+ "name": "search_hcp_web",
33
+ "description": (
34
+ "Search the web for information about a specific Healthcare Professional. "
35
+ "Constructs a targeted query from the HCP's name, specialty, country, and "
36
+ "institution to find institutional profiles, society memberships, conference "
37
+ "appearances, and biographical information. Returns URLs, titles, and snippets. "
38
+ "Use query_focus to steer toward specific profile sections (e.g. 'education', "
39
+ "'advisory board')."
40
+ ),
41
+ "args": SearchHcpWebArgs,
42
+ }
43
+
44
+
45
+ def handle(arguments: dict[str, Any]) -> list[dict[str, Any]]:
46
+ import json
47
+ from datetime import datetime, timezone
48
+
49
+ query_parts = [f'"{arguments["name"]}"']
50
+ if arguments.get("specialty"):
51
+ query_parts.append(arguments["specialty"])
52
+ if arguments.get("country"):
53
+ query_parts.append(arguments["country"])
54
+ if arguments.get("institution"):
55
+ query_parts.append(arguments["institution"])
56
+ if arguments.get("query_focus"):
57
+ query_parts.append(arguments["query_focus"])
58
+ else:
59
+ query_parts.append("doctor OR physician OR professor OR consultant")
60
+
61
+ query = " ".join(query_parts)
62
+ max_results = arguments.get("max_results", 10)
63
+
64
+ results = _web_search(query, max_results)
65
+
66
+ output = {
67
+ "query": query,
68
+ "results": results,
69
+ "search_backend": _backend_name(),
70
+ "searched_at": datetime.now(timezone.utc).isoformat(),
71
+ }
72
+ return [{"type": "text", "text": json.dumps(output, indent=2)}]
73
+
74
+
75
+ def _backend_name() -> str:
76
+ from shared.env import get_env
77
+
78
+ backend = get_env("OPEN_PHARMA_SEARCH_BACKEND", "auto").strip().lower()
79
+ if backend != "auto":
80
+ return backend
81
+ if get_env("SERPER_API_KEY", ""):
82
+ return "serper"
83
+ if get_env("TAVILY_API_KEY", ""):
84
+ return "tavily"
85
+ if get_env("EXA_API_KEY", ""):
86
+ return "exa"
87
+ return "none"
88
+
89
+
90
+ def _web_search(query: str, max_results: int) -> list[dict]:
91
+ backend = _backend_name()
92
+ if backend == "serper":
93
+ return _serper_search(query, max_results)
94
+ if backend == "tavily":
95
+ return _tavily_search(query, max_results)
96
+ if backend == "exa":
97
+ return _exa_search(query, max_results)
98
+ raise RuntimeError("No web search backend configured. Set SERPER_API_KEY, TAVILY_API_KEY, or EXA_API_KEY.")
99
+
100
+
101
+ def _serper_search(query: str, max_results: int) -> list[dict]:
102
+ import json
103
+ import urllib.request
104
+
105
+ from shared.env import get_env
106
+
107
+ body = json.dumps({"q": query, "num": max_results}).encode()
108
+ req = urllib.request.Request(
109
+ "https://google.serper.dev/search",
110
+ data=body,
111
+ headers={
112
+ "X-API-KEY": get_env("SERPER_API_KEY", ""),
113
+ "Content-Type": "application/json",
114
+ },
115
+ )
116
+ with urllib.request.urlopen(req, timeout=15) as resp:
117
+ data = json.loads(resp.read())
118
+
119
+ return [
120
+ {
121
+ "url": r.get("link", ""),
122
+ "title": r.get("title", ""),
123
+ "snippet": r.get("snippet", ""),
124
+ "published_date": r.get("date"),
125
+ "domain": r.get("link", "").split("/")[2] if "/" in r.get("link", "") else None,
126
+ }
127
+ for r in data.get("organic", [])
128
+ ]
129
+
130
+
131
+ def _tavily_search(query: str, max_results: int) -> list[dict]:
132
+ import json
133
+ import urllib.request
134
+
135
+ from shared.env import get_env
136
+
137
+ body = json.dumps(
138
+ {
139
+ "api_key": get_env("TAVILY_API_KEY", ""),
140
+ "query": query,
141
+ "max_results": max_results,
142
+ "search_depth": "advanced",
143
+ }
144
+ ).encode()
145
+ req = urllib.request.Request(
146
+ "https://api.tavily.com/search",
147
+ data=body,
148
+ headers={"Content-Type": "application/json"},
149
+ )
150
+ with urllib.request.urlopen(req, timeout=15) as resp:
151
+ data = json.loads(resp.read())
152
+
153
+ return [
154
+ {
155
+ "url": r.get("url", ""),
156
+ "title": r.get("title", ""),
157
+ "snippet": r.get("content", ""),
158
+ "published_date": r.get("published_date"),
159
+ "domain": r.get("url", "").split("/")[2] if "/" in r.get("url", "") else None,
160
+ }
161
+ for r in data.get("results", [])
162
+ ]
163
+
164
+
165
+ def _exa_search(query: str, max_results: int) -> list[dict]:
166
+ import json
167
+ import urllib.request
168
+
169
+ from shared.env import get_env
170
+
171
+ body = json.dumps(
172
+ {
173
+ "query": query,
174
+ "numResults": max_results,
175
+ "useAutoprompt": True,
176
+ "contents": {"text": {"maxCharacters": 500}},
177
+ }
178
+ ).encode()
179
+ req = urllib.request.Request(
180
+ "https://api.exa.ai/search",
181
+ data=body,
182
+ headers={
183
+ "x-api-key": get_env("EXA_API_KEY", ""),
184
+ "Content-Type": "application/json",
185
+ },
186
+ )
187
+ with urllib.request.urlopen(req, timeout=15) as resp:
188
+ data = json.loads(resp.read())
189
+
190
+ return [
191
+ {
192
+ "url": r.get("url", ""),
193
+ "title": r.get("title", ""),
194
+ "snippet": r.get("text", ""),
195
+ "published_date": r.get("publishedDate"),
196
+ "domain": r.get("url", "").split("/")[2] if "/" in r.get("url", "") else None,
197
+ }
198
+ for r in data.get("results", [])
199
+ ]
@@ -0,0 +1,214 @@
1
+ """search_orcid — query the ORCID public API for researcher profiles."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from pydantic import BaseModel, Field
8
+
9
+
10
+ class SearchOrcidArgs(BaseModel):
11
+ name: str = Field(
12
+ description="Full name of the researcher (e.g. 'Yvonne Lim')",
13
+ )
14
+ affiliation: str | None = Field(
15
+ default=None,
16
+ description="Known institution to help disambiguate (e.g. 'KK Women's and Children's Hospital')",
17
+ )
18
+ orcid_id: str | None = Field(
19
+ default=None,
20
+ description="If the ORCID ID is already known (e.g. '0000-0002-1234-5678'), fetch directly instead of searching",
21
+ )
22
+ max_results: int = Field(
23
+ default=5,
24
+ ge=1,
25
+ le=10,
26
+ description="Maximum candidate profiles to return when searching by name",
27
+ )
28
+
29
+
30
+ TOOL: dict[str, Any] = {
31
+ "name": "search_orcid",
32
+ "description": (
33
+ "Search the ORCID registry for a researcher's profile. ORCID provides "
34
+ "author-curated, globally unique researcher identifiers with verified "
35
+ "affiliations, education history, publication counts, and funding. Call "
36
+ "early in the HCP profiling workflow — it disambiguates common names and "
37
+ "fills education/affiliation gaps cheaply. Pass an orcid_id to fetch a "
38
+ "known profile directly, or search by name and optional affiliation."
39
+ ),
40
+ "args": SearchOrcidArgs,
41
+ }
42
+
43
+
44
+ _BASE = "https://pub.orcid.org/v3.0"
45
+ _HEADERS = {"Accept": "application/json"}
46
+
47
+
48
+ def handle(arguments: dict[str, Any]) -> list[dict[str, Any]]:
49
+ import json
50
+ from datetime import datetime, timezone
51
+
52
+ orcid_id = arguments.get("orcid_id")
53
+ if orcid_id:
54
+ profile = _fetch_profile(orcid_id)
55
+ output = {
56
+ "query": orcid_id,
57
+ "total_found": 1 if profile else 0,
58
+ "profiles": [profile] if profile else [],
59
+ "source": "orcid",
60
+ "searched_at": datetime.now(timezone.utc).isoformat(),
61
+ }
62
+ return [{"type": "text", "text": json.dumps(output, indent=2)}]
63
+
64
+ name = arguments["name"]
65
+ affiliation = arguments.get("affiliation")
66
+ max_results = arguments.get("max_results", 5)
67
+
68
+ query = _build_query(name, affiliation)
69
+ orcid_ids = _search_ids(query, max_results)
70
+
71
+ profiles = []
72
+ for oid in orcid_ids:
73
+ p = _fetch_profile(oid)
74
+ if p:
75
+ profiles.append(p)
76
+
77
+ output = {
78
+ "query": query,
79
+ "total_found": len(orcid_ids),
80
+ "profiles": profiles,
81
+ "source": "orcid",
82
+ "searched_at": datetime.now(timezone.utc).isoformat(),
83
+ }
84
+ return [{"type": "text", "text": json.dumps(output, indent=2)}]
85
+
86
+
87
+ def _build_query(name: str, affiliation: str | None) -> str:
88
+ parts = name.strip().split()
89
+ if len(parts) >= 2:
90
+ family = parts[-1]
91
+ given = " ".join(parts[:-1])
92
+ q = f"family-name:{family} AND given-names:{given}"
93
+ else:
94
+ q = f"family-name:{name}"
95
+
96
+ if affiliation:
97
+ q += f" AND affiliation-org-name:{affiliation}"
98
+ return q
99
+
100
+
101
+ def _search_ids(query: str, max_results: int) -> list[str]:
102
+ import json
103
+ import urllib.parse
104
+ import urllib.request
105
+
106
+ url = f"{_BASE}/search/?q={urllib.parse.quote(query)}&rows={max_results}"
107
+ req = urllib.request.Request(url, headers=_HEADERS)
108
+ try:
109
+ with urllib.request.urlopen(req, timeout=15) as resp:
110
+ data = json.loads(resp.read())
111
+ except Exception:
112
+ return []
113
+
114
+ results = data.get("result", []) or []
115
+ ids = []
116
+ for r in results:
117
+ oid = r.get("orcid-identifier", {}).get("path")
118
+ if oid:
119
+ ids.append(oid)
120
+ return ids
121
+
122
+
123
+ def _fetch_profile(orcid_id: str) -> dict | None:
124
+ import json
125
+ import urllib.request
126
+
127
+ url = f"{_BASE}/{orcid_id}"
128
+ req = urllib.request.Request(url, headers=_HEADERS)
129
+ try:
130
+ with urllib.request.urlopen(req, timeout=15) as resp:
131
+ data = json.loads(resp.read())
132
+ except Exception:
133
+ return None
134
+
135
+ return _extract_profile(orcid_id, data)
136
+
137
+
138
+ def _extract_profile(orcid_id: str, data: dict) -> dict:
139
+ person = data.get("person", {}) or {}
140
+ activities = data.get("activities-summary", {}) or {}
141
+
142
+ name_obj = person.get("name", {}) or {}
143
+ given = (name_obj.get("given-names", {}) or {}).get("value", "")
144
+ family = (name_obj.get("family-name", {}) or {}).get("value", "")
145
+
146
+ bio_obj = person.get("biography", {}) or {}
147
+ biography = bio_obj.get("content", "") or ""
148
+ if len(biography) > 500:
149
+ biography = biography[:497] + "..."
150
+
151
+ education = _extract_affiliations(activities.get("educations", {}) or {}, "education-summary")
152
+ employment = _extract_affiliations(activities.get("employments", {}) or {}, "employment-summary")
153
+
154
+ work_groups = (activities.get("works", {}) or {}).get("group", []) or []
155
+ funding_groups = (activities.get("fundings", {}) or {}).get("group", []) or []
156
+ review_groups = (activities.get("peer-reviews", {}) or {}).get("group", []) or []
157
+
158
+ return {
159
+ "orcid_id": orcid_id,
160
+ "given_names": given,
161
+ "family_name": family,
162
+ "biography": biography or None,
163
+ "education": education,
164
+ "employment": employment,
165
+ "publication_count": len(work_groups),
166
+ "funding_count": len(funding_groups),
167
+ "peer_review_count": len(review_groups),
168
+ "profile_url": f"https://orcid.org/{orcid_id}",
169
+ }
170
+
171
+
172
+ def _extract_affiliations(section: dict, summary_key: str) -> list[dict]:
173
+ groups = section.get("affiliation-group", []) or []
174
+ results = []
175
+ for group in groups:
176
+ summaries = group.get("summaries", []) or []
177
+ for entry in summaries:
178
+ summary = entry.get(summary_key, {}) or {}
179
+ org = (summary.get("organization", {}) or {}).get("name", "")
180
+ role = (summary.get("role-title") or "") if summary_key == "employment-summary" else ""
181
+ degree = (summary.get("role-title") or "") if summary_key == "education-summary" else ""
182
+ dept = summary.get("department-name") or ""
183
+
184
+ start = summary.get("start-date") or {}
185
+ end = summary.get("end-date") or {}
186
+ start_year = _year_from(start)
187
+ end_year = _year_from(end)
188
+
189
+ entry_dict: dict[str, Any] = {"institution": org}
190
+ if summary_key == "education-summary":
191
+ entry_dict["degree"] = degree
192
+ else:
193
+ entry_dict["role"] = role
194
+ entry_dict["department"] = dept or None
195
+ entry_dict["start_year"] = start_year
196
+ entry_dict["end_year"] = end_year
197
+ results.append(entry_dict)
198
+ return results
199
+
200
+
201
+ def _year_from(date_obj: dict | None) -> int | None:
202
+ if not date_obj:
203
+ return None
204
+ year = date_obj.get("year", {})
205
+ if isinstance(year, dict):
206
+ val = year.get("value")
207
+ else:
208
+ val = year
209
+ if val:
210
+ try:
211
+ return int(val)
212
+ except (ValueError, TypeError):
213
+ pass
214
+ return None
@@ -0,0 +1,207 @@
1
+ """search_publications — query PubMed via NCBI E-utilities."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from pydantic import BaseModel, Field
8
+
9
+
10
+ class SearchPublicationsArgs(BaseModel):
11
+ author_name: str = Field(description="Full name of the author (e.g. 'Yvonne Lim')")
12
+ affiliation: str | None = Field(
13
+ default=None,
14
+ description="Institution or affiliation to narrow results (e.g. 'KK Women's and Children's Hospital')",
15
+ )
16
+ keywords: str | None = Field(
17
+ default=None,
18
+ description="Additional search terms: specialty, disease area, or MeSH terms",
19
+ )
20
+ year_from: int | None = Field(default=None, description="Earliest publication year (e.g. 2019)")
21
+ year_to: int | None = Field(default=None, description="Latest publication year (e.g. 2026)")
22
+ max_results: int = Field(
23
+ default=20,
24
+ ge=1,
25
+ le=100,
26
+ description="Maximum publications to return",
27
+ )
28
+
29
+
30
+ TOOL: dict[str, Any] = {
31
+ "name": "search_publications",
32
+ "description": (
33
+ "Search PubMed for publications by a specific author. Returns structured "
34
+ "records with PMID, title, authors, journal, year, abstract, MeSH terms, and "
35
+ "DOI. Use to identify an HCP's research output, publication themes, and "
36
+ "co-author network. Supports filtering by affiliation, keywords, and year range."
37
+ ),
38
+ "args": SearchPublicationsArgs,
39
+ }
40
+
41
+
42
+ def handle(arguments: dict[str, Any]) -> list[dict[str, Any]]:
43
+ import json
44
+ import urllib.parse
45
+ import urllib.request
46
+ from datetime import datetime, timezone
47
+
48
+ from shared.env import get_env
49
+
50
+ api_key = get_env("NCBI_API_KEY", "")
51
+ base = "https://eutils.ncbi.nlm.nih.gov/entrez/eutils"
52
+
53
+ author = arguments["author_name"]
54
+ parts = [f"{author}[Author]"]
55
+ if arguments.get("affiliation"):
56
+ parts.append(f"{arguments['affiliation']}[Affiliation]")
57
+ if arguments.get("keywords"):
58
+ parts.append(arguments["keywords"])
59
+ query = " AND ".join(parts)
60
+
61
+ if arguments.get("year_from") or arguments.get("year_to"):
62
+ mindate = str(arguments.get("year_from", 1900))
63
+ maxdate = str(arguments.get("year_to", 2099))
64
+ query += f" AND {mindate}:{maxdate}[dp]"
65
+
66
+ max_results = arguments.get("max_results", 20)
67
+
68
+ search_params = {
69
+ "db": "pubmed",
70
+ "term": query,
71
+ "retmax": str(max_results),
72
+ "retmode": "json",
73
+ "sort": "date",
74
+ }
75
+ if api_key:
76
+ search_params["api_key"] = api_key
77
+
78
+ search_url = f"{base}/esearch.fcgi?{urllib.parse.urlencode(search_params)}"
79
+ with urllib.request.urlopen(search_url, timeout=30) as resp:
80
+ search_data = json.loads(resp.read())
81
+
82
+ result = search_data.get("esearchresult", {})
83
+ total_count = int(result.get("count", 0))
84
+ id_list = result.get("idlist", [])
85
+
86
+ publications = []
87
+ if id_list:
88
+ fetch_params = {
89
+ "db": "pubmed",
90
+ "id": ",".join(id_list),
91
+ "retmode": "xml",
92
+ "rettype": "abstract",
93
+ }
94
+ if api_key:
95
+ fetch_params["api_key"] = api_key
96
+
97
+ fetch_url = f"{base}/efetch.fcgi?{urllib.parse.urlencode(fetch_params)}"
98
+ with urllib.request.urlopen(fetch_url, timeout=60) as resp:
99
+ xml_data = resp.read()
100
+
101
+ publications = _parse_pubmed_xml(xml_data)
102
+
103
+ output = {
104
+ "query": query,
105
+ "total_count": total_count,
106
+ "publications": publications,
107
+ "source": "pubmed",
108
+ "searched_at": datetime.now(timezone.utc).isoformat(),
109
+ }
110
+ return [{"type": "text", "text": json.dumps(output, indent=2)}]
111
+
112
+
113
+ def _parse_pubmed_xml(xml_bytes: bytes) -> list[dict]:
114
+ """Parse PubMed efetch XML into structured publication records."""
115
+ import xml.etree.ElementTree as ET
116
+
117
+ root = ET.fromstring(xml_bytes)
118
+ pubs = []
119
+
120
+ for article in root.findall(".//PubmedArticle"):
121
+ medline = article.find("MedlineCitation")
122
+ if medline is None:
123
+ continue
124
+
125
+ pmid_el = medline.find("PMID")
126
+ pmid = pmid_el.text if pmid_el is not None else None
127
+
128
+ art = medline.find("Article")
129
+ if art is None:
130
+ continue
131
+
132
+ title_el = art.find("ArticleTitle")
133
+ title = "".join(title_el.itertext()) if title_el is not None else ""
134
+
135
+ journal_el = art.find("Journal/Title")
136
+ journal = journal_el.text if journal_el is not None else ""
137
+
138
+ year = None
139
+ for date_path in [
140
+ "Journal/JournalIssue/PubDate/Year",
141
+ "ArticleDate/Year",
142
+ ]:
143
+ y_el = art.find(date_path)
144
+ if y_el is not None and y_el.text:
145
+ year = int(y_el.text)
146
+ break
147
+ if year is None:
148
+ medline_date = art.find("Journal/JournalIssue/PubDate/MedlineDate")
149
+ if medline_date is not None and medline_date.text:
150
+ import re
151
+
152
+ m = re.search(r"(\d{4})", medline_date.text)
153
+ if m:
154
+ year = int(m.group(1))
155
+
156
+ authors = []
157
+ for au in art.findall("AuthorList/Author"):
158
+ last = au.find("LastName")
159
+ fore = au.find("ForeName")
160
+ if last is not None:
161
+ name = last.text or ""
162
+ if fore is not None and fore.text:
163
+ name += f" {fore.text}"
164
+ authors.append(name)
165
+
166
+ abstract_parts = []
167
+ for abs_text in art.findall("Abstract/AbstractText"):
168
+ label = abs_text.get("Label", "")
169
+ text = "".join(abs_text.itertext())
170
+ if label:
171
+ abstract_parts.append(f"{label}: {text}")
172
+ else:
173
+ abstract_parts.append(text)
174
+ abstract = "\n".join(abstract_parts) if abstract_parts else None
175
+
176
+ doi = None
177
+ for eid in art.findall("ELocationID"):
178
+ if eid.get("EIdType") == "doi":
179
+ doi = eid.text
180
+ break
181
+
182
+ mesh_terms = []
183
+ for mh in medline.findall("MeshHeadingList/MeshHeading/DescriptorName"):
184
+ if mh.text:
185
+ mesh_terms.append(mh.text)
186
+
187
+ pub_types = []
188
+ for pt in art.findall("PublicationTypeList/PublicationType"):
189
+ if pt.text:
190
+ pub_types.append(pt.text)
191
+
192
+ pubs.append(
193
+ {
194
+ "pmid": pmid,
195
+ "title": title,
196
+ "authors": authors,
197
+ "journal": journal,
198
+ "year": year,
199
+ "doi": doi,
200
+ "abstract": abstract[:2000] if abstract and len(abstract) > 2000 else abstract,
201
+ "mesh_terms": mesh_terms,
202
+ "publication_types": pub_types,
203
+ "source_url": f"https://pubmed.ncbi.nlm.nih.gov/{pmid}/" if pmid else "",
204
+ }
205
+ )
206
+
207
+ return pubs