open-pharma-plugins 2.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. mcp_framework.py +495 -0
  2. open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
  3. open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
  4. open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
  5. open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
  6. open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
  7. open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
  8. open_pharma_plugins_campaign_studio/__init__.py +14 -0
  9. open_pharma_plugins_campaign_studio/__main__.py +11 -0
  10. open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
  11. open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
  12. open_pharma_plugins_campaign_studio/_renderer.py +119 -0
  13. open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
  14. open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
  15. open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
  16. open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
  17. open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
  18. open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
  19. open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
  20. open_pharma_plugins_campaign_studio/models/_common.py +12 -0
  21. open_pharma_plugins_campaign_studio/models/brief.py +72 -0
  22. open_pharma_plugins_campaign_studio/models/claims.py +15 -0
  23. open_pharma_plugins_campaign_studio/models/copy.py +47 -0
  24. open_pharma_plugins_campaign_studio/models/journey.py +21 -0
  25. open_pharma_plugins_campaign_studio/models/message.py +25 -0
  26. open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
  27. open_pharma_plugins_campaign_studio/models/validation.py +32 -0
  28. open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
  29. open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
  30. open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
  31. open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
  32. open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
  33. open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
  34. open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
  35. open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
  36. open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
  37. open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
  38. open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
  39. open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
  40. open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
  41. open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
  42. open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
  43. open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
  44. open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
  45. open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
  46. open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
  47. open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
  48. open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
  49. open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
  50. open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
  51. open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
  52. open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
  53. open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
  54. open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
  55. open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
  56. open_pharma_plugins_competitive_intelligence/models.py +525 -0
  57. open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
  58. open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
  59. open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
  60. open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
  61. open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
  62. open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
  63. open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
  64. open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
  65. open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
  66. open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
  67. open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
  68. open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
  69. open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
  70. open_pharma_plugins_field_training/__init__.py +13 -0
  71. open_pharma_plugins_field_training/__main__.py +11 -0
  72. open_pharma_plugins_field_training/_content_store.py +131 -0
  73. open_pharma_plugins_field_training/_grounding.py +75 -0
  74. open_pharma_plugins_field_training/_html_renderers.py +546 -0
  75. open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
  76. open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
  77. open_pharma_plugins_field_training/models.py +265 -0
  78. open_pharma_plugins_field_training/tools/__init__.py +0 -0
  79. open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
  80. open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
  81. open_pharma_plugins_field_training/tools/list_documents.py +51 -0
  82. open_pharma_plugins_field_training/tools/render_output.py +118 -0
  83. open_pharma_plugins_field_training/tools/search_content.py +57 -0
  84. open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
  85. open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
  86. open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
  87. open_pharma_plugins_hcp_intelligence/batch.py +891 -0
  88. open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
  89. open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
  90. open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
  91. open_pharma_plugins_hcp_intelligence/models.py +356 -0
  92. open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
  93. open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
  94. open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
  95. open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
  96. open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
  97. open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
  98. open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
  99. open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
  100. open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
  101. open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
  102. open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
  103. open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
  104. open_pharma_plugins_next_best_engagement/__init__.py +14 -0
  105. open_pharma_plugins_next_best_engagement/__main__.py +11 -0
  106. open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
  107. open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
  108. open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
  109. open_pharma_plugins_next_best_engagement/_universe.py +145 -0
  110. open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
  111. open_pharma_plugins_next_best_engagement/models.py +135 -0
  112. open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
  113. open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
  114. open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
  115. open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
  116. open_pharma_plugins_territory_alignment/__init__.py +14 -0
  117. open_pharma_plugins_territory_alignment/__main__.py +11 -0
  118. open_pharma_plugins_territory_alignment/data.py +300 -0
  119. open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
  120. open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
  121. open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
  122. open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
  123. open_pharma_plugins_territory_alignment/geo.py +175 -0
  124. open_pharma_plugins_territory_alignment/models.py +201 -0
  125. open_pharma_plugins_territory_alignment/scoring.py +125 -0
  126. open_pharma_plugins_territory_alignment/solver.py +504 -0
  127. open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
  128. open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
  129. open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
  130. open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
  131. open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
  132. open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
  133. open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
  134. shared/__init__.py +11 -0
  135. shared/env.py +217 -0
  136. shared/filesystem.py +110 -0
@@ -0,0 +1,891 @@
1
+ """Package-owned batch runtime for hcp-intelligence accounts."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import csv
6
+ import hashlib
7
+ import io
8
+ import json
9
+ import math
10
+ import os
11
+ import random
12
+ import stat
13
+ import sys
14
+ import time
15
+ import urllib.error
16
+ from concurrent.futures import ThreadPoolExecutor, as_completed
17
+ from contextvars import ContextVar
18
+ from dataclasses import dataclass
19
+ from datetime import datetime, timezone
20
+ from pathlib import Path
21
+ from threading import Lock
22
+ from typing import Any, Callable, Literal
23
+
24
+ BUNDLED_INPUT = Path(__file__).parent / "fixtures" / "sample_accounts.csv"
25
+ INPUT_COLUMNS = ("id", "name", "specialty", "country", "account_type", "institution")
26
+ REQUIRED_VALUES = ("id", "name", "country", "account_type")
27
+ DEFAULT_OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1"
28
+ DEFAULT_SYNTHESIS_MODEL = "deepseek/deepseek-v4-flash-0731"
29
+ DEFAULT_REASONING_EFFORT = "high"
30
+ DEFAULT_SYNTHESIS_TIMEOUT_SECONDS = 120.0
31
+
32
+
33
+ class BatchUsageError(ValueError):
34
+ """A user-correctable batch input or path error."""
35
+
36
+
37
+ class SynthesisProviderError(RuntimeError):
38
+ """A sanitized synthesis-provider failure safe to persist."""
39
+
40
+
41
+ @dataclass(frozen=True)
42
+ class BatchOptions:
43
+ input_file: str | Path | None
44
+ output_dir: str | Path
45
+ country: str | None
46
+ account_type: str | None
47
+ ids: tuple[str, ...]
48
+ concurrency: int
49
+ resume: bool
50
+ write_back: bool
51
+ synthesize: bool
52
+ base_url: str
53
+ api_key_env: str
54
+ model: str
55
+ reasoning_effort: Literal["high", "xhigh"]
56
+ synthesis_timeout_seconds: float
57
+
58
+
59
+ @dataclass(frozen=True)
60
+ class BatchPlan:
61
+ input_path: Path
62
+ input_sha256: str
63
+ output_dir: Path
64
+ accounts: tuple[dict[str, str], ...]
65
+ options: BatchOptions
66
+
67
+
68
+ @dataclass(frozen=True)
69
+ class BatchOutcome:
70
+ output_dir: Path
71
+ results: tuple[dict[str, Any], ...]
72
+ manifest: dict[str, Any]
73
+ exit_code: int
74
+
75
+
76
+ # ---------------------------------------------------------------------------
77
+ # Tool imports (lazy so importing the batch runtime has no provider side effects)
78
+ # ---------------------------------------------------------------------------
79
+
80
+ _TOOL_MODULES: dict[str, object] = {}
81
+ _WRITEBACK_LOCK = Lock()
82
+ _ACTIVE_API_KEY_ENV: ContextVar[str] = ContextVar("hcp_batch_api_key_env", default="OPENROUTER_API_KEY")
83
+ _REDACTED_PROVIDER_KEYS = ("EXA_API_KEY", "SERPER_API_KEY", "TAVILY_API_KEY", "NCBI_API_KEY")
84
+
85
+
86
+ def sanitize_error(error: object, api_key_env: str | None = None) -> str:
87
+ """Redact configured provider credentials from an HCP batch error."""
88
+ from shared.env import get_env
89
+
90
+ selected_key = api_key_env or _ACTIVE_API_KEY_ENV.get()
91
+ key_names = dict.fromkeys((selected_key, *_REDACTED_PROVIDER_KEYS))
92
+ credentials = sorted(
93
+ (value for key in key_names if (value := get_env(key, ""))),
94
+ key=len,
95
+ reverse=True,
96
+ )
97
+ safe = str(error)
98
+ for credential in credentials:
99
+ safe = safe.replace(credential, "[REDACTED]")
100
+ return safe
101
+
102
+
103
+ def resolve_user_path(value: str | Path, *, label: str) -> Path:
104
+ """Resolve a user path only after rejecting invisible control characters."""
105
+ raw = str(value)
106
+ if any(ord(char) < 32 or ord(char) == 127 for char in raw):
107
+ raise BatchUsageError(f"{label} contains a control character")
108
+ return Path(raw).expanduser().resolve()
109
+
110
+
111
+ def _load_accounts_from_bytes(payload: bytes) -> list[dict[str, str]]:
112
+ """Parse one captured CSV byte snapshot into the normalized public contract."""
113
+ try:
114
+ text = payload.decode("utf-8-sig")
115
+ except UnicodeDecodeError as exc:
116
+ raise ValueError("input file must be UTF-8 CSV") from exc
117
+
118
+ with io.StringIO(text, newline="") as handle:
119
+ reader = csv.DictReader(handle)
120
+ headers = set(reader.fieldnames or [])
121
+ missing = sorted(set(INPUT_COLUMNS) - headers)
122
+ if missing:
123
+ raise ValueError(f"missing columns: {', '.join(missing)}")
124
+
125
+ accounts: list[dict[str, str]] = []
126
+ seen_ids: set[str] = set()
127
+ for row_number, raw in enumerate(reader, start=2):
128
+ raw_account_id = raw.get("id") or ""
129
+ if any(ord(char) < 32 or ord(char) == 127 for char in raw_account_id):
130
+ raise ValueError(f"row {row_number} account id contains a control character")
131
+ account = {key: (raw.get(key) or "").strip() for key in INPUT_COLUMNS}
132
+ blank = [key for key in REQUIRED_VALUES if not account[key]]
133
+ if blank:
134
+ raise ValueError(f"row {row_number}: blank required values: {', '.join(blank)}")
135
+
136
+ account_type = account["account_type"].upper()
137
+ if account_type not in {"HCP", "HCO"}:
138
+ raise ValueError(f"row {row_number}: account_type must be HCP or HCO")
139
+ account["account_type"] = account_type
140
+
141
+ account_id = account["id"]
142
+ from shared.filesystem import validate_component
143
+
144
+ validate_component(account_id, label=f"row {row_number} account id")
145
+ if account_id in seen_ids:
146
+ raise ValueError(f"row {row_number}: duplicate account id: {account_id}")
147
+ seen_ids.add(account_id)
148
+ accounts.append(account)
149
+
150
+ if not accounts:
151
+ raise ValueError("input file contains no account rows")
152
+ return accounts
153
+
154
+
155
+ def load_accounts_from_csv(path: str | Path) -> list[dict[str, str]]:
156
+ """Load and validate the public batch-input contract before any network calls."""
157
+ source = resolve_user_path(path, label="input path")
158
+ if not source.is_file():
159
+ raise ValueError(f"input file does not exist: {source}")
160
+ try:
161
+ payload = source.read_bytes()
162
+ except OSError as exc:
163
+ raise ValueError(f"could not read input file: {source}") from exc
164
+ return _load_accounts_from_bytes(payload)
165
+
166
+
167
+ def validate_synthesized_profile(profile_json: str, account_type: str) -> dict:
168
+ """Reject malformed or out-of-contract LLM output before persistence."""
169
+ from open_pharma_plugins_hcp_intelligence.models import HcoProfile, HcpProfile
170
+
171
+ data = json.loads(profile_json)
172
+ model = HcpProfile if account_type.upper() == "HCP" else HcoProfile
173
+ return model.model_validate(data).model_dump(mode="json", exclude_unset=True)
174
+
175
+
176
+ def write_batch_json(path: str | Path, data: object) -> Path:
177
+ """Persist a private, atomic JSON artifact."""
178
+ from shared.filesystem import atomic_write_json
179
+
180
+ return atomic_write_json(path, data)
181
+
182
+
183
+ def build_batch_manifest(
184
+ *,
185
+ input_path: str | Path,
186
+ input_sha256: str,
187
+ selected_count: int,
188
+ synthesis: bool,
189
+ model: str,
190
+ base_url: str,
191
+ reasoning_effort: str,
192
+ synthesis_timeout_seconds: float,
193
+ concurrency: int,
194
+ results: list[dict],
195
+ summary_csv: dict[str, object],
196
+ started_at: str,
197
+ completed_at: str,
198
+ ) -> dict:
199
+ """Build an auditable batch summary without duplicating account names."""
200
+ source = Path(input_path).expanduser().resolve()
201
+ statuses = {status: 0 for status in ("completed", "partial", "failed", "skipped")}
202
+ accounts: list[dict] = []
203
+ allowed = ("account_id", "status", "tools_failed", "output_file", "profile_validated", "error")
204
+ for result in results:
205
+ status = result["status"]
206
+ statuses[status] += 1
207
+ accounts.append({key: result[key] for key in allowed if key in result})
208
+
209
+ return {
210
+ "schema_version": 2,
211
+ "input": {
212
+ "path": str(source),
213
+ "sha256": input_sha256,
214
+ },
215
+ "started_at": started_at,
216
+ "completed_at": completed_at,
217
+ "processing": {
218
+ "synthesis": synthesis,
219
+ "model": model if synthesis else None,
220
+ "base_url": base_url if synthesis else None,
221
+ "reasoning_effort": reasoning_effort if synthesis else None,
222
+ "timeout_seconds": synthesis_timeout_seconds if synthesis else None,
223
+ "concurrency": concurrency,
224
+ },
225
+ "summary": {"selected": selected_count, **statuses},
226
+ "accounts": accounts,
227
+ "outputs": {"summary_csv": summary_csv},
228
+ }
229
+
230
+
231
+ def _load_tools():
232
+ if _TOOL_MODULES:
233
+ return
234
+ from open_pharma_plugins_hcp_intelligence.tools import (
235
+ search_clinical_trials,
236
+ search_congresses,
237
+ search_grants,
238
+ search_guidelines,
239
+ search_hco_web,
240
+ search_hcp_web,
241
+ search_orcid,
242
+ search_publications,
243
+ )
244
+
245
+ _TOOL_MODULES.update(
246
+ {
247
+ "search_orcid": search_orcid,
248
+ "search_publications": search_publications,
249
+ "search_guidelines": search_guidelines,
250
+ "search_clinical_trials": search_clinical_trials,
251
+ "search_congresses": search_congresses,
252
+ "search_grants": search_grants,
253
+ "search_hcp_web": search_hcp_web,
254
+ "search_hco_web": search_hco_web,
255
+ }
256
+ )
257
+
258
+
259
+ # ---------------------------------------------------------------------------
260
+ # Retry wrapper
261
+ # ---------------------------------------------------------------------------
262
+
263
+ MAX_RETRIES = 3
264
+
265
+
266
+ def _call_tool(tool_name: str, arguments: dict) -> dict | None:
267
+ module = _TOOL_MODULES[tool_name]
268
+ for attempt in range(MAX_RETRIES + 1):
269
+ try:
270
+ result = module.handle(arguments)
271
+ return json.loads(result[0]["text"])
272
+ except Exception as e:
273
+ retryable = not isinstance(e, urllib.error.HTTPError) or e.code in {408, 425, 429} or e.code >= 500
274
+ if not retryable or attempt == MAX_RETRIES:
275
+ print(f" FAIL {tool_name}: {sanitize_error(e)}", file=sys.stderr)
276
+ return None
277
+ delay = min(2**attempt + random.uniform(0, 1), 30.0)
278
+ print(f" RETRY {tool_name} (attempt {attempt + 1}): {sanitize_error(e)}", file=sys.stderr)
279
+ time.sleep(delay)
280
+ return None
281
+
282
+
283
+ # ---------------------------------------------------------------------------
284
+ # Per-account enrichment
285
+ # ---------------------------------------------------------------------------
286
+
287
+ HCP_TOOLS = [
288
+ "search_orcid",
289
+ "search_publications",
290
+ "search_guidelines",
291
+ "search_clinical_trials",
292
+ "search_congresses",
293
+ "search_grants",
294
+ "search_hcp_web",
295
+ ]
296
+
297
+ HCO_TOOLS = [
298
+ "search_hco_web",
299
+ "search_clinical_trials",
300
+ "search_grants",
301
+ ]
302
+
303
+
304
+ def _enrich_account(account: dict, output_dir: Path) -> dict:
305
+ account_id = account["id"]
306
+ name = account["name"]
307
+ specialty = account.get("specialty", "")
308
+ country = account.get("country", "")
309
+ institution = account.get("institution", "")
310
+ account_type = account.get("account_type", "HCP").upper()
311
+
312
+ print(f" [{account_id}] {name} ({account_type}, {country})")
313
+
314
+ is_hcp = account_type == "HCP"
315
+ tools = HCP_TOOLS if is_hcp else HCO_TOOLS
316
+ results: dict[str, dict | None] = {}
317
+
318
+ for tool_name in tools:
319
+ args = _build_args(tool_name, name, specialty, country, institution, is_hcp)
320
+ if args is None:
321
+ continue
322
+ results[tool_name] = _call_tool(tool_name, args)
323
+ time.sleep(0.3)
324
+
325
+ output = {
326
+ "account": {
327
+ "id": account_id,
328
+ "name": name,
329
+ "specialty": specialty,
330
+ "country": country,
331
+ "account_type": account_type,
332
+ "institution": institution,
333
+ },
334
+ "search_results": {k: v for k, v in results.items() if v is not None},
335
+ "tools_called": list(results.keys()),
336
+ "tools_succeeded": [k for k, v in results.items() if v is not None],
337
+ "tools_failed": [k for k, v in results.items() if v is None],
338
+ }
339
+
340
+ output_file = output_dir / f"{account_id}.json"
341
+ write_batch_json(output_file, output)
342
+ print(f" saved → {output_file}")
343
+
344
+ return output
345
+
346
+
347
+ def _build_args(
348
+ tool_name: str,
349
+ name: str,
350
+ specialty: str,
351
+ country: str,
352
+ institution: str,
353
+ is_hcp: bool,
354
+ ) -> dict | None:
355
+ if tool_name == "search_orcid":
356
+ args = {"name": name, "max_results": 3}
357
+ if institution:
358
+ args["affiliation"] = institution
359
+ return args
360
+
361
+ if tool_name == "search_publications":
362
+ args = {"author_name": name, "max_results": 20}
363
+ if institution:
364
+ args["affiliation"] = institution
365
+ return args
366
+
367
+ if tool_name == "search_guidelines":
368
+ args = {"name": name, "scope": "both", "max_results": 10}
369
+ if specialty:
370
+ args["therapeutic_area"] = specialty
371
+ return args
372
+
373
+ if tool_name == "search_clinical_trials":
374
+ if is_hcp:
375
+ args = {"investigator_name": name, "max_results": 15}
376
+ else:
377
+ args = {"organization_name": name, "max_results": 15}
378
+ if country:
379
+ args["country"] = country
380
+ return args
381
+
382
+ if tool_name == "search_congresses":
383
+ args = {"name": name, "max_results": 15}
384
+ if specialty:
385
+ args["specialty"] = specialty
386
+ return args
387
+
388
+ if tool_name == "search_grants":
389
+ args: dict = {"pi_name": name, "max_results": 15}
390
+ if institution:
391
+ args["institution"] = institution
392
+ if country:
393
+ args["country"] = country
394
+ return args
395
+
396
+ if tool_name == "search_hcp_web":
397
+ args = {"name": name, "max_results": 10}
398
+ if specialty:
399
+ args["specialty"] = specialty
400
+ if country:
401
+ args["country"] = country
402
+ if institution:
403
+ args["institution"] = institution
404
+ return args
405
+
406
+ if tool_name == "search_hco_web":
407
+ args = {"name": name, "max_results": 10}
408
+ if country:
409
+ args["country"] = country
410
+ return args
411
+
412
+ return None
413
+
414
+
415
+ # ---------------------------------------------------------------------------
416
+ # LLM synthesis
417
+ # ---------------------------------------------------------------------------
418
+
419
+ SYNTHESIS_SYSTEM_PROMPT = """\
420
+ You are an HCP/HCO intelligence analyst. Given an account record and raw search \
421
+ results from PubMed, ClinicalTrials.gov, ORCID, NIH RePORTER, web search, and \
422
+ congress/guideline searches, synthesize a structured profile.
423
+
424
+ RULES:
425
+ 1. Follow the supplied JSON Schema exactly. Identity and metadata fields typed as strings or numbers \
426
+ MUST remain plain JSON scalars; do not wrap them as evidenced claims.
427
+ 2. Every field typed as EvidencedClaim MUST contain value, sources, and confidence. Each source \
428
+ requires a URL and access date. Assign confidence: high (2+ authoritative sources), medium \
429
+ (1 authoritative), low (1 informal).
430
+ 3. SourceCitation.source_type MUST be exactly one of: pubmed, clinical_trials, web, registry. \
431
+ Map ORCID to registry, ClinicalTrials.gov to clinical_trials, and congress/guideline pages to web.
432
+ 4. Do NOT fabricate data. If an optional section has no supporting evidence, omit it or leave it empty.
433
+ 5. Set profile_completeness (0.0-1.0) based on how many sections have data.
434
+ 6. Include disambiguation_notes as a plain string if the person has a common name.
435
+
436
+ OUTPUT FORMAT:
437
+ Return ONLY a valid JSON object conforming to the HcpProfile schema (for HCPs) or \
438
+ HcoProfile schema (for HCOs). No markdown, no commentary, just the JSON object.
439
+
440
+ HcpProfile fields: full_name, specialty, country, current_title, designations, \
441
+ affiliations, education, qualifications, society_memberships, professional_roles, \
442
+ editorial_roles, research_interests, publication_summary, key_publications (up to 10), \
443
+ guideline_publications, clinical_trial_involvement, active_grants, congress_activity, \
444
+ regulatory_advisory_roles, orcid_id, profile_completeness, disambiguation_notes, \
445
+ sources_consulted, built_at.
446
+
447
+ HcoProfile fields: name, country, organization_type, clinical_focus_areas, \
448
+ specialist_departments, bed_capacity, annual_patient_volume, staff_count, \
449
+ accreditations, research_focus, active_clinical_trials, institutional_grants, \
450
+ founding_year, key_milestones, notable_affiliations, profile_completeness, \
451
+ sources_consulted, built_at.
452
+
453
+ Each evidenced claim has: value (string), sources (list of {url, source_type, title, \
454
+ accessed_date}), confidence (high/medium/low).\
455
+ """
456
+
457
+
458
+ def _profile_response_format(account_type: str) -> dict:
459
+ """Build the OpenRouter structured-output contract from the Pydantic source of truth."""
460
+ from open_pharma_plugins_hcp_intelligence.models import HcoProfile, HcpProfile
461
+
462
+ is_hcp = account_type.upper() == "HCP"
463
+ profile_model = HcpProfile if is_hcp else HcoProfile
464
+ return {
465
+ "type": "json_schema",
466
+ "json_schema": {
467
+ "name": "hcp_profile" if is_hcp else "hco_profile",
468
+ "strict": True,
469
+ "schema": profile_model.model_json_schema(),
470
+ },
471
+ }
472
+
473
+
474
+ def _synthesize(raw_output: dict, options: BatchOptions) -> str | None:
475
+ try:
476
+ from openai import OpenAI
477
+ except ImportError:
478
+ print(
479
+ " SKIP synthesis: 'openai' package not installed. Install with: pip install openai",
480
+ file=sys.stderr,
481
+ )
482
+ return None
483
+
484
+ from shared.env import get_env
485
+
486
+ api_key = get_env(options.api_key_env, "")
487
+ if not api_key and options.api_key_env != "NONE":
488
+ print(
489
+ f" SKIP synthesis: {options.api_key_env} not set in the process environment or user config.",
490
+ file=sys.stderr,
491
+ )
492
+ return None
493
+
494
+ client = OpenAI(
495
+ base_url=options.base_url,
496
+ api_key=api_key or "not-needed",
497
+ timeout=options.synthesis_timeout_seconds,
498
+ max_retries=0,
499
+ )
500
+
501
+ account = raw_output["account"]
502
+ user_msg = (
503
+ f"Account: {json.dumps(account)}\n\nSearch results:\n{json.dumps(raw_output['search_results'], indent=2)}"
504
+ )
505
+
506
+ try:
507
+ response = client.chat.completions.create(
508
+ model=options.model,
509
+ messages=[
510
+ {"role": "system", "content": SYNTHESIS_SYSTEM_PROMPT},
511
+ {"role": "user", "content": user_msg},
512
+ ],
513
+ temperature=0.1,
514
+ max_tokens=8192,
515
+ response_format=_profile_response_format(account["account_type"]),
516
+ extra_body={
517
+ "reasoning": {"effort": options.reasoning_effort},
518
+ "provider": {"require_parameters": True},
519
+ },
520
+ )
521
+ except Exception as e:
522
+ safe_error = sanitize_error(e, options.api_key_env)
523
+ message = f"synthesis provider failed ({type(e).__name__}): {safe_error}"
524
+ print(f" FAIL synthesis: {message}", file=sys.stderr)
525
+ raise SynthesisProviderError(message) from e
526
+
527
+ choice = response.choices[0]
528
+ if choice.message.content:
529
+ return choice.message.content
530
+
531
+ usage = getattr(response, "usage", None)
532
+ completion_details = getattr(usage, "completion_tokens_details", None)
533
+ finish_reason = getattr(choice, "finish_reason", None) or "unknown"
534
+ completion_tokens = getattr(usage, "completion_tokens", None)
535
+ reasoning_tokens = getattr(completion_details, "reasoning_tokens", None)
536
+ raise ValueError(
537
+ "OpenRouter returned no final content "
538
+ f"(finish_reason={finish_reason}, completion_tokens={completion_tokens}, "
539
+ f"reasoning_tokens={reasoning_tokens})"
540
+ )
541
+
542
+
543
+ # ---------------------------------------------------------------------------
544
+ # Planning and execution
545
+ # ---------------------------------------------------------------------------
546
+
547
+
548
+ def _validate_output_directory(path: Path, *, resume: bool) -> None:
549
+ if path.exists() and not path.is_dir():
550
+ raise BatchUsageError("output path is not a directory")
551
+ if path.exists() and any(path.iterdir()) and not resume:
552
+ raise BatchUsageError("output directory is not empty; pass --resume to reuse it")
553
+
554
+
555
+ def prepare_output_directory(path: Path, *, resume: bool) -> Path:
556
+ """Validate output reuse, then create a private execution directory."""
557
+ from shared.filesystem import ensure_private_dir
558
+
559
+ _validate_output_directory(path, resume=resume)
560
+ return ensure_private_dir(path).resolve()
561
+
562
+
563
+ def plan_batch(options: BatchOptions) -> BatchPlan:
564
+ """Resolve paths, validate the complete input, and apply account filters."""
565
+ if options.concurrency < 1:
566
+ raise BatchUsageError("--concurrency must be at least 1")
567
+ if not math.isfinite(options.synthesis_timeout_seconds) or options.synthesis_timeout_seconds <= 0:
568
+ raise BatchUsageError("--synthesis-timeout-seconds must be greater than 0 and finite")
569
+ account_type = options.account_type.upper() if options.account_type else None
570
+ if account_type not in {None, "HCP", "HCO"}:
571
+ raise BatchUsageError("--account-type must be HCP or HCO")
572
+
573
+ try:
574
+ if options.input_file is None:
575
+ input_path = resolve_user_path(BUNDLED_INPUT, label="input path")
576
+ else:
577
+ input_path = resolve_user_path(options.input_file, label="input path")
578
+ if not input_path.is_file():
579
+ raise ValueError(f"input file does not exist: {input_path}")
580
+ try:
581
+ input_payload = input_path.read_bytes()
582
+ except OSError as exc:
583
+ raise ValueError(f"could not read input file: {input_path}") from exc
584
+ accounts = _load_accounts_from_bytes(input_payload)
585
+ except ValueError as exc:
586
+ raise BatchUsageError(str(exc)) from exc
587
+
588
+ output_dir = resolve_user_path(options.output_dir, label="output path")
589
+
590
+ if options.country:
591
+ accounts = [account for account in accounts if account["country"].lower() == options.country.lower()]
592
+ if account_type:
593
+ accounts = [account for account in accounts if account["account_type"] == account_type]
594
+ if options.ids:
595
+ selected_ids = set(options.ids)
596
+ accounts = [account for account in accounts if account["id"] in selected_ids]
597
+
598
+ input_sha256 = hashlib.sha256(input_payload).hexdigest()
599
+ if accounts:
600
+ planned_outputs = {
601
+ *(output_dir / f"{account['id']}.json" for account in accounts),
602
+ output_dir / "batch_summary.csv",
603
+ output_dir / "batch_manifest.json",
604
+ }
605
+ if input_path in {path.resolve() for path in planned_outputs}:
606
+ raise BatchUsageError("input file collides with planned output")
607
+ _validate_output_directory(output_dir, resume=options.resume)
608
+
609
+ return BatchPlan(
610
+ input_path=input_path,
611
+ input_sha256=input_sha256,
612
+ output_dir=output_dir,
613
+ accounts=tuple(accounts),
614
+ options=options,
615
+ )
616
+
617
+
618
+ def _load_resume_artifact(path: Path, account: dict[str, str]) -> dict[str, Any] | None:
619
+ """Load a regular, canonical artifact owned by exactly one selected account."""
620
+ try:
621
+ metadata = path.lstat()
622
+ except FileNotFoundError:
623
+ return None
624
+ except OSError:
625
+ return None
626
+ if not stat.S_ISREG(metadata.st_mode):
627
+ return None
628
+ try:
629
+ flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0)
630
+ descriptor = os.open(path, flags)
631
+ with os.fdopen(descriptor, "rb") as handle:
632
+ opened_metadata = os.fstat(handle.fileno())
633
+ if (
634
+ not stat.S_ISREG(opened_metadata.st_mode)
635
+ or opened_metadata.st_dev != metadata.st_dev
636
+ or opened_metadata.st_ino != metadata.st_ino
637
+ ):
638
+ return None
639
+ payload = handle.read()
640
+ loaded = json.loads(payload.decode("utf-8"))
641
+ except (OSError, UnicodeDecodeError, json.JSONDecodeError):
642
+ return None
643
+ if not isinstance(loaded, dict) or loaded.get("account") != account:
644
+ return None
645
+ if not isinstance(loaded.get("search_results"), dict):
646
+ return None
647
+ for key in ("tools_called", "tools_succeeded", "tools_failed"):
648
+ values = loaded.get(key)
649
+ if not isinstance(values, list) or not all(isinstance(value, str) for value in values):
650
+ return None
651
+ return loaded
652
+
653
+
654
+ def run_batch(plan: BatchPlan, emit: Callable[[str], None] = print) -> BatchOutcome:
655
+ """Run a planned batch and persist its ordered artifacts and manifest."""
656
+ options = plan.options
657
+ accounts = plan.accounts
658
+ if not accounts:
659
+ return BatchOutcome(
660
+ output_dir=plan.output_dir,
661
+ results=(),
662
+ manifest={},
663
+ exit_code=0,
664
+ )
665
+
666
+ output_dir = prepare_output_directory(plan.output_dir, resume=options.resume)
667
+ _load_tools()
668
+ write_back = options.write_back or options.input_file is None
669
+ if write_back:
670
+ from open_pharma_plugins_hcp_intelligence._crm_store import write_enrichment
671
+ else:
672
+ write_enrichment = None
673
+ validated_artifacts: dict[str, dict[str, Any]] = {}
674
+ artifact_lock = Lock()
675
+
676
+ def _remember_artifact(account_id: str, artifact: dict[str, Any]) -> None:
677
+ with artifact_lock:
678
+ validated_artifacts[account_id] = artifact
679
+
680
+ emit(f"Processing {len(accounts)} account(s), concurrency={options.concurrency}")
681
+ if options.synthesize:
682
+ emit(f"Synthesis: {options.model} via {options.base_url}")
683
+ emit(
684
+ f"Reasoning effort: {options.reasoning_effort}; "
685
+ f"timeout: {options.synthesis_timeout_seconds:g}s; SDK retries: 0"
686
+ )
687
+ emit("")
688
+
689
+ started_at = datetime.now(timezone.utc).isoformat()
690
+
691
+ def _persist_writeback(account_id: str, profile_json: str, status: str) -> None:
692
+ if not write_back:
693
+ return
694
+ assert write_enrichment is not None
695
+ with _WRITEBACK_LOCK:
696
+ write_enrichment(account_id, profile_json, status)
697
+
698
+ def _process_account(account: dict) -> dict:
699
+ account_id = account["id"]
700
+ result_file = output_dir / f"{account_id}.json"
701
+ raw: dict | None = None
702
+ writeback_attempted = False
703
+ artifact_write_failed = False
704
+
705
+ if options.resume:
706
+ existing = _load_resume_artifact(result_file, account)
707
+ if existing is None:
708
+ try:
709
+ result_file.lstat()
710
+ except (FileNotFoundError, OSError):
711
+ pass
712
+ else:
713
+ emit(f" [{account_id}] existing artifact is unusable; processing normally")
714
+ else:
715
+ synthesized_profile = existing.get("synthesized_profile")
716
+ profile_validated = False
717
+ if options.synthesize and synthesized_profile:
718
+ try:
719
+ validate_synthesized_profile(json.dumps(synthesized_profile), account["account_type"])
720
+ except (TypeError, ValueError):
721
+ existing.pop("synthesized_profile", None)
722
+ existing.pop("synthesis_error", None)
723
+ raw = existing
724
+ emit(f" [{account_id}] invalid synthesized profile; reusing raw evidence")
725
+ else:
726
+ profile_validated = True
727
+ if not options.synthesize or profile_validated:
728
+ _remember_artifact(account_id, existing)
729
+ emit(f" [{account_id}] skipped (exists)")
730
+ return {
731
+ "account_id": account_id,
732
+ "status": "skipped",
733
+ "tools_failed": existing.get("tools_failed", []),
734
+ "output_file": str(result_file),
735
+ "profile_validated": profile_validated,
736
+ }
737
+ if raw is None:
738
+ raw = existing
739
+ raw.pop("synthesis_error", None)
740
+ emit(f" [{account_id}] reusing raw evidence for synthesis")
741
+
742
+ try:
743
+ if raw is None:
744
+ raw = _enrich_account(account, output_dir)
745
+ try:
746
+ write_batch_json(result_file, raw)
747
+ except Exception:
748
+ artifact_write_failed = True
749
+ raise
750
+ tools_failed = raw.get("tools_failed", [])
751
+ status = "partial" if tools_failed else "completed"
752
+ profile_validated = False
753
+
754
+ if options.synthesize:
755
+ profile_json = _synthesize(raw, options)
756
+ if not profile_json:
757
+ raise ValueError("synthesis returned no profile")
758
+ profile = validate_synthesized_profile(profile_json, account["account_type"])
759
+ raw["synthesized_profile"] = profile
760
+ profile_validated = True
761
+ try:
762
+ write_batch_json(result_file, raw)
763
+ except Exception:
764
+ artifact_write_failed = True
765
+ raise
766
+ writeback_attempted = write_back
767
+ _persist_writeback(account_id, json.dumps(profile), "enriched")
768
+ emit(" synthesized + schema validated ✓")
769
+ else:
770
+ try:
771
+ write_batch_json(result_file, raw)
772
+ except Exception:
773
+ artifact_write_failed = True
774
+ raise
775
+ writeback_attempted = write_back
776
+ _persist_writeback(account_id, json.dumps(raw["search_results"]), "enriched")
777
+
778
+ _remember_artifact(account_id, raw)
779
+
780
+ return {
781
+ "account_id": account_id,
782
+ "status": status,
783
+ "tools_failed": tools_failed,
784
+ "output_file": str(result_file),
785
+ "profile_validated": profile_validated,
786
+ }
787
+
788
+ except Exception as e:
789
+ safe_error = sanitize_error(e)
790
+ if raw is not None and not artifact_write_failed:
791
+ raw["synthesis_error" if options.synthesize else "processing_error"] = safe_error
792
+ try:
793
+ write_batch_json(result_file, raw)
794
+ except Exception as persist_exc:
795
+ artifact_write_failed = True
796
+ safe_error = f"{safe_error}; artifact persistence failed: {sanitize_error(persist_exc)}"
797
+ else:
798
+ _remember_artifact(account_id, raw)
799
+ if write_back and not writeback_attempted:
800
+ try:
801
+ _persist_writeback(
802
+ account_id,
803
+ json.dumps(raw.get("search_results", {})) if raw is not None else "{}",
804
+ "failed",
805
+ )
806
+ except Exception as writeback_exc:
807
+ safe_error = f"{safe_error}; write-back failed: {sanitize_error(writeback_exc)}"
808
+ print(f" [{account_id}] ERROR: {safe_error}", file=sys.stderr)
809
+ return {
810
+ "account_id": account_id,
811
+ "status": "failed",
812
+ "tools_failed": raw.get("tools_failed", []) if raw is not None else [],
813
+ "output_file": str(result_file),
814
+ "profile_validated": False,
815
+ "error": safe_error,
816
+ }
817
+
818
+ def _process(account: dict) -> dict:
819
+ token = _ACTIVE_API_KEY_ENV.set(options.api_key_env)
820
+ try:
821
+ return _process_account(account)
822
+ finally:
823
+ _ACTIVE_API_KEY_ENV.reset(token)
824
+
825
+ results: list[dict] = []
826
+ if options.concurrency == 1:
827
+ for account in accounts:
828
+ results.append(_process(account))
829
+ else:
830
+ with ThreadPoolExecutor(max_workers=options.concurrency) as pool:
831
+ futures = {pool.submit(_process, a): a for a in accounts}
832
+ for future in as_completed(futures):
833
+ try:
834
+ results.append(future.result())
835
+ except Exception as e:
836
+ account = futures[future]
837
+ safe_error = sanitize_error(e, options.api_key_env)
838
+ print(f" [{account['id']}] UNHANDLED: {safe_error}", file=sys.stderr)
839
+ results.append(
840
+ {
841
+ "account_id": account["id"],
842
+ "status": "failed",
843
+ "tools_failed": [],
844
+ "profile_validated": False,
845
+ "error": safe_error,
846
+ }
847
+ )
848
+
849
+ # -- Summary --
850
+ order = {account["id"]: index for index, account in enumerate(accounts)}
851
+ results.sort(key=lambda result: order[result["account_id"]])
852
+ from open_pharma_plugins_hcp_intelligence import batch_csv
853
+
854
+ summary_path = output_dir / batch_csv.SUMMARY_FILENAME
855
+ csv_failed = False
856
+ try:
857
+ rows = batch_csv.build_summary_rows(accounts, results, validated_artifacts)
858
+ summary_csv = batch_csv.write_summary_csv(summary_path, rows)
859
+ except Exception as exc:
860
+ csv_failed = True
861
+ summary_csv = {
862
+ "status": "failed",
863
+ "path": str(summary_path.resolve()),
864
+ "schema_version": batch_csv.CSV_SCHEMA_VERSION,
865
+ "error": f"CSV export failed ({type(exc).__name__})",
866
+ }
867
+ completed_at = datetime.now(timezone.utc).isoformat()
868
+ manifest = build_batch_manifest(
869
+ input_path=plan.input_path,
870
+ input_sha256=plan.input_sha256,
871
+ selected_count=len(accounts),
872
+ synthesis=options.synthesize,
873
+ model=options.model,
874
+ base_url=options.base_url,
875
+ reasoning_effort=options.reasoning_effort,
876
+ synthesis_timeout_seconds=options.synthesis_timeout_seconds,
877
+ concurrency=options.concurrency,
878
+ results=results,
879
+ summary_csv=summary_csv,
880
+ started_at=started_at,
881
+ completed_at=completed_at,
882
+ )
883
+ write_batch_json(output_dir / "batch_manifest.json", manifest)
884
+ summary = manifest["summary"]
885
+ exit_code = 1 if csv_failed or summary["failed"] or summary["partial"] else 0
886
+ return BatchOutcome(
887
+ output_dir=output_dir,
888
+ results=tuple(results),
889
+ manifest=manifest,
890
+ exit_code=exit_code,
891
+ )