open-pharma-plugins 2.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_framework.py +495 -0
- open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
- open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
- open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
- open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
- open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
- open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
- open_pharma_plugins_campaign_studio/__init__.py +14 -0
- open_pharma_plugins_campaign_studio/__main__.py +11 -0
- open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
- open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
- open_pharma_plugins_campaign_studio/_renderer.py +119 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
- open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
- open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
- open_pharma_plugins_campaign_studio/models/_common.py +12 -0
- open_pharma_plugins_campaign_studio/models/brief.py +72 -0
- open_pharma_plugins_campaign_studio/models/claims.py +15 -0
- open_pharma_plugins_campaign_studio/models/copy.py +47 -0
- open_pharma_plugins_campaign_studio/models/journey.py +21 -0
- open_pharma_plugins_campaign_studio/models/message.py +25 -0
- open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
- open_pharma_plugins_campaign_studio/models/validation.py +32 -0
- open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
- open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
- open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
- open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
- open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
- open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
- open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
- open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
- open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
- open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
- open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
- open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
- open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
- open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
- open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
- open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
- open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
- open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
- open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
- open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
- open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
- open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
- open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
- open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
- open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
- open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
- open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
- open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
- open_pharma_plugins_competitive_intelligence/models.py +525 -0
- open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
- open_pharma_plugins_field_training/__init__.py +13 -0
- open_pharma_plugins_field_training/__main__.py +11 -0
- open_pharma_plugins_field_training/_content_store.py +131 -0
- open_pharma_plugins_field_training/_grounding.py +75 -0
- open_pharma_plugins_field_training/_html_renderers.py +546 -0
- open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
- open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
- open_pharma_plugins_field_training/models.py +265 -0
- open_pharma_plugins_field_training/tools/__init__.py +0 -0
- open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
- open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
- open_pharma_plugins_field_training/tools/list_documents.py +51 -0
- open_pharma_plugins_field_training/tools/render_output.py +118 -0
- open_pharma_plugins_field_training/tools/search_content.py +57 -0
- open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
- open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
- open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
- open_pharma_plugins_hcp_intelligence/batch.py +891 -0
- open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
- open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
- open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
- open_pharma_plugins_hcp_intelligence/models.py +356 -0
- open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
- open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
- open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
- open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
- open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
- open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
- open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
- open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
- open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
- open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
- open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
- open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
- open_pharma_plugins_next_best_engagement/__init__.py +14 -0
- open_pharma_plugins_next_best_engagement/__main__.py +11 -0
- open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
- open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
- open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
- open_pharma_plugins_next_best_engagement/_universe.py +145 -0
- open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
- open_pharma_plugins_next_best_engagement/models.py +135 -0
- open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
- open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
- open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
- open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
- open_pharma_plugins_territory_alignment/__init__.py +14 -0
- open_pharma_plugins_territory_alignment/__main__.py +11 -0
- open_pharma_plugins_territory_alignment/data.py +300 -0
- open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
- open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
- open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
- open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
- open_pharma_plugins_territory_alignment/geo.py +175 -0
- open_pharma_plugins_territory_alignment/models.py +201 -0
- open_pharma_plugins_territory_alignment/scoring.py +125 -0
- open_pharma_plugins_territory_alignment/solver.py +504 -0
- open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
- open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
- open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
- open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
- open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
- open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
- open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
- shared/__init__.py +11 -0
- shared/env.py +217 -0
- shared/filesystem.py +110 -0
|
@@ -0,0 +1,891 @@
|
|
|
1
|
+
"""Package-owned batch runtime for hcp-intelligence accounts."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import csv
|
|
6
|
+
import hashlib
|
|
7
|
+
import io
|
|
8
|
+
import json
|
|
9
|
+
import math
|
|
10
|
+
import os
|
|
11
|
+
import random
|
|
12
|
+
import stat
|
|
13
|
+
import sys
|
|
14
|
+
import time
|
|
15
|
+
import urllib.error
|
|
16
|
+
from concurrent.futures import ThreadPoolExecutor, as_completed
|
|
17
|
+
from contextvars import ContextVar
|
|
18
|
+
from dataclasses import dataclass
|
|
19
|
+
from datetime import datetime, timezone
|
|
20
|
+
from pathlib import Path
|
|
21
|
+
from threading import Lock
|
|
22
|
+
from typing import Any, Callable, Literal
|
|
23
|
+
|
|
24
|
+
BUNDLED_INPUT = Path(__file__).parent / "fixtures" / "sample_accounts.csv"
|
|
25
|
+
INPUT_COLUMNS = ("id", "name", "specialty", "country", "account_type", "institution")
|
|
26
|
+
REQUIRED_VALUES = ("id", "name", "country", "account_type")
|
|
27
|
+
DEFAULT_OPENROUTER_BASE_URL = "https://openrouter.ai/api/v1"
|
|
28
|
+
DEFAULT_SYNTHESIS_MODEL = "deepseek/deepseek-v4-flash-0731"
|
|
29
|
+
DEFAULT_REASONING_EFFORT = "high"
|
|
30
|
+
DEFAULT_SYNTHESIS_TIMEOUT_SECONDS = 120.0
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
class BatchUsageError(ValueError):
|
|
34
|
+
"""A user-correctable batch input or path error."""
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
class SynthesisProviderError(RuntimeError):
|
|
38
|
+
"""A sanitized synthesis-provider failure safe to persist."""
|
|
39
|
+
|
|
40
|
+
|
|
41
|
+
@dataclass(frozen=True)
|
|
42
|
+
class BatchOptions:
|
|
43
|
+
input_file: str | Path | None
|
|
44
|
+
output_dir: str | Path
|
|
45
|
+
country: str | None
|
|
46
|
+
account_type: str | None
|
|
47
|
+
ids: tuple[str, ...]
|
|
48
|
+
concurrency: int
|
|
49
|
+
resume: bool
|
|
50
|
+
write_back: bool
|
|
51
|
+
synthesize: bool
|
|
52
|
+
base_url: str
|
|
53
|
+
api_key_env: str
|
|
54
|
+
model: str
|
|
55
|
+
reasoning_effort: Literal["high", "xhigh"]
|
|
56
|
+
synthesis_timeout_seconds: float
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True)
|
|
60
|
+
class BatchPlan:
|
|
61
|
+
input_path: Path
|
|
62
|
+
input_sha256: str
|
|
63
|
+
output_dir: Path
|
|
64
|
+
accounts: tuple[dict[str, str], ...]
|
|
65
|
+
options: BatchOptions
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
@dataclass(frozen=True)
|
|
69
|
+
class BatchOutcome:
|
|
70
|
+
output_dir: Path
|
|
71
|
+
results: tuple[dict[str, Any], ...]
|
|
72
|
+
manifest: dict[str, Any]
|
|
73
|
+
exit_code: int
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
# ---------------------------------------------------------------------------
|
|
77
|
+
# Tool imports (lazy so importing the batch runtime has no provider side effects)
|
|
78
|
+
# ---------------------------------------------------------------------------
|
|
79
|
+
|
|
80
|
+
_TOOL_MODULES: dict[str, object] = {}
|
|
81
|
+
_WRITEBACK_LOCK = Lock()
|
|
82
|
+
_ACTIVE_API_KEY_ENV: ContextVar[str] = ContextVar("hcp_batch_api_key_env", default="OPENROUTER_API_KEY")
|
|
83
|
+
_REDACTED_PROVIDER_KEYS = ("EXA_API_KEY", "SERPER_API_KEY", "TAVILY_API_KEY", "NCBI_API_KEY")
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def sanitize_error(error: object, api_key_env: str | None = None) -> str:
|
|
87
|
+
"""Redact configured provider credentials from an HCP batch error."""
|
|
88
|
+
from shared.env import get_env
|
|
89
|
+
|
|
90
|
+
selected_key = api_key_env or _ACTIVE_API_KEY_ENV.get()
|
|
91
|
+
key_names = dict.fromkeys((selected_key, *_REDACTED_PROVIDER_KEYS))
|
|
92
|
+
credentials = sorted(
|
|
93
|
+
(value for key in key_names if (value := get_env(key, ""))),
|
|
94
|
+
key=len,
|
|
95
|
+
reverse=True,
|
|
96
|
+
)
|
|
97
|
+
safe = str(error)
|
|
98
|
+
for credential in credentials:
|
|
99
|
+
safe = safe.replace(credential, "[REDACTED]")
|
|
100
|
+
return safe
|
|
101
|
+
|
|
102
|
+
|
|
103
|
+
def resolve_user_path(value: str | Path, *, label: str) -> Path:
|
|
104
|
+
"""Resolve a user path only after rejecting invisible control characters."""
|
|
105
|
+
raw = str(value)
|
|
106
|
+
if any(ord(char) < 32 or ord(char) == 127 for char in raw):
|
|
107
|
+
raise BatchUsageError(f"{label} contains a control character")
|
|
108
|
+
return Path(raw).expanduser().resolve()
|
|
109
|
+
|
|
110
|
+
|
|
111
|
+
def _load_accounts_from_bytes(payload: bytes) -> list[dict[str, str]]:
|
|
112
|
+
"""Parse one captured CSV byte snapshot into the normalized public contract."""
|
|
113
|
+
try:
|
|
114
|
+
text = payload.decode("utf-8-sig")
|
|
115
|
+
except UnicodeDecodeError as exc:
|
|
116
|
+
raise ValueError("input file must be UTF-8 CSV") from exc
|
|
117
|
+
|
|
118
|
+
with io.StringIO(text, newline="") as handle:
|
|
119
|
+
reader = csv.DictReader(handle)
|
|
120
|
+
headers = set(reader.fieldnames or [])
|
|
121
|
+
missing = sorted(set(INPUT_COLUMNS) - headers)
|
|
122
|
+
if missing:
|
|
123
|
+
raise ValueError(f"missing columns: {', '.join(missing)}")
|
|
124
|
+
|
|
125
|
+
accounts: list[dict[str, str]] = []
|
|
126
|
+
seen_ids: set[str] = set()
|
|
127
|
+
for row_number, raw in enumerate(reader, start=2):
|
|
128
|
+
raw_account_id = raw.get("id") or ""
|
|
129
|
+
if any(ord(char) < 32 or ord(char) == 127 for char in raw_account_id):
|
|
130
|
+
raise ValueError(f"row {row_number} account id contains a control character")
|
|
131
|
+
account = {key: (raw.get(key) or "").strip() for key in INPUT_COLUMNS}
|
|
132
|
+
blank = [key for key in REQUIRED_VALUES if not account[key]]
|
|
133
|
+
if blank:
|
|
134
|
+
raise ValueError(f"row {row_number}: blank required values: {', '.join(blank)}")
|
|
135
|
+
|
|
136
|
+
account_type = account["account_type"].upper()
|
|
137
|
+
if account_type not in {"HCP", "HCO"}:
|
|
138
|
+
raise ValueError(f"row {row_number}: account_type must be HCP or HCO")
|
|
139
|
+
account["account_type"] = account_type
|
|
140
|
+
|
|
141
|
+
account_id = account["id"]
|
|
142
|
+
from shared.filesystem import validate_component
|
|
143
|
+
|
|
144
|
+
validate_component(account_id, label=f"row {row_number} account id")
|
|
145
|
+
if account_id in seen_ids:
|
|
146
|
+
raise ValueError(f"row {row_number}: duplicate account id: {account_id}")
|
|
147
|
+
seen_ids.add(account_id)
|
|
148
|
+
accounts.append(account)
|
|
149
|
+
|
|
150
|
+
if not accounts:
|
|
151
|
+
raise ValueError("input file contains no account rows")
|
|
152
|
+
return accounts
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def load_accounts_from_csv(path: str | Path) -> list[dict[str, str]]:
|
|
156
|
+
"""Load and validate the public batch-input contract before any network calls."""
|
|
157
|
+
source = resolve_user_path(path, label="input path")
|
|
158
|
+
if not source.is_file():
|
|
159
|
+
raise ValueError(f"input file does not exist: {source}")
|
|
160
|
+
try:
|
|
161
|
+
payload = source.read_bytes()
|
|
162
|
+
except OSError as exc:
|
|
163
|
+
raise ValueError(f"could not read input file: {source}") from exc
|
|
164
|
+
return _load_accounts_from_bytes(payload)
|
|
165
|
+
|
|
166
|
+
|
|
167
|
+
def validate_synthesized_profile(profile_json: str, account_type: str) -> dict:
|
|
168
|
+
"""Reject malformed or out-of-contract LLM output before persistence."""
|
|
169
|
+
from open_pharma_plugins_hcp_intelligence.models import HcoProfile, HcpProfile
|
|
170
|
+
|
|
171
|
+
data = json.loads(profile_json)
|
|
172
|
+
model = HcpProfile if account_type.upper() == "HCP" else HcoProfile
|
|
173
|
+
return model.model_validate(data).model_dump(mode="json", exclude_unset=True)
|
|
174
|
+
|
|
175
|
+
|
|
176
|
+
def write_batch_json(path: str | Path, data: object) -> Path:
|
|
177
|
+
"""Persist a private, atomic JSON artifact."""
|
|
178
|
+
from shared.filesystem import atomic_write_json
|
|
179
|
+
|
|
180
|
+
return atomic_write_json(path, data)
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
def build_batch_manifest(
|
|
184
|
+
*,
|
|
185
|
+
input_path: str | Path,
|
|
186
|
+
input_sha256: str,
|
|
187
|
+
selected_count: int,
|
|
188
|
+
synthesis: bool,
|
|
189
|
+
model: str,
|
|
190
|
+
base_url: str,
|
|
191
|
+
reasoning_effort: str,
|
|
192
|
+
synthesis_timeout_seconds: float,
|
|
193
|
+
concurrency: int,
|
|
194
|
+
results: list[dict],
|
|
195
|
+
summary_csv: dict[str, object],
|
|
196
|
+
started_at: str,
|
|
197
|
+
completed_at: str,
|
|
198
|
+
) -> dict:
|
|
199
|
+
"""Build an auditable batch summary without duplicating account names."""
|
|
200
|
+
source = Path(input_path).expanduser().resolve()
|
|
201
|
+
statuses = {status: 0 for status in ("completed", "partial", "failed", "skipped")}
|
|
202
|
+
accounts: list[dict] = []
|
|
203
|
+
allowed = ("account_id", "status", "tools_failed", "output_file", "profile_validated", "error")
|
|
204
|
+
for result in results:
|
|
205
|
+
status = result["status"]
|
|
206
|
+
statuses[status] += 1
|
|
207
|
+
accounts.append({key: result[key] for key in allowed if key in result})
|
|
208
|
+
|
|
209
|
+
return {
|
|
210
|
+
"schema_version": 2,
|
|
211
|
+
"input": {
|
|
212
|
+
"path": str(source),
|
|
213
|
+
"sha256": input_sha256,
|
|
214
|
+
},
|
|
215
|
+
"started_at": started_at,
|
|
216
|
+
"completed_at": completed_at,
|
|
217
|
+
"processing": {
|
|
218
|
+
"synthesis": synthesis,
|
|
219
|
+
"model": model if synthesis else None,
|
|
220
|
+
"base_url": base_url if synthesis else None,
|
|
221
|
+
"reasoning_effort": reasoning_effort if synthesis else None,
|
|
222
|
+
"timeout_seconds": synthesis_timeout_seconds if synthesis else None,
|
|
223
|
+
"concurrency": concurrency,
|
|
224
|
+
},
|
|
225
|
+
"summary": {"selected": selected_count, **statuses},
|
|
226
|
+
"accounts": accounts,
|
|
227
|
+
"outputs": {"summary_csv": summary_csv},
|
|
228
|
+
}
|
|
229
|
+
|
|
230
|
+
|
|
231
|
+
def _load_tools():
|
|
232
|
+
if _TOOL_MODULES:
|
|
233
|
+
return
|
|
234
|
+
from open_pharma_plugins_hcp_intelligence.tools import (
|
|
235
|
+
search_clinical_trials,
|
|
236
|
+
search_congresses,
|
|
237
|
+
search_grants,
|
|
238
|
+
search_guidelines,
|
|
239
|
+
search_hco_web,
|
|
240
|
+
search_hcp_web,
|
|
241
|
+
search_orcid,
|
|
242
|
+
search_publications,
|
|
243
|
+
)
|
|
244
|
+
|
|
245
|
+
_TOOL_MODULES.update(
|
|
246
|
+
{
|
|
247
|
+
"search_orcid": search_orcid,
|
|
248
|
+
"search_publications": search_publications,
|
|
249
|
+
"search_guidelines": search_guidelines,
|
|
250
|
+
"search_clinical_trials": search_clinical_trials,
|
|
251
|
+
"search_congresses": search_congresses,
|
|
252
|
+
"search_grants": search_grants,
|
|
253
|
+
"search_hcp_web": search_hcp_web,
|
|
254
|
+
"search_hco_web": search_hco_web,
|
|
255
|
+
}
|
|
256
|
+
)
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
# ---------------------------------------------------------------------------
|
|
260
|
+
# Retry wrapper
|
|
261
|
+
# ---------------------------------------------------------------------------
|
|
262
|
+
|
|
263
|
+
MAX_RETRIES = 3
|
|
264
|
+
|
|
265
|
+
|
|
266
|
+
def _call_tool(tool_name: str, arguments: dict) -> dict | None:
|
|
267
|
+
module = _TOOL_MODULES[tool_name]
|
|
268
|
+
for attempt in range(MAX_RETRIES + 1):
|
|
269
|
+
try:
|
|
270
|
+
result = module.handle(arguments)
|
|
271
|
+
return json.loads(result[0]["text"])
|
|
272
|
+
except Exception as e:
|
|
273
|
+
retryable = not isinstance(e, urllib.error.HTTPError) or e.code in {408, 425, 429} or e.code >= 500
|
|
274
|
+
if not retryable or attempt == MAX_RETRIES:
|
|
275
|
+
print(f" FAIL {tool_name}: {sanitize_error(e)}", file=sys.stderr)
|
|
276
|
+
return None
|
|
277
|
+
delay = min(2**attempt + random.uniform(0, 1), 30.0)
|
|
278
|
+
print(f" RETRY {tool_name} (attempt {attempt + 1}): {sanitize_error(e)}", file=sys.stderr)
|
|
279
|
+
time.sleep(delay)
|
|
280
|
+
return None
|
|
281
|
+
|
|
282
|
+
|
|
283
|
+
# ---------------------------------------------------------------------------
|
|
284
|
+
# Per-account enrichment
|
|
285
|
+
# ---------------------------------------------------------------------------
|
|
286
|
+
|
|
287
|
+
HCP_TOOLS = [
|
|
288
|
+
"search_orcid",
|
|
289
|
+
"search_publications",
|
|
290
|
+
"search_guidelines",
|
|
291
|
+
"search_clinical_trials",
|
|
292
|
+
"search_congresses",
|
|
293
|
+
"search_grants",
|
|
294
|
+
"search_hcp_web",
|
|
295
|
+
]
|
|
296
|
+
|
|
297
|
+
HCO_TOOLS = [
|
|
298
|
+
"search_hco_web",
|
|
299
|
+
"search_clinical_trials",
|
|
300
|
+
"search_grants",
|
|
301
|
+
]
|
|
302
|
+
|
|
303
|
+
|
|
304
|
+
def _enrich_account(account: dict, output_dir: Path) -> dict:
|
|
305
|
+
account_id = account["id"]
|
|
306
|
+
name = account["name"]
|
|
307
|
+
specialty = account.get("specialty", "")
|
|
308
|
+
country = account.get("country", "")
|
|
309
|
+
institution = account.get("institution", "")
|
|
310
|
+
account_type = account.get("account_type", "HCP").upper()
|
|
311
|
+
|
|
312
|
+
print(f" [{account_id}] {name} ({account_type}, {country})")
|
|
313
|
+
|
|
314
|
+
is_hcp = account_type == "HCP"
|
|
315
|
+
tools = HCP_TOOLS if is_hcp else HCO_TOOLS
|
|
316
|
+
results: dict[str, dict | None] = {}
|
|
317
|
+
|
|
318
|
+
for tool_name in tools:
|
|
319
|
+
args = _build_args(tool_name, name, specialty, country, institution, is_hcp)
|
|
320
|
+
if args is None:
|
|
321
|
+
continue
|
|
322
|
+
results[tool_name] = _call_tool(tool_name, args)
|
|
323
|
+
time.sleep(0.3)
|
|
324
|
+
|
|
325
|
+
output = {
|
|
326
|
+
"account": {
|
|
327
|
+
"id": account_id,
|
|
328
|
+
"name": name,
|
|
329
|
+
"specialty": specialty,
|
|
330
|
+
"country": country,
|
|
331
|
+
"account_type": account_type,
|
|
332
|
+
"institution": institution,
|
|
333
|
+
},
|
|
334
|
+
"search_results": {k: v for k, v in results.items() if v is not None},
|
|
335
|
+
"tools_called": list(results.keys()),
|
|
336
|
+
"tools_succeeded": [k for k, v in results.items() if v is not None],
|
|
337
|
+
"tools_failed": [k for k, v in results.items() if v is None],
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
output_file = output_dir / f"{account_id}.json"
|
|
341
|
+
write_batch_json(output_file, output)
|
|
342
|
+
print(f" saved → {output_file}")
|
|
343
|
+
|
|
344
|
+
return output
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
def _build_args(
|
|
348
|
+
tool_name: str,
|
|
349
|
+
name: str,
|
|
350
|
+
specialty: str,
|
|
351
|
+
country: str,
|
|
352
|
+
institution: str,
|
|
353
|
+
is_hcp: bool,
|
|
354
|
+
) -> dict | None:
|
|
355
|
+
if tool_name == "search_orcid":
|
|
356
|
+
args = {"name": name, "max_results": 3}
|
|
357
|
+
if institution:
|
|
358
|
+
args["affiliation"] = institution
|
|
359
|
+
return args
|
|
360
|
+
|
|
361
|
+
if tool_name == "search_publications":
|
|
362
|
+
args = {"author_name": name, "max_results": 20}
|
|
363
|
+
if institution:
|
|
364
|
+
args["affiliation"] = institution
|
|
365
|
+
return args
|
|
366
|
+
|
|
367
|
+
if tool_name == "search_guidelines":
|
|
368
|
+
args = {"name": name, "scope": "both", "max_results": 10}
|
|
369
|
+
if specialty:
|
|
370
|
+
args["therapeutic_area"] = specialty
|
|
371
|
+
return args
|
|
372
|
+
|
|
373
|
+
if tool_name == "search_clinical_trials":
|
|
374
|
+
if is_hcp:
|
|
375
|
+
args = {"investigator_name": name, "max_results": 15}
|
|
376
|
+
else:
|
|
377
|
+
args = {"organization_name": name, "max_results": 15}
|
|
378
|
+
if country:
|
|
379
|
+
args["country"] = country
|
|
380
|
+
return args
|
|
381
|
+
|
|
382
|
+
if tool_name == "search_congresses":
|
|
383
|
+
args = {"name": name, "max_results": 15}
|
|
384
|
+
if specialty:
|
|
385
|
+
args["specialty"] = specialty
|
|
386
|
+
return args
|
|
387
|
+
|
|
388
|
+
if tool_name == "search_grants":
|
|
389
|
+
args: dict = {"pi_name": name, "max_results": 15}
|
|
390
|
+
if institution:
|
|
391
|
+
args["institution"] = institution
|
|
392
|
+
if country:
|
|
393
|
+
args["country"] = country
|
|
394
|
+
return args
|
|
395
|
+
|
|
396
|
+
if tool_name == "search_hcp_web":
|
|
397
|
+
args = {"name": name, "max_results": 10}
|
|
398
|
+
if specialty:
|
|
399
|
+
args["specialty"] = specialty
|
|
400
|
+
if country:
|
|
401
|
+
args["country"] = country
|
|
402
|
+
if institution:
|
|
403
|
+
args["institution"] = institution
|
|
404
|
+
return args
|
|
405
|
+
|
|
406
|
+
if tool_name == "search_hco_web":
|
|
407
|
+
args = {"name": name, "max_results": 10}
|
|
408
|
+
if country:
|
|
409
|
+
args["country"] = country
|
|
410
|
+
return args
|
|
411
|
+
|
|
412
|
+
return None
|
|
413
|
+
|
|
414
|
+
|
|
415
|
+
# ---------------------------------------------------------------------------
|
|
416
|
+
# LLM synthesis
|
|
417
|
+
# ---------------------------------------------------------------------------
|
|
418
|
+
|
|
419
|
+
SYNTHESIS_SYSTEM_PROMPT = """\
|
|
420
|
+
You are an HCP/HCO intelligence analyst. Given an account record and raw search \
|
|
421
|
+
results from PubMed, ClinicalTrials.gov, ORCID, NIH RePORTER, web search, and \
|
|
422
|
+
congress/guideline searches, synthesize a structured profile.
|
|
423
|
+
|
|
424
|
+
RULES:
|
|
425
|
+
1. Follow the supplied JSON Schema exactly. Identity and metadata fields typed as strings or numbers \
|
|
426
|
+
MUST remain plain JSON scalars; do not wrap them as evidenced claims.
|
|
427
|
+
2. Every field typed as EvidencedClaim MUST contain value, sources, and confidence. Each source \
|
|
428
|
+
requires a URL and access date. Assign confidence: high (2+ authoritative sources), medium \
|
|
429
|
+
(1 authoritative), low (1 informal).
|
|
430
|
+
3. SourceCitation.source_type MUST be exactly one of: pubmed, clinical_trials, web, registry. \
|
|
431
|
+
Map ORCID to registry, ClinicalTrials.gov to clinical_trials, and congress/guideline pages to web.
|
|
432
|
+
4. Do NOT fabricate data. If an optional section has no supporting evidence, omit it or leave it empty.
|
|
433
|
+
5. Set profile_completeness (0.0-1.0) based on how many sections have data.
|
|
434
|
+
6. Include disambiguation_notes as a plain string if the person has a common name.
|
|
435
|
+
|
|
436
|
+
OUTPUT FORMAT:
|
|
437
|
+
Return ONLY a valid JSON object conforming to the HcpProfile schema (for HCPs) or \
|
|
438
|
+
HcoProfile schema (for HCOs). No markdown, no commentary, just the JSON object.
|
|
439
|
+
|
|
440
|
+
HcpProfile fields: full_name, specialty, country, current_title, designations, \
|
|
441
|
+
affiliations, education, qualifications, society_memberships, professional_roles, \
|
|
442
|
+
editorial_roles, research_interests, publication_summary, key_publications (up to 10), \
|
|
443
|
+
guideline_publications, clinical_trial_involvement, active_grants, congress_activity, \
|
|
444
|
+
regulatory_advisory_roles, orcid_id, profile_completeness, disambiguation_notes, \
|
|
445
|
+
sources_consulted, built_at.
|
|
446
|
+
|
|
447
|
+
HcoProfile fields: name, country, organization_type, clinical_focus_areas, \
|
|
448
|
+
specialist_departments, bed_capacity, annual_patient_volume, staff_count, \
|
|
449
|
+
accreditations, research_focus, active_clinical_trials, institutional_grants, \
|
|
450
|
+
founding_year, key_milestones, notable_affiliations, profile_completeness, \
|
|
451
|
+
sources_consulted, built_at.
|
|
452
|
+
|
|
453
|
+
Each evidenced claim has: value (string), sources (list of {url, source_type, title, \
|
|
454
|
+
accessed_date}), confidence (high/medium/low).\
|
|
455
|
+
"""
|
|
456
|
+
|
|
457
|
+
|
|
458
|
+
def _profile_response_format(account_type: str) -> dict:
|
|
459
|
+
"""Build the OpenRouter structured-output contract from the Pydantic source of truth."""
|
|
460
|
+
from open_pharma_plugins_hcp_intelligence.models import HcoProfile, HcpProfile
|
|
461
|
+
|
|
462
|
+
is_hcp = account_type.upper() == "HCP"
|
|
463
|
+
profile_model = HcpProfile if is_hcp else HcoProfile
|
|
464
|
+
return {
|
|
465
|
+
"type": "json_schema",
|
|
466
|
+
"json_schema": {
|
|
467
|
+
"name": "hcp_profile" if is_hcp else "hco_profile",
|
|
468
|
+
"strict": True,
|
|
469
|
+
"schema": profile_model.model_json_schema(),
|
|
470
|
+
},
|
|
471
|
+
}
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def _synthesize(raw_output: dict, options: BatchOptions) -> str | None:
|
|
475
|
+
try:
|
|
476
|
+
from openai import OpenAI
|
|
477
|
+
except ImportError:
|
|
478
|
+
print(
|
|
479
|
+
" SKIP synthesis: 'openai' package not installed. Install with: pip install openai",
|
|
480
|
+
file=sys.stderr,
|
|
481
|
+
)
|
|
482
|
+
return None
|
|
483
|
+
|
|
484
|
+
from shared.env import get_env
|
|
485
|
+
|
|
486
|
+
api_key = get_env(options.api_key_env, "")
|
|
487
|
+
if not api_key and options.api_key_env != "NONE":
|
|
488
|
+
print(
|
|
489
|
+
f" SKIP synthesis: {options.api_key_env} not set in the process environment or user config.",
|
|
490
|
+
file=sys.stderr,
|
|
491
|
+
)
|
|
492
|
+
return None
|
|
493
|
+
|
|
494
|
+
client = OpenAI(
|
|
495
|
+
base_url=options.base_url,
|
|
496
|
+
api_key=api_key or "not-needed",
|
|
497
|
+
timeout=options.synthesis_timeout_seconds,
|
|
498
|
+
max_retries=0,
|
|
499
|
+
)
|
|
500
|
+
|
|
501
|
+
account = raw_output["account"]
|
|
502
|
+
user_msg = (
|
|
503
|
+
f"Account: {json.dumps(account)}\n\nSearch results:\n{json.dumps(raw_output['search_results'], indent=2)}"
|
|
504
|
+
)
|
|
505
|
+
|
|
506
|
+
try:
|
|
507
|
+
response = client.chat.completions.create(
|
|
508
|
+
model=options.model,
|
|
509
|
+
messages=[
|
|
510
|
+
{"role": "system", "content": SYNTHESIS_SYSTEM_PROMPT},
|
|
511
|
+
{"role": "user", "content": user_msg},
|
|
512
|
+
],
|
|
513
|
+
temperature=0.1,
|
|
514
|
+
max_tokens=8192,
|
|
515
|
+
response_format=_profile_response_format(account["account_type"]),
|
|
516
|
+
extra_body={
|
|
517
|
+
"reasoning": {"effort": options.reasoning_effort},
|
|
518
|
+
"provider": {"require_parameters": True},
|
|
519
|
+
},
|
|
520
|
+
)
|
|
521
|
+
except Exception as e:
|
|
522
|
+
safe_error = sanitize_error(e, options.api_key_env)
|
|
523
|
+
message = f"synthesis provider failed ({type(e).__name__}): {safe_error}"
|
|
524
|
+
print(f" FAIL synthesis: {message}", file=sys.stderr)
|
|
525
|
+
raise SynthesisProviderError(message) from e
|
|
526
|
+
|
|
527
|
+
choice = response.choices[0]
|
|
528
|
+
if choice.message.content:
|
|
529
|
+
return choice.message.content
|
|
530
|
+
|
|
531
|
+
usage = getattr(response, "usage", None)
|
|
532
|
+
completion_details = getattr(usage, "completion_tokens_details", None)
|
|
533
|
+
finish_reason = getattr(choice, "finish_reason", None) or "unknown"
|
|
534
|
+
completion_tokens = getattr(usage, "completion_tokens", None)
|
|
535
|
+
reasoning_tokens = getattr(completion_details, "reasoning_tokens", None)
|
|
536
|
+
raise ValueError(
|
|
537
|
+
"OpenRouter returned no final content "
|
|
538
|
+
f"(finish_reason={finish_reason}, completion_tokens={completion_tokens}, "
|
|
539
|
+
f"reasoning_tokens={reasoning_tokens})"
|
|
540
|
+
)
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
# ---------------------------------------------------------------------------
|
|
544
|
+
# Planning and execution
|
|
545
|
+
# ---------------------------------------------------------------------------
|
|
546
|
+
|
|
547
|
+
|
|
548
|
+
def _validate_output_directory(path: Path, *, resume: bool) -> None:
|
|
549
|
+
if path.exists() and not path.is_dir():
|
|
550
|
+
raise BatchUsageError("output path is not a directory")
|
|
551
|
+
if path.exists() and any(path.iterdir()) and not resume:
|
|
552
|
+
raise BatchUsageError("output directory is not empty; pass --resume to reuse it")
|
|
553
|
+
|
|
554
|
+
|
|
555
|
+
def prepare_output_directory(path: Path, *, resume: bool) -> Path:
|
|
556
|
+
"""Validate output reuse, then create a private execution directory."""
|
|
557
|
+
from shared.filesystem import ensure_private_dir
|
|
558
|
+
|
|
559
|
+
_validate_output_directory(path, resume=resume)
|
|
560
|
+
return ensure_private_dir(path).resolve()
|
|
561
|
+
|
|
562
|
+
|
|
563
|
+
def plan_batch(options: BatchOptions) -> BatchPlan:
|
|
564
|
+
"""Resolve paths, validate the complete input, and apply account filters."""
|
|
565
|
+
if options.concurrency < 1:
|
|
566
|
+
raise BatchUsageError("--concurrency must be at least 1")
|
|
567
|
+
if not math.isfinite(options.synthesis_timeout_seconds) or options.synthesis_timeout_seconds <= 0:
|
|
568
|
+
raise BatchUsageError("--synthesis-timeout-seconds must be greater than 0 and finite")
|
|
569
|
+
account_type = options.account_type.upper() if options.account_type else None
|
|
570
|
+
if account_type not in {None, "HCP", "HCO"}:
|
|
571
|
+
raise BatchUsageError("--account-type must be HCP or HCO")
|
|
572
|
+
|
|
573
|
+
try:
|
|
574
|
+
if options.input_file is None:
|
|
575
|
+
input_path = resolve_user_path(BUNDLED_INPUT, label="input path")
|
|
576
|
+
else:
|
|
577
|
+
input_path = resolve_user_path(options.input_file, label="input path")
|
|
578
|
+
if not input_path.is_file():
|
|
579
|
+
raise ValueError(f"input file does not exist: {input_path}")
|
|
580
|
+
try:
|
|
581
|
+
input_payload = input_path.read_bytes()
|
|
582
|
+
except OSError as exc:
|
|
583
|
+
raise ValueError(f"could not read input file: {input_path}") from exc
|
|
584
|
+
accounts = _load_accounts_from_bytes(input_payload)
|
|
585
|
+
except ValueError as exc:
|
|
586
|
+
raise BatchUsageError(str(exc)) from exc
|
|
587
|
+
|
|
588
|
+
output_dir = resolve_user_path(options.output_dir, label="output path")
|
|
589
|
+
|
|
590
|
+
if options.country:
|
|
591
|
+
accounts = [account for account in accounts if account["country"].lower() == options.country.lower()]
|
|
592
|
+
if account_type:
|
|
593
|
+
accounts = [account for account in accounts if account["account_type"] == account_type]
|
|
594
|
+
if options.ids:
|
|
595
|
+
selected_ids = set(options.ids)
|
|
596
|
+
accounts = [account for account in accounts if account["id"] in selected_ids]
|
|
597
|
+
|
|
598
|
+
input_sha256 = hashlib.sha256(input_payload).hexdigest()
|
|
599
|
+
if accounts:
|
|
600
|
+
planned_outputs = {
|
|
601
|
+
*(output_dir / f"{account['id']}.json" for account in accounts),
|
|
602
|
+
output_dir / "batch_summary.csv",
|
|
603
|
+
output_dir / "batch_manifest.json",
|
|
604
|
+
}
|
|
605
|
+
if input_path in {path.resolve() for path in planned_outputs}:
|
|
606
|
+
raise BatchUsageError("input file collides with planned output")
|
|
607
|
+
_validate_output_directory(output_dir, resume=options.resume)
|
|
608
|
+
|
|
609
|
+
return BatchPlan(
|
|
610
|
+
input_path=input_path,
|
|
611
|
+
input_sha256=input_sha256,
|
|
612
|
+
output_dir=output_dir,
|
|
613
|
+
accounts=tuple(accounts),
|
|
614
|
+
options=options,
|
|
615
|
+
)
|
|
616
|
+
|
|
617
|
+
|
|
618
|
+
def _load_resume_artifact(path: Path, account: dict[str, str]) -> dict[str, Any] | None:
|
|
619
|
+
"""Load a regular, canonical artifact owned by exactly one selected account."""
|
|
620
|
+
try:
|
|
621
|
+
metadata = path.lstat()
|
|
622
|
+
except FileNotFoundError:
|
|
623
|
+
return None
|
|
624
|
+
except OSError:
|
|
625
|
+
return None
|
|
626
|
+
if not stat.S_ISREG(metadata.st_mode):
|
|
627
|
+
return None
|
|
628
|
+
try:
|
|
629
|
+
flags = os.O_RDONLY | getattr(os, "O_NOFOLLOW", 0)
|
|
630
|
+
descriptor = os.open(path, flags)
|
|
631
|
+
with os.fdopen(descriptor, "rb") as handle:
|
|
632
|
+
opened_metadata = os.fstat(handle.fileno())
|
|
633
|
+
if (
|
|
634
|
+
not stat.S_ISREG(opened_metadata.st_mode)
|
|
635
|
+
or opened_metadata.st_dev != metadata.st_dev
|
|
636
|
+
or opened_metadata.st_ino != metadata.st_ino
|
|
637
|
+
):
|
|
638
|
+
return None
|
|
639
|
+
payload = handle.read()
|
|
640
|
+
loaded = json.loads(payload.decode("utf-8"))
|
|
641
|
+
except (OSError, UnicodeDecodeError, json.JSONDecodeError):
|
|
642
|
+
return None
|
|
643
|
+
if not isinstance(loaded, dict) or loaded.get("account") != account:
|
|
644
|
+
return None
|
|
645
|
+
if not isinstance(loaded.get("search_results"), dict):
|
|
646
|
+
return None
|
|
647
|
+
for key in ("tools_called", "tools_succeeded", "tools_failed"):
|
|
648
|
+
values = loaded.get(key)
|
|
649
|
+
if not isinstance(values, list) or not all(isinstance(value, str) for value in values):
|
|
650
|
+
return None
|
|
651
|
+
return loaded
|
|
652
|
+
|
|
653
|
+
|
|
654
|
+
def run_batch(plan: BatchPlan, emit: Callable[[str], None] = print) -> BatchOutcome:
|
|
655
|
+
"""Run a planned batch and persist its ordered artifacts and manifest."""
|
|
656
|
+
options = plan.options
|
|
657
|
+
accounts = plan.accounts
|
|
658
|
+
if not accounts:
|
|
659
|
+
return BatchOutcome(
|
|
660
|
+
output_dir=plan.output_dir,
|
|
661
|
+
results=(),
|
|
662
|
+
manifest={},
|
|
663
|
+
exit_code=0,
|
|
664
|
+
)
|
|
665
|
+
|
|
666
|
+
output_dir = prepare_output_directory(plan.output_dir, resume=options.resume)
|
|
667
|
+
_load_tools()
|
|
668
|
+
write_back = options.write_back or options.input_file is None
|
|
669
|
+
if write_back:
|
|
670
|
+
from open_pharma_plugins_hcp_intelligence._crm_store import write_enrichment
|
|
671
|
+
else:
|
|
672
|
+
write_enrichment = None
|
|
673
|
+
validated_artifacts: dict[str, dict[str, Any]] = {}
|
|
674
|
+
artifact_lock = Lock()
|
|
675
|
+
|
|
676
|
+
def _remember_artifact(account_id: str, artifact: dict[str, Any]) -> None:
|
|
677
|
+
with artifact_lock:
|
|
678
|
+
validated_artifacts[account_id] = artifact
|
|
679
|
+
|
|
680
|
+
emit(f"Processing {len(accounts)} account(s), concurrency={options.concurrency}")
|
|
681
|
+
if options.synthesize:
|
|
682
|
+
emit(f"Synthesis: {options.model} via {options.base_url}")
|
|
683
|
+
emit(
|
|
684
|
+
f"Reasoning effort: {options.reasoning_effort}; "
|
|
685
|
+
f"timeout: {options.synthesis_timeout_seconds:g}s; SDK retries: 0"
|
|
686
|
+
)
|
|
687
|
+
emit("")
|
|
688
|
+
|
|
689
|
+
started_at = datetime.now(timezone.utc).isoformat()
|
|
690
|
+
|
|
691
|
+
def _persist_writeback(account_id: str, profile_json: str, status: str) -> None:
|
|
692
|
+
if not write_back:
|
|
693
|
+
return
|
|
694
|
+
assert write_enrichment is not None
|
|
695
|
+
with _WRITEBACK_LOCK:
|
|
696
|
+
write_enrichment(account_id, profile_json, status)
|
|
697
|
+
|
|
698
|
+
def _process_account(account: dict) -> dict:
|
|
699
|
+
account_id = account["id"]
|
|
700
|
+
result_file = output_dir / f"{account_id}.json"
|
|
701
|
+
raw: dict | None = None
|
|
702
|
+
writeback_attempted = False
|
|
703
|
+
artifact_write_failed = False
|
|
704
|
+
|
|
705
|
+
if options.resume:
|
|
706
|
+
existing = _load_resume_artifact(result_file, account)
|
|
707
|
+
if existing is None:
|
|
708
|
+
try:
|
|
709
|
+
result_file.lstat()
|
|
710
|
+
except (FileNotFoundError, OSError):
|
|
711
|
+
pass
|
|
712
|
+
else:
|
|
713
|
+
emit(f" [{account_id}] existing artifact is unusable; processing normally")
|
|
714
|
+
else:
|
|
715
|
+
synthesized_profile = existing.get("synthesized_profile")
|
|
716
|
+
profile_validated = False
|
|
717
|
+
if options.synthesize and synthesized_profile:
|
|
718
|
+
try:
|
|
719
|
+
validate_synthesized_profile(json.dumps(synthesized_profile), account["account_type"])
|
|
720
|
+
except (TypeError, ValueError):
|
|
721
|
+
existing.pop("synthesized_profile", None)
|
|
722
|
+
existing.pop("synthesis_error", None)
|
|
723
|
+
raw = existing
|
|
724
|
+
emit(f" [{account_id}] invalid synthesized profile; reusing raw evidence")
|
|
725
|
+
else:
|
|
726
|
+
profile_validated = True
|
|
727
|
+
if not options.synthesize or profile_validated:
|
|
728
|
+
_remember_artifact(account_id, existing)
|
|
729
|
+
emit(f" [{account_id}] skipped (exists)")
|
|
730
|
+
return {
|
|
731
|
+
"account_id": account_id,
|
|
732
|
+
"status": "skipped",
|
|
733
|
+
"tools_failed": existing.get("tools_failed", []),
|
|
734
|
+
"output_file": str(result_file),
|
|
735
|
+
"profile_validated": profile_validated,
|
|
736
|
+
}
|
|
737
|
+
if raw is None:
|
|
738
|
+
raw = existing
|
|
739
|
+
raw.pop("synthesis_error", None)
|
|
740
|
+
emit(f" [{account_id}] reusing raw evidence for synthesis")
|
|
741
|
+
|
|
742
|
+
try:
|
|
743
|
+
if raw is None:
|
|
744
|
+
raw = _enrich_account(account, output_dir)
|
|
745
|
+
try:
|
|
746
|
+
write_batch_json(result_file, raw)
|
|
747
|
+
except Exception:
|
|
748
|
+
artifact_write_failed = True
|
|
749
|
+
raise
|
|
750
|
+
tools_failed = raw.get("tools_failed", [])
|
|
751
|
+
status = "partial" if tools_failed else "completed"
|
|
752
|
+
profile_validated = False
|
|
753
|
+
|
|
754
|
+
if options.synthesize:
|
|
755
|
+
profile_json = _synthesize(raw, options)
|
|
756
|
+
if not profile_json:
|
|
757
|
+
raise ValueError("synthesis returned no profile")
|
|
758
|
+
profile = validate_synthesized_profile(profile_json, account["account_type"])
|
|
759
|
+
raw["synthesized_profile"] = profile
|
|
760
|
+
profile_validated = True
|
|
761
|
+
try:
|
|
762
|
+
write_batch_json(result_file, raw)
|
|
763
|
+
except Exception:
|
|
764
|
+
artifact_write_failed = True
|
|
765
|
+
raise
|
|
766
|
+
writeback_attempted = write_back
|
|
767
|
+
_persist_writeback(account_id, json.dumps(profile), "enriched")
|
|
768
|
+
emit(" synthesized + schema validated ✓")
|
|
769
|
+
else:
|
|
770
|
+
try:
|
|
771
|
+
write_batch_json(result_file, raw)
|
|
772
|
+
except Exception:
|
|
773
|
+
artifact_write_failed = True
|
|
774
|
+
raise
|
|
775
|
+
writeback_attempted = write_back
|
|
776
|
+
_persist_writeback(account_id, json.dumps(raw["search_results"]), "enriched")
|
|
777
|
+
|
|
778
|
+
_remember_artifact(account_id, raw)
|
|
779
|
+
|
|
780
|
+
return {
|
|
781
|
+
"account_id": account_id,
|
|
782
|
+
"status": status,
|
|
783
|
+
"tools_failed": tools_failed,
|
|
784
|
+
"output_file": str(result_file),
|
|
785
|
+
"profile_validated": profile_validated,
|
|
786
|
+
}
|
|
787
|
+
|
|
788
|
+
except Exception as e:
|
|
789
|
+
safe_error = sanitize_error(e)
|
|
790
|
+
if raw is not None and not artifact_write_failed:
|
|
791
|
+
raw["synthesis_error" if options.synthesize else "processing_error"] = safe_error
|
|
792
|
+
try:
|
|
793
|
+
write_batch_json(result_file, raw)
|
|
794
|
+
except Exception as persist_exc:
|
|
795
|
+
artifact_write_failed = True
|
|
796
|
+
safe_error = f"{safe_error}; artifact persistence failed: {sanitize_error(persist_exc)}"
|
|
797
|
+
else:
|
|
798
|
+
_remember_artifact(account_id, raw)
|
|
799
|
+
if write_back and not writeback_attempted:
|
|
800
|
+
try:
|
|
801
|
+
_persist_writeback(
|
|
802
|
+
account_id,
|
|
803
|
+
json.dumps(raw.get("search_results", {})) if raw is not None else "{}",
|
|
804
|
+
"failed",
|
|
805
|
+
)
|
|
806
|
+
except Exception as writeback_exc:
|
|
807
|
+
safe_error = f"{safe_error}; write-back failed: {sanitize_error(writeback_exc)}"
|
|
808
|
+
print(f" [{account_id}] ERROR: {safe_error}", file=sys.stderr)
|
|
809
|
+
return {
|
|
810
|
+
"account_id": account_id,
|
|
811
|
+
"status": "failed",
|
|
812
|
+
"tools_failed": raw.get("tools_failed", []) if raw is not None else [],
|
|
813
|
+
"output_file": str(result_file),
|
|
814
|
+
"profile_validated": False,
|
|
815
|
+
"error": safe_error,
|
|
816
|
+
}
|
|
817
|
+
|
|
818
|
+
def _process(account: dict) -> dict:
|
|
819
|
+
token = _ACTIVE_API_KEY_ENV.set(options.api_key_env)
|
|
820
|
+
try:
|
|
821
|
+
return _process_account(account)
|
|
822
|
+
finally:
|
|
823
|
+
_ACTIVE_API_KEY_ENV.reset(token)
|
|
824
|
+
|
|
825
|
+
results: list[dict] = []
|
|
826
|
+
if options.concurrency == 1:
|
|
827
|
+
for account in accounts:
|
|
828
|
+
results.append(_process(account))
|
|
829
|
+
else:
|
|
830
|
+
with ThreadPoolExecutor(max_workers=options.concurrency) as pool:
|
|
831
|
+
futures = {pool.submit(_process, a): a for a in accounts}
|
|
832
|
+
for future in as_completed(futures):
|
|
833
|
+
try:
|
|
834
|
+
results.append(future.result())
|
|
835
|
+
except Exception as e:
|
|
836
|
+
account = futures[future]
|
|
837
|
+
safe_error = sanitize_error(e, options.api_key_env)
|
|
838
|
+
print(f" [{account['id']}] UNHANDLED: {safe_error}", file=sys.stderr)
|
|
839
|
+
results.append(
|
|
840
|
+
{
|
|
841
|
+
"account_id": account["id"],
|
|
842
|
+
"status": "failed",
|
|
843
|
+
"tools_failed": [],
|
|
844
|
+
"profile_validated": False,
|
|
845
|
+
"error": safe_error,
|
|
846
|
+
}
|
|
847
|
+
)
|
|
848
|
+
|
|
849
|
+
# -- Summary --
|
|
850
|
+
order = {account["id"]: index for index, account in enumerate(accounts)}
|
|
851
|
+
results.sort(key=lambda result: order[result["account_id"]])
|
|
852
|
+
from open_pharma_plugins_hcp_intelligence import batch_csv
|
|
853
|
+
|
|
854
|
+
summary_path = output_dir / batch_csv.SUMMARY_FILENAME
|
|
855
|
+
csv_failed = False
|
|
856
|
+
try:
|
|
857
|
+
rows = batch_csv.build_summary_rows(accounts, results, validated_artifacts)
|
|
858
|
+
summary_csv = batch_csv.write_summary_csv(summary_path, rows)
|
|
859
|
+
except Exception as exc:
|
|
860
|
+
csv_failed = True
|
|
861
|
+
summary_csv = {
|
|
862
|
+
"status": "failed",
|
|
863
|
+
"path": str(summary_path.resolve()),
|
|
864
|
+
"schema_version": batch_csv.CSV_SCHEMA_VERSION,
|
|
865
|
+
"error": f"CSV export failed ({type(exc).__name__})",
|
|
866
|
+
}
|
|
867
|
+
completed_at = datetime.now(timezone.utc).isoformat()
|
|
868
|
+
manifest = build_batch_manifest(
|
|
869
|
+
input_path=plan.input_path,
|
|
870
|
+
input_sha256=plan.input_sha256,
|
|
871
|
+
selected_count=len(accounts),
|
|
872
|
+
synthesis=options.synthesize,
|
|
873
|
+
model=options.model,
|
|
874
|
+
base_url=options.base_url,
|
|
875
|
+
reasoning_effort=options.reasoning_effort,
|
|
876
|
+
synthesis_timeout_seconds=options.synthesis_timeout_seconds,
|
|
877
|
+
concurrency=options.concurrency,
|
|
878
|
+
results=results,
|
|
879
|
+
summary_csv=summary_csv,
|
|
880
|
+
started_at=started_at,
|
|
881
|
+
completed_at=completed_at,
|
|
882
|
+
)
|
|
883
|
+
write_batch_json(output_dir / "batch_manifest.json", manifest)
|
|
884
|
+
summary = manifest["summary"]
|
|
885
|
+
exit_code = 1 if csv_failed or summary["failed"] or summary["partial"] else 0
|
|
886
|
+
return BatchOutcome(
|
|
887
|
+
output_dir=output_dir,
|
|
888
|
+
results=tuple(results),
|
|
889
|
+
manifest=manifest,
|
|
890
|
+
exit_code=exit_code,
|
|
891
|
+
)
|