open-pharma-plugins 2.2.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- mcp_framework.py +495 -0
- open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
- open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
- open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
- open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
- open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
- open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
- open_pharma_plugins_campaign_studio/__init__.py +14 -0
- open_pharma_plugins_campaign_studio/__main__.py +11 -0
- open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
- open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
- open_pharma_plugins_campaign_studio/_renderer.py +119 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
- open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
- open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
- open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
- open_pharma_plugins_campaign_studio/models/_common.py +12 -0
- open_pharma_plugins_campaign_studio/models/brief.py +72 -0
- open_pharma_plugins_campaign_studio/models/claims.py +15 -0
- open_pharma_plugins_campaign_studio/models/copy.py +47 -0
- open_pharma_plugins_campaign_studio/models/journey.py +21 -0
- open_pharma_plugins_campaign_studio/models/message.py +25 -0
- open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
- open_pharma_plugins_campaign_studio/models/validation.py +32 -0
- open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
- open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
- open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
- open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
- open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
- open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
- open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
- open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
- open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
- open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
- open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
- open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
- open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
- open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
- open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
- open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
- open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
- open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
- open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
- open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
- open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
- open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
- open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
- open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
- open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
- open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
- open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
- open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
- open_pharma_plugins_competitive_intelligence/models.py +525 -0
- open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
- open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
- open_pharma_plugins_field_training/__init__.py +13 -0
- open_pharma_plugins_field_training/__main__.py +11 -0
- open_pharma_plugins_field_training/_content_store.py +131 -0
- open_pharma_plugins_field_training/_grounding.py +75 -0
- open_pharma_plugins_field_training/_html_renderers.py +546 -0
- open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
- open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
- open_pharma_plugins_field_training/models.py +265 -0
- open_pharma_plugins_field_training/tools/__init__.py +0 -0
- open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
- open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
- open_pharma_plugins_field_training/tools/list_documents.py +51 -0
- open_pharma_plugins_field_training/tools/render_output.py +118 -0
- open_pharma_plugins_field_training/tools/search_content.py +57 -0
- open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
- open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
- open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
- open_pharma_plugins_hcp_intelligence/batch.py +891 -0
- open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
- open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
- open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
- open_pharma_plugins_hcp_intelligence/models.py +356 -0
- open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
- open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
- open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
- open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
- open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
- open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
- open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
- open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
- open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
- open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
- open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
- open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
- open_pharma_plugins_next_best_engagement/__init__.py +14 -0
- open_pharma_plugins_next_best_engagement/__main__.py +11 -0
- open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
- open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
- open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
- open_pharma_plugins_next_best_engagement/_universe.py +145 -0
- open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
- open_pharma_plugins_next_best_engagement/models.py +135 -0
- open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
- open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
- open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
- open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
- open_pharma_plugins_territory_alignment/__init__.py +14 -0
- open_pharma_plugins_territory_alignment/__main__.py +11 -0
- open_pharma_plugins_territory_alignment/data.py +300 -0
- open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
- open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
- open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
- open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
- open_pharma_plugins_territory_alignment/geo.py +175 -0
- open_pharma_plugins_territory_alignment/models.py +201 -0
- open_pharma_plugins_territory_alignment/scoring.py +125 -0
- open_pharma_plugins_territory_alignment/solver.py +504 -0
- open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
- open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
- open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
- open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
- open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
- open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
- open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
- shared/__init__.py +11 -0
- shared/env.py +217 -0
- shared/filesystem.py +110 -0
|
@@ -0,0 +1,221 @@
|
|
|
1
|
+
"""Batch enrichment for hcp-intelligence accounts.
|
|
2
|
+
|
|
3
|
+
Calls tool handle() functions directly (no MCP server) to gather data from
|
|
4
|
+
PubMed, ClinicalTrials.gov, ORCID, NIH RePORTER, and web search, then
|
|
5
|
+
optionally synthesizes structured profiles through OpenRouter's OpenAI-compatible API.
|
|
6
|
+
|
|
7
|
+
Usage:
|
|
8
|
+
# Dry run — list what would be processed
|
|
9
|
+
python3 scripts/batch_enrich.py --dry-run
|
|
10
|
+
|
|
11
|
+
# Enrich one account (raw data only)
|
|
12
|
+
python3 scripts/batch_enrich.py --ids HCP-AU-001
|
|
13
|
+
|
|
14
|
+
# Enrich all Singapore HCPs with schema-validated DeepSeek synthesis via OpenRouter
|
|
15
|
+
python3 scripts/batch_enrich.py --country Singapore --account-type HCP \\
|
|
16
|
+
--synthesize
|
|
17
|
+
|
|
18
|
+
# Process a user-curated CSV into a private output directory
|
|
19
|
+
python3 scripts/batch_enrich.py --input-file ./accounts.csv \\
|
|
20
|
+
--output-dir ./data/hcp-intelligence --synthesize
|
|
21
|
+
|
|
22
|
+
# Resume after interruption
|
|
23
|
+
python3 scripts/batch_enrich.py --resume --synthesize
|
|
24
|
+
"""
|
|
25
|
+
|
|
26
|
+
from __future__ import annotations
|
|
27
|
+
|
|
28
|
+
import argparse
|
|
29
|
+
import sys
|
|
30
|
+
from typing import Literal, Sequence, cast
|
|
31
|
+
|
|
32
|
+
from open_pharma_plugins_hcp_intelligence.batch import (
|
|
33
|
+
DEFAULT_OPENROUTER_BASE_URL,
|
|
34
|
+
DEFAULT_REASONING_EFFORT,
|
|
35
|
+
DEFAULT_SYNTHESIS_MODEL,
|
|
36
|
+
DEFAULT_SYNTHESIS_TIMEOUT_SECONDS,
|
|
37
|
+
HCO_TOOLS,
|
|
38
|
+
HCP_TOOLS,
|
|
39
|
+
BatchOptions,
|
|
40
|
+
BatchOutcome,
|
|
41
|
+
BatchPlan,
|
|
42
|
+
BatchUsageError,
|
|
43
|
+
plan_batch,
|
|
44
|
+
run_batch,
|
|
45
|
+
)
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def build_parser() -> argparse.ArgumentParser:
|
|
49
|
+
parser = argparse.ArgumentParser(
|
|
50
|
+
description="Batch enrichment for hcp-intelligence accounts.",
|
|
51
|
+
formatter_class=argparse.RawDescriptionHelpFormatter,
|
|
52
|
+
epilog=__doc__,
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
filtering = parser.add_argument_group("filtering")
|
|
56
|
+
filtering.add_argument(
|
|
57
|
+
"--input-file",
|
|
58
|
+
help="Account CSV using the bundled fixture columns (default: bundled sample_accounts.csv)",
|
|
59
|
+
)
|
|
60
|
+
filtering.add_argument("--country", help="Filter by country")
|
|
61
|
+
filtering.add_argument(
|
|
62
|
+
"--account-type",
|
|
63
|
+
type=str.upper,
|
|
64
|
+
choices=("HCP", "HCO"),
|
|
65
|
+
help="Filter by HCP or HCO",
|
|
66
|
+
)
|
|
67
|
+
filtering.add_argument("--ids", nargs="+", help="Process specific account IDs only")
|
|
68
|
+
|
|
69
|
+
execution = parser.add_argument_group("execution")
|
|
70
|
+
execution.add_argument("--concurrency", type=int, default=5, help="Parallel accounts (default: 5)")
|
|
71
|
+
execution.add_argument("--resume", action="store_true", help="Skip accounts whose output JSON already exists")
|
|
72
|
+
execution.add_argument("--dry-run", action="store_true", help="List accounts without processing")
|
|
73
|
+
execution.add_argument(
|
|
74
|
+
"--write-back",
|
|
75
|
+
action="store_true",
|
|
76
|
+
help="Persist results to the demo enrichment store (off by default for --input-file)",
|
|
77
|
+
)
|
|
78
|
+
|
|
79
|
+
synthesis = parser.add_argument_group("synthesis")
|
|
80
|
+
synthesis.add_argument("--synthesize", action="store_true", help="Enable LLM profile synthesis")
|
|
81
|
+
synthesis.add_argument(
|
|
82
|
+
"--base-url",
|
|
83
|
+
default=None,
|
|
84
|
+
help="OpenAI-compatible API base URL (default: OPENROUTER_BASE_URL or OpenRouter)",
|
|
85
|
+
)
|
|
86
|
+
synthesis.add_argument(
|
|
87
|
+
"--api-key-env",
|
|
88
|
+
default="OPENROUTER_API_KEY",
|
|
89
|
+
help="Config/environment key holding the API key (default: OPENROUTER_API_KEY)",
|
|
90
|
+
)
|
|
91
|
+
synthesis.add_argument(
|
|
92
|
+
"--model",
|
|
93
|
+
default=DEFAULT_SYNTHESIS_MODEL,
|
|
94
|
+
help=f"Model name (default: {DEFAULT_SYNTHESIS_MODEL})",
|
|
95
|
+
)
|
|
96
|
+
synthesis.add_argument(
|
|
97
|
+
"--reasoning-effort",
|
|
98
|
+
choices=("high", "xhigh"),
|
|
99
|
+
default=DEFAULT_REASONING_EFFORT,
|
|
100
|
+
help=f"Reasoning effort for extraction/synthesis (default: {DEFAULT_REASONING_EFFORT})",
|
|
101
|
+
)
|
|
102
|
+
synthesis.add_argument(
|
|
103
|
+
"--synthesis-timeout-seconds",
|
|
104
|
+
type=float,
|
|
105
|
+
default=DEFAULT_SYNTHESIS_TIMEOUT_SECONDS,
|
|
106
|
+
help=f"Per-request synthesis timeout (default: {DEFAULT_SYNTHESIS_TIMEOUT_SECONDS:g})",
|
|
107
|
+
)
|
|
108
|
+
|
|
109
|
+
output = parser.add_argument_group("output")
|
|
110
|
+
output.add_argument(
|
|
111
|
+
"--output-dir",
|
|
112
|
+
default="./batch_output",
|
|
113
|
+
help="Directory for raw results (default: ./batch_output)",
|
|
114
|
+
)
|
|
115
|
+
return parser
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def _options_from_args(args: argparse.Namespace) -> BatchOptions:
|
|
119
|
+
from shared.env import get_env
|
|
120
|
+
|
|
121
|
+
base_url = args.base_url or get_env("OPENROUTER_BASE_URL", DEFAULT_OPENROUTER_BASE_URL)
|
|
122
|
+
return BatchOptions(
|
|
123
|
+
input_file=args.input_file,
|
|
124
|
+
output_dir=args.output_dir,
|
|
125
|
+
country=args.country,
|
|
126
|
+
account_type=args.account_type,
|
|
127
|
+
ids=tuple(args.ids or ()),
|
|
128
|
+
concurrency=args.concurrency,
|
|
129
|
+
resume=args.resume,
|
|
130
|
+
write_back=args.write_back,
|
|
131
|
+
synthesize=args.synthesize,
|
|
132
|
+
base_url=base_url or DEFAULT_OPENROUTER_BASE_URL,
|
|
133
|
+
api_key_env=args.api_key_env,
|
|
134
|
+
model=args.model,
|
|
135
|
+
reasoning_effort=cast(Literal["high", "xhigh"], args.reasoning_effort),
|
|
136
|
+
synthesis_timeout_seconds=args.synthesis_timeout_seconds,
|
|
137
|
+
)
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
def render_dry_run(plan: BatchPlan) -> None:
|
|
141
|
+
hcp_count = sum(account["account_type"] == "HCP" for account in plan.accounts)
|
|
142
|
+
hco_count = sum(account["account_type"] == "HCO" for account in plan.accounts)
|
|
143
|
+
print(f"Input: {plan.input_path}")
|
|
144
|
+
print(f"Output: {plan.output_dir}")
|
|
145
|
+
print(f"Selected: {len(plan.accounts)} total (HCP: {hcp_count}, HCO: {hco_count})")
|
|
146
|
+
print(f"Would process {len(plan.accounts)} account(s):\n")
|
|
147
|
+
for account in plan.accounts:
|
|
148
|
+
tools = HCP_TOOLS if account.get("account_type", "").upper() == "HCP" else HCO_TOOLS
|
|
149
|
+
print(
|
|
150
|
+
f" {account['id']:16s} {account['name']:40s} "
|
|
151
|
+
f"{account['account_type']:4s} {account['country']:12s} ({len(tools)} tools)"
|
|
152
|
+
)
|
|
153
|
+
if plan.options.synthesize:
|
|
154
|
+
print(f"\nSynthesis: {plan.options.model} via {plan.options.base_url}")
|
|
155
|
+
print(f"Provider: {plan.options.base_url}")
|
|
156
|
+
print(f"Model: {plan.options.model}")
|
|
157
|
+
print(
|
|
158
|
+
f"Reasoning effort: {plan.options.reasoning_effort}; "
|
|
159
|
+
f"timeout: {plan.options.synthesis_timeout_seconds:g}s; SDK retries: 0"
|
|
160
|
+
)
|
|
161
|
+
print(f"Timeout: {plan.options.synthesis_timeout_seconds:g}s")
|
|
162
|
+
else:
|
|
163
|
+
print("\nSynthesis: disabled")
|
|
164
|
+
print(f"Provider: {plan.options.base_url}")
|
|
165
|
+
print(f"Model: {plan.options.model}")
|
|
166
|
+
print(f"Reasoning effort: {plan.options.reasoning_effort}")
|
|
167
|
+
print(f"Timeout: {plan.options.synthesis_timeout_seconds:g}s")
|
|
168
|
+
print("SDK retries: 0")
|
|
169
|
+
|
|
170
|
+
print("\nPlanned artifacts:")
|
|
171
|
+
for account in plan.accounts:
|
|
172
|
+
print(f" {account['id']}.json")
|
|
173
|
+
print(" batch_summary.csv")
|
|
174
|
+
print(" batch_manifest.json")
|
|
175
|
+
print("\nNo external calls were made.")
|
|
176
|
+
print("Data sharing: execution would send search query terms to configured public data providers.")
|
|
177
|
+
if plan.options.synthesize:
|
|
178
|
+
print(
|
|
179
|
+
"Data sharing (synthesis): selected account fields and gathered evidence "
|
|
180
|
+
f"would be sent to {plan.options.base_url}."
|
|
181
|
+
)
|
|
182
|
+
if not plan.accounts:
|
|
183
|
+
print("No accounts match the filters.")
|
|
184
|
+
|
|
185
|
+
|
|
186
|
+
def render_outcome(outcome: BatchOutcome) -> None:
|
|
187
|
+
if not outcome.manifest:
|
|
188
|
+
print("No accounts match the filters.")
|
|
189
|
+
return
|
|
190
|
+
|
|
191
|
+
summary = outcome.manifest["summary"]
|
|
192
|
+
print(
|
|
193
|
+
"\nDone. "
|
|
194
|
+
f"completed={summary['completed']} partial={summary['partial']} "
|
|
195
|
+
f"failed={summary['failed']} skipped={summary['skipped']}"
|
|
196
|
+
)
|
|
197
|
+
summary_csv = outcome.manifest["outputs"]["summary_csv"]
|
|
198
|
+
print(f"Summary CSV: {summary_csv['path']}")
|
|
199
|
+
print(f"Manifest: {(outcome.output_dir / 'batch_manifest.json').resolve()}")
|
|
200
|
+
print(f"Output directory: {outcome.output_dir.resolve()}")
|
|
201
|
+
if summary_csv["status"] == "failed":
|
|
202
|
+
print(summary_csv["error"], file=sys.stderr)
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def main(argv: Sequence[str] | None = None) -> int:
|
|
206
|
+
parser = build_parser()
|
|
207
|
+
args = parser.parse_args(argv)
|
|
208
|
+
try:
|
|
209
|
+
plan = plan_batch(_options_from_args(args))
|
|
210
|
+
if args.dry_run:
|
|
211
|
+
render_dry_run(plan)
|
|
212
|
+
return 0
|
|
213
|
+
outcome = run_batch(plan)
|
|
214
|
+
except BatchUsageError as exc:
|
|
215
|
+
parser.error(str(exc))
|
|
216
|
+
render_outcome(outcome)
|
|
217
|
+
return outcome.exit_code
|
|
218
|
+
|
|
219
|
+
|
|
220
|
+
if __name__ == "__main__":
|
|
221
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Stable CSV projection for HCP Intelligence batch artifacts."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import codecs
|
|
6
|
+
import csv
|
|
7
|
+
import hashlib
|
|
8
|
+
import io
|
|
9
|
+
from collections.abc import Iterable, Mapping, Sequence
|
|
10
|
+
from pathlib import Path
|
|
11
|
+
from typing import Any
|
|
12
|
+
|
|
13
|
+
from pydantic import ValidationError
|
|
14
|
+
|
|
15
|
+
from open_pharma_plugins_hcp_intelligence.models import HcoProfile, HcpProfile
|
|
16
|
+
from shared.filesystem import atomic_write_bytes
|
|
17
|
+
|
|
18
|
+
CSV_SCHEMA_VERSION = 1
|
|
19
|
+
SUMMARY_FILENAME = "batch_summary.csv"
|
|
20
|
+
SUMMARY_COLUMNS = (
|
|
21
|
+
"account_id",
|
|
22
|
+
"account_type",
|
|
23
|
+
"input_name",
|
|
24
|
+
"input_specialty",
|
|
25
|
+
"input_country",
|
|
26
|
+
"input_institution",
|
|
27
|
+
"status",
|
|
28
|
+
"profile_validated",
|
|
29
|
+
"profile_completeness",
|
|
30
|
+
"profile_name",
|
|
31
|
+
"profile_specialty",
|
|
32
|
+
"profile_country",
|
|
33
|
+
"current_title",
|
|
34
|
+
"organization_type",
|
|
35
|
+
"affiliations",
|
|
36
|
+
"qualifications",
|
|
37
|
+
"research_or_clinical_focus",
|
|
38
|
+
"professional_roles",
|
|
39
|
+
"key_publication_count",
|
|
40
|
+
"clinical_trial_count",
|
|
41
|
+
"active_grant_count",
|
|
42
|
+
"congress_activity_count",
|
|
43
|
+
"source_count",
|
|
44
|
+
"source_urls",
|
|
45
|
+
"tools_failed",
|
|
46
|
+
"error",
|
|
47
|
+
"json_file",
|
|
48
|
+
)
|
|
49
|
+
|
|
50
|
+
_NUMERIC_COLUMNS = {
|
|
51
|
+
"profile_completeness",
|
|
52
|
+
"key_publication_count",
|
|
53
|
+
"clinical_trial_count",
|
|
54
|
+
"active_grant_count",
|
|
55
|
+
"congress_activity_count",
|
|
56
|
+
"source_count",
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _join_unique(values: Iterable[str]) -> str:
|
|
61
|
+
return " | ".join(dict.fromkeys(value for value in values if value))
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
def _safe_text(value: object) -> str:
|
|
65
|
+
text = "" if value is None else str(value)
|
|
66
|
+
return f"'{text}" if text.startswith(("=", "+", "-", "@")) else text
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def _result_values(value: object) -> Iterable[str]:
|
|
70
|
+
if isinstance(value, str):
|
|
71
|
+
return (value,)
|
|
72
|
+
if isinstance(value, Sequence):
|
|
73
|
+
return (str(item) for item in value if item is not None)
|
|
74
|
+
return ()
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _blank_row(account: Mapping[str, str], result: Mapping[str, Any]) -> dict[str, str | int | float]:
|
|
78
|
+
account_id = account["id"]
|
|
79
|
+
return {
|
|
80
|
+
"account_id": account_id,
|
|
81
|
+
"account_type": account["account_type"],
|
|
82
|
+
"input_name": account["name"],
|
|
83
|
+
"input_specialty": account.get("specialty", ""),
|
|
84
|
+
"input_country": account["country"],
|
|
85
|
+
"input_institution": account.get("institution", ""),
|
|
86
|
+
"status": result.get("status", ""),
|
|
87
|
+
"profile_validated": "false",
|
|
88
|
+
"profile_completeness": "",
|
|
89
|
+
"profile_name": "",
|
|
90
|
+
"profile_specialty": "",
|
|
91
|
+
"profile_country": "",
|
|
92
|
+
"current_title": "",
|
|
93
|
+
"organization_type": "",
|
|
94
|
+
"affiliations": "",
|
|
95
|
+
"qualifications": "",
|
|
96
|
+
"research_or_clinical_focus": "",
|
|
97
|
+
"professional_roles": "",
|
|
98
|
+
"key_publication_count": "",
|
|
99
|
+
"clinical_trial_count": "",
|
|
100
|
+
"active_grant_count": "",
|
|
101
|
+
"congress_activity_count": "",
|
|
102
|
+
"source_count": "",
|
|
103
|
+
"source_urls": "",
|
|
104
|
+
"tools_failed": _join_unique(_result_values(result.get("tools_failed", []))),
|
|
105
|
+
"error": result.get("error", ""),
|
|
106
|
+
"json_file": f"{account_id}.json",
|
|
107
|
+
}
|
|
108
|
+
|
|
109
|
+
|
|
110
|
+
def _populate_hcp(row: dict[str, str | int | float], profile: HcpProfile) -> None:
|
|
111
|
+
row.update(
|
|
112
|
+
{
|
|
113
|
+
"profile_validated": "true",
|
|
114
|
+
"profile_completeness": profile.profile_completeness,
|
|
115
|
+
"profile_name": profile.full_name,
|
|
116
|
+
"profile_specialty": profile.specialty,
|
|
117
|
+
"profile_country": profile.country,
|
|
118
|
+
"current_title": profile.current_title.value if profile.current_title else "",
|
|
119
|
+
"affiliations": _join_unique(claim.value for claim in profile.affiliations),
|
|
120
|
+
"qualifications": _join_unique(claim.value for claim in profile.qualifications),
|
|
121
|
+
"research_or_clinical_focus": _join_unique(claim.value for claim in profile.research_interests),
|
|
122
|
+
"professional_roles": _join_unique(claim.value for claim in profile.professional_roles),
|
|
123
|
+
"key_publication_count": len(profile.key_publications),
|
|
124
|
+
"clinical_trial_count": len(profile.clinical_trial_involvement),
|
|
125
|
+
"active_grant_count": len(profile.active_grants),
|
|
126
|
+
"congress_activity_count": len(profile.congress_activity),
|
|
127
|
+
}
|
|
128
|
+
)
|
|
129
|
+
|
|
130
|
+
|
|
131
|
+
def _populate_hco(row: dict[str, str | int | float], profile: HcoProfile) -> None:
|
|
132
|
+
row.update(
|
|
133
|
+
{
|
|
134
|
+
"profile_validated": "true",
|
|
135
|
+
"profile_completeness": profile.profile_completeness,
|
|
136
|
+
"profile_name": profile.name,
|
|
137
|
+
"profile_country": profile.country,
|
|
138
|
+
"organization_type": profile.organization_type.value if profile.organization_type else "",
|
|
139
|
+
"affiliations": _join_unique(claim.value for claim in profile.notable_affiliations),
|
|
140
|
+
"qualifications": _join_unique(claim.value for claim in profile.accreditations),
|
|
141
|
+
"research_or_clinical_focus": _join_unique(
|
|
142
|
+
claim.value for claim in (*profile.clinical_focus_areas, *profile.research_focus)
|
|
143
|
+
),
|
|
144
|
+
"key_publication_count": 0,
|
|
145
|
+
"clinical_trial_count": len(profile.active_clinical_trials),
|
|
146
|
+
"active_grant_count": len(profile.institutional_grants),
|
|
147
|
+
"congress_activity_count": 0,
|
|
148
|
+
}
|
|
149
|
+
)
|
|
150
|
+
|
|
151
|
+
|
|
152
|
+
def build_summary_rows(
|
|
153
|
+
accounts: Sequence[Mapping[str, str]],
|
|
154
|
+
results: Sequence[Mapping[str, Any]],
|
|
155
|
+
artifacts: Mapping[str, Mapping[str, Any]],
|
|
156
|
+
) -> list[dict[str, str | int | float]]:
|
|
157
|
+
"""Build rows from artifacts already ownership-validated by the batch engine."""
|
|
158
|
+
results_by_id = {result["account_id"]: result for result in results}
|
|
159
|
+
rows: list[dict[str, str | int | float]] = []
|
|
160
|
+
for account in accounts:
|
|
161
|
+
account_id = account["id"]
|
|
162
|
+
result = results_by_id.get(account_id, {})
|
|
163
|
+
row = _blank_row(account, result)
|
|
164
|
+
artifact = artifacts.get(account_id, {})
|
|
165
|
+
raw_profile = artifact.get("synthesized_profile")
|
|
166
|
+
if raw_profile is not None:
|
|
167
|
+
try:
|
|
168
|
+
if account["account_type"] == "HCP":
|
|
169
|
+
profile = HcpProfile.model_validate(raw_profile)
|
|
170
|
+
_populate_hcp(row, profile)
|
|
171
|
+
sources = profile.sources_consulted
|
|
172
|
+
else:
|
|
173
|
+
hco_profile = HcoProfile.model_validate(raw_profile)
|
|
174
|
+
_populate_hco(row, hco_profile)
|
|
175
|
+
sources = hco_profile.sources_consulted
|
|
176
|
+
except ValidationError:
|
|
177
|
+
pass
|
|
178
|
+
else:
|
|
179
|
+
source_urls = list(dict.fromkeys(source.url for source in sources if source.url))
|
|
180
|
+
row["source_count"] = len(source_urls)
|
|
181
|
+
row["source_urls"] = _join_unique(source_urls)
|
|
182
|
+
|
|
183
|
+
rows.append(
|
|
184
|
+
{
|
|
185
|
+
column: row[column] if column in _NUMERIC_COLUMNS else _safe_text(row[column])
|
|
186
|
+
for column in SUMMARY_COLUMNS
|
|
187
|
+
}
|
|
188
|
+
)
|
|
189
|
+
return rows
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
def write_summary_csv(path: Path, rows: Sequence[Mapping[str, object]]) -> dict[str, object]:
|
|
193
|
+
"""Atomically write the stable UTF-8 BOM CSV and return exact-byte metadata."""
|
|
194
|
+
buffer = io.StringIO(newline="")
|
|
195
|
+
writer = csv.DictWriter(buffer, fieldnames=SUMMARY_COLUMNS, lineterminator="\r\n")
|
|
196
|
+
writer.writeheader()
|
|
197
|
+
writer.writerows(rows)
|
|
198
|
+
payload = codecs.BOM_UTF8 + buffer.getvalue().encode("utf-8")
|
|
199
|
+
written = atomic_write_bytes(path, payload)
|
|
200
|
+
return {
|
|
201
|
+
"status": "completed",
|
|
202
|
+
"path": str(written.resolve()),
|
|
203
|
+
"schema_version": CSV_SCHEMA_VERSION,
|
|
204
|
+
"row_count": len(rows),
|
|
205
|
+
"sha256": hashlib.sha256(payload).hexdigest(),
|
|
206
|
+
}
|
|
@@ -0,0 +1,27 @@
|
|
|
1
|
+
id,name,specialty,country,account_type,institution
|
|
2
|
+
HCP-AU-001,Sarah Chen,Medical Oncology,Australia,HCP,Peter MacCallum Cancer Centre
|
|
3
|
+
HCP-AU-002,James Mitchell,Radiation Oncology,Australia,HCP,Chris O'Brien Lifehouse
|
|
4
|
+
HCP-AU-003,Priya Sharma,Haematology-Oncology,Australia,HCP,Royal Melbourne Hospital
|
|
5
|
+
HCP-AU-004,David Lee,Surgical Oncology,Australia,HCP,Westmead Hospital
|
|
6
|
+
HCP-AU-005,Emily Watson,Paediatric Oncology,Australia,HCP,The Children's Hospital at Westmead
|
|
7
|
+
HCP-AU-006,Michael Zhou,Medical Oncology,Australia,HCP,Royal Prince Alfred Hospital
|
|
8
|
+
HCP-AU-007,Rachel O'Brien,Breast Oncology,Australia,HCP,Mater Hospital Brisbane
|
|
9
|
+
HCP-AU-008,Andrew Tan,Immuno-Oncology,Australia,HCP,Olivia Newton-John Cancer Centre
|
|
10
|
+
HCP-AU-009,Lisa Nguyen,Gynaecological Oncology,Australia,HCP,Royal Women's Hospital Melbourne
|
|
11
|
+
HCP-AU-010,Thomas Baker,Thoracic Oncology,Australia,HCP,The Prince Charles Hospital
|
|
12
|
+
HCP-SG-001,Wei Lin Tan,Medical Oncology,Singapore,HCP,National Cancer Centre Singapore
|
|
13
|
+
HCP-SG-002,David Lee,Radiation Oncology,Singapore,HCP,National University Cancer Institute
|
|
14
|
+
HCP-SG-003,Mei Ling Wong,Haematology-Oncology,Singapore,HCP,Singapore General Hospital
|
|
15
|
+
HCP-SG-004,Rajesh Kumar,Surgical Oncology,Singapore,HCP,National University Hospital
|
|
16
|
+
HCP-SG-005,Nur Aisyah Rahman,Paediatric Oncology,Singapore,HCP,KK Women's and Children's Hospital
|
|
17
|
+
HCP-SG-006,Cheng Wei Lim,Medical Oncology,Singapore,HCP,Parkway Cancer Centre
|
|
18
|
+
HCP-SG-007,Hui Fen Ong,Breast Oncology,Singapore,HCP,National Cancer Centre Singapore
|
|
19
|
+
HCP-SG-008,Arun Patel,Thoracic Oncology,Singapore,HCP,National University Cancer Institute
|
|
20
|
+
HCO-AU-001,Peter MacCallum Cancer Centre,Oncology,Australia,HCO,
|
|
21
|
+
HCO-AU-002,Chris O'Brien Lifehouse,Oncology,Australia,HCO,
|
|
22
|
+
HCO-AU-003,Olivia Newton-John Cancer Wellness and Research Centre,Oncology,Australia,HCO,
|
|
23
|
+
HCO-AU-004,Royal Melbourne Hospital,General - Oncology Services,Australia,HCO,
|
|
24
|
+
HCO-SG-001,National Cancer Centre Singapore,Oncology,Singapore,HCO,
|
|
25
|
+
HCO-SG-002,National University Cancer Institute Singapore,Oncology,Singapore,HCO,
|
|
26
|
+
HCO-SG-003,Parkway Cancer Centre,Oncology,Singapore,HCO,
|
|
27
|
+
HCO-SG-004,Mount Elizabeth Hospital,General - Oncology Department,Singapore,HCO,
|