open-pharma-plugins 2.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. mcp_framework.py +495 -0
  2. open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
  3. open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
  4. open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
  5. open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
  6. open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
  7. open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
  8. open_pharma_plugins_campaign_studio/__init__.py +14 -0
  9. open_pharma_plugins_campaign_studio/__main__.py +11 -0
  10. open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
  11. open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
  12. open_pharma_plugins_campaign_studio/_renderer.py +119 -0
  13. open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
  14. open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
  15. open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
  16. open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
  17. open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
  18. open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
  19. open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
  20. open_pharma_plugins_campaign_studio/models/_common.py +12 -0
  21. open_pharma_plugins_campaign_studio/models/brief.py +72 -0
  22. open_pharma_plugins_campaign_studio/models/claims.py +15 -0
  23. open_pharma_plugins_campaign_studio/models/copy.py +47 -0
  24. open_pharma_plugins_campaign_studio/models/journey.py +21 -0
  25. open_pharma_plugins_campaign_studio/models/message.py +25 -0
  26. open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
  27. open_pharma_plugins_campaign_studio/models/validation.py +32 -0
  28. open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
  29. open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
  30. open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
  31. open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
  32. open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
  33. open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
  34. open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
  35. open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
  36. open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
  37. open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
  38. open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
  39. open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
  40. open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
  41. open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
  42. open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
  43. open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
  44. open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
  45. open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
  46. open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
  47. open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
  48. open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
  49. open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
  50. open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
  51. open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
  52. open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
  53. open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
  54. open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
  55. open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
  56. open_pharma_plugins_competitive_intelligence/models.py +525 -0
  57. open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
  58. open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
  59. open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
  60. open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
  61. open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
  62. open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
  63. open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
  64. open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
  65. open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
  66. open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
  67. open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
  68. open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
  69. open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
  70. open_pharma_plugins_field_training/__init__.py +13 -0
  71. open_pharma_plugins_field_training/__main__.py +11 -0
  72. open_pharma_plugins_field_training/_content_store.py +131 -0
  73. open_pharma_plugins_field_training/_grounding.py +75 -0
  74. open_pharma_plugins_field_training/_html_renderers.py +546 -0
  75. open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
  76. open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
  77. open_pharma_plugins_field_training/models.py +265 -0
  78. open_pharma_plugins_field_training/tools/__init__.py +0 -0
  79. open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
  80. open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
  81. open_pharma_plugins_field_training/tools/list_documents.py +51 -0
  82. open_pharma_plugins_field_training/tools/render_output.py +118 -0
  83. open_pharma_plugins_field_training/tools/search_content.py +57 -0
  84. open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
  85. open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
  86. open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
  87. open_pharma_plugins_hcp_intelligence/batch.py +891 -0
  88. open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
  89. open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
  90. open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
  91. open_pharma_plugins_hcp_intelligence/models.py +356 -0
  92. open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
  93. open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
  94. open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
  95. open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
  96. open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
  97. open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
  98. open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
  99. open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
  100. open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
  101. open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
  102. open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
  103. open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
  104. open_pharma_plugins_next_best_engagement/__init__.py +14 -0
  105. open_pharma_plugins_next_best_engagement/__main__.py +11 -0
  106. open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
  107. open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
  108. open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
  109. open_pharma_plugins_next_best_engagement/_universe.py +145 -0
  110. open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
  111. open_pharma_plugins_next_best_engagement/models.py +135 -0
  112. open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
  113. open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
  114. open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
  115. open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
  116. open_pharma_plugins_territory_alignment/__init__.py +14 -0
  117. open_pharma_plugins_territory_alignment/__main__.py +11 -0
  118. open_pharma_plugins_territory_alignment/data.py +300 -0
  119. open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
  120. open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
  121. open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
  122. open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
  123. open_pharma_plugins_territory_alignment/geo.py +175 -0
  124. open_pharma_plugins_territory_alignment/models.py +201 -0
  125. open_pharma_plugins_territory_alignment/scoring.py +125 -0
  126. open_pharma_plugins_territory_alignment/solver.py +504 -0
  127. open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
  128. open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
  129. open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
  130. open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
  131. open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
  132. open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
  133. open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
  134. shared/__init__.py +11 -0
  135. shared/env.py +217 -0
  136. shared/filesystem.py +110 -0
@@ -0,0 +1,221 @@
1
+ """Batch enrichment for hcp-intelligence accounts.
2
+
3
+ Calls tool handle() functions directly (no MCP server) to gather data from
4
+ PubMed, ClinicalTrials.gov, ORCID, NIH RePORTER, and web search, then
5
+ optionally synthesizes structured profiles through OpenRouter's OpenAI-compatible API.
6
+
7
+ Usage:
8
+ # Dry run — list what would be processed
9
+ python3 scripts/batch_enrich.py --dry-run
10
+
11
+ # Enrich one account (raw data only)
12
+ python3 scripts/batch_enrich.py --ids HCP-AU-001
13
+
14
+ # Enrich all Singapore HCPs with schema-validated DeepSeek synthesis via OpenRouter
15
+ python3 scripts/batch_enrich.py --country Singapore --account-type HCP \\
16
+ --synthesize
17
+
18
+ # Process a user-curated CSV into a private output directory
19
+ python3 scripts/batch_enrich.py --input-file ./accounts.csv \\
20
+ --output-dir ./data/hcp-intelligence --synthesize
21
+
22
+ # Resume after interruption
23
+ python3 scripts/batch_enrich.py --resume --synthesize
24
+ """
25
+
26
+ from __future__ import annotations
27
+
28
+ import argparse
29
+ import sys
30
+ from typing import Literal, Sequence, cast
31
+
32
+ from open_pharma_plugins_hcp_intelligence.batch import (
33
+ DEFAULT_OPENROUTER_BASE_URL,
34
+ DEFAULT_REASONING_EFFORT,
35
+ DEFAULT_SYNTHESIS_MODEL,
36
+ DEFAULT_SYNTHESIS_TIMEOUT_SECONDS,
37
+ HCO_TOOLS,
38
+ HCP_TOOLS,
39
+ BatchOptions,
40
+ BatchOutcome,
41
+ BatchPlan,
42
+ BatchUsageError,
43
+ plan_batch,
44
+ run_batch,
45
+ )
46
+
47
+
48
+ def build_parser() -> argparse.ArgumentParser:
49
+ parser = argparse.ArgumentParser(
50
+ description="Batch enrichment for hcp-intelligence accounts.",
51
+ formatter_class=argparse.RawDescriptionHelpFormatter,
52
+ epilog=__doc__,
53
+ )
54
+
55
+ filtering = parser.add_argument_group("filtering")
56
+ filtering.add_argument(
57
+ "--input-file",
58
+ help="Account CSV using the bundled fixture columns (default: bundled sample_accounts.csv)",
59
+ )
60
+ filtering.add_argument("--country", help="Filter by country")
61
+ filtering.add_argument(
62
+ "--account-type",
63
+ type=str.upper,
64
+ choices=("HCP", "HCO"),
65
+ help="Filter by HCP or HCO",
66
+ )
67
+ filtering.add_argument("--ids", nargs="+", help="Process specific account IDs only")
68
+
69
+ execution = parser.add_argument_group("execution")
70
+ execution.add_argument("--concurrency", type=int, default=5, help="Parallel accounts (default: 5)")
71
+ execution.add_argument("--resume", action="store_true", help="Skip accounts whose output JSON already exists")
72
+ execution.add_argument("--dry-run", action="store_true", help="List accounts without processing")
73
+ execution.add_argument(
74
+ "--write-back",
75
+ action="store_true",
76
+ help="Persist results to the demo enrichment store (off by default for --input-file)",
77
+ )
78
+
79
+ synthesis = parser.add_argument_group("synthesis")
80
+ synthesis.add_argument("--synthesize", action="store_true", help="Enable LLM profile synthesis")
81
+ synthesis.add_argument(
82
+ "--base-url",
83
+ default=None,
84
+ help="OpenAI-compatible API base URL (default: OPENROUTER_BASE_URL or OpenRouter)",
85
+ )
86
+ synthesis.add_argument(
87
+ "--api-key-env",
88
+ default="OPENROUTER_API_KEY",
89
+ help="Config/environment key holding the API key (default: OPENROUTER_API_KEY)",
90
+ )
91
+ synthesis.add_argument(
92
+ "--model",
93
+ default=DEFAULT_SYNTHESIS_MODEL,
94
+ help=f"Model name (default: {DEFAULT_SYNTHESIS_MODEL})",
95
+ )
96
+ synthesis.add_argument(
97
+ "--reasoning-effort",
98
+ choices=("high", "xhigh"),
99
+ default=DEFAULT_REASONING_EFFORT,
100
+ help=f"Reasoning effort for extraction/synthesis (default: {DEFAULT_REASONING_EFFORT})",
101
+ )
102
+ synthesis.add_argument(
103
+ "--synthesis-timeout-seconds",
104
+ type=float,
105
+ default=DEFAULT_SYNTHESIS_TIMEOUT_SECONDS,
106
+ help=f"Per-request synthesis timeout (default: {DEFAULT_SYNTHESIS_TIMEOUT_SECONDS:g})",
107
+ )
108
+
109
+ output = parser.add_argument_group("output")
110
+ output.add_argument(
111
+ "--output-dir",
112
+ default="./batch_output",
113
+ help="Directory for raw results (default: ./batch_output)",
114
+ )
115
+ return parser
116
+
117
+
118
+ def _options_from_args(args: argparse.Namespace) -> BatchOptions:
119
+ from shared.env import get_env
120
+
121
+ base_url = args.base_url or get_env("OPENROUTER_BASE_URL", DEFAULT_OPENROUTER_BASE_URL)
122
+ return BatchOptions(
123
+ input_file=args.input_file,
124
+ output_dir=args.output_dir,
125
+ country=args.country,
126
+ account_type=args.account_type,
127
+ ids=tuple(args.ids or ()),
128
+ concurrency=args.concurrency,
129
+ resume=args.resume,
130
+ write_back=args.write_back,
131
+ synthesize=args.synthesize,
132
+ base_url=base_url or DEFAULT_OPENROUTER_BASE_URL,
133
+ api_key_env=args.api_key_env,
134
+ model=args.model,
135
+ reasoning_effort=cast(Literal["high", "xhigh"], args.reasoning_effort),
136
+ synthesis_timeout_seconds=args.synthesis_timeout_seconds,
137
+ )
138
+
139
+
140
+ def render_dry_run(plan: BatchPlan) -> None:
141
+ hcp_count = sum(account["account_type"] == "HCP" for account in plan.accounts)
142
+ hco_count = sum(account["account_type"] == "HCO" for account in plan.accounts)
143
+ print(f"Input: {plan.input_path}")
144
+ print(f"Output: {plan.output_dir}")
145
+ print(f"Selected: {len(plan.accounts)} total (HCP: {hcp_count}, HCO: {hco_count})")
146
+ print(f"Would process {len(plan.accounts)} account(s):\n")
147
+ for account in plan.accounts:
148
+ tools = HCP_TOOLS if account.get("account_type", "").upper() == "HCP" else HCO_TOOLS
149
+ print(
150
+ f" {account['id']:16s} {account['name']:40s} "
151
+ f"{account['account_type']:4s} {account['country']:12s} ({len(tools)} tools)"
152
+ )
153
+ if plan.options.synthesize:
154
+ print(f"\nSynthesis: {plan.options.model} via {plan.options.base_url}")
155
+ print(f"Provider: {plan.options.base_url}")
156
+ print(f"Model: {plan.options.model}")
157
+ print(
158
+ f"Reasoning effort: {plan.options.reasoning_effort}; "
159
+ f"timeout: {plan.options.synthesis_timeout_seconds:g}s; SDK retries: 0"
160
+ )
161
+ print(f"Timeout: {plan.options.synthesis_timeout_seconds:g}s")
162
+ else:
163
+ print("\nSynthesis: disabled")
164
+ print(f"Provider: {plan.options.base_url}")
165
+ print(f"Model: {plan.options.model}")
166
+ print(f"Reasoning effort: {plan.options.reasoning_effort}")
167
+ print(f"Timeout: {plan.options.synthesis_timeout_seconds:g}s")
168
+ print("SDK retries: 0")
169
+
170
+ print("\nPlanned artifacts:")
171
+ for account in plan.accounts:
172
+ print(f" {account['id']}.json")
173
+ print(" batch_summary.csv")
174
+ print(" batch_manifest.json")
175
+ print("\nNo external calls were made.")
176
+ print("Data sharing: execution would send search query terms to configured public data providers.")
177
+ if plan.options.synthesize:
178
+ print(
179
+ "Data sharing (synthesis): selected account fields and gathered evidence "
180
+ f"would be sent to {plan.options.base_url}."
181
+ )
182
+ if not plan.accounts:
183
+ print("No accounts match the filters.")
184
+
185
+
186
+ def render_outcome(outcome: BatchOutcome) -> None:
187
+ if not outcome.manifest:
188
+ print("No accounts match the filters.")
189
+ return
190
+
191
+ summary = outcome.manifest["summary"]
192
+ print(
193
+ "\nDone. "
194
+ f"completed={summary['completed']} partial={summary['partial']} "
195
+ f"failed={summary['failed']} skipped={summary['skipped']}"
196
+ )
197
+ summary_csv = outcome.manifest["outputs"]["summary_csv"]
198
+ print(f"Summary CSV: {summary_csv['path']}")
199
+ print(f"Manifest: {(outcome.output_dir / 'batch_manifest.json').resolve()}")
200
+ print(f"Output directory: {outcome.output_dir.resolve()}")
201
+ if summary_csv["status"] == "failed":
202
+ print(summary_csv["error"], file=sys.stderr)
203
+
204
+
205
+ def main(argv: Sequence[str] | None = None) -> int:
206
+ parser = build_parser()
207
+ args = parser.parse_args(argv)
208
+ try:
209
+ plan = plan_batch(_options_from_args(args))
210
+ if args.dry_run:
211
+ render_dry_run(plan)
212
+ return 0
213
+ outcome = run_batch(plan)
214
+ except BatchUsageError as exc:
215
+ parser.error(str(exc))
216
+ render_outcome(outcome)
217
+ return outcome.exit_code
218
+
219
+
220
+ if __name__ == "__main__":
221
+ raise SystemExit(main())
@@ -0,0 +1,206 @@
1
+ """Stable CSV projection for HCP Intelligence batch artifacts."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import codecs
6
+ import csv
7
+ import hashlib
8
+ import io
9
+ from collections.abc import Iterable, Mapping, Sequence
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from pydantic import ValidationError
14
+
15
+ from open_pharma_plugins_hcp_intelligence.models import HcoProfile, HcpProfile
16
+ from shared.filesystem import atomic_write_bytes
17
+
18
+ CSV_SCHEMA_VERSION = 1
19
+ SUMMARY_FILENAME = "batch_summary.csv"
20
+ SUMMARY_COLUMNS = (
21
+ "account_id",
22
+ "account_type",
23
+ "input_name",
24
+ "input_specialty",
25
+ "input_country",
26
+ "input_institution",
27
+ "status",
28
+ "profile_validated",
29
+ "profile_completeness",
30
+ "profile_name",
31
+ "profile_specialty",
32
+ "profile_country",
33
+ "current_title",
34
+ "organization_type",
35
+ "affiliations",
36
+ "qualifications",
37
+ "research_or_clinical_focus",
38
+ "professional_roles",
39
+ "key_publication_count",
40
+ "clinical_trial_count",
41
+ "active_grant_count",
42
+ "congress_activity_count",
43
+ "source_count",
44
+ "source_urls",
45
+ "tools_failed",
46
+ "error",
47
+ "json_file",
48
+ )
49
+
50
+ _NUMERIC_COLUMNS = {
51
+ "profile_completeness",
52
+ "key_publication_count",
53
+ "clinical_trial_count",
54
+ "active_grant_count",
55
+ "congress_activity_count",
56
+ "source_count",
57
+ }
58
+
59
+
60
+ def _join_unique(values: Iterable[str]) -> str:
61
+ return " | ".join(dict.fromkeys(value for value in values if value))
62
+
63
+
64
+ def _safe_text(value: object) -> str:
65
+ text = "" if value is None else str(value)
66
+ return f"'{text}" if text.startswith(("=", "+", "-", "@")) else text
67
+
68
+
69
+ def _result_values(value: object) -> Iterable[str]:
70
+ if isinstance(value, str):
71
+ return (value,)
72
+ if isinstance(value, Sequence):
73
+ return (str(item) for item in value if item is not None)
74
+ return ()
75
+
76
+
77
+ def _blank_row(account: Mapping[str, str], result: Mapping[str, Any]) -> dict[str, str | int | float]:
78
+ account_id = account["id"]
79
+ return {
80
+ "account_id": account_id,
81
+ "account_type": account["account_type"],
82
+ "input_name": account["name"],
83
+ "input_specialty": account.get("specialty", ""),
84
+ "input_country": account["country"],
85
+ "input_institution": account.get("institution", ""),
86
+ "status": result.get("status", ""),
87
+ "profile_validated": "false",
88
+ "profile_completeness": "",
89
+ "profile_name": "",
90
+ "profile_specialty": "",
91
+ "profile_country": "",
92
+ "current_title": "",
93
+ "organization_type": "",
94
+ "affiliations": "",
95
+ "qualifications": "",
96
+ "research_or_clinical_focus": "",
97
+ "professional_roles": "",
98
+ "key_publication_count": "",
99
+ "clinical_trial_count": "",
100
+ "active_grant_count": "",
101
+ "congress_activity_count": "",
102
+ "source_count": "",
103
+ "source_urls": "",
104
+ "tools_failed": _join_unique(_result_values(result.get("tools_failed", []))),
105
+ "error": result.get("error", ""),
106
+ "json_file": f"{account_id}.json",
107
+ }
108
+
109
+
110
+ def _populate_hcp(row: dict[str, str | int | float], profile: HcpProfile) -> None:
111
+ row.update(
112
+ {
113
+ "profile_validated": "true",
114
+ "profile_completeness": profile.profile_completeness,
115
+ "profile_name": profile.full_name,
116
+ "profile_specialty": profile.specialty,
117
+ "profile_country": profile.country,
118
+ "current_title": profile.current_title.value if profile.current_title else "",
119
+ "affiliations": _join_unique(claim.value for claim in profile.affiliations),
120
+ "qualifications": _join_unique(claim.value for claim in profile.qualifications),
121
+ "research_or_clinical_focus": _join_unique(claim.value for claim in profile.research_interests),
122
+ "professional_roles": _join_unique(claim.value for claim in profile.professional_roles),
123
+ "key_publication_count": len(profile.key_publications),
124
+ "clinical_trial_count": len(profile.clinical_trial_involvement),
125
+ "active_grant_count": len(profile.active_grants),
126
+ "congress_activity_count": len(profile.congress_activity),
127
+ }
128
+ )
129
+
130
+
131
+ def _populate_hco(row: dict[str, str | int | float], profile: HcoProfile) -> None:
132
+ row.update(
133
+ {
134
+ "profile_validated": "true",
135
+ "profile_completeness": profile.profile_completeness,
136
+ "profile_name": profile.name,
137
+ "profile_country": profile.country,
138
+ "organization_type": profile.organization_type.value if profile.organization_type else "",
139
+ "affiliations": _join_unique(claim.value for claim in profile.notable_affiliations),
140
+ "qualifications": _join_unique(claim.value for claim in profile.accreditations),
141
+ "research_or_clinical_focus": _join_unique(
142
+ claim.value for claim in (*profile.clinical_focus_areas, *profile.research_focus)
143
+ ),
144
+ "key_publication_count": 0,
145
+ "clinical_trial_count": len(profile.active_clinical_trials),
146
+ "active_grant_count": len(profile.institutional_grants),
147
+ "congress_activity_count": 0,
148
+ }
149
+ )
150
+
151
+
152
+ def build_summary_rows(
153
+ accounts: Sequence[Mapping[str, str]],
154
+ results: Sequence[Mapping[str, Any]],
155
+ artifacts: Mapping[str, Mapping[str, Any]],
156
+ ) -> list[dict[str, str | int | float]]:
157
+ """Build rows from artifacts already ownership-validated by the batch engine."""
158
+ results_by_id = {result["account_id"]: result for result in results}
159
+ rows: list[dict[str, str | int | float]] = []
160
+ for account in accounts:
161
+ account_id = account["id"]
162
+ result = results_by_id.get(account_id, {})
163
+ row = _blank_row(account, result)
164
+ artifact = artifacts.get(account_id, {})
165
+ raw_profile = artifact.get("synthesized_profile")
166
+ if raw_profile is not None:
167
+ try:
168
+ if account["account_type"] == "HCP":
169
+ profile = HcpProfile.model_validate(raw_profile)
170
+ _populate_hcp(row, profile)
171
+ sources = profile.sources_consulted
172
+ else:
173
+ hco_profile = HcoProfile.model_validate(raw_profile)
174
+ _populate_hco(row, hco_profile)
175
+ sources = hco_profile.sources_consulted
176
+ except ValidationError:
177
+ pass
178
+ else:
179
+ source_urls = list(dict.fromkeys(source.url for source in sources if source.url))
180
+ row["source_count"] = len(source_urls)
181
+ row["source_urls"] = _join_unique(source_urls)
182
+
183
+ rows.append(
184
+ {
185
+ column: row[column] if column in _NUMERIC_COLUMNS else _safe_text(row[column])
186
+ for column in SUMMARY_COLUMNS
187
+ }
188
+ )
189
+ return rows
190
+
191
+
192
+ def write_summary_csv(path: Path, rows: Sequence[Mapping[str, object]]) -> dict[str, object]:
193
+ """Atomically write the stable UTF-8 BOM CSV and return exact-byte metadata."""
194
+ buffer = io.StringIO(newline="")
195
+ writer = csv.DictWriter(buffer, fieldnames=SUMMARY_COLUMNS, lineterminator="\r\n")
196
+ writer.writeheader()
197
+ writer.writerows(rows)
198
+ payload = codecs.BOM_UTF8 + buffer.getvalue().encode("utf-8")
199
+ written = atomic_write_bytes(path, payload)
200
+ return {
201
+ "status": "completed",
202
+ "path": str(written.resolve()),
203
+ "schema_version": CSV_SCHEMA_VERSION,
204
+ "row_count": len(rows),
205
+ "sha256": hashlib.sha256(payload).hexdigest(),
206
+ }
@@ -0,0 +1,27 @@
1
+ id,name,specialty,country,account_type,institution
2
+ HCP-AU-001,Sarah Chen,Medical Oncology,Australia,HCP,Peter MacCallum Cancer Centre
3
+ HCP-AU-002,James Mitchell,Radiation Oncology,Australia,HCP,Chris O'Brien Lifehouse
4
+ HCP-AU-003,Priya Sharma,Haematology-Oncology,Australia,HCP,Royal Melbourne Hospital
5
+ HCP-AU-004,David Lee,Surgical Oncology,Australia,HCP,Westmead Hospital
6
+ HCP-AU-005,Emily Watson,Paediatric Oncology,Australia,HCP,The Children's Hospital at Westmead
7
+ HCP-AU-006,Michael Zhou,Medical Oncology,Australia,HCP,Royal Prince Alfred Hospital
8
+ HCP-AU-007,Rachel O'Brien,Breast Oncology,Australia,HCP,Mater Hospital Brisbane
9
+ HCP-AU-008,Andrew Tan,Immuno-Oncology,Australia,HCP,Olivia Newton-John Cancer Centre
10
+ HCP-AU-009,Lisa Nguyen,Gynaecological Oncology,Australia,HCP,Royal Women's Hospital Melbourne
11
+ HCP-AU-010,Thomas Baker,Thoracic Oncology,Australia,HCP,The Prince Charles Hospital
12
+ HCP-SG-001,Wei Lin Tan,Medical Oncology,Singapore,HCP,National Cancer Centre Singapore
13
+ HCP-SG-002,David Lee,Radiation Oncology,Singapore,HCP,National University Cancer Institute
14
+ HCP-SG-003,Mei Ling Wong,Haematology-Oncology,Singapore,HCP,Singapore General Hospital
15
+ HCP-SG-004,Rajesh Kumar,Surgical Oncology,Singapore,HCP,National University Hospital
16
+ HCP-SG-005,Nur Aisyah Rahman,Paediatric Oncology,Singapore,HCP,KK Women's and Children's Hospital
17
+ HCP-SG-006,Cheng Wei Lim,Medical Oncology,Singapore,HCP,Parkway Cancer Centre
18
+ HCP-SG-007,Hui Fen Ong,Breast Oncology,Singapore,HCP,National Cancer Centre Singapore
19
+ HCP-SG-008,Arun Patel,Thoracic Oncology,Singapore,HCP,National University Cancer Institute
20
+ HCO-AU-001,Peter MacCallum Cancer Centre,Oncology,Australia,HCO,
21
+ HCO-AU-002,Chris O'Brien Lifehouse,Oncology,Australia,HCO,
22
+ HCO-AU-003,Olivia Newton-John Cancer Wellness and Research Centre,Oncology,Australia,HCO,
23
+ HCO-AU-004,Royal Melbourne Hospital,General - Oncology Services,Australia,HCO,
24
+ HCO-SG-001,National Cancer Centre Singapore,Oncology,Singapore,HCO,
25
+ HCO-SG-002,National University Cancer Institute Singapore,Oncology,Singapore,HCO,
26
+ HCO-SG-003,Parkway Cancer Centre,Oncology,Singapore,HCO,
27
+ HCO-SG-004,Mount Elizabeth Hospital,General - Oncology Department,Singapore,HCO,