open-pharma-plugins 2.2.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (136) hide show
  1. mcp_framework.py +495 -0
  2. open_pharma_plugins-2.2.0.dist-info/METADATA +135 -0
  3. open_pharma_plugins-2.2.0.dist-info/RECORD +136 -0
  4. open_pharma_plugins-2.2.0.dist-info/WHEEL +5 -0
  5. open_pharma_plugins-2.2.0.dist-info/entry_points.txt +8 -0
  6. open_pharma_plugins-2.2.0.dist-info/licenses/LICENSE +202 -0
  7. open_pharma_plugins-2.2.0.dist-info/top_level.txt +8 -0
  8. open_pharma_plugins_campaign_studio/__init__.py +14 -0
  9. open_pharma_plugins_campaign_studio/__main__.py +11 -0
  10. open_pharma_plugins_campaign_studio/_campaign_store.py +162 -0
  11. open_pharma_plugins_campaign_studio/_claim_engine.py +262 -0
  12. open_pharma_plugins_campaign_studio/_renderer.py +119 -0
  13. open_pharma_plugins_campaign_studio/fixtures/brand_kit/legal.json +21 -0
  14. open_pharma_plugins_campaign_studio/fixtures/brand_kit/logo.svg +4 -0
  15. open_pharma_plugins_campaign_studio/fixtures/brand_kit/palette.json +11 -0
  16. open_pharma_plugins_campaign_studio/fixtures/brand_kit/product.png +1 -0
  17. open_pharma_plugins_campaign_studio/fixtures/brand_kit/typography.json +14 -0
  18. open_pharma_plugins_campaign_studio/fixtures/sample_approved_claims.json +119 -0
  19. open_pharma_plugins_campaign_studio/models/__init__.py +34 -0
  20. open_pharma_plugins_campaign_studio/models/_common.py +12 -0
  21. open_pharma_plugins_campaign_studio/models/brief.py +72 -0
  22. open_pharma_plugins_campaign_studio/models/claims.py +15 -0
  23. open_pharma_plugins_campaign_studio/models/copy.py +47 -0
  24. open_pharma_plugins_campaign_studio/models/journey.py +21 -0
  25. open_pharma_plugins_campaign_studio/models/message.py +25 -0
  26. open_pharma_plugins_campaign_studio/models/mlr.py +29 -0
  27. open_pharma_plugins_campaign_studio/models/validation.py +32 -0
  28. open_pharma_plugins_campaign_studio/policy/rules.json +79 -0
  29. open_pharma_plugins_campaign_studio/templates/banner.svg.j2 +26 -0
  30. open_pharma_plugins_campaign_studio/templates/email.html.j2 +54 -0
  31. open_pharma_plugins_campaign_studio/tools/__init__.py +0 -0
  32. open_pharma_plugins_campaign_studio/tools/create_campaign_brief.py +212 -0
  33. open_pharma_plugins_campaign_studio/tools/generate_audience_journey.py +129 -0
  34. open_pharma_plugins_campaign_studio/tools/generate_channel_copy.py +199 -0
  35. open_pharma_plugins_campaign_studio/tools/generate_message_architecture.py +121 -0
  36. open_pharma_plugins_campaign_studio/tools/package_mlr_submission.py +230 -0
  37. open_pharma_plugins_campaign_studio/tools/render_banner.py +99 -0
  38. open_pharma_plugins_campaign_studio/tools/render_email.py +101 -0
  39. open_pharma_plugins_campaign_studio/tools/render_poster.py +222 -0
  40. open_pharma_plugins_campaign_studio/tools/retrieve_approved_claims.py +71 -0
  41. open_pharma_plugins_campaign_studio/tools/retrieve_brand_components.py +76 -0
  42. open_pharma_plugins_campaign_studio/tools/validate_claims_and_fair_balance.py +285 -0
  43. open_pharma_plugins_competitive_intelligence/__init__.py +13 -0
  44. open_pharma_plugins_competitive_intelligence/__main__.py +11 -0
  45. open_pharma_plugins_competitive_intelligence/_artifacts.py +87 -0
  46. open_pharma_plugins_competitive_intelligence/_cache.py +144 -0
  47. open_pharma_plugins_competitive_intelligence/_clinical_trials.py +569 -0
  48. open_pharma_plugins_competitive_intelligence/_dailymed.py +260 -0
  49. open_pharma_plugins_competitive_intelligence/_fda.py +255 -0
  50. open_pharma_plugins_competitive_intelligence/_pubmed.py +342 -0
  51. open_pharma_plugins_competitive_intelligence/_regulatory.py +140 -0
  52. open_pharma_plugins_competitive_intelligence/_runs.py +278 -0
  53. open_pharma_plugins_competitive_intelligence/_transport.py +83 -0
  54. open_pharma_plugins_competitive_intelligence/_watchlist.py +113 -0
  55. open_pharma_plugins_competitive_intelligence/_web_search.py +331 -0
  56. open_pharma_plugins_competitive_intelligence/models.py +525 -0
  57. open_pharma_plugins_competitive_intelligence/tools/__init__.py +0 -0
  58. open_pharma_plugins_competitive_intelligence/tools/ci_extract_events.py +247 -0
  59. open_pharma_plugins_competitive_intelligence/tools/ci_landscape.py +195 -0
  60. open_pharma_plugins_competitive_intelligence/tools/ci_refresh.py +101 -0
  61. open_pharma_plugins_competitive_intelligence/tools/ci_report.py +402 -0
  62. open_pharma_plugins_competitive_intelligence/tools/ci_scan_news.py +48 -0
  63. open_pharma_plugins_competitive_intelligence/tools/ci_scan_publications.py +46 -0
  64. open_pharma_plugins_competitive_intelligence/tools/ci_scan_regulatory.py +64 -0
  65. open_pharma_plugins_competitive_intelligence/tools/ci_scan_trials.py +85 -0
  66. open_pharma_plugins_competitive_intelligence/tools/ci_status.py +106 -0
  67. open_pharma_plugins_competitive_intelligence/tools/ci_timeline.py +453 -0
  68. open_pharma_plugins_competitive_intelligence/tools/ci_track.py +121 -0
  69. open_pharma_plugins_competitive_intelligence/tools/ci_trial_detail.py +55 -0
  70. open_pharma_plugins_field_training/__init__.py +13 -0
  71. open_pharma_plugins_field_training/__main__.py +11 -0
  72. open_pharma_plugins_field_training/_content_store.py +131 -0
  73. open_pharma_plugins_field_training/_grounding.py +75 -0
  74. open_pharma_plugins_field_training/_html_renderers.py +546 -0
  75. open_pharma_plugins_field_training/fixtures/sample_product_message.pdf +156 -0
  76. open_pharma_plugins_field_training/fixtures/sample_training_deck.pptx +0 -0
  77. open_pharma_plugins_field_training/models.py +265 -0
  78. open_pharma_plugins_field_training/tools/__init__.py +0 -0
  79. open_pharma_plugins_field_training/tools/get_document_page.py +67 -0
  80. open_pharma_plugins_field_training/tools/ingest_document.py +147 -0
  81. open_pharma_plugins_field_training/tools/list_documents.py +51 -0
  82. open_pharma_plugins_field_training/tools/render_output.py +118 -0
  83. open_pharma_plugins_field_training/tools/search_content.py +57 -0
  84. open_pharma_plugins_hcp_intelligence/__init__.py +14 -0
  85. open_pharma_plugins_hcp_intelligence/__main__.py +11 -0
  86. open_pharma_plugins_hcp_intelligence/_crm_store.py +70 -0
  87. open_pharma_plugins_hcp_intelligence/batch.py +891 -0
  88. open_pharma_plugins_hcp_intelligence/batch_cli.py +221 -0
  89. open_pharma_plugins_hcp_intelligence/batch_csv.py +206 -0
  90. open_pharma_plugins_hcp_intelligence/fixtures/sample_accounts.csv +27 -0
  91. open_pharma_plugins_hcp_intelligence/models.py +356 -0
  92. open_pharma_plugins_hcp_intelligence/tools/__init__.py +0 -0
  93. open_pharma_plugins_hcp_intelligence/tools/get_account.py +48 -0
  94. open_pharma_plugins_hcp_intelligence/tools/list_accounts.py +66 -0
  95. open_pharma_plugins_hcp_intelligence/tools/search_clinical_trials.py +175 -0
  96. open_pharma_plugins_hcp_intelligence/tools/search_congresses.py +134 -0
  97. open_pharma_plugins_hcp_intelligence/tools/search_grants.py +181 -0
  98. open_pharma_plugins_hcp_intelligence/tools/search_guidelines.py +238 -0
  99. open_pharma_plugins_hcp_intelligence/tools/search_hco_web.py +73 -0
  100. open_pharma_plugins_hcp_intelligence/tools/search_hcp_web.py +199 -0
  101. open_pharma_plugins_hcp_intelligence/tools/search_orcid.py +214 -0
  102. open_pharma_plugins_hcp_intelligence/tools/search_publications.py +207 -0
  103. open_pharma_plugins_hcp_intelligence/tools/update_account.py +84 -0
  104. open_pharma_plugins_next_best_engagement/__init__.py +14 -0
  105. open_pharma_plugins_next_best_engagement/__main__.py +11 -0
  106. open_pharma_plugins_next_best_engagement/_optimizer.py +400 -0
  107. open_pharma_plugins_next_best_engagement/_renderer.py +149 -0
  108. open_pharma_plugins_next_best_engagement/_scoring.py +82 -0
  109. open_pharma_plugins_next_best_engagement/_universe.py +145 -0
  110. open_pharma_plugins_next_best_engagement/fixtures/sample_universe.csv +81 -0
  111. open_pharma_plugins_next_best_engagement/models.py +135 -0
  112. open_pharma_plugins_next_best_engagement/tools/__init__.py +0 -0
  113. open_pharma_plugins_next_best_engagement/tools/load_universe.py +47 -0
  114. open_pharma_plugins_next_best_engagement/tools/recommend_engagements.py +80 -0
  115. open_pharma_plugins_next_best_engagement/tools/render_plan.py +90 -0
  116. open_pharma_plugins_territory_alignment/__init__.py +14 -0
  117. open_pharma_plugins_territory_alignment/__main__.py +11 -0
  118. open_pharma_plugins_territory_alignment/data.py +300 -0
  119. open_pharma_plugins_territory_alignment/fixtures/constraints.csv +11 -0
  120. open_pharma_plugins_territory_alignment/fixtures/current_alignment.csv +81 -0
  121. open_pharma_plugins_territory_alignment/fixtures/hcps.csv +81 -0
  122. open_pharma_plugins_territory_alignment/fixtures/reps.csv +9 -0
  123. open_pharma_plugins_territory_alignment/geo.py +175 -0
  124. open_pharma_plugins_territory_alignment/models.py +201 -0
  125. open_pharma_plugins_territory_alignment/scoring.py +125 -0
  126. open_pharma_plugins_territory_alignment/solver.py +504 -0
  127. open_pharma_plugins_territory_alignment/tools/__init__.py +0 -0
  128. open_pharma_plugins_territory_alignment/tools/ta_align.py +138 -0
  129. open_pharma_plugins_territory_alignment/tools/ta_cluster.py +247 -0
  130. open_pharma_plugins_territory_alignment/tools/ta_compare.py +173 -0
  131. open_pharma_plugins_territory_alignment/tools/ta_evaluate.py +119 -0
  132. open_pharma_plugins_territory_alignment/tools/ta_status.py +34 -0
  133. open_pharma_plugins_territory_alignment/tools/ta_visualize.py +504 -0
  134. shared/__init__.py +11 -0
  135. shared/env.py +217 -0
  136. shared/filesystem.py +110 -0
@@ -0,0 +1,285 @@
1
+ """validate_claims_and_fair_balance — compliance gate before rendering."""
2
+
3
+ from __future__ import annotations
4
+
5
+ from typing import Any
6
+
7
+ from pydantic import BaseModel, Field
8
+
9
+
10
+ class ValidateClaimsArgs(BaseModel):
11
+ campaign_brief_id: str = Field(description="Links to the campaign brief")
12
+ channels: list[str] | None = Field(
13
+ default=None,
14
+ description="Validate specific channels; all if omitted",
15
+ )
16
+
17
+
18
+ TOOL: dict[str, Any] = {
19
+ "name": "validate_claims_and_fair_balance",
20
+ "description": (
21
+ "Run the compliance gate on generated channel copy. Reads copy and "
22
+ "approved claims from the campaign directory and performs three checks: "
23
+ "(1) claim grounding — fuzzy-match each copy statement against approved "
24
+ "claims; (2) fair balance — ratio of safety to efficacy content; "
25
+ "(3) policy compliance — jurisdiction-required elements, prohibited "
26
+ "language patterns. Persists claim-map.json, policy-checks.json, and "
27
+ "source-evidence.json. ALL checks must pass before rendering."
28
+ ),
29
+ "args": ValidateClaimsArgs,
30
+ }
31
+
32
+
33
+ def handle(arguments: dict[str, Any]) -> list[dict[str, Any]]:
34
+ import json
35
+ from datetime import datetime, timezone
36
+
37
+ from .._campaign_store import (
38
+ load_artifact,
39
+ load_brief,
40
+ save_validation_artifact,
41
+ )
42
+ from .._claim_engine import (
43
+ check_fair_balance,
44
+ check_prohibited_language,
45
+ check_required_elements,
46
+ fuzzy_match,
47
+ is_claim_citation_exempt,
48
+ load_policy_rules,
49
+ )
50
+ from .._renderer import validation_input_fingerprint
51
+
52
+ campaign_brief_id = arguments["campaign_brief_id"]
53
+ requested_channels = arguments.get("channels")
54
+
55
+ brief = load_brief(campaign_brief_id)
56
+ if not brief:
57
+ return [
58
+ {
59
+ "type": "text",
60
+ "text": json.dumps({"error": f"Campaign brief '{campaign_brief_id}' not found."}),
61
+ }
62
+ ]
63
+
64
+ claims_data = load_artifact(campaign_brief_id, "approved-claims.json")
65
+ if not claims_data:
66
+ return [
67
+ {
68
+ "type": "text",
69
+ "text": json.dumps({"error": "No approved-claims.json found. Run retrieve_approved_claims first."}),
70
+ }
71
+ ]
72
+
73
+ channels = requested_channels or brief.get("channels", [])
74
+ jurisdiction = brief.get("policy_jurisdiction", "FDA")
75
+ mode = brief.get("mode", "promotional")
76
+
77
+ rules = load_policy_rules(jurisdiction)
78
+
79
+ all_claim_results: list[dict] = []
80
+ all_policy_checks: list[dict] = []
81
+ all_copy_blocks: list[dict] = []
82
+ all_text_parts: list[str] = []
83
+ claim_by_id = {claim.get("claim_id"): claim for claim in claims_data}
84
+ used_claim_ids: set[str] = set()
85
+ brand_kit = _load_brand_legal(campaign_brief_id)
86
+
87
+ for channel in channels:
88
+ copy_artifact = load_artifact(campaign_brief_id, f"copy-{channel}.json")
89
+ if not copy_artifact:
90
+ all_policy_checks.append(
91
+ {
92
+ "check_name": f"copy_{channel}_exists",
93
+ "result": "fail",
94
+ "detail": f"No copy-{channel}.json found. Run generate_channel_copy first.",
95
+ }
96
+ )
97
+ continue
98
+
99
+ copy_data = copy_artifact.get("copy", {})
100
+ blocks = _extract_all_blocks(copy_data, channel)
101
+ all_copy_blocks.extend(blocks)
102
+
103
+ for block_name, block in blocks:
104
+ text = block.get("text", "")
105
+ all_text_parts.append(text)
106
+
107
+ claim_ids = block.get("claim_ids", [])
108
+ if not claim_ids:
109
+ if mode == "promotional" and text and not is_claim_citation_exempt(block_name, text, brief, brand_kit):
110
+ all_policy_checks.append(
111
+ {
112
+ "check_name": "missing_claim_citation",
113
+ "result": "fail",
114
+ "detail": f"{channel}.{block_name}: promotional copy has no approved claim ID.",
115
+ }
116
+ )
117
+ continue
118
+
119
+ for claim_id in claim_ids:
120
+ claim = claim_by_id.get(claim_id)
121
+ used_claim_ids.add(claim_id)
122
+ if claim is None:
123
+ all_claim_results.append(
124
+ {
125
+ "claim_id": claim_id,
126
+ "declared_claim_id": claim_id,
127
+ "statement": text,
128
+ "status": "not_found",
129
+ "matched_claim_text": None,
130
+ "similarity_score": 0.0,
131
+ "deviation": "Declared claim ID is not in the approved claims set",
132
+ }
133
+ )
134
+ continue
135
+ match_result = fuzzy_match(text, [claim])
136
+ match_result["declared_claim_id"] = claim_id
137
+ match_result["statement"] = text
138
+ all_claim_results.append(match_result)
139
+
140
+ combined_text = " ".join(all_text_parts)
141
+
142
+ # Fair balance check
143
+ fb = check_fair_balance(
144
+ [b for _, b in _extract_flat_blocks(channels, campaign_brief_id)],
145
+ claims_data,
146
+ rules.get("min_safety_ratio", 0.3),
147
+ )
148
+ all_policy_checks.append(fb)
149
+
150
+ # Prohibited language check
151
+ prohibited = rules.get("prohibited_patterns", [])
152
+ all_policy_checks.extend(check_prohibited_language(combined_text, prohibited))
153
+
154
+ # Non-promotional language check
155
+ if mode in ("non_promotional", "disease_awareness"):
156
+ np_prohibited = rules.get("non_promotional_prohibited", [])
157
+ all_policy_checks.extend(check_prohibited_language(combined_text, np_prohibited))
158
+
159
+ # Required elements check
160
+ required = rules.get("required_elements", [])
161
+ all_policy_checks.extend(check_required_elements(combined_text, brand_kit, required))
162
+
163
+ overall_pass = all(c.get("result") != "fail" for c in all_policy_checks) and all(
164
+ c.get("status") == "approved" for c in all_claim_results
165
+ )
166
+
167
+ # Build source evidence
168
+ source_evidence = []
169
+ seen_sources: set[str] = set()
170
+ for claim in claims_data:
171
+ if claim.get("claim_id") not in used_claim_ids:
172
+ continue
173
+ key = f"{claim.get('source_document')}:{claim.get('source_reference')}"
174
+ if key not in seen_sources:
175
+ seen_sources.add(key)
176
+ source_evidence.append(
177
+ {
178
+ "document_id": claim.get("claim_id", ""),
179
+ "document_name": claim.get("source_document", ""),
180
+ "page_number": None,
181
+ "excerpt": claim.get("source_reference", ""),
182
+ }
183
+ )
184
+
185
+ # Build claim map
186
+ claim_map: dict[str, list[str]] = {}
187
+ for result in all_claim_results:
188
+ stmt = result.get("statement", "")[:60]
189
+ cid = result.get("declared_claim_id")
190
+ if cid:
191
+ claim_map.setdefault(stmt, []).append(cid)
192
+
193
+ # Persist
194
+ report = {
195
+ "campaign_brief_id": campaign_brief_id,
196
+ "channels_validated": channels,
197
+ "claims_checked": all_claim_results,
198
+ "policy_checks": all_policy_checks,
199
+ "overall_pass": overall_pass,
200
+ "input_fingerprint": validation_input_fingerprint(campaign_brief_id, channels),
201
+ "generated_at": datetime.now(timezone.utc).isoformat(),
202
+ }
203
+
204
+ save_validation_artifact(campaign_brief_id, "policy-checks.json", report)
205
+ save_validation_artifact(campaign_brief_id, "claim-map.json", claim_map)
206
+ save_validation_artifact(campaign_brief_id, "source-evidence.json", source_evidence)
207
+
208
+ summary = {
209
+ "campaign_brief_id": campaign_brief_id,
210
+ "overall_pass": overall_pass,
211
+ "channels_validated": channels,
212
+ "claims_total": len(all_claim_results),
213
+ "claims_approved": sum(1 for c in all_claim_results if c.get("status") == "approved"),
214
+ "claims_needs_review": sum(1 for c in all_claim_results if c.get("status") == "needs_review"),
215
+ "claims_not_found": sum(1 for c in all_claim_results if c.get("status") == "not_found"),
216
+ "policy_pass": sum(1 for c in all_policy_checks if c.get("result") == "pass"),
217
+ "policy_warn": sum(1 for c in all_policy_checks if c.get("result") == "warn"),
218
+ "policy_fail": sum(1 for c in all_policy_checks if c.get("result") == "fail"),
219
+ "failures": [c for c in all_policy_checks if c.get("result") == "fail"]
220
+ + [c for c in all_claim_results if c.get("status") != "approved"],
221
+ }
222
+ return [{"type": "text", "text": json.dumps(summary, indent=2)}]
223
+
224
+
225
+ def _extract_all_blocks(copy_data: dict, channel: str) -> list[tuple[str, dict]]:
226
+ """Extract named copy blocks from channel-specific copy data."""
227
+ blocks: list[tuple[str, dict]] = []
228
+
229
+ simple = {
230
+ "email": ["subject", "preheader", "headline", "cta"],
231
+ "banner": ["headline", "sub_headline", "cta"],
232
+ "poster": ["headline", "subhead", "cta"],
233
+ }
234
+ for field in simple.get(channel, []):
235
+ val = copy_data.get(field)
236
+ if val and isinstance(val, dict):
237
+ blocks.append((field, val))
238
+
239
+ list_fields = {
240
+ "email": ["body"],
241
+ "poster": ["body", "bullet_points"],
242
+ }
243
+ for field in list_fields.get(channel, []):
244
+ items = copy_data.get(field)
245
+ if items and isinstance(items, list):
246
+ for i, item in enumerate(items):
247
+ if isinstance(item, dict):
248
+ blocks.append((f"{field}[{i}]", item))
249
+
250
+ return blocks
251
+
252
+
253
+ def _extract_flat_blocks(channels: list[str], campaign_brief_id: str) -> list[tuple[str, dict]]:
254
+ """Load and extract blocks from all channel copy files."""
255
+ from .._campaign_store import load_artifact
256
+
257
+ blocks: list[tuple[str, dict]] = []
258
+ for channel in channels:
259
+ copy_artifact = load_artifact(campaign_brief_id, f"copy-{channel}.json")
260
+ if copy_artifact:
261
+ blocks.extend(_extract_all_blocks(copy_artifact.get("copy", {}), channel))
262
+ return blocks
263
+
264
+
265
+ def _load_brand_legal(campaign_brief_id: str) -> dict:
266
+ """Load legal content from brand kit."""
267
+ import json
268
+ from importlib.resources import files
269
+ from pathlib import Path
270
+
271
+ from .._campaign_store import load_brief
272
+
273
+ brief = load_brief(campaign_brief_id)
274
+ kit_path = None
275
+ if brief:
276
+ kit_path = brief.get("brand_kit_path")
277
+
278
+ if kit_path and Path(kit_path).is_dir():
279
+ legal_path = Path(kit_path) / "legal.json"
280
+ else:
281
+ legal_path = Path(str(files("open_pharma_plugins_campaign_studio") / "fixtures" / "brand_kit" / "legal.json"))
282
+
283
+ if legal_path.exists():
284
+ return json.loads(legal_path.read_text())
285
+ return {}
@@ -0,0 +1,13 @@
1
+ from mcp_framework import build_registry
2
+
3
+ __version__ = "1.1.0"
4
+
5
+ SPECS, get_handler, list_tools = build_registry(__name__, ["tools"])
6
+
7
+ USAGE_NOTE = (
8
+ "Competitive Intelligence tracks competitor trial pipelines, FDA "
9
+ "regulatory events, label changes, news, and publications. Maintains "
10
+ "a persistent watchlist and generates shareable briefing reports."
11
+ )
12
+
13
+ SYSTEM_DEPS = []
@@ -0,0 +1,11 @@
1
+ from pathlib import Path
2
+
3
+ from mcp_framework import run_main
4
+
5
+
6
+ def main() -> None:
7
+ run_main(__package__ or Path(__file__).resolve().parent.name)
8
+
9
+
10
+ if __name__ == "__main__":
11
+ main()
@@ -0,0 +1,87 @@
1
+ """Private, non-overwriting artifact primitives for CI run projections."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ import re
8
+ import secrets
9
+ from datetime import datetime, timezone
10
+ from pathlib import Path
11
+ from typing import Any
12
+
13
+ from shared.filesystem import (
14
+ atomic_write_bytes,
15
+ atomic_write_text,
16
+ contained_path,
17
+ ensure_private_dir,
18
+ validate_component,
19
+ )
20
+
21
+ from ._watchlist import reports_dir
22
+ from .models import ArtifactManifest, ArtifactRecord
23
+
24
+ _FORMULA_PREFIXES = ("=", "+", "-", "@", "\t", "\r")
25
+ _MAX_ATTEMPTS = 3
26
+
27
+
28
+ def sanitize_display_stem(value: str | None, *, default: str) -> str:
29
+ clean = re.sub(r"[^A-Za-z0-9_-]", "_", value or "")
30
+ clean = clean[:100]
31
+ return clean if clean not in {"", ".", ".."} else default
32
+
33
+
34
+ def safe_csv_cell(value: Any) -> str:
35
+ text = "" if value is None else str(value)
36
+ candidate = text.lstrip(" \u00a0")
37
+ return f"'{text}" if candidate.startswith(_FORMULA_PREFIXES) else text
38
+
39
+
40
+ def create_artifact_dir(run_id: str, *, now: datetime | None = None) -> Path:
41
+ validate_component(run_id, label="run id")
42
+ generated_at = _utc(now)
43
+ run_root = ensure_private_dir(contained_path(reports_dir(), run_id))
44
+ for _attempt in range(_MAX_ATTEMPTS):
45
+ artifact_id = f"artifact_{generated_at.strftime('%Y%m%dT%H%M%SZ')}_{secrets.token_hex(4)}"
46
+ candidate = contained_path(run_root, artifact_id)
47
+ try:
48
+ candidate.mkdir(mode=0o700, exist_ok=False)
49
+ return candidate
50
+ except FileExistsError:
51
+ continue
52
+ raise FileExistsError("could not allocate a unique artifact directory after three attempts")
53
+
54
+
55
+ def write_artifact(
56
+ output_dir: Path,
57
+ relative_path: str,
58
+ content: str | bytes,
59
+ *,
60
+ media_type: str,
61
+ ) -> ArtifactRecord:
62
+ validate_component(relative_path, label="artifact path")
63
+ path = contained_path(output_dir, relative_path)
64
+ payload = content.encode("utf-8") if isinstance(content, str) else content
65
+ atomic_write_bytes(path, payload)
66
+ return ArtifactRecord(
67
+ relative_path=relative_path,
68
+ media_type=media_type,
69
+ byte_size=len(payload),
70
+ sha256=hashlib.sha256(payload).hexdigest(),
71
+ )
72
+
73
+
74
+ def write_manifest(output_dir: Path, manifest: ArtifactManifest) -> Path:
75
+ path = contained_path(output_dir, "manifest.json")
76
+ atomic_write_text(
77
+ path,
78
+ json.dumps(manifest.model_dump(mode="json"), indent=2, ensure_ascii=False),
79
+ )
80
+ return path
81
+
82
+
83
+ def _utc(value: datetime | None) -> datetime:
84
+ now = value or datetime.now(timezone.utc)
85
+ if now.tzinfo is None:
86
+ raise ValueError("now must be timezone-aware")
87
+ return now.astimezone(timezone.utc)
@@ -0,0 +1,144 @@
1
+ """Schema-versioned, credential-free response cache for CI provider calls."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import hashlib
6
+ import json
7
+ from dataclasses import dataclass
8
+ from datetime import datetime, timezone
9
+ from pathlib import Path
10
+ from typing import Any, Mapping
11
+
12
+ from shared.filesystem import atomic_write_json, contained_path, ensure_private_dir, sanitize_mapping, sanitize_url
13
+
14
+ from .models import CacheStatus
15
+
16
+ _SCHEMA_VERSION = 2
17
+ _DEFAULT_TTL_HOURS = 24
18
+
19
+
20
+ @dataclass(frozen=True)
21
+ class CacheLookup:
22
+ status: CacheStatus
23
+ payload: Any | None
24
+ cached_at: datetime | None
25
+
26
+
27
+ def _cache_dir() -> Path:
28
+ from shared.env import get_env
29
+
30
+ root = ensure_private_dir(
31
+ get_env(
32
+ "OPEN_PHARMA_CI_DATA_DIR",
33
+ str(Path.home() / ".open-pharma-plugins" / "competitive-intelligence"),
34
+ )
35
+ )
36
+ return ensure_private_dir(contained_path(root, "cache"))
37
+
38
+
39
+ def _ttl_hours() -> int:
40
+ try:
41
+ from shared.env import get_env
42
+
43
+ value = get_env("CI_CACHE_TTL_HOURS", str(_DEFAULT_TTL_HOURS))
44
+ parsed = int(value) if value else _DEFAULT_TTL_HOURS
45
+ return parsed if parsed >= 0 else _DEFAULT_TTL_HOURS
46
+ except (TypeError, ValueError):
47
+ return _DEFAULT_TTL_HOURS
48
+
49
+
50
+ def _cache_key(namespace: str, params: Mapping[str, Any] | None = None) -> str:
51
+ clean_namespace = _sanitize_namespace(namespace)
52
+ clean_params = _sanitize_value(params or {})
53
+ raw = clean_namespace + "|" + json.dumps(clean_params, sort_keys=True, default=str)
54
+ return hashlib.sha256(raw.encode()).hexdigest()[:24]
55
+
56
+
57
+ def cache_lookup(namespace: str, params: Mapping[str, Any] | None = None) -> CacheLookup:
58
+ ttl_hours = _ttl_hours()
59
+ if ttl_hours == 0:
60
+ return CacheLookup(status=CacheStatus.DISABLED, payload=None, cached_at=None)
61
+ key = _cache_key(namespace, params)
62
+ path = contained_path(_cache_dir(), f"{key}.json")
63
+ if not path.exists():
64
+ return CacheLookup(status=CacheStatus.MISS, payload=None, cached_at=None)
65
+ try:
66
+ data = json.loads(path.read_text())
67
+ if data.get("schema_version") != _SCHEMA_VERSION:
68
+ return CacheLookup(status=CacheStatus.MISS, payload=None, cached_at=None)
69
+ cached_at = datetime.fromisoformat(str(data["cached_at"]).replace("Z", "+00:00"))
70
+ if cached_at.tzinfo is None:
71
+ raise ValueError("cache timestamp must include a timezone")
72
+ cached_at = cached_at.astimezone(timezone.utc)
73
+ age_hours = (datetime.now(timezone.utc) - cached_at).total_seconds() / 3600
74
+ if age_hours > ttl_hours:
75
+ return CacheLookup(status=CacheStatus.MISS, payload=None, cached_at=None)
76
+ return CacheLookup(status=CacheStatus.HIT, payload=data.get("payload"), cached_at=cached_at)
77
+ except (KeyError, TypeError, ValueError, json.JSONDecodeError, OSError):
78
+ return CacheLookup(status=CacheStatus.MISS, payload=None, cached_at=None)
79
+
80
+
81
+ def cache_store(namespace: str, params: Mapping[str, Any] | None, payload: Any) -> None:
82
+ if _ttl_hours() == 0:
83
+ return
84
+ clean_namespace = _sanitize_namespace(namespace)
85
+ clean_params = _sanitize_value(params or {})
86
+ key = _cache_key(clean_namespace, clean_params)
87
+ path = contained_path(_cache_dir(), f"{key}.json")
88
+ data = {
89
+ "schema_version": _SCHEMA_VERSION,
90
+ "cached_at": datetime.now(timezone.utc).isoformat().replace("+00:00", "Z"),
91
+ "namespace": clean_namespace,
92
+ "params": clean_params,
93
+ "payload": _sanitize_value(payload),
94
+ }
95
+ atomic_write_json(path, data)
96
+
97
+
98
+ def cache_stats() -> dict[str, Any]:
99
+ d = _cache_dir()
100
+ files = list(d.glob("*.json"))
101
+ total_bytes = sum(f.stat().st_size for f in files)
102
+ recognized = 0
103
+ for path in files:
104
+ try:
105
+ if json.loads(path.read_text()).get("schema_version") == _SCHEMA_VERSION:
106
+ recognized += 1
107
+ except (json.JSONDecodeError, OSError, AttributeError):
108
+ pass
109
+ return {
110
+ "cache_dir": str(d),
111
+ "schema_version": _SCHEMA_VERSION,
112
+ "entry_count": recognized,
113
+ "ignored_entry_count": len(files) - recognized,
114
+ "total_bytes": total_bytes,
115
+ }
116
+
117
+
118
+ def cache_clear() -> int:
119
+ d = _cache_dir()
120
+ removed = 0
121
+ for path in d.glob("*.json"):
122
+ try:
123
+ if json.loads(path.read_text()).get("schema_version") != _SCHEMA_VERSION:
124
+ continue
125
+ except (json.JSONDecodeError, OSError, AttributeError):
126
+ continue
127
+ path.unlink(missing_ok=True)
128
+ removed += 1
129
+ return removed
130
+
131
+
132
+ def _sanitize_namespace(namespace: str) -> str:
133
+ return sanitize_url(namespace) if "://" in namespace else namespace
134
+
135
+
136
+ def _sanitize_value(value: Any) -> Any:
137
+ sanitized = sanitize_mapping(value)
138
+ if isinstance(sanitized, dict):
139
+ return {key: _sanitize_value(item) for key, item in sanitized.items()}
140
+ if isinstance(sanitized, list):
141
+ return [_sanitize_value(item) for item in sanitized]
142
+ if isinstance(sanitized, str) and "://" in sanitized:
143
+ return sanitize_url(sanitized)
144
+ return sanitized