ictrp-mcp-server 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +65 -0
  2. package/LICENSE +37 -0
  3. package/README.md +208 -0
  4. package/README_ZH.md +189 -0
  5. package/dist/cli/setup-cli.d.ts +14 -0
  6. package/dist/cli/setup-cli.js +230 -0
  7. package/dist/cli/setup-cli.js.map +1 -0
  8. package/dist/index.d.ts +18 -0
  9. package/dist/index.js +477 -0
  10. package/dist/index.js.map +1 -0
  11. package/dist/runtime/bootstrap.d.ts +99 -0
  12. package/dist/runtime/bootstrap.js +350 -0
  13. package/dist/runtime/bootstrap.js.map +1 -0
  14. package/dist/runtime/env-probe.d.ts +108 -0
  15. package/dist/runtime/env-probe.js +479 -0
  16. package/dist/runtime/env-probe.js.map +1 -0
  17. package/dist/runtime/sidecar-client.d.ts +50 -0
  18. package/dist/runtime/sidecar-client.js +120 -0
  19. package/dist/runtime/sidecar-client.js.map +1 -0
  20. package/dist/runtime/supervisor.d.ts +47 -0
  21. package/dist/runtime/supervisor.js +248 -0
  22. package/dist/runtime/supervisor.js.map +1 -0
  23. package/package.json +60 -0
  24. package/sidecar/ictrp_sidecar.py +602 -0
  25. package/sidecar/vendor/ictrp_mcp/__init__.py +3 -0
  26. package/sidecar/vendor/ictrp_mcp/cache/__init__.py +0 -0
  27. package/sidecar/vendor/ictrp_mcp/cache/store.py +313 -0
  28. package/sidecar/vendor/ictrp_mcp/data/__init__.py +0 -0
  29. package/sidecar/vendor/ictrp_mcp/data/columns.py +108 -0
  30. package/sidecar/vendor/ictrp_mcp/data/jsonio.py +213 -0
  31. package/sidecar/vendor/ictrp_mcp/data/normalize.py +348 -0
  32. package/sidecar/vendor/ictrp_mcp/data/query.py +307 -0
  33. package/sidecar/vendor/ictrp_mcp/errors.py +123 -0
  34. package/sidecar/vendor/ictrp_mcp/ictrp/__init__.py +0 -0
  35. package/sidecar/vendor/ictrp_mcp/ictrp/export_guard.py +269 -0
  36. package/sidecar/vendor/ictrp_mcp/ictrp/htmlstate.py +143 -0
  37. package/sidecar/vendor/ictrp_mcp/ictrp/session.py +245 -0
  38. package/sidecar/vendor/ictrp_mcp/offline.py +133 -0
  39. package/sidecar/vendor/ictrp_mcp/provenance.py +182 -0
  40. package/sidecar/vendor/ictrp_mcp/server.py +368 -0
  41. package/sidecar/vendor/ictrp_mcp/tools.py +712 -0
  42. package/sidecar/vendor/pyproject.toml +25 -0
@@ -0,0 +1,712 @@
1
+ """Tool implementations.
2
+
3
+ Each function returns a plain dict that already carries provenance. The MCP layer
4
+ in `server.py` only handles transport concerns.
5
+
6
+ Design rules enforced here:
7
+
8
+ * No tool reports zero results unless a validated export genuinely contained zero
9
+ rows. Failures raise `IctrpError` and never degrade into an empty list.
10
+ * Every response carries `provenance`, including the local-only tools, because a
11
+ response that omits it cannot be attributed.
12
+ * Local tools (`ictrp_filter`, `ictrp_field_query`, `ictrp_registry_summary`,
13
+ `ictrp_find_duplicates`, `ictrp_export`) do not touch the network. Refining a
14
+ search is therefore free and repeatable.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import csv
20
+ import io
21
+ import json
22
+ import os
23
+ import time
24
+ from pathlib import Path
25
+ from typing import Any
26
+
27
+ from . import offline
28
+ from .cache.store import SetStore, query_key
29
+ from .data import query as q
30
+ from .data.columns import CSV_COLUMNS
31
+ from .data.jsonio import snapshot_from_trials
32
+ from .errors import ErrorCode, IctrpError
33
+ from .ictrp.session import IctrpSession
34
+ from .provenance import INCOMPLETENESS_NOTICE, Provenance, derive_set_provenance
35
+
36
+ #: Fields shown by default when a caller does not ask for specific ones. Chosen to
37
+ #: be the high-coverage, decision-relevant columns rather than all 58.
38
+ DEFAULT_FIELDS: tuple[str, ...] = (
39
+ "trial_id",
40
+ "source_register",
41
+ "public_title",
42
+ "recruitment_status",
43
+ "phase_code",
44
+ "registration_date",
45
+ "condition",
46
+ "countries",
47
+ "target_size_total",
48
+ "last_refreshed_display",
49
+ )
50
+
51
+ #: How long a cached set is served before a refresh is attempted automatically.
52
+ #: Short enough that a weekly upstream refresh is picked up promptly, long enough
53
+ #: that a session of repeated queries makes one export rather than many.
54
+ DEFAULT_REFRESH_AFTER_SECONDS = 7 * 24 * 60 * 60
55
+
56
+ ENV_REFRESH_AFTER = "ICTRP_REFRESH_AFTER_SECONDS"
57
+ ENV_AUTO_REFRESH = "ICTRP_AUTO_REFRESH"
58
+
59
+
60
+ def _refresh_after_seconds() -> int:
61
+ raw = os.environ.get(ENV_REFRESH_AFTER)
62
+ if raw:
63
+ try:
64
+ return max(0, int(raw))
65
+ except ValueError:
66
+ pass
67
+ return DEFAULT_REFRESH_AFTER_SECONDS
68
+
69
+
70
+ def _auto_refresh_enabled() -> bool:
71
+ """Auto-refresh is on unless explicitly disabled.
72
+
73
+ "Off" exists for reproducible runs and for offline deployments, where reaching
74
+ the network at an unpredictable moment is worse than serving older data.
75
+ """
76
+ raw = os.environ.get(ENV_AUTO_REFRESH, "").strip().lower()
77
+ return raw not in {"0", "false", "no", "off"}
78
+
79
+
80
+ class IctrpService:
81
+ """Holds the cache and executes tools against it."""
82
+
83
+ #: How long a refresh failure suppresses further automatic attempts, so a
84
+ #: failing upstream is not retried on every single call.
85
+ REFRESH_FAILURE_BACKOFF_SECONDS = 15 * 60
86
+
87
+ def __init__(self, store: SetStore | None = None) -> None:
88
+ self.store = store or SetStore()
89
+ self._refresh_failures: dict[str, tuple[float, str]] = {}
90
+
91
+ # ---- network-backed ---------------------------------------------------
92
+
93
+ async def search(
94
+ self,
95
+ keyword: str,
96
+ *,
97
+ limit: int = 50,
98
+ offset: int = 0,
99
+ fields: list[str] | None = None,
100
+ filters: list[dict[str, Any]] | None = None,
101
+ sort_by: str | None = None,
102
+ descending: bool = False,
103
+ refresh: bool = False,
104
+ ) -> dict[str, Any]:
105
+ """Materialize a search, then return a page of it.
106
+
107
+ The set stays queryable afterwards, so subsequent refinement costs nothing.
108
+
109
+ Refresh policy: `refresh=True` forces an upstream search. Otherwise a cached
110
+ set is reused until it is older than the refresh interval, at which point an
111
+ automatic refresh is attempted -- and, if that fails, the cached set is
112
+ served anyway with the failure recorded in provenance. Serving known-stale
113
+ data is preferable to failing a query, provided the staleness is stated.
114
+ """
115
+ if not keyword or not keyword.strip():
116
+ raise IctrpError(
117
+ ErrorCode.INVALID_ARGUMENT,
118
+ "keyword is required",
119
+ hint="Provide a search term, e.g. 'pancreatic cancer'.",
120
+ )
121
+ if limit < 1 or limit > 1000:
122
+ raise IctrpError(
123
+ ErrorCode.INVALID_ARGUMENT,
124
+ "limit must be between 1 and 1000",
125
+ )
126
+ if offset < 0:
127
+ raise IctrpError(ErrorCode.INVALID_ARGUMENT, "offset must be >= 0")
128
+
129
+ cache_hit = False
130
+ refresh_error: str | None = None
131
+ # Always look for a cached set, even when a refresh was forced. A forced
132
+ # refresh means "try to get fresh data", not "throw away what we have":
133
+ # if the attempt fails, the cached set is still the best answer
134
+ # available, and the code below is written to serve it with the failure
135
+ # recorded. Skipping this lookup made `previous` None, so the failure
136
+ # handler re-raised and an explicit refresh on a flaky network
137
+ # destroyed usable cached data instead of falling back to it.
138
+ materialized = self._cached_set(keyword)
139
+
140
+ should_refresh = refresh or (materialized is not None and self._should_refresh(materialized))
141
+ if should_refresh and self._auto_refresh_allows(keyword, force=refresh):
142
+ previous = materialized
143
+ try:
144
+ materialized = await self._materialize(keyword)
145
+ self._refresh_failures.pop(query_key(keyword), None)
146
+ except IctrpError as exc:
147
+ if previous is None:
148
+ raise
149
+ # We already hold data. Report the failure alongside it rather than
150
+ # discarding a usable set.
151
+ materialized = previous
152
+ refresh_error = f"{exc.code.value}: {exc.message}"
153
+ self._refresh_failures[query_key(keyword)] = (time.time(), refresh_error)
154
+
155
+ if materialized is None:
156
+ materialized = await self._materialize(keyword)
157
+ elif not should_refresh:
158
+ cache_hit = True
159
+ else:
160
+ cache_hit = refresh_error is not None
161
+
162
+ trials = materialized.trials
163
+ filtered = q.apply_filters(trials, filters)
164
+ ordered = q.sort_trials(filtered, sort_by, descending)
165
+ page = ordered[offset : offset + limit]
166
+
167
+ provenance = Provenance.from_dict(materialized.provenance) if materialized.provenance else Provenance.now()
168
+ provenance.cache_hit = cache_hit
169
+ provenance.rows_returned = materialized.rows_returned
170
+ provenance.upstream_reported_total = materialized.reported_total
171
+ provenance.records_incomplete = materialized.is_incomplete
172
+ provenance.estimated_missing = materialized.missing_estimate
173
+ self._annotate_age(provenance, materialized)
174
+ if refresh_error:
175
+ provenance.notes.append(
176
+ f"Automatic refresh failed and an older cached set was served instead. {refresh_error}"
177
+ )
178
+
179
+ selected = list(fields) if fields else list(DEFAULT_FIELDS)
180
+ return {
181
+ "status": "ok",
182
+ "set_id": materialized.set_id,
183
+ "matched_rows_returned": len(ordered),
184
+ "upstream_reported_total": materialized.reported_total,
185
+ "records_incomplete": materialized.is_incomplete,
186
+ "estimated_missing": materialized.missing_estimate,
187
+ "counts_are_of_retrieved_rows_not_of_matching_trials": True,
188
+ "offset": offset,
189
+ "limit": limit,
190
+ "trials": [{"trial_id": t.get("trial_id"), **{f: t.get(f) for f in selected}} for t in page],
191
+ "provenance": provenance.to_dict(),
192
+ }
193
+
194
+ def _should_refresh(self, materialized) -> bool:
195
+ return materialized.age_seconds() > _refresh_after_seconds()
196
+
197
+ def _auto_refresh_allows(self, keyword: str, *, force: bool) -> bool:
198
+ """Whether a refresh may be attempted right now.
199
+
200
+ A forced refresh always proceeds -- that is what forcing means, and the
201
+ caller has explicitly accepted the cost. Note this must be an early
202
+ `return True`, not a fall-through: an inverted test here made
203
+ `refresh=True` a silent no-op, so the one escape hatch out of the
204
+ backoff window below did not exist.
205
+
206
+ An automatic one is skipped when it is disabled, or when a recent
207
+ attempt already failed -- repeatedly retrying a failing upstream would
208
+ turn one outage into a request storm.
209
+ """
210
+ if force:
211
+ return True
212
+ if not _auto_refresh_enabled():
213
+ return False
214
+ failure = self._refresh_failures.get(query_key(keyword))
215
+ if failure is None:
216
+ return True
217
+ failed_at, _ = failure
218
+ return (time.time() - failed_at) > self.REFRESH_FAILURE_BACKOFF_SECONDS
219
+
220
+ def _annotate_age(self, provenance: Provenance, materialized) -> None:
221
+ """Record how old the served data is, whatever its origin."""
222
+ age_days = materialized.age_seconds() / 86400.0
223
+ provenance.notes.append(
224
+ f"Data age: {age_days:.2f} days (set created {_iso(materialized.created_at)})."
225
+ )
226
+ if materialized.age_seconds() > _refresh_after_seconds():
227
+ provenance.notes.append(
228
+ "This set is older than the configured refresh interval "
229
+ f"({_refresh_after_seconds()}s); it was served because a refresh was "
230
+ "not permitted or did not succeed."
231
+ )
232
+
233
+ async def _materialize(self, keyword: str):
234
+ """Run the upstream chain, falling back to a local snapshot when possible.
235
+
236
+ The fallback is deliberately narrow: snapshots are consulted only after an
237
+ upstream attempt has failed, never in preference to it. A deployment that
238
+ can reach the network should always get live data; a deployment that cannot
239
+ should get clearly-labelled older data instead of an error.
240
+ """
241
+ try:
242
+ return await self._fetch_upstream(keyword)
243
+ except IctrpError as exc:
244
+ if not _is_offline_fallback_worthy(exc):
245
+ raise
246
+ bundled = offline.load_bundle(keyword)
247
+ if bundled is None:
248
+ raise
249
+ snapshot, path = bundled
250
+ provenance = dict(snapshot.provenance)
251
+ provenance.update(offline.snapshot_provenance(snapshot, path))
252
+ notes = list(provenance.get("notes") or [])
253
+ notes.append(
254
+ f"Upstream request failed ({exc.code.value}); this response was served "
255
+ f"from a local snapshot instead. Upstream error: {exc.message}"
256
+ )
257
+ provenance["notes"] = notes
258
+ return self.store.adopt(
259
+ keyword=keyword,
260
+ trials=snapshot.trials,
261
+ provenance=provenance,
262
+ created_at=snapshot.created_at,
263
+ source_label="snapshot",
264
+ )
265
+
266
+ async def _fetch_upstream(self, keyword: str):
267
+ session = IctrpSession.create()
268
+ try:
269
+ await session.load_form()
270
+ await session.search(keyword)
271
+ payload = await session.export_csv()
272
+ steps = session.provenance_steps()
273
+ response_date = session.response_date
274
+ finally:
275
+ await session.aclose()
276
+
277
+ provenance = derive_set_provenance(
278
+ keyword=keyword,
279
+ trials=[], # filled in by the store; only used for date extraction below
280
+ reported_total=payload.reported_total,
281
+ response_date=response_date,
282
+ steps=steps,
283
+ )
284
+
285
+ # Build trial dicts once to derive export/last-refreshed dates for provenance.
286
+ from .data.normalize import to_trial
287
+
288
+ trials = [to_trial(row, payload.header) for row in payload.rows]
289
+ export_dates = sorted({t.get("export_date_raw") for t in trials if t.get("export_date_raw")})
290
+ refreshed = sorted(
291
+ {t.get("last_refreshed_display") for t in trials if t.get("last_refreshed_display")}
292
+ )
293
+ if export_dates:
294
+ provenance.ictrp_export_date = export_dates[0]
295
+ provenance.notes = [n for n in provenance.notes if "Export date" not in n]
296
+
297
+ raw = _serialize_csv(payload.header, payload.rows)
298
+ materialized = self.store.store(
299
+ keyword=keyword,
300
+ payload_raw=raw,
301
+ header=payload.header,
302
+ rows=payload.rows,
303
+ reported_total=payload.reported_total,
304
+ provenance=provenance.to_dict(),
305
+ )
306
+
307
+ if payload.is_empty:
308
+ raise IctrpError(
309
+ ErrorCode.NO_RESULTS,
310
+ f"The portal returned a valid export with no matching trials for {keyword!r}",
311
+ hint=(
312
+ "This is a genuine zero: the export was structurally valid and "
313
+ "contained no data rows."
314
+ ),
315
+ )
316
+ return materialized
317
+
318
+ def _cached_set(self, keyword: str):
319
+ candidate = query_key(keyword)
320
+ try:
321
+ return self.store.get(candidate)
322
+ except IctrpError:
323
+ return None
324
+
325
+ # ---- local-only -------------------------------------------------------
326
+
327
+ def filter_set(
328
+ self,
329
+ set_id: str,
330
+ *,
331
+ filters: list[dict[str, Any]] | None = None,
332
+ sort_by: str | None = None,
333
+ descending: bool = False,
334
+ limit: int = 50,
335
+ offset: int = 0,
336
+ fields: list[str] | None = None,
337
+ ) -> dict[str, Any]:
338
+ materialized = self.store.get(set_id)
339
+ filtered = q.apply_filters(materialized.trials, filters)
340
+ ordered = q.sort_trials(filtered, sort_by, descending)
341
+ page = ordered[offset : offset + limit]
342
+ selected = list(fields) if fields else list(DEFAULT_FIELDS)
343
+
344
+ provenance = Provenance.from_dict(materialized.provenance)
345
+ provenance.cache_hit = True
346
+ provenance.rows_returned = materialized.rows_returned
347
+ provenance.upstream_reported_total = materialized.reported_total
348
+ provenance.records_incomplete = materialized.is_incomplete
349
+ provenance.estimated_missing = materialized.missing_estimate
350
+
351
+ return {
352
+ "status": "ok",
353
+ "set_id": set_id,
354
+ "matched_rows_returned": len(ordered),
355
+ "of_rows_in_set": materialized.rows_returned,
356
+ "counts_are_of_retrieved_rows_not_of_matching_trials": True,
357
+ "offset": offset,
358
+ "limit": limit,
359
+ "trials": [{"trial_id": t.get("trial_id"), **{f: t.get(f) for f in selected}} for t in page],
360
+ "provenance": provenance.to_dict(),
361
+ "note": "This operation was served entirely from the cached set; no upstream request was made.",
362
+ }
363
+
364
+ def field_query(
365
+ self,
366
+ *,
367
+ field: str,
368
+ set_id: str | None = None,
369
+ keyword: str | None = None,
370
+ limit: int = 50,
371
+ ) -> dict[str, Any]:
372
+ materialized = self._resolve(set_id, keyword)
373
+ values = q.facet(materialized.trials, field, limit=limit)
374
+ coverage = q.field_coverage(materialized.trials, [field])[field]
375
+
376
+ provenance = Provenance.from_dict(materialized.provenance)
377
+ provenance.cache_hit = True
378
+ return {
379
+ "status": "ok",
380
+ "set_id": materialized.set_id,
381
+ "field": field,
382
+ "coverage": coverage,
383
+ "distinct_values": values,
384
+ "provenance": provenance.to_dict(),
385
+ "note": "Served from the cached set; no upstream request was made.",
386
+ }
387
+
388
+ def registry_summary(
389
+ self,
390
+ *,
391
+ set_id: str | None = None,
392
+ keyword: str | None = None,
393
+ group_by: list[str] | None = None,
394
+ ) -> dict[str, Any]:
395
+ materialized = self._resolve(set_id, keyword)
396
+ groups = group_by or ["source_register"]
397
+ facets = {g: q.facet(materialized.trials, g) for g in groups}
398
+
399
+ provenance = Provenance.from_dict(materialized.provenance)
400
+ provenance.cache_hit = True
401
+ return {
402
+ "status": "ok",
403
+ "set_id": materialized.set_id,
404
+ "rows_in_set": materialized.rows_returned,
405
+ "upstream_reported_total": materialized.reported_total,
406
+ "records_incomplete": materialized.is_incomplete,
407
+ "facets": facets,
408
+ "field_coverage": q.field_coverage(
409
+ materialized.trials,
410
+ [
411
+ "trial_id", "public_title", "scientific_title", "condition",
412
+ "intervention", "primary_outcome", "secondary_outcome",
413
+ "inclusion_criteria", "exclusion_criteria",
414
+ "target_size_total", "countries", "inclusion_agemin",
415
+ "inclusion_agemax", "inclusion_gender", "ethics_status",
416
+ "secondary_id", "results_yes_no",
417
+ ],
418
+ ),
419
+ "provenance": provenance.to_dict(),
420
+ "note": "Served from the cached set; no upstream request was made.",
421
+ }
422
+
423
+ def find_duplicates(
424
+ self,
425
+ *,
426
+ set_id: str | None = None,
427
+ keyword: str | None = None,
428
+ ) -> dict[str, Any]:
429
+ materialized = self._resolve(set_id, keyword)
430
+ groups = q.find_duplicates(materialized.trials)
431
+ provenance = Provenance.from_dict(materialized.provenance)
432
+ provenance.cache_hit = True
433
+ return {
434
+ "status": "ok",
435
+ "set_id": materialized.set_id,
436
+ "rows_in_set": materialized.rows_returned,
437
+ "candidate_groups": groups,
438
+ "group_count": len(groups),
439
+ "method": (
440
+ "Identifier cross-references only (Secondary ID). Title similarity is "
441
+ "deliberately not used; multi-centre trials share near-identical titles."
442
+ ),
443
+ "provenance": provenance.to_dict(),
444
+ "note": "Served from the cached set; no upstream request was made.",
445
+ }
446
+
447
+ def export_records(
448
+ self,
449
+ *,
450
+ set_id: str | None = None,
451
+ keyword: str | None = None,
452
+ fmt: str = "json",
453
+ fields: list[str] | None = None,
454
+ filters: list[dict[str, Any]] | None = None,
455
+ include_provenance_header: bool = True,
456
+ ) -> dict[str, Any]:
457
+ materialized = self._resolve(set_id, keyword)
458
+ rows = q.apply_filters(materialized.trials, filters)
459
+ selected = list(fields) if fields else list(materialized.trials[0].keys()) if materialized.trials else []
460
+ provenance = Provenance.from_dict(materialized.provenance)
461
+ provenance.cache_hit = True
462
+
463
+ if fmt == "csv":
464
+ buffer = io.StringIO()
465
+ writer = csv.writer(buffer)
466
+ writer.writerow(selected)
467
+ for row in rows:
468
+ writer.writerow([_cell(row.get(f)) for f in selected])
469
+ body = buffer.getvalue()
470
+ elif fmt == "jsonl":
471
+ body = "\n".join(
472
+ json.dumps({f: row.get(f) for f in selected}, ensure_ascii=False) for row in rows
473
+ )
474
+ elif fmt == "markdown":
475
+ lines = ["| " + " | ".join(selected) + " |", "|" + "---|" * len(selected)]
476
+ for row in rows:
477
+ lines.append("| " + " | ".join(_cell(row.get(f)) for f in selected) + " |")
478
+ body = "\n".join(lines)
479
+ elif fmt == "json":
480
+ body = json.dumps(
481
+ [{f: row.get(f) for f in selected} for row in rows],
482
+ ensure_ascii=False,
483
+ indent=2,
484
+ )
485
+ else:
486
+ raise IctrpError(
487
+ ErrorCode.INVALID_ARGUMENT,
488
+ f"unsupported format {fmt!r}",
489
+ hint="Supported: csv, json, jsonl, markdown.",
490
+ )
491
+
492
+ header = ""
493
+ if include_provenance_header:
494
+ header = (
495
+ f"# Source: WHO ICTRP ({provenance.source_url})\n"
496
+ f"# Retrieved: {provenance.retrieved_at}\n"
497
+ f"# Rows in this export: {len(rows)}\n"
498
+ )
499
+ if provenance.upstream_reported_total is not None:
500
+ header += f"# Portal-reported matches: {provenance.upstream_reported_total}\n"
501
+ if provenance.records_incomplete:
502
+ header += f"# WARNING: {INCOMPLETENESS_NOTICE}\n"
503
+
504
+ return {
505
+ "status": "ok",
506
+ "set_id": materialized.set_id,
507
+ "format": fmt,
508
+ "row_count": len(rows),
509
+ "content": header + body,
510
+ "provenance": provenance.to_dict(),
511
+ "note": "Served from the cached set; no upstream request was made.",
512
+ }
513
+
514
+ def cache_status(self, *, action: str = "list", set_id: str | None = None) -> dict[str, Any]:
515
+ if action == "list":
516
+ sets = self.store.list_sets()
517
+ return {
518
+ "status": "ok",
519
+ "action": "list",
520
+ "sets": sets,
521
+ "set_count": len(sets),
522
+ "cache_dir": str(self.store.cache_dir),
523
+ "max_in_memory_sets": 3,
524
+ "refresh_after_seconds": _refresh_after_seconds(),
525
+ "auto_refresh": _auto_refresh_enabled(),
526
+ "bundle_configured": {
527
+ "ICTRP_BUNDLE_PATH": os.environ.get(offline.ENV_BUNDLE_PATH),
528
+ "ICTRP_BUNDLE_DIR": os.environ.get(offline.ENV_BUNDLE_DIR),
529
+ "max_age_days": offline.max_age_days(),
530
+ },
531
+ "note": (
532
+ "Only a few sets stay parsed in memory; others are re-read from "
533
+ "disk on demand. Raw CSV is cached because re-exporting is expensive."
534
+ ),
535
+ }
536
+ if action == "purge":
537
+ removed = self.store.purge(set_id)
538
+ return {"status": "ok", "action": "purge", "removed": removed, "set_id": set_id}
539
+ raise IctrpError(
540
+ ErrorCode.INVALID_ARGUMENT,
541
+ f"unsupported action {action!r}",
542
+ hint="Supported: list, purge.",
543
+ )
544
+
545
+ # ---- snapshot / bundle ------------------------------------------------
546
+
547
+ def snapshot(
548
+ self,
549
+ *,
550
+ set_id: str | None = None,
551
+ keyword: str | None = None,
552
+ path: str | None = None,
553
+ if_stale: bool = True,
554
+ ) -> dict[str, Any]:
555
+ """Write a cached set to a canonical JSON snapshot.
556
+
557
+ Intended for two uses: producing a dataset to ship inside a packaged
558
+ application, and refreshing such a dataset on a schedule. `if_stale=False`
559
+ forces a rewrite even when the existing snapshot is fresh, which is what a
560
+ release pipeline wants.
561
+ """
562
+ materialized = self._resolve(set_id, keyword)
563
+ if not materialized.trials:
564
+ raise IctrpError(
565
+ ErrorCode.NO_RESULTS,
566
+ "refusing to write an empty snapshot",
567
+ hint=(
568
+ "A snapshot is a distribution artifact; an empty one would make a "
569
+ "packaged application look like it had no data."
570
+ ),
571
+ )
572
+
573
+ target = self._snapshot_target(materialized.keyword, path)
574
+
575
+ if if_stale and target.is_file():
576
+ try:
577
+ existing = offline.load_bundle(materialized.keyword)
578
+ if existing is not None:
579
+ snapshot_obj, existing_path = existing
580
+ if existing_path == target and not snapshot_obj.stale(offline.max_age_days()):
581
+ return {
582
+ "status": "ok",
583
+ "action": "skipped",
584
+ "reason": "snapshot is still within its freshness limit",
585
+ "path": str(target),
586
+ "age_days": round(snapshot_obj.age_days, 2),
587
+ "trial_count": len(snapshot_obj.trials),
588
+ }
589
+ except IctrpError:
590
+ # An unreadable existing file is a reason to overwrite it, not to fail.
591
+ pass
592
+
593
+ provenance = Provenance.from_dict(materialized.provenance)
594
+ provenance.cache_hit = True
595
+ snapshot_obj = snapshot_from_trials(
596
+ trials=materialized.trials,
597
+ keyword=materialized.keyword,
598
+ provenance=provenance.to_dict(),
599
+ created_at=materialized.created_at,
600
+ )
601
+ size = snapshot_obj.write(target)
602
+
603
+ return {
604
+ "status": "ok",
605
+ "action": "written",
606
+ "path": str(target),
607
+ "trial_count": len(snapshot_obj.trials),
608
+ "bytes": size,
609
+ "data_created_at": _iso(materialized.created_at),
610
+ "provenance": provenance.to_dict(),
611
+ "terms_notice": (
612
+ "This file contains WHO ICTRP data. Redistributing it is a decision "
613
+ "about the ICTRP terms of use, not a technical default. See "
614
+ "docs/BUNDLE.md."
615
+ ),
616
+ }
617
+
618
+ def bundle_status(self, *, keyword: str | None = None) -> dict[str, Any]:
619
+ """Report which snapshot would serve a query, and how old it is."""
620
+ if not keyword:
621
+ raise IctrpError(
622
+ ErrorCode.INVALID_ARGUMENT,
623
+ "keyword is required to locate a snapshot",
624
+ )
625
+ candidates = [str(p) for p in offline.candidate_paths(keyword)]
626
+ bundled = offline.load_bundle(keyword)
627
+ out: dict[str, Any] = {
628
+ "status": "ok",
629
+ "keyword": keyword,
630
+ "candidates_checked": candidates,
631
+ "bundle_found": bundled is not None,
632
+ "configured": {
633
+ "ICTRP_BUNDLE_PATH": os.environ.get(offline.ENV_BUNDLE_PATH),
634
+ "ICTRP_BUNDLE_DIR": os.environ.get(offline.ENV_BUNDLE_DIR),
635
+ "max_age_days": offline.max_age_days(),
636
+ },
637
+ "auto_refresh": _auto_refresh_enabled(),
638
+ "refresh_after_seconds": _refresh_after_seconds(),
639
+ }
640
+ if bundled is not None:
641
+ snapshot_obj, path = bundled
642
+ out["snapshot"] = offline.snapshot_provenance(snapshot_obj, path)
643
+ out["snapshot"]["keyword"] = snapshot_obj.keyword
644
+ else:
645
+ out["hint"] = (
646
+ "No snapshot for this keyword. Run ictrp_search then ictrp_snapshot to "
647
+ "create one, or set ICTRP_BUNDLE_PATH / ICTRP_BUNDLE_DIR."
648
+ )
649
+ return out
650
+
651
+ def _snapshot_target(self, keyword: str, path: str | None) -> Path:
652
+ if path:
653
+ return Path(path).expanduser()
654
+ directory = offline.bundle_dir()
655
+ if directory:
656
+ return directory / f"{offline.snapshot_slug(keyword)}.json"
657
+ return self.store.cache_dir / "snapshots" / f"{offline.snapshot_slug(keyword)}.json"
658
+
659
+ # ---- internals --------------------------------------------------------
660
+
661
+ def _resolve(self, set_id: str | None, keyword: str | None):
662
+ if set_id:
663
+ return self.store.get(set_id)
664
+ if keyword:
665
+ materialized = self._cached_set(keyword)
666
+ if materialized is not None:
667
+ return materialized
668
+ raise IctrpError(
669
+ ErrorCode.CACHE_MISS,
670
+ f"no cached result set for keyword {keyword!r}",
671
+ hint="Run ictrp_search with this keyword first, or pass a set_id.",
672
+ )
673
+ raise IctrpError(
674
+ ErrorCode.INVALID_ARGUMENT,
675
+ "provide either set_id or keyword",
676
+ )
677
+
678
+
679
+ def _cell(value: Any) -> str:
680
+ if value is None:
681
+ return ""
682
+ if isinstance(value, (list, dict)):
683
+ return json.dumps(value, ensure_ascii=False)
684
+ return str(value)
685
+
686
+
687
+ def _iso(epoch: float) -> str:
688
+ return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(epoch))
689
+
690
+
691
+ #: Failures that justify falling back to a local snapshot. A missing result or a
692
+ #: malformed request will not be fixed by older data, so those propagate.
693
+ _OFFLINE_FALLBACK_CODES = frozenset(
694
+ {
695
+ ErrorCode.UPSTREAM_BLOCKED,
696
+ ErrorCode.UPSTREAM_ERROR,
697
+ ErrorCode.SESSION_FAILED,
698
+ ErrorCode.UPSTREAM_CONTRACT_DRIFT,
699
+ }
700
+ )
701
+
702
+
703
+ def _is_offline_fallback_worthy(exc: IctrpError) -> bool:
704
+ return exc.code in _OFFLINE_FALLBACK_CODES
705
+
706
+
707
+ def _serialize_csv(header: tuple[str, ...], rows: list[list[str]]) -> bytes:
708
+ buffer = io.StringIO()
709
+ writer = csv.writer(buffer)
710
+ writer.writerow(header)
711
+ writer.writerows(rows)
712
+ return buffer.getvalue().encode("utf-8")