ictrp-mcp-server 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/LICENSE +37 -0
- package/README.md +208 -0
- package/README_ZH.md +189 -0
- package/dist/cli/setup-cli.d.ts +14 -0
- package/dist/cli/setup-cli.js +230 -0
- package/dist/cli/setup-cli.js.map +1 -0
- package/dist/index.d.ts +18 -0
- package/dist/index.js +477 -0
- package/dist/index.js.map +1 -0
- package/dist/runtime/bootstrap.d.ts +99 -0
- package/dist/runtime/bootstrap.js +350 -0
- package/dist/runtime/bootstrap.js.map +1 -0
- package/dist/runtime/env-probe.d.ts +108 -0
- package/dist/runtime/env-probe.js +479 -0
- package/dist/runtime/env-probe.js.map +1 -0
- package/dist/runtime/sidecar-client.d.ts +50 -0
- package/dist/runtime/sidecar-client.js +120 -0
- package/dist/runtime/sidecar-client.js.map +1 -0
- package/dist/runtime/supervisor.d.ts +47 -0
- package/dist/runtime/supervisor.js +248 -0
- package/dist/runtime/supervisor.js.map +1 -0
- package/package.json +60 -0
- package/sidecar/ictrp_sidecar.py +602 -0
- package/sidecar/vendor/ictrp_mcp/__init__.py +3 -0
- package/sidecar/vendor/ictrp_mcp/cache/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/cache/store.py +313 -0
- package/sidecar/vendor/ictrp_mcp/data/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/data/columns.py +108 -0
- package/sidecar/vendor/ictrp_mcp/data/jsonio.py +213 -0
- package/sidecar/vendor/ictrp_mcp/data/normalize.py +348 -0
- package/sidecar/vendor/ictrp_mcp/data/query.py +307 -0
- package/sidecar/vendor/ictrp_mcp/errors.py +123 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/export_guard.py +269 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/htmlstate.py +143 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/session.py +245 -0
- package/sidecar/vendor/ictrp_mcp/offline.py +133 -0
- package/sidecar/vendor/ictrp_mcp/provenance.py +182 -0
- package/sidecar/vendor/ictrp_mcp/server.py +368 -0
- package/sidecar/vendor/ictrp_mcp/tools.py +712 -0
- package/sidecar/vendor/pyproject.toml +25 -0
|
@@ -0,0 +1,712 @@
|
|
|
1
|
+
"""Tool implementations.
|
|
2
|
+
|
|
3
|
+
Each function returns a plain dict that already carries provenance. The MCP layer
|
|
4
|
+
in `server.py` only handles transport concerns.
|
|
5
|
+
|
|
6
|
+
Design rules enforced here:
|
|
7
|
+
|
|
8
|
+
* No tool reports zero results unless a validated export genuinely contained zero
|
|
9
|
+
rows. Failures raise `IctrpError` and never degrade into an empty list.
|
|
10
|
+
* Every response carries `provenance`, including the local-only tools, because a
|
|
11
|
+
response that omits it cannot be attributed.
|
|
12
|
+
* Local tools (`ictrp_filter`, `ictrp_field_query`, `ictrp_registry_summary`,
|
|
13
|
+
`ictrp_find_duplicates`, `ictrp_export`) do not touch the network. Refining a
|
|
14
|
+
search is therefore free and repeatable.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import csv
|
|
20
|
+
import io
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import time
|
|
24
|
+
from pathlib import Path
|
|
25
|
+
from typing import Any
|
|
26
|
+
|
|
27
|
+
from . import offline
|
|
28
|
+
from .cache.store import SetStore, query_key
|
|
29
|
+
from .data import query as q
|
|
30
|
+
from .data.columns import CSV_COLUMNS
|
|
31
|
+
from .data.jsonio import snapshot_from_trials
|
|
32
|
+
from .errors import ErrorCode, IctrpError
|
|
33
|
+
from .ictrp.session import IctrpSession
|
|
34
|
+
from .provenance import INCOMPLETENESS_NOTICE, Provenance, derive_set_provenance
|
|
35
|
+
|
|
36
|
+
#: Fields shown by default when a caller does not ask for specific ones. Chosen to
|
|
37
|
+
#: be the high-coverage, decision-relevant columns rather than all 58.
|
|
38
|
+
DEFAULT_FIELDS: tuple[str, ...] = (
|
|
39
|
+
"trial_id",
|
|
40
|
+
"source_register",
|
|
41
|
+
"public_title",
|
|
42
|
+
"recruitment_status",
|
|
43
|
+
"phase_code",
|
|
44
|
+
"registration_date",
|
|
45
|
+
"condition",
|
|
46
|
+
"countries",
|
|
47
|
+
"target_size_total",
|
|
48
|
+
"last_refreshed_display",
|
|
49
|
+
)
|
|
50
|
+
|
|
51
|
+
#: How long a cached set is served before a refresh is attempted automatically.
|
|
52
|
+
#: Short enough that a weekly upstream refresh is picked up promptly, long enough
|
|
53
|
+
#: that a session of repeated queries makes one export rather than many.
|
|
54
|
+
DEFAULT_REFRESH_AFTER_SECONDS = 7 * 24 * 60 * 60
|
|
55
|
+
|
|
56
|
+
ENV_REFRESH_AFTER = "ICTRP_REFRESH_AFTER_SECONDS"
|
|
57
|
+
ENV_AUTO_REFRESH = "ICTRP_AUTO_REFRESH"
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def _refresh_after_seconds() -> int:
|
|
61
|
+
raw = os.environ.get(ENV_REFRESH_AFTER)
|
|
62
|
+
if raw:
|
|
63
|
+
try:
|
|
64
|
+
return max(0, int(raw))
|
|
65
|
+
except ValueError:
|
|
66
|
+
pass
|
|
67
|
+
return DEFAULT_REFRESH_AFTER_SECONDS
|
|
68
|
+
|
|
69
|
+
|
|
70
|
+
def _auto_refresh_enabled() -> bool:
|
|
71
|
+
"""Auto-refresh is on unless explicitly disabled.
|
|
72
|
+
|
|
73
|
+
"Off" exists for reproducible runs and for offline deployments, where reaching
|
|
74
|
+
the network at an unpredictable moment is worse than serving older data.
|
|
75
|
+
"""
|
|
76
|
+
raw = os.environ.get(ENV_AUTO_REFRESH, "").strip().lower()
|
|
77
|
+
return raw not in {"0", "false", "no", "off"}
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
class IctrpService:
|
|
81
|
+
"""Holds the cache and executes tools against it."""
|
|
82
|
+
|
|
83
|
+
#: How long a refresh failure suppresses further automatic attempts, so a
|
|
84
|
+
#: failing upstream is not retried on every single call.
|
|
85
|
+
REFRESH_FAILURE_BACKOFF_SECONDS = 15 * 60
|
|
86
|
+
|
|
87
|
+
def __init__(self, store: SetStore | None = None) -> None:
|
|
88
|
+
self.store = store or SetStore()
|
|
89
|
+
self._refresh_failures: dict[str, tuple[float, str]] = {}
|
|
90
|
+
|
|
91
|
+
# ---- network-backed ---------------------------------------------------
|
|
92
|
+
|
|
93
|
+
async def search(
|
|
94
|
+
self,
|
|
95
|
+
keyword: str,
|
|
96
|
+
*,
|
|
97
|
+
limit: int = 50,
|
|
98
|
+
offset: int = 0,
|
|
99
|
+
fields: list[str] | None = None,
|
|
100
|
+
filters: list[dict[str, Any]] | None = None,
|
|
101
|
+
sort_by: str | None = None,
|
|
102
|
+
descending: bool = False,
|
|
103
|
+
refresh: bool = False,
|
|
104
|
+
) -> dict[str, Any]:
|
|
105
|
+
"""Materialize a search, then return a page of it.
|
|
106
|
+
|
|
107
|
+
The set stays queryable afterwards, so subsequent refinement costs nothing.
|
|
108
|
+
|
|
109
|
+
Refresh policy: `refresh=True` forces an upstream search. Otherwise a cached
|
|
110
|
+
set is reused until it is older than the refresh interval, at which point an
|
|
111
|
+
automatic refresh is attempted -- and, if that fails, the cached set is
|
|
112
|
+
served anyway with the failure recorded in provenance. Serving known-stale
|
|
113
|
+
data is preferable to failing a query, provided the staleness is stated.
|
|
114
|
+
"""
|
|
115
|
+
if not keyword or not keyword.strip():
|
|
116
|
+
raise IctrpError(
|
|
117
|
+
ErrorCode.INVALID_ARGUMENT,
|
|
118
|
+
"keyword is required",
|
|
119
|
+
hint="Provide a search term, e.g. 'pancreatic cancer'.",
|
|
120
|
+
)
|
|
121
|
+
if limit < 1 or limit > 1000:
|
|
122
|
+
raise IctrpError(
|
|
123
|
+
ErrorCode.INVALID_ARGUMENT,
|
|
124
|
+
"limit must be between 1 and 1000",
|
|
125
|
+
)
|
|
126
|
+
if offset < 0:
|
|
127
|
+
raise IctrpError(ErrorCode.INVALID_ARGUMENT, "offset must be >= 0")
|
|
128
|
+
|
|
129
|
+
cache_hit = False
|
|
130
|
+
refresh_error: str | None = None
|
|
131
|
+
# Always look for a cached set, even when a refresh was forced. A forced
|
|
132
|
+
# refresh means "try to get fresh data", not "throw away what we have":
|
|
133
|
+
# if the attempt fails, the cached set is still the best answer
|
|
134
|
+
# available, and the code below is written to serve it with the failure
|
|
135
|
+
# recorded. Skipping this lookup made `previous` None, so the failure
|
|
136
|
+
# handler re-raised and an explicit refresh on a flaky network
|
|
137
|
+
# destroyed usable cached data instead of falling back to it.
|
|
138
|
+
materialized = self._cached_set(keyword)
|
|
139
|
+
|
|
140
|
+
should_refresh = refresh or (materialized is not None and self._should_refresh(materialized))
|
|
141
|
+
if should_refresh and self._auto_refresh_allows(keyword, force=refresh):
|
|
142
|
+
previous = materialized
|
|
143
|
+
try:
|
|
144
|
+
materialized = await self._materialize(keyword)
|
|
145
|
+
self._refresh_failures.pop(query_key(keyword), None)
|
|
146
|
+
except IctrpError as exc:
|
|
147
|
+
if previous is None:
|
|
148
|
+
raise
|
|
149
|
+
# We already hold data. Report the failure alongside it rather than
|
|
150
|
+
# discarding a usable set.
|
|
151
|
+
materialized = previous
|
|
152
|
+
refresh_error = f"{exc.code.value}: {exc.message}"
|
|
153
|
+
self._refresh_failures[query_key(keyword)] = (time.time(), refresh_error)
|
|
154
|
+
|
|
155
|
+
if materialized is None:
|
|
156
|
+
materialized = await self._materialize(keyword)
|
|
157
|
+
elif not should_refresh:
|
|
158
|
+
cache_hit = True
|
|
159
|
+
else:
|
|
160
|
+
cache_hit = refresh_error is not None
|
|
161
|
+
|
|
162
|
+
trials = materialized.trials
|
|
163
|
+
filtered = q.apply_filters(trials, filters)
|
|
164
|
+
ordered = q.sort_trials(filtered, sort_by, descending)
|
|
165
|
+
page = ordered[offset : offset + limit]
|
|
166
|
+
|
|
167
|
+
provenance = Provenance.from_dict(materialized.provenance) if materialized.provenance else Provenance.now()
|
|
168
|
+
provenance.cache_hit = cache_hit
|
|
169
|
+
provenance.rows_returned = materialized.rows_returned
|
|
170
|
+
provenance.upstream_reported_total = materialized.reported_total
|
|
171
|
+
provenance.records_incomplete = materialized.is_incomplete
|
|
172
|
+
provenance.estimated_missing = materialized.missing_estimate
|
|
173
|
+
self._annotate_age(provenance, materialized)
|
|
174
|
+
if refresh_error:
|
|
175
|
+
provenance.notes.append(
|
|
176
|
+
f"Automatic refresh failed and an older cached set was served instead. {refresh_error}"
|
|
177
|
+
)
|
|
178
|
+
|
|
179
|
+
selected = list(fields) if fields else list(DEFAULT_FIELDS)
|
|
180
|
+
return {
|
|
181
|
+
"status": "ok",
|
|
182
|
+
"set_id": materialized.set_id,
|
|
183
|
+
"matched_rows_returned": len(ordered),
|
|
184
|
+
"upstream_reported_total": materialized.reported_total,
|
|
185
|
+
"records_incomplete": materialized.is_incomplete,
|
|
186
|
+
"estimated_missing": materialized.missing_estimate,
|
|
187
|
+
"counts_are_of_retrieved_rows_not_of_matching_trials": True,
|
|
188
|
+
"offset": offset,
|
|
189
|
+
"limit": limit,
|
|
190
|
+
"trials": [{"trial_id": t.get("trial_id"), **{f: t.get(f) for f in selected}} for t in page],
|
|
191
|
+
"provenance": provenance.to_dict(),
|
|
192
|
+
}
|
|
193
|
+
|
|
194
|
+
def _should_refresh(self, materialized) -> bool:
|
|
195
|
+
return materialized.age_seconds() > _refresh_after_seconds()
|
|
196
|
+
|
|
197
|
+
def _auto_refresh_allows(self, keyword: str, *, force: bool) -> bool:
|
|
198
|
+
"""Whether a refresh may be attempted right now.
|
|
199
|
+
|
|
200
|
+
A forced refresh always proceeds -- that is what forcing means, and the
|
|
201
|
+
caller has explicitly accepted the cost. Note this must be an early
|
|
202
|
+
`return True`, not a fall-through: an inverted test here made
|
|
203
|
+
`refresh=True` a silent no-op, so the one escape hatch out of the
|
|
204
|
+
backoff window below did not exist.
|
|
205
|
+
|
|
206
|
+
An automatic one is skipped when it is disabled, or when a recent
|
|
207
|
+
attempt already failed -- repeatedly retrying a failing upstream would
|
|
208
|
+
turn one outage into a request storm.
|
|
209
|
+
"""
|
|
210
|
+
if force:
|
|
211
|
+
return True
|
|
212
|
+
if not _auto_refresh_enabled():
|
|
213
|
+
return False
|
|
214
|
+
failure = self._refresh_failures.get(query_key(keyword))
|
|
215
|
+
if failure is None:
|
|
216
|
+
return True
|
|
217
|
+
failed_at, _ = failure
|
|
218
|
+
return (time.time() - failed_at) > self.REFRESH_FAILURE_BACKOFF_SECONDS
|
|
219
|
+
|
|
220
|
+
def _annotate_age(self, provenance: Provenance, materialized) -> None:
|
|
221
|
+
"""Record how old the served data is, whatever its origin."""
|
|
222
|
+
age_days = materialized.age_seconds() / 86400.0
|
|
223
|
+
provenance.notes.append(
|
|
224
|
+
f"Data age: {age_days:.2f} days (set created {_iso(materialized.created_at)})."
|
|
225
|
+
)
|
|
226
|
+
if materialized.age_seconds() > _refresh_after_seconds():
|
|
227
|
+
provenance.notes.append(
|
|
228
|
+
"This set is older than the configured refresh interval "
|
|
229
|
+
f"({_refresh_after_seconds()}s); it was served because a refresh was "
|
|
230
|
+
"not permitted or did not succeed."
|
|
231
|
+
)
|
|
232
|
+
|
|
233
|
+
async def _materialize(self, keyword: str):
|
|
234
|
+
"""Run the upstream chain, falling back to a local snapshot when possible.
|
|
235
|
+
|
|
236
|
+
The fallback is deliberately narrow: snapshots are consulted only after an
|
|
237
|
+
upstream attempt has failed, never in preference to it. A deployment that
|
|
238
|
+
can reach the network should always get live data; a deployment that cannot
|
|
239
|
+
should get clearly-labelled older data instead of an error.
|
|
240
|
+
"""
|
|
241
|
+
try:
|
|
242
|
+
return await self._fetch_upstream(keyword)
|
|
243
|
+
except IctrpError as exc:
|
|
244
|
+
if not _is_offline_fallback_worthy(exc):
|
|
245
|
+
raise
|
|
246
|
+
bundled = offline.load_bundle(keyword)
|
|
247
|
+
if bundled is None:
|
|
248
|
+
raise
|
|
249
|
+
snapshot, path = bundled
|
|
250
|
+
provenance = dict(snapshot.provenance)
|
|
251
|
+
provenance.update(offline.snapshot_provenance(snapshot, path))
|
|
252
|
+
notes = list(provenance.get("notes") or [])
|
|
253
|
+
notes.append(
|
|
254
|
+
f"Upstream request failed ({exc.code.value}); this response was served "
|
|
255
|
+
f"from a local snapshot instead. Upstream error: {exc.message}"
|
|
256
|
+
)
|
|
257
|
+
provenance["notes"] = notes
|
|
258
|
+
return self.store.adopt(
|
|
259
|
+
keyword=keyword,
|
|
260
|
+
trials=snapshot.trials,
|
|
261
|
+
provenance=provenance,
|
|
262
|
+
created_at=snapshot.created_at,
|
|
263
|
+
source_label="snapshot",
|
|
264
|
+
)
|
|
265
|
+
|
|
266
|
+
async def _fetch_upstream(self, keyword: str):
|
|
267
|
+
session = IctrpSession.create()
|
|
268
|
+
try:
|
|
269
|
+
await session.load_form()
|
|
270
|
+
await session.search(keyword)
|
|
271
|
+
payload = await session.export_csv()
|
|
272
|
+
steps = session.provenance_steps()
|
|
273
|
+
response_date = session.response_date
|
|
274
|
+
finally:
|
|
275
|
+
await session.aclose()
|
|
276
|
+
|
|
277
|
+
provenance = derive_set_provenance(
|
|
278
|
+
keyword=keyword,
|
|
279
|
+
trials=[], # filled in by the store; only used for date extraction below
|
|
280
|
+
reported_total=payload.reported_total,
|
|
281
|
+
response_date=response_date,
|
|
282
|
+
steps=steps,
|
|
283
|
+
)
|
|
284
|
+
|
|
285
|
+
# Build trial dicts once to derive export/last-refreshed dates for provenance.
|
|
286
|
+
from .data.normalize import to_trial
|
|
287
|
+
|
|
288
|
+
trials = [to_trial(row, payload.header) for row in payload.rows]
|
|
289
|
+
export_dates = sorted({t.get("export_date_raw") for t in trials if t.get("export_date_raw")})
|
|
290
|
+
refreshed = sorted(
|
|
291
|
+
{t.get("last_refreshed_display") for t in trials if t.get("last_refreshed_display")}
|
|
292
|
+
)
|
|
293
|
+
if export_dates:
|
|
294
|
+
provenance.ictrp_export_date = export_dates[0]
|
|
295
|
+
provenance.notes = [n for n in provenance.notes if "Export date" not in n]
|
|
296
|
+
|
|
297
|
+
raw = _serialize_csv(payload.header, payload.rows)
|
|
298
|
+
materialized = self.store.store(
|
|
299
|
+
keyword=keyword,
|
|
300
|
+
payload_raw=raw,
|
|
301
|
+
header=payload.header,
|
|
302
|
+
rows=payload.rows,
|
|
303
|
+
reported_total=payload.reported_total,
|
|
304
|
+
provenance=provenance.to_dict(),
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
if payload.is_empty:
|
|
308
|
+
raise IctrpError(
|
|
309
|
+
ErrorCode.NO_RESULTS,
|
|
310
|
+
f"The portal returned a valid export with no matching trials for {keyword!r}",
|
|
311
|
+
hint=(
|
|
312
|
+
"This is a genuine zero: the export was structurally valid and "
|
|
313
|
+
"contained no data rows."
|
|
314
|
+
),
|
|
315
|
+
)
|
|
316
|
+
return materialized
|
|
317
|
+
|
|
318
|
+
def _cached_set(self, keyword: str):
|
|
319
|
+
candidate = query_key(keyword)
|
|
320
|
+
try:
|
|
321
|
+
return self.store.get(candidate)
|
|
322
|
+
except IctrpError:
|
|
323
|
+
return None
|
|
324
|
+
|
|
325
|
+
# ---- local-only -------------------------------------------------------
|
|
326
|
+
|
|
327
|
+
def filter_set(
|
|
328
|
+
self,
|
|
329
|
+
set_id: str,
|
|
330
|
+
*,
|
|
331
|
+
filters: list[dict[str, Any]] | None = None,
|
|
332
|
+
sort_by: str | None = None,
|
|
333
|
+
descending: bool = False,
|
|
334
|
+
limit: int = 50,
|
|
335
|
+
offset: int = 0,
|
|
336
|
+
fields: list[str] | None = None,
|
|
337
|
+
) -> dict[str, Any]:
|
|
338
|
+
materialized = self.store.get(set_id)
|
|
339
|
+
filtered = q.apply_filters(materialized.trials, filters)
|
|
340
|
+
ordered = q.sort_trials(filtered, sort_by, descending)
|
|
341
|
+
page = ordered[offset : offset + limit]
|
|
342
|
+
selected = list(fields) if fields else list(DEFAULT_FIELDS)
|
|
343
|
+
|
|
344
|
+
provenance = Provenance.from_dict(materialized.provenance)
|
|
345
|
+
provenance.cache_hit = True
|
|
346
|
+
provenance.rows_returned = materialized.rows_returned
|
|
347
|
+
provenance.upstream_reported_total = materialized.reported_total
|
|
348
|
+
provenance.records_incomplete = materialized.is_incomplete
|
|
349
|
+
provenance.estimated_missing = materialized.missing_estimate
|
|
350
|
+
|
|
351
|
+
return {
|
|
352
|
+
"status": "ok",
|
|
353
|
+
"set_id": set_id,
|
|
354
|
+
"matched_rows_returned": len(ordered),
|
|
355
|
+
"of_rows_in_set": materialized.rows_returned,
|
|
356
|
+
"counts_are_of_retrieved_rows_not_of_matching_trials": True,
|
|
357
|
+
"offset": offset,
|
|
358
|
+
"limit": limit,
|
|
359
|
+
"trials": [{"trial_id": t.get("trial_id"), **{f: t.get(f) for f in selected}} for t in page],
|
|
360
|
+
"provenance": provenance.to_dict(),
|
|
361
|
+
"note": "This operation was served entirely from the cached set; no upstream request was made.",
|
|
362
|
+
}
|
|
363
|
+
|
|
364
|
+
def field_query(
|
|
365
|
+
self,
|
|
366
|
+
*,
|
|
367
|
+
field: str,
|
|
368
|
+
set_id: str | None = None,
|
|
369
|
+
keyword: str | None = None,
|
|
370
|
+
limit: int = 50,
|
|
371
|
+
) -> dict[str, Any]:
|
|
372
|
+
materialized = self._resolve(set_id, keyword)
|
|
373
|
+
values = q.facet(materialized.trials, field, limit=limit)
|
|
374
|
+
coverage = q.field_coverage(materialized.trials, [field])[field]
|
|
375
|
+
|
|
376
|
+
provenance = Provenance.from_dict(materialized.provenance)
|
|
377
|
+
provenance.cache_hit = True
|
|
378
|
+
return {
|
|
379
|
+
"status": "ok",
|
|
380
|
+
"set_id": materialized.set_id,
|
|
381
|
+
"field": field,
|
|
382
|
+
"coverage": coverage,
|
|
383
|
+
"distinct_values": values,
|
|
384
|
+
"provenance": provenance.to_dict(),
|
|
385
|
+
"note": "Served from the cached set; no upstream request was made.",
|
|
386
|
+
}
|
|
387
|
+
|
|
388
|
+
def registry_summary(
|
|
389
|
+
self,
|
|
390
|
+
*,
|
|
391
|
+
set_id: str | None = None,
|
|
392
|
+
keyword: str | None = None,
|
|
393
|
+
group_by: list[str] | None = None,
|
|
394
|
+
) -> dict[str, Any]:
|
|
395
|
+
materialized = self._resolve(set_id, keyword)
|
|
396
|
+
groups = group_by or ["source_register"]
|
|
397
|
+
facets = {g: q.facet(materialized.trials, g) for g in groups}
|
|
398
|
+
|
|
399
|
+
provenance = Provenance.from_dict(materialized.provenance)
|
|
400
|
+
provenance.cache_hit = True
|
|
401
|
+
return {
|
|
402
|
+
"status": "ok",
|
|
403
|
+
"set_id": materialized.set_id,
|
|
404
|
+
"rows_in_set": materialized.rows_returned,
|
|
405
|
+
"upstream_reported_total": materialized.reported_total,
|
|
406
|
+
"records_incomplete": materialized.is_incomplete,
|
|
407
|
+
"facets": facets,
|
|
408
|
+
"field_coverage": q.field_coverage(
|
|
409
|
+
materialized.trials,
|
|
410
|
+
[
|
|
411
|
+
"trial_id", "public_title", "scientific_title", "condition",
|
|
412
|
+
"intervention", "primary_outcome", "secondary_outcome",
|
|
413
|
+
"inclusion_criteria", "exclusion_criteria",
|
|
414
|
+
"target_size_total", "countries", "inclusion_agemin",
|
|
415
|
+
"inclusion_agemax", "inclusion_gender", "ethics_status",
|
|
416
|
+
"secondary_id", "results_yes_no",
|
|
417
|
+
],
|
|
418
|
+
),
|
|
419
|
+
"provenance": provenance.to_dict(),
|
|
420
|
+
"note": "Served from the cached set; no upstream request was made.",
|
|
421
|
+
}
|
|
422
|
+
|
|
423
|
+
def find_duplicates(
|
|
424
|
+
self,
|
|
425
|
+
*,
|
|
426
|
+
set_id: str | None = None,
|
|
427
|
+
keyword: str | None = None,
|
|
428
|
+
) -> dict[str, Any]:
|
|
429
|
+
materialized = self._resolve(set_id, keyword)
|
|
430
|
+
groups = q.find_duplicates(materialized.trials)
|
|
431
|
+
provenance = Provenance.from_dict(materialized.provenance)
|
|
432
|
+
provenance.cache_hit = True
|
|
433
|
+
return {
|
|
434
|
+
"status": "ok",
|
|
435
|
+
"set_id": materialized.set_id,
|
|
436
|
+
"rows_in_set": materialized.rows_returned,
|
|
437
|
+
"candidate_groups": groups,
|
|
438
|
+
"group_count": len(groups),
|
|
439
|
+
"method": (
|
|
440
|
+
"Identifier cross-references only (Secondary ID). Title similarity is "
|
|
441
|
+
"deliberately not used; multi-centre trials share near-identical titles."
|
|
442
|
+
),
|
|
443
|
+
"provenance": provenance.to_dict(),
|
|
444
|
+
"note": "Served from the cached set; no upstream request was made.",
|
|
445
|
+
}
|
|
446
|
+
|
|
447
|
+
def export_records(
|
|
448
|
+
self,
|
|
449
|
+
*,
|
|
450
|
+
set_id: str | None = None,
|
|
451
|
+
keyword: str | None = None,
|
|
452
|
+
fmt: str = "json",
|
|
453
|
+
fields: list[str] | None = None,
|
|
454
|
+
filters: list[dict[str, Any]] | None = None,
|
|
455
|
+
include_provenance_header: bool = True,
|
|
456
|
+
) -> dict[str, Any]:
|
|
457
|
+
materialized = self._resolve(set_id, keyword)
|
|
458
|
+
rows = q.apply_filters(materialized.trials, filters)
|
|
459
|
+
selected = list(fields) if fields else list(materialized.trials[0].keys()) if materialized.trials else []
|
|
460
|
+
provenance = Provenance.from_dict(materialized.provenance)
|
|
461
|
+
provenance.cache_hit = True
|
|
462
|
+
|
|
463
|
+
if fmt == "csv":
|
|
464
|
+
buffer = io.StringIO()
|
|
465
|
+
writer = csv.writer(buffer)
|
|
466
|
+
writer.writerow(selected)
|
|
467
|
+
for row in rows:
|
|
468
|
+
writer.writerow([_cell(row.get(f)) for f in selected])
|
|
469
|
+
body = buffer.getvalue()
|
|
470
|
+
elif fmt == "jsonl":
|
|
471
|
+
body = "\n".join(
|
|
472
|
+
json.dumps({f: row.get(f) for f in selected}, ensure_ascii=False) for row in rows
|
|
473
|
+
)
|
|
474
|
+
elif fmt == "markdown":
|
|
475
|
+
lines = ["| " + " | ".join(selected) + " |", "|" + "---|" * len(selected)]
|
|
476
|
+
for row in rows:
|
|
477
|
+
lines.append("| " + " | ".join(_cell(row.get(f)) for f in selected) + " |")
|
|
478
|
+
body = "\n".join(lines)
|
|
479
|
+
elif fmt == "json":
|
|
480
|
+
body = json.dumps(
|
|
481
|
+
[{f: row.get(f) for f in selected} for row in rows],
|
|
482
|
+
ensure_ascii=False,
|
|
483
|
+
indent=2,
|
|
484
|
+
)
|
|
485
|
+
else:
|
|
486
|
+
raise IctrpError(
|
|
487
|
+
ErrorCode.INVALID_ARGUMENT,
|
|
488
|
+
f"unsupported format {fmt!r}",
|
|
489
|
+
hint="Supported: csv, json, jsonl, markdown.",
|
|
490
|
+
)
|
|
491
|
+
|
|
492
|
+
header = ""
|
|
493
|
+
if include_provenance_header:
|
|
494
|
+
header = (
|
|
495
|
+
f"# Source: WHO ICTRP ({provenance.source_url})\n"
|
|
496
|
+
f"# Retrieved: {provenance.retrieved_at}\n"
|
|
497
|
+
f"# Rows in this export: {len(rows)}\n"
|
|
498
|
+
)
|
|
499
|
+
if provenance.upstream_reported_total is not None:
|
|
500
|
+
header += f"# Portal-reported matches: {provenance.upstream_reported_total}\n"
|
|
501
|
+
if provenance.records_incomplete:
|
|
502
|
+
header += f"# WARNING: {INCOMPLETENESS_NOTICE}\n"
|
|
503
|
+
|
|
504
|
+
return {
|
|
505
|
+
"status": "ok",
|
|
506
|
+
"set_id": materialized.set_id,
|
|
507
|
+
"format": fmt,
|
|
508
|
+
"row_count": len(rows),
|
|
509
|
+
"content": header + body,
|
|
510
|
+
"provenance": provenance.to_dict(),
|
|
511
|
+
"note": "Served from the cached set; no upstream request was made.",
|
|
512
|
+
}
|
|
513
|
+
|
|
514
|
+
def cache_status(self, *, action: str = "list", set_id: str | None = None) -> dict[str, Any]:
|
|
515
|
+
if action == "list":
|
|
516
|
+
sets = self.store.list_sets()
|
|
517
|
+
return {
|
|
518
|
+
"status": "ok",
|
|
519
|
+
"action": "list",
|
|
520
|
+
"sets": sets,
|
|
521
|
+
"set_count": len(sets),
|
|
522
|
+
"cache_dir": str(self.store.cache_dir),
|
|
523
|
+
"max_in_memory_sets": 3,
|
|
524
|
+
"refresh_after_seconds": _refresh_after_seconds(),
|
|
525
|
+
"auto_refresh": _auto_refresh_enabled(),
|
|
526
|
+
"bundle_configured": {
|
|
527
|
+
"ICTRP_BUNDLE_PATH": os.environ.get(offline.ENV_BUNDLE_PATH),
|
|
528
|
+
"ICTRP_BUNDLE_DIR": os.environ.get(offline.ENV_BUNDLE_DIR),
|
|
529
|
+
"max_age_days": offline.max_age_days(),
|
|
530
|
+
},
|
|
531
|
+
"note": (
|
|
532
|
+
"Only a few sets stay parsed in memory; others are re-read from "
|
|
533
|
+
"disk on demand. Raw CSV is cached because re-exporting is expensive."
|
|
534
|
+
),
|
|
535
|
+
}
|
|
536
|
+
if action == "purge":
|
|
537
|
+
removed = self.store.purge(set_id)
|
|
538
|
+
return {"status": "ok", "action": "purge", "removed": removed, "set_id": set_id}
|
|
539
|
+
raise IctrpError(
|
|
540
|
+
ErrorCode.INVALID_ARGUMENT,
|
|
541
|
+
f"unsupported action {action!r}",
|
|
542
|
+
hint="Supported: list, purge.",
|
|
543
|
+
)
|
|
544
|
+
|
|
545
|
+
# ---- snapshot / bundle ------------------------------------------------
|
|
546
|
+
|
|
547
|
+
def snapshot(
|
|
548
|
+
self,
|
|
549
|
+
*,
|
|
550
|
+
set_id: str | None = None,
|
|
551
|
+
keyword: str | None = None,
|
|
552
|
+
path: str | None = None,
|
|
553
|
+
if_stale: bool = True,
|
|
554
|
+
) -> dict[str, Any]:
|
|
555
|
+
"""Write a cached set to a canonical JSON snapshot.
|
|
556
|
+
|
|
557
|
+
Intended for two uses: producing a dataset to ship inside a packaged
|
|
558
|
+
application, and refreshing such a dataset on a schedule. `if_stale=False`
|
|
559
|
+
forces a rewrite even when the existing snapshot is fresh, which is what a
|
|
560
|
+
release pipeline wants.
|
|
561
|
+
"""
|
|
562
|
+
materialized = self._resolve(set_id, keyword)
|
|
563
|
+
if not materialized.trials:
|
|
564
|
+
raise IctrpError(
|
|
565
|
+
ErrorCode.NO_RESULTS,
|
|
566
|
+
"refusing to write an empty snapshot",
|
|
567
|
+
hint=(
|
|
568
|
+
"A snapshot is a distribution artifact; an empty one would make a "
|
|
569
|
+
"packaged application look like it had no data."
|
|
570
|
+
),
|
|
571
|
+
)
|
|
572
|
+
|
|
573
|
+
target = self._snapshot_target(materialized.keyword, path)
|
|
574
|
+
|
|
575
|
+
if if_stale and target.is_file():
|
|
576
|
+
try:
|
|
577
|
+
existing = offline.load_bundle(materialized.keyword)
|
|
578
|
+
if existing is not None:
|
|
579
|
+
snapshot_obj, existing_path = existing
|
|
580
|
+
if existing_path == target and not snapshot_obj.stale(offline.max_age_days()):
|
|
581
|
+
return {
|
|
582
|
+
"status": "ok",
|
|
583
|
+
"action": "skipped",
|
|
584
|
+
"reason": "snapshot is still within its freshness limit",
|
|
585
|
+
"path": str(target),
|
|
586
|
+
"age_days": round(snapshot_obj.age_days, 2),
|
|
587
|
+
"trial_count": len(snapshot_obj.trials),
|
|
588
|
+
}
|
|
589
|
+
except IctrpError:
|
|
590
|
+
# An unreadable existing file is a reason to overwrite it, not to fail.
|
|
591
|
+
pass
|
|
592
|
+
|
|
593
|
+
provenance = Provenance.from_dict(materialized.provenance)
|
|
594
|
+
provenance.cache_hit = True
|
|
595
|
+
snapshot_obj = snapshot_from_trials(
|
|
596
|
+
trials=materialized.trials,
|
|
597
|
+
keyword=materialized.keyword,
|
|
598
|
+
provenance=provenance.to_dict(),
|
|
599
|
+
created_at=materialized.created_at,
|
|
600
|
+
)
|
|
601
|
+
size = snapshot_obj.write(target)
|
|
602
|
+
|
|
603
|
+
return {
|
|
604
|
+
"status": "ok",
|
|
605
|
+
"action": "written",
|
|
606
|
+
"path": str(target),
|
|
607
|
+
"trial_count": len(snapshot_obj.trials),
|
|
608
|
+
"bytes": size,
|
|
609
|
+
"data_created_at": _iso(materialized.created_at),
|
|
610
|
+
"provenance": provenance.to_dict(),
|
|
611
|
+
"terms_notice": (
|
|
612
|
+
"This file contains WHO ICTRP data. Redistributing it is a decision "
|
|
613
|
+
"about the ICTRP terms of use, not a technical default. See "
|
|
614
|
+
"docs/BUNDLE.md."
|
|
615
|
+
),
|
|
616
|
+
}
|
|
617
|
+
|
|
618
|
+
def bundle_status(self, *, keyword: str | None = None) -> dict[str, Any]:
|
|
619
|
+
"""Report which snapshot would serve a query, and how old it is."""
|
|
620
|
+
if not keyword:
|
|
621
|
+
raise IctrpError(
|
|
622
|
+
ErrorCode.INVALID_ARGUMENT,
|
|
623
|
+
"keyword is required to locate a snapshot",
|
|
624
|
+
)
|
|
625
|
+
candidates = [str(p) for p in offline.candidate_paths(keyword)]
|
|
626
|
+
bundled = offline.load_bundle(keyword)
|
|
627
|
+
out: dict[str, Any] = {
|
|
628
|
+
"status": "ok",
|
|
629
|
+
"keyword": keyword,
|
|
630
|
+
"candidates_checked": candidates,
|
|
631
|
+
"bundle_found": bundled is not None,
|
|
632
|
+
"configured": {
|
|
633
|
+
"ICTRP_BUNDLE_PATH": os.environ.get(offline.ENV_BUNDLE_PATH),
|
|
634
|
+
"ICTRP_BUNDLE_DIR": os.environ.get(offline.ENV_BUNDLE_DIR),
|
|
635
|
+
"max_age_days": offline.max_age_days(),
|
|
636
|
+
},
|
|
637
|
+
"auto_refresh": _auto_refresh_enabled(),
|
|
638
|
+
"refresh_after_seconds": _refresh_after_seconds(),
|
|
639
|
+
}
|
|
640
|
+
if bundled is not None:
|
|
641
|
+
snapshot_obj, path = bundled
|
|
642
|
+
out["snapshot"] = offline.snapshot_provenance(snapshot_obj, path)
|
|
643
|
+
out["snapshot"]["keyword"] = snapshot_obj.keyword
|
|
644
|
+
else:
|
|
645
|
+
out["hint"] = (
|
|
646
|
+
"No snapshot for this keyword. Run ictrp_search then ictrp_snapshot to "
|
|
647
|
+
"create one, or set ICTRP_BUNDLE_PATH / ICTRP_BUNDLE_DIR."
|
|
648
|
+
)
|
|
649
|
+
return out
|
|
650
|
+
|
|
651
|
+
def _snapshot_target(self, keyword: str, path: str | None) -> Path:
|
|
652
|
+
if path:
|
|
653
|
+
return Path(path).expanduser()
|
|
654
|
+
directory = offline.bundle_dir()
|
|
655
|
+
if directory:
|
|
656
|
+
return directory / f"{offline.snapshot_slug(keyword)}.json"
|
|
657
|
+
return self.store.cache_dir / "snapshots" / f"{offline.snapshot_slug(keyword)}.json"
|
|
658
|
+
|
|
659
|
+
# ---- internals --------------------------------------------------------
|
|
660
|
+
|
|
661
|
+
def _resolve(self, set_id: str | None, keyword: str | None):
|
|
662
|
+
if set_id:
|
|
663
|
+
return self.store.get(set_id)
|
|
664
|
+
if keyword:
|
|
665
|
+
materialized = self._cached_set(keyword)
|
|
666
|
+
if materialized is not None:
|
|
667
|
+
return materialized
|
|
668
|
+
raise IctrpError(
|
|
669
|
+
ErrorCode.CACHE_MISS,
|
|
670
|
+
f"no cached result set for keyword {keyword!r}",
|
|
671
|
+
hint="Run ictrp_search with this keyword first, or pass a set_id.",
|
|
672
|
+
)
|
|
673
|
+
raise IctrpError(
|
|
674
|
+
ErrorCode.INVALID_ARGUMENT,
|
|
675
|
+
"provide either set_id or keyword",
|
|
676
|
+
)
|
|
677
|
+
|
|
678
|
+
|
|
679
|
+
def _cell(value: Any) -> str:
|
|
680
|
+
if value is None:
|
|
681
|
+
return ""
|
|
682
|
+
if isinstance(value, (list, dict)):
|
|
683
|
+
return json.dumps(value, ensure_ascii=False)
|
|
684
|
+
return str(value)
|
|
685
|
+
|
|
686
|
+
|
|
687
|
+
def _iso(epoch: float) -> str:
|
|
688
|
+
return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(epoch))
|
|
689
|
+
|
|
690
|
+
|
|
691
|
+
#: Failures that justify falling back to a local snapshot. A missing result or a
|
|
692
|
+
#: malformed request will not be fixed by older data, so those propagate.
|
|
693
|
+
_OFFLINE_FALLBACK_CODES = frozenset(
|
|
694
|
+
{
|
|
695
|
+
ErrorCode.UPSTREAM_BLOCKED,
|
|
696
|
+
ErrorCode.UPSTREAM_ERROR,
|
|
697
|
+
ErrorCode.SESSION_FAILED,
|
|
698
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
699
|
+
}
|
|
700
|
+
)
|
|
701
|
+
|
|
702
|
+
|
|
703
|
+
def _is_offline_fallback_worthy(exc: IctrpError) -> bool:
|
|
704
|
+
return exc.code in _OFFLINE_FALLBACK_CODES
|
|
705
|
+
|
|
706
|
+
|
|
707
|
+
def _serialize_csv(header: tuple[str, ...], rows: list[list[str]]) -> bytes:
|
|
708
|
+
buffer = io.StringIO()
|
|
709
|
+
writer = csv.writer(buffer)
|
|
710
|
+
writer.writerow(header)
|
|
711
|
+
writer.writerows(rows)
|
|
712
|
+
return buffer.getvalue().encode("utf-8")
|