ictrp-mcp-server 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/LICENSE +37 -0
- package/README.md +208 -0
- package/README_ZH.md +189 -0
- package/dist/cli/setup-cli.d.ts +14 -0
- package/dist/cli/setup-cli.js +230 -0
- package/dist/cli/setup-cli.js.map +1 -0
- package/dist/index.d.ts +18 -0
- package/dist/index.js +477 -0
- package/dist/index.js.map +1 -0
- package/dist/runtime/bootstrap.d.ts +99 -0
- package/dist/runtime/bootstrap.js +350 -0
- package/dist/runtime/bootstrap.js.map +1 -0
- package/dist/runtime/env-probe.d.ts +108 -0
- package/dist/runtime/env-probe.js +479 -0
- package/dist/runtime/env-probe.js.map +1 -0
- package/dist/runtime/sidecar-client.d.ts +50 -0
- package/dist/runtime/sidecar-client.js +120 -0
- package/dist/runtime/sidecar-client.js.map +1 -0
- package/dist/runtime/supervisor.d.ts +47 -0
- package/dist/runtime/supervisor.js +248 -0
- package/dist/runtime/supervisor.js.map +1 -0
- package/package.json +60 -0
- package/sidecar/ictrp_sidecar.py +602 -0
- package/sidecar/vendor/ictrp_mcp/__init__.py +3 -0
- package/sidecar/vendor/ictrp_mcp/cache/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/cache/store.py +313 -0
- package/sidecar/vendor/ictrp_mcp/data/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/data/columns.py +108 -0
- package/sidecar/vendor/ictrp_mcp/data/jsonio.py +213 -0
- package/sidecar/vendor/ictrp_mcp/data/normalize.py +348 -0
- package/sidecar/vendor/ictrp_mcp/data/query.py +307 -0
- package/sidecar/vendor/ictrp_mcp/errors.py +123 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/export_guard.py +269 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/htmlstate.py +143 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/session.py +245 -0
- package/sidecar/vendor/ictrp_mcp/offline.py +133 -0
- package/sidecar/vendor/ictrp_mcp/provenance.py +182 -0
- package/sidecar/vendor/ictrp_mcp/server.py +368 -0
- package/sidecar/vendor/ictrp_mcp/tools.py +712 -0
- package/sidecar/vendor/pyproject.toml +25 -0
|
@@ -0,0 +1,313 @@
|
|
|
1
|
+
"""Caching for materialized result sets.
|
|
2
|
+
|
|
3
|
+
The governing constraint is cost: a single export can be tens of megabytes, and
|
|
4
|
+
the chain behind it is three round trips to a public service we should not hammer.
|
|
5
|
+
So a search result is materialized once and then queried locally.
|
|
6
|
+
|
|
7
|
+
Two deliberate choices:
|
|
8
|
+
|
|
9
|
+
* Raw CSV bytes are written to disk instead of holding every row in memory for
|
|
10
|
+
every live set. A large set is only parsed when it is actually used.
|
|
11
|
+
* Only a small number of sets stay in memory. Older ones are dropped (and can be
|
|
12
|
+
re-read from disk), which bounds the process footprint.
|
|
13
|
+
|
|
14
|
+
Nothing here caches a failure as if it were data. A set only enters the store
|
|
15
|
+
after it has been validated.
|
|
16
|
+
"""
|
|
17
|
+
|
|
18
|
+
from __future__ import annotations
|
|
19
|
+
|
|
20
|
+
import hashlib
|
|
21
|
+
import json
|
|
22
|
+
import os
|
|
23
|
+
import re
|
|
24
|
+
import time
|
|
25
|
+
from dataclasses import dataclass, field
|
|
26
|
+
from pathlib import Path
|
|
27
|
+
from typing import Any
|
|
28
|
+
|
|
29
|
+
from ..errors import ErrorCode, IctrpError
|
|
30
|
+
from ..data.normalize import to_trial
|
|
31
|
+
|
|
32
|
+
#: Parsed sets held in memory at once.
|
|
33
|
+
MAX_MATERIALIZED_SETS = 3
|
|
34
|
+
|
|
35
|
+
#: How long a materialized set stays queryable.
|
|
36
|
+
DEFAULT_SET_TTL_SECONDS = 7 * 24 * 60 * 60
|
|
37
|
+
|
|
38
|
+
#: How long a single-trial lookup stays fresh.
|
|
39
|
+
DEFAULT_TRIAL_TTL_SECONDS = 24 * 60 * 60
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def normalize_query(keyword: str) -> str:
|
|
43
|
+
"""Collapse a keyword to a stable cache key component."""
|
|
44
|
+
return re.sub(r"\s+", " ", keyword.strip().lower())
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def query_key(keyword: str) -> str:
|
|
48
|
+
digest = hashlib.sha256(normalize_query(keyword).encode()).hexdigest()[:16]
|
|
49
|
+
return f"search:{digest}"
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def trial_key(trial_id: str) -> str:
|
|
53
|
+
return f"trial:{trial_id.strip().upper()}"
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
def default_cache_dir() -> Path:
|
|
57
|
+
override = os.environ.get("ICTRP_CACHE_DIR")
|
|
58
|
+
if override:
|
|
59
|
+
return Path(override)
|
|
60
|
+
return Path.home() / ".cache" / "ictrp-mcp-service"
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
@dataclass
|
|
64
|
+
class MaterializedSet:
|
|
65
|
+
"""One export, parsed into canonical trials."""
|
|
66
|
+
|
|
67
|
+
set_id: str
|
|
68
|
+
keyword: str
|
|
69
|
+
trials: list[dict[str, Any]]
|
|
70
|
+
header: tuple[str, ...]
|
|
71
|
+
reported_total: int | None
|
|
72
|
+
csv_bytes: int
|
|
73
|
+
created_at: float
|
|
74
|
+
provenance: dict[str, Any] = field(default_factory=dict)
|
|
75
|
+
|
|
76
|
+
@property
|
|
77
|
+
def rows_returned(self) -> int:
|
|
78
|
+
"""How many rows we actually hold.
|
|
79
|
+
|
|
80
|
+
Never described as a total. The portal's own figure lives separately in
|
|
81
|
+
`reported_total`, and the two are known to differ by 0.4%-29%.
|
|
82
|
+
"""
|
|
83
|
+
return len(self.trials)
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def is_incomplete(self) -> bool:
|
|
87
|
+
"""True when the portal reported more matches than the export delivered."""
|
|
88
|
+
if self.reported_total is None:
|
|
89
|
+
return False
|
|
90
|
+
return self.reported_total > self.rows_returned
|
|
91
|
+
|
|
92
|
+
@property
|
|
93
|
+
def missing_estimate(self) -> int | None:
|
|
94
|
+
if self.reported_total is None:
|
|
95
|
+
return None
|
|
96
|
+
return max(0, self.reported_total - self.rows_returned)
|
|
97
|
+
|
|
98
|
+
def age_seconds(self) -> float:
|
|
99
|
+
return time.time() - self.created_at
|
|
100
|
+
|
|
101
|
+
def summary(self) -> dict[str, Any]:
|
|
102
|
+
return {
|
|
103
|
+
"set_id": self.set_id,
|
|
104
|
+
"keyword": self.keyword,
|
|
105
|
+
"rows_returned": self.rows_returned,
|
|
106
|
+
"upstream_reported_total": self.reported_total,
|
|
107
|
+
"records_incomplete": self.is_incomplete,
|
|
108
|
+
"estimated_missing": self.missing_estimate,
|
|
109
|
+
"csv_bytes": self.csv_bytes,
|
|
110
|
+
"age_seconds": round(self.age_seconds(), 1),
|
|
111
|
+
"created_at": _iso(self.created_at),
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
|
|
115
|
+
def _iso(epoch: float) -> str:
|
|
116
|
+
return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(epoch))
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
class SetStore:
|
|
120
|
+
"""Holds materialized sets in memory, backed by raw CSV on disk."""
|
|
121
|
+
|
|
122
|
+
def __init__(self, cache_dir: Path | None = None, ttl_seconds: int = DEFAULT_SET_TTL_SECONDS):
|
|
123
|
+
self.cache_dir = cache_dir or default_cache_dir()
|
|
124
|
+
self.ttl_seconds = ttl_seconds
|
|
125
|
+
self._sets: dict[str, MaterializedSet] = {}
|
|
126
|
+
self._order: list[str] = []
|
|
127
|
+
|
|
128
|
+
def _raw_path(self, cache_key: str) -> Path:
|
|
129
|
+
safe = re.sub(r"[^A-Za-z0-9_.-]", "_", cache_key)
|
|
130
|
+
return self.cache_dir / f"{safe}.csv"
|
|
131
|
+
|
|
132
|
+
def _meta_path(self, cache_key: str) -> Path:
|
|
133
|
+
safe = re.sub(r"[^A-Za-z0-9_.-]", "_", cache_key)
|
|
134
|
+
return self.cache_dir / f"{safe}.json"
|
|
135
|
+
|
|
136
|
+
def _ensure_dir(self) -> None:
|
|
137
|
+
try:
|
|
138
|
+
self.cache_dir.mkdir(parents=True, exist_ok=True)
|
|
139
|
+
except OSError:
|
|
140
|
+
# Fall back to a temp location rather than failing the request.
|
|
141
|
+
self.cache_dir = Path("/tmp/ictrp-mcp-service")
|
|
142
|
+
self.cache_dir.mkdir(parents=True, exist_ok=True)
|
|
143
|
+
|
|
144
|
+
def store(
|
|
145
|
+
self,
|
|
146
|
+
*,
|
|
147
|
+
keyword: str,
|
|
148
|
+
payload_raw: bytes,
|
|
149
|
+
header: tuple[str, ...],
|
|
150
|
+
rows: list[list[str]],
|
|
151
|
+
reported_total: int | None,
|
|
152
|
+
provenance: dict[str, Any],
|
|
153
|
+
) -> MaterializedSet:
|
|
154
|
+
"""Persist and materialize a validated export."""
|
|
155
|
+
cache_key = query_key(keyword)
|
|
156
|
+
trials = [to_trial(row, header) for row in rows]
|
|
157
|
+
materialized = MaterializedSet(
|
|
158
|
+
set_id=cache_key,
|
|
159
|
+
keyword=keyword,
|
|
160
|
+
trials=trials,
|
|
161
|
+
header=header,
|
|
162
|
+
reported_total=reported_total,
|
|
163
|
+
csv_bytes=len(payload_raw),
|
|
164
|
+
created_at=time.time(),
|
|
165
|
+
provenance=dict(provenance),
|
|
166
|
+
)
|
|
167
|
+
|
|
168
|
+
self._ensure_dir()
|
|
169
|
+
try:
|
|
170
|
+
self._raw_path(cache_key).write_bytes(payload_raw)
|
|
171
|
+
self._meta_path(cache_key).write_text(
|
|
172
|
+
json.dumps(
|
|
173
|
+
{
|
|
174
|
+
"keyword": keyword,
|
|
175
|
+
"header": list(header),
|
|
176
|
+
"reported_total": reported_total,
|
|
177
|
+
"created_at": materialized.created_at,
|
|
178
|
+
"csv_bytes": materialized.csv_bytes,
|
|
179
|
+
"provenance": provenance,
|
|
180
|
+
}
|
|
181
|
+
),
|
|
182
|
+
encoding="utf-8",
|
|
183
|
+
)
|
|
184
|
+
except OSError:
|
|
185
|
+
# Disk caching is best-effort; the in-memory set is still usable.
|
|
186
|
+
pass
|
|
187
|
+
|
|
188
|
+
self._remember(materialized)
|
|
189
|
+
return materialized
|
|
190
|
+
|
|
191
|
+
def adopt(
|
|
192
|
+
self,
|
|
193
|
+
*,
|
|
194
|
+
keyword: str,
|
|
195
|
+
trials: list[dict[str, Any]],
|
|
196
|
+
provenance: dict[str, Any],
|
|
197
|
+
created_at: float | None = None,
|
|
198
|
+
source_label: str = "snapshot",
|
|
199
|
+
) -> MaterializedSet:
|
|
200
|
+
"""Register an externally built set (e.g. from an offline snapshot).
|
|
201
|
+
|
|
202
|
+
`created_at` is carried over from the snapshot so the set's age reflects
|
|
203
|
+
when the data was retrieved, not when it was loaded. A snapshot read today
|
|
204
|
+
from a month-old file must still report as a month old.
|
|
205
|
+
|
|
206
|
+
The set is memory-resident only. Writing it back as a raw CSV would
|
|
207
|
+
fabricate an export we never received, and `_reload_from_disk` would then
|
|
208
|
+
present it as a genuine export on the next run.
|
|
209
|
+
"""
|
|
210
|
+
materialized = MaterializedSet(
|
|
211
|
+
set_id=query_key(keyword),
|
|
212
|
+
keyword=keyword,
|
|
213
|
+
trials=list(trials),
|
|
214
|
+
header=(),
|
|
215
|
+
reported_total=provenance.get("upstream_reported_total"),
|
|
216
|
+
csv_bytes=0,
|
|
217
|
+
created_at=created_at if created_at is not None else time.time(),
|
|
218
|
+
provenance=dict(provenance),
|
|
219
|
+
)
|
|
220
|
+
materialized.provenance.setdefault("notes", [])
|
|
221
|
+
self._remember(materialized)
|
|
222
|
+
return materialized
|
|
223
|
+
|
|
224
|
+
def _remember(self, materialized: MaterializedSet) -> None:
|
|
225
|
+
self._sets[materialized.set_id] = materialized
|
|
226
|
+
if materialized.set_id in self._order:
|
|
227
|
+
self._order.remove(materialized.set_id)
|
|
228
|
+
self._order.append(materialized.set_id)
|
|
229
|
+
while len(self._order) > MAX_MATERIALIZED_SETS:
|
|
230
|
+
evicted = self._order.pop(0)
|
|
231
|
+
self._sets.pop(evicted, None)
|
|
232
|
+
|
|
233
|
+
def get(self, set_id: str) -> MaterializedSet:
|
|
234
|
+
"""Return a set, reloading skipped ones from disk when possible."""
|
|
235
|
+
materialized = self._sets.get(set_id)
|
|
236
|
+
if materialized is not None and materialized.age_seconds() <= self.ttl_seconds:
|
|
237
|
+
return materialized
|
|
238
|
+
if materialized is not None:
|
|
239
|
+
self._sets.pop(set_id, None)
|
|
240
|
+
if set_id in self._order:
|
|
241
|
+
self._order.remove(set_id)
|
|
242
|
+
|
|
243
|
+
reloaded = self._reload_from_disk(set_id)
|
|
244
|
+
if reloaded is not None:
|
|
245
|
+
self._remember(reloaded)
|
|
246
|
+
return reloaded
|
|
247
|
+
|
|
248
|
+
raise IctrpError(
|
|
249
|
+
ErrorCode.CACHE_MISS,
|
|
250
|
+
f"result set {set_id!r} is not available",
|
|
251
|
+
hint=(
|
|
252
|
+
"Result sets are held for a limited time and only a few at once. "
|
|
253
|
+
"Run the search again to re-materialize it."
|
|
254
|
+
),
|
|
255
|
+
)
|
|
256
|
+
|
|
257
|
+
def _reload_from_disk(self, set_id: str) -> MaterializedSet | None:
|
|
258
|
+
meta_path = self._meta_path(set_id)
|
|
259
|
+
raw_path = self._raw_path(set_id)
|
|
260
|
+
if not meta_path.exists() or not raw_path.exists():
|
|
261
|
+
return None
|
|
262
|
+
try:
|
|
263
|
+
meta = json.loads(meta_path.read_text(encoding="utf-8"))
|
|
264
|
+
raw = raw_path.read_bytes()
|
|
265
|
+
except (OSError, json.JSONDecodeError):
|
|
266
|
+
return None
|
|
267
|
+
|
|
268
|
+
created_at = float(meta.get("created_at", 0.0))
|
|
269
|
+
if time.time() - created_at > self.ttl_seconds:
|
|
270
|
+
return None
|
|
271
|
+
|
|
272
|
+
header, rows = _parse_raw(raw)
|
|
273
|
+
if not header:
|
|
274
|
+
return None
|
|
275
|
+
return MaterializedSet(
|
|
276
|
+
set_id=set_id,
|
|
277
|
+
keyword=meta.get("keyword", ""),
|
|
278
|
+
trials=[to_trial(r, header) for r in rows],
|
|
279
|
+
header=header,
|
|
280
|
+
reported_total=meta.get("reported_total"),
|
|
281
|
+
csv_bytes=meta.get("csv_bytes", len(raw)),
|
|
282
|
+
created_at=created_at,
|
|
283
|
+
provenance=meta.get("provenance", {}) or {},
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
def list_sets(self) -> list[dict[str, Any]]:
|
|
287
|
+
return [self._sets[s].summary() for s in self._order if s in self._sets]
|
|
288
|
+
|
|
289
|
+
def purge(self, set_id: str | None = None) -> int:
|
|
290
|
+
if set_id is None:
|
|
291
|
+
count = len(self._sets)
|
|
292
|
+
self._sets.clear()
|
|
293
|
+
self._order.clear()
|
|
294
|
+
return count
|
|
295
|
+
removed = 1 if self._sets.pop(set_id, None) is not None else 0
|
|
296
|
+
if set_id in self._order:
|
|
297
|
+
self._order.remove(set_id)
|
|
298
|
+
return removed
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def _parse_raw(raw: bytes) -> tuple[tuple[str, ...], list[list[str]]]:
|
|
302
|
+
import csv as _csv
|
|
303
|
+
import io as _io
|
|
304
|
+
|
|
305
|
+
text = raw.decode("utf-8-sig", "replace")
|
|
306
|
+
reader = _csv.reader(_io.StringIO(text))
|
|
307
|
+
try:
|
|
308
|
+
header = tuple(h.strip() for h in next(reader))
|
|
309
|
+
except StopIteration:
|
|
310
|
+
return (), []
|
|
311
|
+
rows = [r for r in reader if r and any(c.strip() for c in r)]
|
|
312
|
+
width = len(header)
|
|
313
|
+
return header, [(r + [""] * width)[:width] for r in rows]
|
|
File without changes
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
"""The 58 ICTRP CSV column headers, as literal strings.
|
|
2
|
+
|
|
3
|
+
These are reproduced VERBATIM from the live export, including three
|
|
4
|
+
misspellings present in the upstream data:
|
|
5
|
+
|
|
6
|
+
'Inclusion agemin' (not 'agemin' -> 'age_min')
|
|
7
|
+
'Inclusion agemax'
|
|
8
|
+
'Date enrollement' (not 'enrollment')
|
|
9
|
+
|
|
10
|
+
Do NOT "correct" these. The parser matches on the exact upstream strings; a
|
|
11
|
+
helpful-looking normalisation here would silently produce all-null fields.
|
|
12
|
+
`tests/test_columns.py` asserts the exact set so an upstream rename surfaces as
|
|
13
|
+
a failing test rather than a silently degraded response.
|
|
14
|
+
"""
|
|
15
|
+
|
|
16
|
+
from __future__ import annotations
|
|
17
|
+
|
|
18
|
+
#: Column order as returned by the live export. Order is not load-bearing for
|
|
19
|
+
#: parsing (we match by name) but is asserted for drift detection.
|
|
20
|
+
CSV_COLUMNS: tuple[str, ...] = (
|
|
21
|
+
"TrialID",
|
|
22
|
+
"Last Refreshed on",
|
|
23
|
+
"Public title",
|
|
24
|
+
"Scientific title",
|
|
25
|
+
"Acronym",
|
|
26
|
+
"Primary sponsor",
|
|
27
|
+
"Date registration",
|
|
28
|
+
"Date registration3",
|
|
29
|
+
"Export date",
|
|
30
|
+
"Source Register",
|
|
31
|
+
"web address",
|
|
32
|
+
"Recruitment Status",
|
|
33
|
+
"other records",
|
|
34
|
+
"Inclusion agemin",
|
|
35
|
+
"Inclusion agemax",
|
|
36
|
+
"Inclusion gender",
|
|
37
|
+
"Date enrollement",
|
|
38
|
+
"Target size",
|
|
39
|
+
"Study type",
|
|
40
|
+
"Study design",
|
|
41
|
+
"Phase",
|
|
42
|
+
"Countries",
|
|
43
|
+
"Contact Firstname",
|
|
44
|
+
"Contact Lastname",
|
|
45
|
+
"Contact Address",
|
|
46
|
+
"Contact Email",
|
|
47
|
+
"Contact Tel",
|
|
48
|
+
"Contact Affiliation",
|
|
49
|
+
"Inclusion Criteria",
|
|
50
|
+
"Exclusion Criteria",
|
|
51
|
+
"Condition",
|
|
52
|
+
"Intervention",
|
|
53
|
+
"Primary outcome",
|
|
54
|
+
"Secondary outcome",
|
|
55
|
+
"Secondary ID",
|
|
56
|
+
"Source Name",
|
|
57
|
+
"Secondary Sponsor",
|
|
58
|
+
"Ethics Status",
|
|
59
|
+
"Ethics Approval Date",
|
|
60
|
+
"Ethics Contact Name",
|
|
61
|
+
"Ethics Contact Address",
|
|
62
|
+
"Ethics Contact Phone",
|
|
63
|
+
"Ethics Contact Email",
|
|
64
|
+
"results yes no",
|
|
65
|
+
"results date posted",
|
|
66
|
+
"results url link",
|
|
67
|
+
"results url protocol",
|
|
68
|
+
"results date completed",
|
|
69
|
+
"results date first publication",
|
|
70
|
+
"results summary",
|
|
71
|
+
"results baseline char",
|
|
72
|
+
"results adverse events",
|
|
73
|
+
"results outcome measures",
|
|
74
|
+
"results ipd plan",
|
|
75
|
+
"results ipd description",
|
|
76
|
+
"Prospective registration",
|
|
77
|
+
"Bridging flag truefalse",
|
|
78
|
+
"Bridged type",
|
|
79
|
+
)
|
|
80
|
+
|
|
81
|
+
#: Columns without which a response cannot be treated as a valid ICTRP export.
|
|
82
|
+
#: A CSV missing any of these is `UpstreamContractDrift`, never "no results".
|
|
83
|
+
REQUIRED_COLUMNS: frozenset[str] = frozenset(
|
|
84
|
+
{
|
|
85
|
+
"TrialID",
|
|
86
|
+
"Source Register",
|
|
87
|
+
"Public title",
|
|
88
|
+
"Scientific title",
|
|
89
|
+
"Recruitment Status",
|
|
90
|
+
"Date registration",
|
|
91
|
+
"Last Refreshed on",
|
|
92
|
+
"Export date",
|
|
93
|
+
"Phase",
|
|
94
|
+
"Condition",
|
|
95
|
+
}
|
|
96
|
+
)
|
|
97
|
+
|
|
98
|
+
#: Columns where a new value indicates upstream vocabulary drift and should be
|
|
99
|
+
#: surfaced rather than silently absorbed into a facet list.
|
|
100
|
+
FACET_COLUMNS: frozenset[str] = frozenset(
|
|
101
|
+
{"Source Register", "Phase", "Recruitment Status", "Study type", "Inclusion gender", "Countries"}
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
EXPECTED_COLUMN_COUNT = len(CSV_COLUMNS)
|
|
105
|
+
|
|
106
|
+
#: Sanity assertion at import time: the tuple is the shape we measured.
|
|
107
|
+
assert EXPECTED_COLUMN_COUNT == 58, f"expected 58 columns, got {EXPECTED_COLUMN_COUNT}"
|
|
108
|
+
assert REQUIRED_COLUMNS <= set(CSV_COLUMNS), "REQUIRED_COLUMNS contains an unknown column"
|
|
@@ -0,0 +1,213 @@
|
|
|
1
|
+
"""Canonical-JSON persistence for result sets.
|
|
2
|
+
|
|
3
|
+
Why this exists
|
|
4
|
+
---------------
|
|
5
|
+
Two independent needs converge on the same format:
|
|
6
|
+
|
|
7
|
+
1. **Distribution.** A packaged application (pi-desktop and friends) wants trials
|
|
8
|
+
available without a network round trip on first launch. That requires a file
|
|
9
|
+
the application can ship and read directly.
|
|
10
|
+
2. **Durability.** The CSV cache in `cache/store.py` stores raw export bytes,
|
|
11
|
+
which are tied to one export's exact header. Canonical JSON is decoupled from
|
|
12
|
+
the upstream column set, so a snapshot stays readable across parser changes.
|
|
13
|
+
|
|
14
|
+
The format is deliberately boring: one JSON object, a `trials` array of canonical
|
|
15
|
+
trial dicts, and a `snapshot` block of provenance. No schema version bump will be
|
|
16
|
+
silently ignored -- a reader refuses a `format_version` it does not know.
|
|
17
|
+
|
|
18
|
+
Licensing note
|
|
19
|
+
--------------
|
|
20
|
+
WHO ICTRP terms (section 4c) state you "shall not assert any proprietary rights
|
|
21
|
+
to any portion of the ICTRP database", and 4d bars commercial use of extracted
|
|
22
|
+
information. A snapshot produced here is not our property, and redistributing it
|
|
23
|
+
is the operator's decision, not a technical default. `snapshot.attribution` and
|
|
24
|
+
`snapshot.terms_notice` are therefore written into every file so a snapshot
|
|
25
|
+
cannot be separated from its terms. See docs/BUNDLE.md.
|
|
26
|
+
"""
|
|
27
|
+
|
|
28
|
+
from __future__ import annotations
|
|
29
|
+
|
|
30
|
+
import json
|
|
31
|
+
import time
|
|
32
|
+
from dataclasses import dataclass
|
|
33
|
+
from pathlib import Path
|
|
34
|
+
from typing import Any, Iterable
|
|
35
|
+
|
|
36
|
+
from ..errors import ErrorCode, IctrpError
|
|
37
|
+
|
|
38
|
+
#: Bump only for a breaking change to the object shape. Readers reject unknown values.
|
|
39
|
+
FORMAT_VERSION = 1
|
|
40
|
+
|
|
41
|
+
#: Snapshot kind value. Distinguishes a shipped bundle from a user cache dump.
|
|
42
|
+
KIND_BUNDLE = "ictrp-trial-set"
|
|
43
|
+
|
|
44
|
+
TERMS_NOTICE = (
|
|
45
|
+
"Data from the WHO International Clinical Trials Registry Platform (ICTRP). "
|
|
46
|
+
"ICTRP data are publicly available for download from the ICTRP Search Portal "
|
|
47
|
+
"at no charge. WHO updates ICTRP weekly. Under the ICTRP terms of use you may "
|
|
48
|
+
"not assert proprietary rights to any portion of the ICTRP database, and may "
|
|
49
|
+
"not use the data for marketing, promotional or commercial purposes. This "
|
|
50
|
+
"snapshot is not a WHO product and is not endorsed by WHO."
|
|
51
|
+
)
|
|
52
|
+
|
|
53
|
+
|
|
54
|
+
@dataclass
|
|
55
|
+
class Snapshot:
|
|
56
|
+
"""A trial set plus the provenance needed to describe it honestly."""
|
|
57
|
+
|
|
58
|
+
trials: list[dict[str, Any]]
|
|
59
|
+
keyword: str
|
|
60
|
+
created_at: float
|
|
61
|
+
provenance: dict[str, Any]
|
|
62
|
+
format_version: int = FORMAT_VERSION
|
|
63
|
+
kind: str = KIND_BUNDLE
|
|
64
|
+
|
|
65
|
+
@property
|
|
66
|
+
def age_seconds(self) -> float:
|
|
67
|
+
return time.time() - self.created_at
|
|
68
|
+
|
|
69
|
+
@property
|
|
70
|
+
def age_days(self) -> float:
|
|
71
|
+
return self.age_seconds / 86400.0
|
|
72
|
+
|
|
73
|
+
def to_dict(self) -> dict[str, Any]:
|
|
74
|
+
return {
|
|
75
|
+
"kind": self.kind,
|
|
76
|
+
"format_version": self.format_version,
|
|
77
|
+
"snapshot": {
|
|
78
|
+
"keyword": self.keyword,
|
|
79
|
+
"created_at": _iso(self.created_at),
|
|
80
|
+
"created_at_epoch": self.created_at,
|
|
81
|
+
"trial_count": len(self.trials),
|
|
82
|
+
"attribution": (
|
|
83
|
+
"Source: WHO International Clinical Trials Registry Platform (ICTRP)."
|
|
84
|
+
),
|
|
85
|
+
"terms_notice": TERMS_NOTICE,
|
|
86
|
+
"provenance": self.provenance,
|
|
87
|
+
},
|
|
88
|
+
"trials": self.trials,
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
def write(self, path: Path) -> int:
|
|
92
|
+
"""Write the snapshot, returning bytes written."""
|
|
93
|
+
path.parent.mkdir(parents=True, exist_ok=True)
|
|
94
|
+
body = json.dumps(self.to_dict(), ensure_ascii=False, indent=1)
|
|
95
|
+
tmp = path.with_suffix(path.suffix + ".tmp")
|
|
96
|
+
tmp.write_text(body, encoding="utf-8")
|
|
97
|
+
tmp.replace(path)
|
|
98
|
+
return len(body.encode("utf-8"))
|
|
99
|
+
|
|
100
|
+
@classmethod
|
|
101
|
+
def read(cls, path: Path) -> "Snapshot":
|
|
102
|
+
"""Load a snapshot, refusing anything structurally unexpected.
|
|
103
|
+
|
|
104
|
+
Every failure raises rather than returning an empty set: an unreadable
|
|
105
|
+
bundle must never look like "no trials matched".
|
|
106
|
+
"""
|
|
107
|
+
try:
|
|
108
|
+
raw = path.read_text(encoding="utf-8")
|
|
109
|
+
except OSError as exc:
|
|
110
|
+
raise IctrpError(
|
|
111
|
+
ErrorCode.CACHE_MISS,
|
|
112
|
+
f"snapshot {str(path)!r} could not be read: {exc}",
|
|
113
|
+
hint="Check the path and that the file was shipped with the application.",
|
|
114
|
+
) from exc
|
|
115
|
+
|
|
116
|
+
try:
|
|
117
|
+
data = json.loads(raw)
|
|
118
|
+
except json.JSONDecodeError as exc:
|
|
119
|
+
raise IctrpError(
|
|
120
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
121
|
+
f"snapshot {str(path)!r} is not valid JSON: {exc}",
|
|
122
|
+
hint="The file is truncated or was not produced by this service.",
|
|
123
|
+
) from exc
|
|
124
|
+
|
|
125
|
+
if not isinstance(data, dict):
|
|
126
|
+
raise IctrpError(
|
|
127
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
128
|
+
f"snapshot {str(path)!r} must be a JSON object",
|
|
129
|
+
)
|
|
130
|
+
|
|
131
|
+
version = data.get("format_version")
|
|
132
|
+
if version != FORMAT_VERSION:
|
|
133
|
+
raise IctrpError(
|
|
134
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
135
|
+
f"snapshot {str(path)!r} has format_version {version!r}; "
|
|
136
|
+
f"this build reads version {FORMAT_VERSION}",
|
|
137
|
+
hint="Regenerate the snapshot with a matching build.",
|
|
138
|
+
)
|
|
139
|
+
|
|
140
|
+
trials = data.get("trials")
|
|
141
|
+
if not isinstance(trials, list):
|
|
142
|
+
raise IctrpError(
|
|
143
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
144
|
+
f"snapshot {str(path)!r} has no 'trials' array",
|
|
145
|
+
)
|
|
146
|
+
for index, trial in enumerate(trials[:5]):
|
|
147
|
+
if not isinstance(trial, dict):
|
|
148
|
+
raise IctrpError(
|
|
149
|
+
ErrorCode.UPSTREAM_CONTRACT_DRIFT,
|
|
150
|
+
f"snapshot {str(path)!r} trial #{index} is not an object",
|
|
151
|
+
)
|
|
152
|
+
|
|
153
|
+
meta = data.get("snapshot") or {}
|
|
154
|
+
created_epoch = meta.get("created_at_epoch")
|
|
155
|
+
if not isinstance(created_epoch, (int, float)):
|
|
156
|
+
created_epoch = _parse_iso_epoch(meta.get("created_at"))
|
|
157
|
+
|
|
158
|
+
return cls(
|
|
159
|
+
trials=trials,
|
|
160
|
+
keyword=str(meta.get("keyword", "")),
|
|
161
|
+
created_at=float(created_epoch),
|
|
162
|
+
provenance=meta.get("provenance") or {},
|
|
163
|
+
format_version=FORMAT_VERSION,
|
|
164
|
+
kind=str(data.get("kind", KIND_BUNDLE)),
|
|
165
|
+
)
|
|
166
|
+
|
|
167
|
+
def stale(self, max_age_days: float) -> bool:
|
|
168
|
+
"""True when the snapshot is older than the permitted age."""
|
|
169
|
+
if max_age_days <= 0:
|
|
170
|
+
return False
|
|
171
|
+
return self.age_days > max_age_days
|
|
172
|
+
|
|
173
|
+
|
|
174
|
+
def snapshot_from_trials(
|
|
175
|
+
*,
|
|
176
|
+
trials: Iterable[dict[str, Any]],
|
|
177
|
+
keyword: str,
|
|
178
|
+
provenance: dict[str, Any],
|
|
179
|
+
created_at: float | None = None,
|
|
180
|
+
) -> Snapshot:
|
|
181
|
+
return Snapshot(
|
|
182
|
+
trials=list(trials),
|
|
183
|
+
keyword=keyword,
|
|
184
|
+
created_at=created_at if created_at is not None else time.time(),
|
|
185
|
+
provenance=dict(provenance),
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def locate_bundle(paths: Iterable[Path]) -> Path | None:
|
|
190
|
+
"""Return the first existing snapshot path, or None.
|
|
191
|
+
|
|
192
|
+
Ordered by the caller: an explicit environment override should come first, a
|
|
193
|
+
shipped bundle last.
|
|
194
|
+
"""
|
|
195
|
+
for candidate in paths:
|
|
196
|
+
if candidate and candidate.is_file():
|
|
197
|
+
return candidate
|
|
198
|
+
return None
|
|
199
|
+
|
|
200
|
+
|
|
201
|
+
def _iso(epoch: float) -> str:
|
|
202
|
+
return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(epoch))
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
def _parse_iso_epoch(value: Any) -> float:
|
|
206
|
+
if not isinstance(value, str):
|
|
207
|
+
return 0.0
|
|
208
|
+
for fmt in ("%Y-%m-%dT%H:%M:%SZ", "%Y-%m-%dT%H:%M:%S"):
|
|
209
|
+
try:
|
|
210
|
+
return time.mktime(time.strptime(value, fmt))
|
|
211
|
+
except ValueError:
|
|
212
|
+
continue
|
|
213
|
+
return 0.0
|