ictrp-mcp-server 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +65 -0
  2. package/LICENSE +37 -0
  3. package/README.md +208 -0
  4. package/README_ZH.md +189 -0
  5. package/dist/cli/setup-cli.d.ts +14 -0
  6. package/dist/cli/setup-cli.js +230 -0
  7. package/dist/cli/setup-cli.js.map +1 -0
  8. package/dist/index.d.ts +18 -0
  9. package/dist/index.js +477 -0
  10. package/dist/index.js.map +1 -0
  11. package/dist/runtime/bootstrap.d.ts +99 -0
  12. package/dist/runtime/bootstrap.js +350 -0
  13. package/dist/runtime/bootstrap.js.map +1 -0
  14. package/dist/runtime/env-probe.d.ts +108 -0
  15. package/dist/runtime/env-probe.js +479 -0
  16. package/dist/runtime/env-probe.js.map +1 -0
  17. package/dist/runtime/sidecar-client.d.ts +50 -0
  18. package/dist/runtime/sidecar-client.js +120 -0
  19. package/dist/runtime/sidecar-client.js.map +1 -0
  20. package/dist/runtime/supervisor.d.ts +47 -0
  21. package/dist/runtime/supervisor.js +248 -0
  22. package/dist/runtime/supervisor.js.map +1 -0
  23. package/package.json +60 -0
  24. package/sidecar/ictrp_sidecar.py +602 -0
  25. package/sidecar/vendor/ictrp_mcp/__init__.py +3 -0
  26. package/sidecar/vendor/ictrp_mcp/cache/__init__.py +0 -0
  27. package/sidecar/vendor/ictrp_mcp/cache/store.py +313 -0
  28. package/sidecar/vendor/ictrp_mcp/data/__init__.py +0 -0
  29. package/sidecar/vendor/ictrp_mcp/data/columns.py +108 -0
  30. package/sidecar/vendor/ictrp_mcp/data/jsonio.py +213 -0
  31. package/sidecar/vendor/ictrp_mcp/data/normalize.py +348 -0
  32. package/sidecar/vendor/ictrp_mcp/data/query.py +307 -0
  33. package/sidecar/vendor/ictrp_mcp/errors.py +123 -0
  34. package/sidecar/vendor/ictrp_mcp/ictrp/__init__.py +0 -0
  35. package/sidecar/vendor/ictrp_mcp/ictrp/export_guard.py +269 -0
  36. package/sidecar/vendor/ictrp_mcp/ictrp/htmlstate.py +143 -0
  37. package/sidecar/vendor/ictrp_mcp/ictrp/session.py +245 -0
  38. package/sidecar/vendor/ictrp_mcp/offline.py +133 -0
  39. package/sidecar/vendor/ictrp_mcp/provenance.py +182 -0
  40. package/sidecar/vendor/ictrp_mcp/server.py +368 -0
  41. package/sidecar/vendor/ictrp_mcp/tools.py +712 -0
  42. package/sidecar/vendor/pyproject.toml +25 -0
@@ -0,0 +1,313 @@
1
+ """Caching for materialized result sets.
2
+
3
+ The governing constraint is cost: a single export can be tens of megabytes, and
4
+ the chain behind it is three round trips to a public service we should not hammer.
5
+ So a search result is materialized once and then queried locally.
6
+
7
+ Two deliberate choices:
8
+
9
+ * Raw CSV bytes are written to disk instead of holding every row in memory for
10
+ every live set. A large set is only parsed when it is actually used.
11
+ * Only a small number of sets stay in memory. Older ones are dropped (and can be
12
+ re-read from disk), which bounds the process footprint.
13
+
14
+ Nothing here caches a failure as if it were data. A set only enters the store
15
+ after it has been validated.
16
+ """
17
+
18
+ from __future__ import annotations
19
+
20
+ import hashlib
21
+ import json
22
+ import os
23
+ import re
24
+ import time
25
+ from dataclasses import dataclass, field
26
+ from pathlib import Path
27
+ from typing import Any
28
+
29
+ from ..errors import ErrorCode, IctrpError
30
+ from ..data.normalize import to_trial
31
+
32
+ #: Parsed sets held in memory at once.
33
+ MAX_MATERIALIZED_SETS = 3
34
+
35
+ #: How long a materialized set stays queryable.
36
+ DEFAULT_SET_TTL_SECONDS = 7 * 24 * 60 * 60
37
+
38
+ #: How long a single-trial lookup stays fresh.
39
+ DEFAULT_TRIAL_TTL_SECONDS = 24 * 60 * 60
40
+
41
+
42
+ def normalize_query(keyword: str) -> str:
43
+ """Collapse a keyword to a stable cache key component."""
44
+ return re.sub(r"\s+", " ", keyword.strip().lower())
45
+
46
+
47
+ def query_key(keyword: str) -> str:
48
+ digest = hashlib.sha256(normalize_query(keyword).encode()).hexdigest()[:16]
49
+ return f"search:{digest}"
50
+
51
+
52
+ def trial_key(trial_id: str) -> str:
53
+ return f"trial:{trial_id.strip().upper()}"
54
+
55
+
56
+ def default_cache_dir() -> Path:
57
+ override = os.environ.get("ICTRP_CACHE_DIR")
58
+ if override:
59
+ return Path(override)
60
+ return Path.home() / ".cache" / "ictrp-mcp-service"
61
+
62
+
63
+ @dataclass
64
+ class MaterializedSet:
65
+ """One export, parsed into canonical trials."""
66
+
67
+ set_id: str
68
+ keyword: str
69
+ trials: list[dict[str, Any]]
70
+ header: tuple[str, ...]
71
+ reported_total: int | None
72
+ csv_bytes: int
73
+ created_at: float
74
+ provenance: dict[str, Any] = field(default_factory=dict)
75
+
76
+ @property
77
+ def rows_returned(self) -> int:
78
+ """How many rows we actually hold.
79
+
80
+ Never described as a total. The portal's own figure lives separately in
81
+ `reported_total`, and the two are known to differ by 0.4%-29%.
82
+ """
83
+ return len(self.trials)
84
+
85
+ @property
86
+ def is_incomplete(self) -> bool:
87
+ """True when the portal reported more matches than the export delivered."""
88
+ if self.reported_total is None:
89
+ return False
90
+ return self.reported_total > self.rows_returned
91
+
92
+ @property
93
+ def missing_estimate(self) -> int | None:
94
+ if self.reported_total is None:
95
+ return None
96
+ return max(0, self.reported_total - self.rows_returned)
97
+
98
+ def age_seconds(self) -> float:
99
+ return time.time() - self.created_at
100
+
101
+ def summary(self) -> dict[str, Any]:
102
+ return {
103
+ "set_id": self.set_id,
104
+ "keyword": self.keyword,
105
+ "rows_returned": self.rows_returned,
106
+ "upstream_reported_total": self.reported_total,
107
+ "records_incomplete": self.is_incomplete,
108
+ "estimated_missing": self.missing_estimate,
109
+ "csv_bytes": self.csv_bytes,
110
+ "age_seconds": round(self.age_seconds(), 1),
111
+ "created_at": _iso(self.created_at),
112
+ }
113
+
114
+
115
+ def _iso(epoch: float) -> str:
116
+ return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(epoch))
117
+
118
+
119
+ class SetStore:
120
+ """Holds materialized sets in memory, backed by raw CSV on disk."""
121
+
122
+ def __init__(self, cache_dir: Path | None = None, ttl_seconds: int = DEFAULT_SET_TTL_SECONDS):
123
+ self.cache_dir = cache_dir or default_cache_dir()
124
+ self.ttl_seconds = ttl_seconds
125
+ self._sets: dict[str, MaterializedSet] = {}
126
+ self._order: list[str] = []
127
+
128
+ def _raw_path(self, cache_key: str) -> Path:
129
+ safe = re.sub(r"[^A-Za-z0-9_.-]", "_", cache_key)
130
+ return self.cache_dir / f"{safe}.csv"
131
+
132
+ def _meta_path(self, cache_key: str) -> Path:
133
+ safe = re.sub(r"[^A-Za-z0-9_.-]", "_", cache_key)
134
+ return self.cache_dir / f"{safe}.json"
135
+
136
+ def _ensure_dir(self) -> None:
137
+ try:
138
+ self.cache_dir.mkdir(parents=True, exist_ok=True)
139
+ except OSError:
140
+ # Fall back to a temp location rather than failing the request.
141
+ self.cache_dir = Path("/tmp/ictrp-mcp-service")
142
+ self.cache_dir.mkdir(parents=True, exist_ok=True)
143
+
144
+ def store(
145
+ self,
146
+ *,
147
+ keyword: str,
148
+ payload_raw: bytes,
149
+ header: tuple[str, ...],
150
+ rows: list[list[str]],
151
+ reported_total: int | None,
152
+ provenance: dict[str, Any],
153
+ ) -> MaterializedSet:
154
+ """Persist and materialize a validated export."""
155
+ cache_key = query_key(keyword)
156
+ trials = [to_trial(row, header) for row in rows]
157
+ materialized = MaterializedSet(
158
+ set_id=cache_key,
159
+ keyword=keyword,
160
+ trials=trials,
161
+ header=header,
162
+ reported_total=reported_total,
163
+ csv_bytes=len(payload_raw),
164
+ created_at=time.time(),
165
+ provenance=dict(provenance),
166
+ )
167
+
168
+ self._ensure_dir()
169
+ try:
170
+ self._raw_path(cache_key).write_bytes(payload_raw)
171
+ self._meta_path(cache_key).write_text(
172
+ json.dumps(
173
+ {
174
+ "keyword": keyword,
175
+ "header": list(header),
176
+ "reported_total": reported_total,
177
+ "created_at": materialized.created_at,
178
+ "csv_bytes": materialized.csv_bytes,
179
+ "provenance": provenance,
180
+ }
181
+ ),
182
+ encoding="utf-8",
183
+ )
184
+ except OSError:
185
+ # Disk caching is best-effort; the in-memory set is still usable.
186
+ pass
187
+
188
+ self._remember(materialized)
189
+ return materialized
190
+
191
+ def adopt(
192
+ self,
193
+ *,
194
+ keyword: str,
195
+ trials: list[dict[str, Any]],
196
+ provenance: dict[str, Any],
197
+ created_at: float | None = None,
198
+ source_label: str = "snapshot",
199
+ ) -> MaterializedSet:
200
+ """Register an externally built set (e.g. from an offline snapshot).
201
+
202
+ `created_at` is carried over from the snapshot so the set's age reflects
203
+ when the data was retrieved, not when it was loaded. A snapshot read today
204
+ from a month-old file must still report as a month old.
205
+
206
+ The set is memory-resident only. Writing it back as a raw CSV would
207
+ fabricate an export we never received, and `_reload_from_disk` would then
208
+ present it as a genuine export on the next run.
209
+ """
210
+ materialized = MaterializedSet(
211
+ set_id=query_key(keyword),
212
+ keyword=keyword,
213
+ trials=list(trials),
214
+ header=(),
215
+ reported_total=provenance.get("upstream_reported_total"),
216
+ csv_bytes=0,
217
+ created_at=created_at if created_at is not None else time.time(),
218
+ provenance=dict(provenance),
219
+ )
220
+ materialized.provenance.setdefault("notes", [])
221
+ self._remember(materialized)
222
+ return materialized
223
+
224
+ def _remember(self, materialized: MaterializedSet) -> None:
225
+ self._sets[materialized.set_id] = materialized
226
+ if materialized.set_id in self._order:
227
+ self._order.remove(materialized.set_id)
228
+ self._order.append(materialized.set_id)
229
+ while len(self._order) > MAX_MATERIALIZED_SETS:
230
+ evicted = self._order.pop(0)
231
+ self._sets.pop(evicted, None)
232
+
233
+ def get(self, set_id: str) -> MaterializedSet:
234
+ """Return a set, reloading skipped ones from disk when possible."""
235
+ materialized = self._sets.get(set_id)
236
+ if materialized is not None and materialized.age_seconds() <= self.ttl_seconds:
237
+ return materialized
238
+ if materialized is not None:
239
+ self._sets.pop(set_id, None)
240
+ if set_id in self._order:
241
+ self._order.remove(set_id)
242
+
243
+ reloaded = self._reload_from_disk(set_id)
244
+ if reloaded is not None:
245
+ self._remember(reloaded)
246
+ return reloaded
247
+
248
+ raise IctrpError(
249
+ ErrorCode.CACHE_MISS,
250
+ f"result set {set_id!r} is not available",
251
+ hint=(
252
+ "Result sets are held for a limited time and only a few at once. "
253
+ "Run the search again to re-materialize it."
254
+ ),
255
+ )
256
+
257
+ def _reload_from_disk(self, set_id: str) -> MaterializedSet | None:
258
+ meta_path = self._meta_path(set_id)
259
+ raw_path = self._raw_path(set_id)
260
+ if not meta_path.exists() or not raw_path.exists():
261
+ return None
262
+ try:
263
+ meta = json.loads(meta_path.read_text(encoding="utf-8"))
264
+ raw = raw_path.read_bytes()
265
+ except (OSError, json.JSONDecodeError):
266
+ return None
267
+
268
+ created_at = float(meta.get("created_at", 0.0))
269
+ if time.time() - created_at > self.ttl_seconds:
270
+ return None
271
+
272
+ header, rows = _parse_raw(raw)
273
+ if not header:
274
+ return None
275
+ return MaterializedSet(
276
+ set_id=set_id,
277
+ keyword=meta.get("keyword", ""),
278
+ trials=[to_trial(r, header) for r in rows],
279
+ header=header,
280
+ reported_total=meta.get("reported_total"),
281
+ csv_bytes=meta.get("csv_bytes", len(raw)),
282
+ created_at=created_at,
283
+ provenance=meta.get("provenance", {}) or {},
284
+ )
285
+
286
+ def list_sets(self) -> list[dict[str, Any]]:
287
+ return [self._sets[s].summary() for s in self._order if s in self._sets]
288
+
289
+ def purge(self, set_id: str | None = None) -> int:
290
+ if set_id is None:
291
+ count = len(self._sets)
292
+ self._sets.clear()
293
+ self._order.clear()
294
+ return count
295
+ removed = 1 if self._sets.pop(set_id, None) is not None else 0
296
+ if set_id in self._order:
297
+ self._order.remove(set_id)
298
+ return removed
299
+
300
+
301
+ def _parse_raw(raw: bytes) -> tuple[tuple[str, ...], list[list[str]]]:
302
+ import csv as _csv
303
+ import io as _io
304
+
305
+ text = raw.decode("utf-8-sig", "replace")
306
+ reader = _csv.reader(_io.StringIO(text))
307
+ try:
308
+ header = tuple(h.strip() for h in next(reader))
309
+ except StopIteration:
310
+ return (), []
311
+ rows = [r for r in reader if r and any(c.strip() for c in r)]
312
+ width = len(header)
313
+ return header, [(r + [""] * width)[:width] for r in rows]
File without changes
@@ -0,0 +1,108 @@
1
+ """The 58 ICTRP CSV column headers, as literal strings.
2
+
3
+ These are reproduced VERBATIM from the live export, including three
4
+ misspellings present in the upstream data:
5
+
6
+ 'Inclusion agemin' (not 'agemin' -> 'age_min')
7
+ 'Inclusion agemax'
8
+ 'Date enrollement' (not 'enrollment')
9
+
10
+ Do NOT "correct" these. The parser matches on the exact upstream strings; a
11
+ helpful-looking normalisation here would silently produce all-null fields.
12
+ `tests/test_columns.py` asserts the exact set so an upstream rename surfaces as
13
+ a failing test rather than a silently degraded response.
14
+ """
15
+
16
+ from __future__ import annotations
17
+
18
+ #: Column order as returned by the live export. Order is not load-bearing for
19
+ #: parsing (we match by name) but is asserted for drift detection.
20
+ CSV_COLUMNS: tuple[str, ...] = (
21
+ "TrialID",
22
+ "Last Refreshed on",
23
+ "Public title",
24
+ "Scientific title",
25
+ "Acronym",
26
+ "Primary sponsor",
27
+ "Date registration",
28
+ "Date registration3",
29
+ "Export date",
30
+ "Source Register",
31
+ "web address",
32
+ "Recruitment Status",
33
+ "other records",
34
+ "Inclusion agemin",
35
+ "Inclusion agemax",
36
+ "Inclusion gender",
37
+ "Date enrollement",
38
+ "Target size",
39
+ "Study type",
40
+ "Study design",
41
+ "Phase",
42
+ "Countries",
43
+ "Contact Firstname",
44
+ "Contact Lastname",
45
+ "Contact Address",
46
+ "Contact Email",
47
+ "Contact Tel",
48
+ "Contact Affiliation",
49
+ "Inclusion Criteria",
50
+ "Exclusion Criteria",
51
+ "Condition",
52
+ "Intervention",
53
+ "Primary outcome",
54
+ "Secondary outcome",
55
+ "Secondary ID",
56
+ "Source Name",
57
+ "Secondary Sponsor",
58
+ "Ethics Status",
59
+ "Ethics Approval Date",
60
+ "Ethics Contact Name",
61
+ "Ethics Contact Address",
62
+ "Ethics Contact Phone",
63
+ "Ethics Contact Email",
64
+ "results yes no",
65
+ "results date posted",
66
+ "results url link",
67
+ "results url protocol",
68
+ "results date completed",
69
+ "results date first publication",
70
+ "results summary",
71
+ "results baseline char",
72
+ "results adverse events",
73
+ "results outcome measures",
74
+ "results ipd plan",
75
+ "results ipd description",
76
+ "Prospective registration",
77
+ "Bridging flag truefalse",
78
+ "Bridged type",
79
+ )
80
+
81
+ #: Columns without which a response cannot be treated as a valid ICTRP export.
82
+ #: A CSV missing any of these is `UpstreamContractDrift`, never "no results".
83
+ REQUIRED_COLUMNS: frozenset[str] = frozenset(
84
+ {
85
+ "TrialID",
86
+ "Source Register",
87
+ "Public title",
88
+ "Scientific title",
89
+ "Recruitment Status",
90
+ "Date registration",
91
+ "Last Refreshed on",
92
+ "Export date",
93
+ "Phase",
94
+ "Condition",
95
+ }
96
+ )
97
+
98
+ #: Columns where a new value indicates upstream vocabulary drift and should be
99
+ #: surfaced rather than silently absorbed into a facet list.
100
+ FACET_COLUMNS: frozenset[str] = frozenset(
101
+ {"Source Register", "Phase", "Recruitment Status", "Study type", "Inclusion gender", "Countries"}
102
+ )
103
+
104
+ EXPECTED_COLUMN_COUNT = len(CSV_COLUMNS)
105
+
106
+ #: Sanity assertion at import time: the tuple is the shape we measured.
107
+ assert EXPECTED_COLUMN_COUNT == 58, f"expected 58 columns, got {EXPECTED_COLUMN_COUNT}"
108
+ assert REQUIRED_COLUMNS <= set(CSV_COLUMNS), "REQUIRED_COLUMNS contains an unknown column"
@@ -0,0 +1,213 @@
1
+ """Canonical-JSON persistence for result sets.
2
+
3
+ Why this exists
4
+ ---------------
5
+ Two independent needs converge on the same format:
6
+
7
+ 1. **Distribution.** A packaged application (pi-desktop and friends) wants trials
8
+ available without a network round trip on first launch. That requires a file
9
+ the application can ship and read directly.
10
+ 2. **Durability.** The CSV cache in `cache/store.py` stores raw export bytes,
11
+ which are tied to one export's exact header. Canonical JSON is decoupled from
12
+ the upstream column set, so a snapshot stays readable across parser changes.
13
+
14
+ The format is deliberately boring: one JSON object, a `trials` array of canonical
15
+ trial dicts, and a `snapshot` block of provenance. No schema version bump will be
16
+ silently ignored -- a reader refuses a `format_version` it does not know.
17
+
18
+ Licensing note
19
+ --------------
20
+ WHO ICTRP terms (section 4c) state you "shall not assert any proprietary rights
21
+ to any portion of the ICTRP database", and 4d bars commercial use of extracted
22
+ information. A snapshot produced here is not our property, and redistributing it
23
+ is the operator's decision, not a technical default. `snapshot.attribution` and
24
+ `snapshot.terms_notice` are therefore written into every file so a snapshot
25
+ cannot be separated from its terms. See docs/BUNDLE.md.
26
+ """
27
+
28
+ from __future__ import annotations
29
+
30
+ import json
31
+ import time
32
+ from dataclasses import dataclass
33
+ from pathlib import Path
34
+ from typing import Any, Iterable
35
+
36
+ from ..errors import ErrorCode, IctrpError
37
+
38
+ #: Bump only for a breaking change to the object shape. Readers reject unknown values.
39
+ FORMAT_VERSION = 1
40
+
41
+ #: Snapshot kind value. Distinguishes a shipped bundle from a user cache dump.
42
+ KIND_BUNDLE = "ictrp-trial-set"
43
+
44
+ TERMS_NOTICE = (
45
+ "Data from the WHO International Clinical Trials Registry Platform (ICTRP). "
46
+ "ICTRP data are publicly available for download from the ICTRP Search Portal "
47
+ "at no charge. WHO updates ICTRP weekly. Under the ICTRP terms of use you may "
48
+ "not assert proprietary rights to any portion of the ICTRP database, and may "
49
+ "not use the data for marketing, promotional or commercial purposes. This "
50
+ "snapshot is not a WHO product and is not endorsed by WHO."
51
+ )
52
+
53
+
54
+ @dataclass
55
+ class Snapshot:
56
+ """A trial set plus the provenance needed to describe it honestly."""
57
+
58
+ trials: list[dict[str, Any]]
59
+ keyword: str
60
+ created_at: float
61
+ provenance: dict[str, Any]
62
+ format_version: int = FORMAT_VERSION
63
+ kind: str = KIND_BUNDLE
64
+
65
+ @property
66
+ def age_seconds(self) -> float:
67
+ return time.time() - self.created_at
68
+
69
+ @property
70
+ def age_days(self) -> float:
71
+ return self.age_seconds / 86400.0
72
+
73
+ def to_dict(self) -> dict[str, Any]:
74
+ return {
75
+ "kind": self.kind,
76
+ "format_version": self.format_version,
77
+ "snapshot": {
78
+ "keyword": self.keyword,
79
+ "created_at": _iso(self.created_at),
80
+ "created_at_epoch": self.created_at,
81
+ "trial_count": len(self.trials),
82
+ "attribution": (
83
+ "Source: WHO International Clinical Trials Registry Platform (ICTRP)."
84
+ ),
85
+ "terms_notice": TERMS_NOTICE,
86
+ "provenance": self.provenance,
87
+ },
88
+ "trials": self.trials,
89
+ }
90
+
91
+ def write(self, path: Path) -> int:
92
+ """Write the snapshot, returning bytes written."""
93
+ path.parent.mkdir(parents=True, exist_ok=True)
94
+ body = json.dumps(self.to_dict(), ensure_ascii=False, indent=1)
95
+ tmp = path.with_suffix(path.suffix + ".tmp")
96
+ tmp.write_text(body, encoding="utf-8")
97
+ tmp.replace(path)
98
+ return len(body.encode("utf-8"))
99
+
100
+ @classmethod
101
+ def read(cls, path: Path) -> "Snapshot":
102
+ """Load a snapshot, refusing anything structurally unexpected.
103
+
104
+ Every failure raises rather than returning an empty set: an unreadable
105
+ bundle must never look like "no trials matched".
106
+ """
107
+ try:
108
+ raw = path.read_text(encoding="utf-8")
109
+ except OSError as exc:
110
+ raise IctrpError(
111
+ ErrorCode.CACHE_MISS,
112
+ f"snapshot {str(path)!r} could not be read: {exc}",
113
+ hint="Check the path and that the file was shipped with the application.",
114
+ ) from exc
115
+
116
+ try:
117
+ data = json.loads(raw)
118
+ except json.JSONDecodeError as exc:
119
+ raise IctrpError(
120
+ ErrorCode.UPSTREAM_CONTRACT_DRIFT,
121
+ f"snapshot {str(path)!r} is not valid JSON: {exc}",
122
+ hint="The file is truncated or was not produced by this service.",
123
+ ) from exc
124
+
125
+ if not isinstance(data, dict):
126
+ raise IctrpError(
127
+ ErrorCode.UPSTREAM_CONTRACT_DRIFT,
128
+ f"snapshot {str(path)!r} must be a JSON object",
129
+ )
130
+
131
+ version = data.get("format_version")
132
+ if version != FORMAT_VERSION:
133
+ raise IctrpError(
134
+ ErrorCode.UPSTREAM_CONTRACT_DRIFT,
135
+ f"snapshot {str(path)!r} has format_version {version!r}; "
136
+ f"this build reads version {FORMAT_VERSION}",
137
+ hint="Regenerate the snapshot with a matching build.",
138
+ )
139
+
140
+ trials = data.get("trials")
141
+ if not isinstance(trials, list):
142
+ raise IctrpError(
143
+ ErrorCode.UPSTREAM_CONTRACT_DRIFT,
144
+ f"snapshot {str(path)!r} has no 'trials' array",
145
+ )
146
+ for index, trial in enumerate(trials[:5]):
147
+ if not isinstance(trial, dict):
148
+ raise IctrpError(
149
+ ErrorCode.UPSTREAM_CONTRACT_DRIFT,
150
+ f"snapshot {str(path)!r} trial #{index} is not an object",
151
+ )
152
+
153
+ meta = data.get("snapshot") or {}
154
+ created_epoch = meta.get("created_at_epoch")
155
+ if not isinstance(created_epoch, (int, float)):
156
+ created_epoch = _parse_iso_epoch(meta.get("created_at"))
157
+
158
+ return cls(
159
+ trials=trials,
160
+ keyword=str(meta.get("keyword", "")),
161
+ created_at=float(created_epoch),
162
+ provenance=meta.get("provenance") or {},
163
+ format_version=FORMAT_VERSION,
164
+ kind=str(data.get("kind", KIND_BUNDLE)),
165
+ )
166
+
167
+ def stale(self, max_age_days: float) -> bool:
168
+ """True when the snapshot is older than the permitted age."""
169
+ if max_age_days <= 0:
170
+ return False
171
+ return self.age_days > max_age_days
172
+
173
+
174
+ def snapshot_from_trials(
175
+ *,
176
+ trials: Iterable[dict[str, Any]],
177
+ keyword: str,
178
+ provenance: dict[str, Any],
179
+ created_at: float | None = None,
180
+ ) -> Snapshot:
181
+ return Snapshot(
182
+ trials=list(trials),
183
+ keyword=keyword,
184
+ created_at=created_at if created_at is not None else time.time(),
185
+ provenance=dict(provenance),
186
+ )
187
+
188
+
189
+ def locate_bundle(paths: Iterable[Path]) -> Path | None:
190
+ """Return the first existing snapshot path, or None.
191
+
192
+ Ordered by the caller: an explicit environment override should come first, a
193
+ shipped bundle last.
194
+ """
195
+ for candidate in paths:
196
+ if candidate and candidate.is_file():
197
+ return candidate
198
+ return None
199
+
200
+
201
+ def _iso(epoch: float) -> str:
202
+ return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(epoch))
203
+
204
+
205
+ def _parse_iso_epoch(value: Any) -> float:
206
+ if not isinstance(value, str):
207
+ return 0.0
208
+ for fmt in ("%Y-%m-%dT%H:%M:%SZ", "%Y-%m-%dT%H:%M:%S"):
209
+ try:
210
+ return time.mktime(time.strptime(value, fmt))
211
+ except ValueError:
212
+ continue
213
+ return 0.0