ictrp-mcp-server 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +65 -0
  2. package/LICENSE +37 -0
  3. package/README.md +208 -0
  4. package/README_ZH.md +189 -0
  5. package/dist/cli/setup-cli.d.ts +14 -0
  6. package/dist/cli/setup-cli.js +230 -0
  7. package/dist/cli/setup-cli.js.map +1 -0
  8. package/dist/index.d.ts +18 -0
  9. package/dist/index.js +477 -0
  10. package/dist/index.js.map +1 -0
  11. package/dist/runtime/bootstrap.d.ts +99 -0
  12. package/dist/runtime/bootstrap.js +350 -0
  13. package/dist/runtime/bootstrap.js.map +1 -0
  14. package/dist/runtime/env-probe.d.ts +108 -0
  15. package/dist/runtime/env-probe.js +479 -0
  16. package/dist/runtime/env-probe.js.map +1 -0
  17. package/dist/runtime/sidecar-client.d.ts +50 -0
  18. package/dist/runtime/sidecar-client.js +120 -0
  19. package/dist/runtime/sidecar-client.js.map +1 -0
  20. package/dist/runtime/supervisor.d.ts +47 -0
  21. package/dist/runtime/supervisor.js +248 -0
  22. package/dist/runtime/supervisor.js.map +1 -0
  23. package/package.json +60 -0
  24. package/sidecar/ictrp_sidecar.py +602 -0
  25. package/sidecar/vendor/ictrp_mcp/__init__.py +3 -0
  26. package/sidecar/vendor/ictrp_mcp/cache/__init__.py +0 -0
  27. package/sidecar/vendor/ictrp_mcp/cache/store.py +313 -0
  28. package/sidecar/vendor/ictrp_mcp/data/__init__.py +0 -0
  29. package/sidecar/vendor/ictrp_mcp/data/columns.py +108 -0
  30. package/sidecar/vendor/ictrp_mcp/data/jsonio.py +213 -0
  31. package/sidecar/vendor/ictrp_mcp/data/normalize.py +348 -0
  32. package/sidecar/vendor/ictrp_mcp/data/query.py +307 -0
  33. package/sidecar/vendor/ictrp_mcp/errors.py +123 -0
  34. package/sidecar/vendor/ictrp_mcp/ictrp/__init__.py +0 -0
  35. package/sidecar/vendor/ictrp_mcp/ictrp/export_guard.py +269 -0
  36. package/sidecar/vendor/ictrp_mcp/ictrp/htmlstate.py +143 -0
  37. package/sidecar/vendor/ictrp_mcp/ictrp/session.py +245 -0
  38. package/sidecar/vendor/ictrp_mcp/offline.py +133 -0
  39. package/sidecar/vendor/ictrp_mcp/provenance.py +182 -0
  40. package/sidecar/vendor/ictrp_mcp/server.py +368 -0
  41. package/sidecar/vendor/ictrp_mcp/tools.py +712 -0
  42. package/sidecar/vendor/pyproject.toml +25 -0
@@ -0,0 +1,245 @@
1
+ """The verified three-step request chain.
2
+
3
+ GET Default.aspx -> harvest form state
4
+ POST Default.aspx -> search, harvest results-page form state
5
+ POST Default.aspx -> export CSV
6
+
7
+ Each step depends on state from the previous one. This is the only part of the
8
+ project carried over from prior verification; everything around it is new.
9
+
10
+ Measured facts this module encodes (docs/MEASUREMENTS.md):
11
+
12
+ * The search POST requires `__EVENTVALIDATION` and the cookie set from the GET.
13
+ Omitting either produces an error page rather than results.
14
+ * The export POST must be built from the results page's own hidden inputs. It
15
+ works with `Button7=Export to CSV`.
16
+ * A successful export is `200` + `application/vnd.ms-excel` +
17
+ `attachment;filename=IctrpResults.csv`.
18
+ """
19
+
20
+ from __future__ import annotations
21
+
22
+ from dataclasses import dataclass, field
23
+
24
+ import httpx
25
+
26
+ from ..errors import ErrorCode, IctrpError
27
+ from . import htmlstate
28
+ from .export_guard import (
29
+ CsvPayload,
30
+ classify_export,
31
+ classify_form_page,
32
+ parse_reported_total,
33
+ )
34
+
35
+ BASE_URL = "https://trialsearch.who.int/Default.aspx"
36
+
37
+ DEFAULT_HEADERS = {
38
+ "User-Agent": (
39
+ "Mozilla/5.0 (Macintosh; Intel Mac OS X 10_15_7) AppleWebKit/537.36 "
40
+ "(KHTML, like Gecko) Chrome/141.0.0.0 Safari/537.36"
41
+ ),
42
+ "Accept": "text/html,application/xhtml+xml,application/xml;q=0.9,*/*;q=0.8",
43
+ "Accept-Language": "en-US,en;q=0.9",
44
+ "Cache-Control": "no-cache",
45
+ "Pragma": "no-cache",
46
+ }
47
+
48
+ #: The export button. Verified working.
49
+ EXPORT_CONTROL = "Button7"
50
+
51
+ #: The search submit button and text box on the form page.
52
+ SEARCH_BUTTON = "Button1"
53
+ SEARCH_TEXTBOX = "TextBox1"
54
+
55
+
56
+ def _check_present(state: htmlstate.FormState, *, step: str) -> None:
57
+ missing = state.missing_state()
58
+ if missing:
59
+ raise IctrpError(
60
+ ErrorCode.SESSION_FAILED,
61
+ f"{step}: form state incomplete (missing {', '.join(missing)})",
62
+ detail=(
63
+ "The page did not carry the hidden inputs this step needs. "
64
+ "Either the page was an error/interstitial, or the portal changed."
65
+ ),
66
+ )
67
+
68
+
69
+ @dataclass
70
+ class IctrpSession:
71
+ """One logical search+export exchange.
72
+
73
+ A fresh session per search keeps state handling simple and avoids reusing a
74
+ `__VIEWSTATE` past its lifetime. Redirects are NOT followed automatically,
75
+ because a `302` to `/NoAccess.aspx` is a failure we must classify rather
76
+ than silently follow.
77
+ """
78
+
79
+ client: httpx.AsyncClient
80
+ base_url: str = BASE_URL
81
+ last_html: str = ""
82
+ form_state: htmlstate.FormState | None = None
83
+ results_state: htmlstate.FormState | None = None
84
+ reported_total: int | None = None
85
+ response_date: str | None = None
86
+ steps: list[str] = field(default_factory=list)
87
+
88
+ @classmethod
89
+ def create(cls, *, timeout: float = 60.0) -> "IctrpSession":
90
+ client = httpx.AsyncClient(
91
+ headers=DEFAULT_HEADERS,
92
+ timeout=timeout,
93
+ follow_redirects=False,
94
+ )
95
+ return cls(client=client)
96
+
97
+ async def __aenter__(self) -> "IctrpSession":
98
+ return self
99
+
100
+ async def __aexit__(self, *exc: object) -> None:
101
+ await self.aclose()
102
+
103
+ async def aclose(self) -> None:
104
+ await self.client.aclose()
105
+
106
+ # ---- step 1 -----------------------------------------------------------
107
+
108
+ async def load_form(self) -> htmlstate.FormState:
109
+ try:
110
+ response = await self.client.get(self.base_url)
111
+ except httpx.HTTPError as exc:
112
+ raise IctrpError(
113
+ ErrorCode.UPSTREAM_ERROR,
114
+ f"Could not reach the ICTRP search portal: {exc}",
115
+ hint="Check network connectivity, then retry.",
116
+ ) from exc
117
+
118
+ html = classify_form_page(status=response.status_code, body=response.content)
119
+ self.steps.append(f"GET {self.base_url} -> {response.status_code}")
120
+ state = htmlstate.extract_form_state(html)
121
+ _check_present(state, step="load_form")
122
+ self.form_state = state
123
+ self.last_html = html
124
+ return state
125
+
126
+ # ---- step 2 -----------------------------------------------------------
127
+
128
+ async def search(self, keyword: str) -> str:
129
+ if not keyword or not keyword.strip():
130
+ raise IctrpError(
131
+ ErrorCode.INVALID_ARGUMENT,
132
+ "keyword must be a non-empty string",
133
+ )
134
+ if self.form_state is None:
135
+ await self.load_form()
136
+ assert self.form_state is not None
137
+
138
+ body = self.form_state.merged_with(
139
+ {SEARCH_TEXTBOX: keyword, SEARCH_BUTTON: "Search"}
140
+ )
141
+ try:
142
+ response = await self.client.post(
143
+ self.base_url,
144
+ data=body,
145
+ headers={
146
+ "Content-Type": "application/x-www-form-urlencoded",
147
+ "Referer": self.base_url,
148
+ },
149
+ )
150
+ except httpx.HTTPError as exc:
151
+ raise IctrpError(
152
+ ErrorCode.UPSTREAM_ERROR,
153
+ f"Search request failed: {exc}",
154
+ ) from exc
155
+
156
+ location = response.headers.get("location")
157
+ if location and "/noaccess.aspx" in location.lower():
158
+ raise IctrpError(
159
+ ErrorCode.UPSTREAM_CONTRACT_DRIFT,
160
+ "Search POST was redirected to NoAccess.aspx",
161
+ upstream_status=response.status_code,
162
+ hint=(
163
+ "The search POST was rejected. Verify that __EVENTVALIDATION "
164
+ "and the GET-set cookies were replayed."
165
+ ),
166
+ )
167
+ if response.status_code in (403, 405, 429, 503):
168
+ raise IctrpError(
169
+ ErrorCode.UPSTREAM_BLOCKED,
170
+ f"Search was refused (HTTP {response.status_code})",
171
+ upstream_status=response.status_code,
172
+ hint="The portal blocked this request. This is not an empty result set.",
173
+ )
174
+
175
+ html = response.text
176
+ self.steps.append(f"POST search {keyword!r} -> {response.status_code}")
177
+
178
+ state = htmlstate.extract_form_state(html)
179
+ if state.missing_state():
180
+ raise IctrpError(
181
+ ErrorCode.UPSTREAM_CONTRACT_DRIFT,
182
+ "Results page did not carry WebForms state",
183
+ upstream_status=response.status_code,
184
+ hint=(
185
+ "The search did not produce a results page. The response may be "
186
+ "an error page rather than a result set."
187
+ ),
188
+ )
189
+ self.results_state = state
190
+ self.last_html = html
191
+ self.reported_total = parse_reported_total(html)
192
+ return html
193
+
194
+ # ---- step 3 -----------------------------------------------------------
195
+
196
+ async def export_csv(self) -> CsvPayload:
197
+ if self.results_state is None:
198
+ raise IctrpError(
199
+ ErrorCode.SESSION_FAILED,
200
+ "export_csv called before a successful search",
201
+ )
202
+
203
+ body = htmlstate.build_export_body(self.last_html, export_control=EXPORT_CONTROL)
204
+ # Fail loudly here rather than upstream, so the measured 302 failure mode
205
+ # can never be reintroduced silently.
206
+ htmlstate.assert_export_body_excludes_search_controls(body)
207
+
208
+ try:
209
+ response = await self.client.post(
210
+ self.base_url,
211
+ data=body,
212
+ headers={
213
+ "Content-Type": "application/x-www-form-urlencoded",
214
+ "Referer": self.base_url,
215
+ },
216
+ )
217
+ except httpx.HTTPError as exc:
218
+ raise IctrpError(
219
+ ErrorCode.UPSTREAM_ERROR,
220
+ f"Export request failed: {exc}",
221
+ ) from exc
222
+
223
+ self.steps.append(f"POST export -> {response.status_code}")
224
+ self.response_date = response.headers.get("date")
225
+
226
+ return classify_export(
227
+ status=response.status_code,
228
+ content_type=response.headers.get("content-type"),
229
+ content_disposition=response.headers.get("content-disposition"),
230
+ body=response.content,
231
+ location=response.headers.get("location"),
232
+ reported_total=self.reported_total,
233
+ )
234
+
235
+ def provenance_steps(self) -> list[str]:
236
+ return list(self.steps)
237
+
238
+
239
+ async def run_chain(keyword: str, *, timeout: float = 120.0) -> tuple[CsvPayload, IctrpSession]:
240
+ """Run the full chain and return the payload plus the session for provenance."""
241
+ session = IctrpSession.create(timeout=timeout)
242
+ await session.load_form()
243
+ await session.search(keyword)
244
+ payload = await session.export_csv()
245
+ return payload, session
@@ -0,0 +1,133 @@
1
+ """Offline bundle support.
2
+
3
+ A packaged desktop application cannot assume a working network on first launch,
4
+ and re-running a three-request upstream chain on every cold start is both slow and
5
+ rude to a public service. So the service can read a pre-built snapshot instead.
6
+
7
+ Resolution order for the active offline source:
8
+
9
+ 1. `ICTRP_BUNDLE_PATH` -- an explicit file, used by tests and by operators who
10
+ want to pin a specific dataset.
11
+ 2. `ICTRP_BUNDLE_DIR` -- a directory holding `<keyword-slug>.json` snapshots.
12
+ 3. The user cache directory, which is where `ictrp_snapshot` writes by default.
13
+
14
+ A bundle is never silent. Every response served from one carries
15
+ `provenance.offline_snapshot` with the snapshot's age, so a stale shipped dataset
16
+ is visible in the output rather than hidden behind a plausible-looking result.
17
+ """
18
+
19
+ from __future__ import annotations
20
+
21
+ import os
22
+ import re
23
+ from pathlib import Path
24
+
25
+ from .data.jsonio import Snapshot, locate_bundle
26
+ from .errors import ErrorCode, IctrpError
27
+
28
+ #: A shipped snapshot older than this is flagged loudly. WHO refreshes weekly, so
29
+ #: four weeks is well past the point where the data should have been rebuilt.
30
+ DEFAULT_MAX_AGE_DAYS = 28.0
31
+
32
+ ENV_BUNDLE_PATH = "ICTRP_BUNDLE_PATH"
33
+ ENV_BUNDLE_DIR = "ICTRP_BUNDLE_DIR"
34
+ ENV_MAX_AGE_DAYS = "ICTRP_BUNDLE_MAX_AGE_DAYS"
35
+
36
+
37
+ def snapshot_slug(keyword: str) -> str:
38
+ """Filesystem-safe name for a keyword."""
39
+ slug = re.sub(r"[^a-z0-9]+", "-", keyword.strip().lower()).strip("-")
40
+ return slug or "all"
41
+
42
+
43
+ def bundle_dir() -> Path | None:
44
+ override = os.environ.get(ENV_BUNDLE_DIR)
45
+ if override:
46
+ return Path(override)
47
+ return None
48
+
49
+
50
+ def max_age_days() -> float:
51
+ raw = os.environ.get(ENV_MAX_AGE_DAYS)
52
+ if raw:
53
+ try:
54
+ return float(raw)
55
+ except ValueError:
56
+ return DEFAULT_MAX_AGE_DAYS
57
+ return DEFAULT_MAX_AGE_DAYS
58
+
59
+
60
+ def candidate_paths(keyword: str) -> list[Path]:
61
+ """Where a snapshot for this keyword might live, best candidate first."""
62
+ candidates: list[Path] = []
63
+
64
+ explicit = os.environ.get(ENV_BUNDLE_PATH)
65
+ if explicit:
66
+ # An explicit path wins outright, but only if it is for this keyword --
67
+ # otherwise a pinned bundle would silently answer unrelated searches.
68
+ path = Path(explicit)
69
+ if path.is_dir():
70
+ candidates.append(path / f"{snapshot_slug(keyword)}.json")
71
+ else:
72
+ candidates.append(path)
73
+
74
+ directory = bundle_dir()
75
+ if directory:
76
+ candidates.append(directory / f"{snapshot_slug(keyword)}.json")
77
+
78
+ return candidates
79
+
80
+
81
+ def load_bundle(keyword: str) -> tuple[Snapshot, Path] | None:
82
+ """Load a snapshot for this keyword, or None when there is not one.
83
+
84
+ Returns None rather than raising: "no bundle here" is the normal case for a
85
+ live installation, and the caller falls through to the network.
86
+ """
87
+ path = locate_bundle(candidate_paths(keyword))
88
+ if path is None:
89
+ return None
90
+
91
+ if path.is_dir():
92
+ raise IctrpError(
93
+ ErrorCode.INVALID_ARGUMENT,
94
+ f"bundle path {str(path)!r} is a directory",
95
+ hint="Point ICTRP_BUNDLE_PATH at a .json snapshot file.",
96
+ )
97
+
98
+ snapshot = Snapshot.read(path)
99
+
100
+ # A single-file bundle configured by path is assumed to be for the keyword it
101
+ # was written for, but a mismatched keyword means it cannot answer this query.
102
+ if snapshot.keyword and snapshot.keyword.strip().lower() != keyword.strip().lower():
103
+ return None
104
+
105
+ return snapshot, path
106
+
107
+
108
+ def snapshot_provenance(snapshot: Snapshot, path: Path) -> dict:
109
+ """Provenance fields describing an offline-served response."""
110
+ stale = snapshot.stale(max_age_days())
111
+ info: dict = {
112
+ "offline_snapshot": True,
113
+ "snapshot_path": str(path),
114
+ "snapshot_created_at": _iso(snapshot.created_at),
115
+ "snapshot_age_days": round(snapshot.age_days, 2),
116
+ "snapshot_trial_count": len(snapshot.trials),
117
+ }
118
+ if stale:
119
+ info["snapshot_stale"] = True
120
+ info["snapshot_stale_warning"] = (
121
+ f"This snapshot is {snapshot.age_days:.1f} days old, beyond the "
122
+ f"{max_age_days():.0f}-day freshness limit. WHO refreshes ICTRP weekly. "
123
+ "Refresh the snapshot before relying on this data."
124
+ )
125
+ else:
126
+ info["snapshot_stale"] = False
127
+ return info
128
+
129
+
130
+ def _iso(epoch: float) -> str:
131
+ import time
132
+
133
+ return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime(epoch))
@@ -0,0 +1,182 @@
1
+ """Provenance.
2
+
3
+ Every tool response carries this. The point is that a caller can always answer:
4
+ where did this come from, when did WHO last process it, and what is missing.
5
+
6
+ Two things are kept rigorously separate, because conflating them is the most
7
+ likely way for this service to mislead:
8
+
9
+ * `rows_returned` -- what we actually hold.
10
+ * `upstream_reported_total` -- what the portal claimed matched.
11
+
12
+ The export is provably incomplete (docs/MEASUREMENTS.md section 2), so neither
13
+ number alone may be presented as "how many trials exist". When they disagree, the
14
+ response says so explicitly.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import time
20
+ from dataclasses import dataclass, field
21
+ from typing import Any
22
+
23
+ SOURCE_NAME = "WHO ICTRP"
24
+
25
+ SOURCE_URL = "https://trialsearch.who.int/"
26
+
27
+ #: Attribution wording required by the ICTRP terms of use.
28
+ ATTRIBUTION = (
29
+ "Data source: WHO International Clinical Trials Registry Platform (ICTRP). "
30
+ "ICTRP data are publicly available for download from the ICTRP Search Portal. "
31
+ "WHO updates ICTRP weekly."
32
+ )
33
+
34
+ #: Stated on every response so the incompleteness can never be missed.
35
+ INCOMPLETENESS_NOTICE = (
36
+ "The ICTRP CSV export is known to omit records that the search portal itself "
37
+ "reports as matches (measured shortfalls from 0.4% to 29% depending on query). "
38
+ "A record absent from a result set is therefore NOT evidence that it does not "
39
+ "exist. Treat counts from this service as counts of retrieved rows, not as "
40
+ "complete counts of matching trials."
41
+ )
42
+
43
+
44
+ @dataclass
45
+ class Provenance:
46
+ """Where a response's data came from and how current it is."""
47
+
48
+ source: str = SOURCE_NAME
49
+ source_url: str = SOURCE_URL
50
+ retrieved_at: str = ""
51
+ upstream_reported_total: int | None = None
52
+ rows_returned: int | None = None
53
+ records_incomplete: bool = False
54
+ estimated_missing: int | None = None
55
+ ictrp_export_date: str | None = None
56
+ ictrp_last_refreshed: str | None = None
57
+ request_steps: list[str] = field(default_factory=list)
58
+ cache_hit: bool = False
59
+ notes: list[str] = field(default_factory=list)
60
+ # Provenance keys this dataclass does not model, carried through untouched.
61
+ #
62
+ # The offline path is the reason this exists. A snapshot provenance carries
63
+ # `offline_snapshot`, `snapshot_path`, `snapshot_age_days` and friends, and
64
+ # those are exactly the fields that tell a caller "this data did not come
65
+ # from a live query". `search()` round-trips through this class before
66
+ # answering, so anything not modelled here is silently dropped -- which
67
+ # would quietly strip the offline labelling and leave snapshot data
68
+ # looking indistinguishable from live data. Modelling the whole set is not
69
+ # an option: the snapshot shape is free to grow.
70
+ extra: dict[str, Any] = field(default_factory=dict)
71
+
72
+ @classmethod
73
+ def now(cls, **kwargs: Any) -> "Provenance":
74
+ return cls(retrieved_at=_utc_now(), **kwargs)
75
+
76
+ @classmethod
77
+ def from_dict(cls, data: dict[str, Any]) -> "Provenance":
78
+ """Rebuild from `to_dict()` output.
79
+
80
+ `to_dict()` is a presentation shape: it adds derived keys like
81
+ `attribution` and `incompleteness_notice`, which are not constructor
82
+ fields. They are accepted and ignored here so a stored provenance can be
83
+ round-tripped without the caller stripping them first.
84
+
85
+ Anything else this class does not model is preserved in `extra` rather
86
+ than discarded, so a round-trip does not lose provenance detail that
87
+ some other layer added.
88
+ """
89
+ known = {
90
+ "source", "source_url", "retrieved_at", "upstream_reported_total",
91
+ "rows_returned", "records_incomplete", "estimated_missing",
92
+ "ictrp_export_date", "ictrp_last_refreshed", "request_steps",
93
+ "cache_hit", "notes",
94
+ }
95
+ # Derived on every serialization, so never re-adopted from input.
96
+ derived = {"attribution", "incompleteness_notice"}
97
+ extra = {
98
+ k: v
99
+ for k, v in data.items()
100
+ if k not in known and k not in derived
101
+ }
102
+ return cls(**{k: v for k, v in data.items() if k in known}, extra=extra)
103
+
104
+ def to_dict(self) -> dict[str, Any]:
105
+ """Serialized form. Attribution and the incompleteness notice always ride along."""
106
+ out: dict[str, Any] = {
107
+ "source": self.source,
108
+ "source_url": self.source_url,
109
+ "retrieved_at": self.retrieved_at,
110
+ "attribution": ATTRIBUTION,
111
+ }
112
+ if self.upstream_reported_total is not None:
113
+ out["upstream_reported_total"] = self.upstream_reported_total
114
+ if self.rows_returned is not None:
115
+ out["rows_returned"] = self.rows_returned
116
+ out["records_incomplete"] = self.records_incomplete
117
+ if self.estimated_missing is not None:
118
+ out["estimated_missing"] = self.estimated_missing
119
+ if self.ictrp_export_date:
120
+ out["ictrp_export_date"] = self.ictrp_export_date
121
+ if self.ictrp_last_refreshed:
122
+ out["ictrp_last_refreshed"] = self.ictrp_last_refreshed
123
+ if self.request_steps:
124
+ out["request_steps"] = list(self.request_steps)
125
+ out["cache_hit"] = self.cache_hit
126
+ if self.records_incomplete:
127
+ out["incompleteness_notice"] = INCOMPLETENESS_NOTICE
128
+ elif not self.cache_hit:
129
+ # Even a complete-looking response carries the caveat: we can never
130
+ # prove from a single export that nothing was withheld.
131
+ out["incompleteness_notice"] = INCOMPLETENESS_NOTICE
132
+ if self.notes:
133
+ out["notes"] = list(self.notes)
134
+ # Keys this class does not model, restored verbatim.
135
+ for key, value in self.extra.items():
136
+ out.setdefault(key, value)
137
+ return out
138
+
139
+
140
+ def _utc_now() -> str:
141
+ return time.strftime("%Y-%m-%dT%H:%M:%SZ", time.gmtime())
142
+
143
+
144
+ def derive_set_provenance(
145
+ *,
146
+ keyword: str,
147
+ trials: list[dict[str, Any]],
148
+ reported_total: int | None,
149
+ response_date: str | None,
150
+ steps: list[str],
151
+ ) -> Provenance:
152
+ """Build provenance from a freshly materialized export."""
153
+ rows = len(trials)
154
+ export_dates = sorted(
155
+ {t.get("export_date_raw") for t in trials if t.get("export_date_raw")}
156
+ )
157
+ refreshed = sorted(
158
+ {t.get("last_refreshed_display") for t in trials if t.get("last_refreshed_display")}
159
+ )
160
+
161
+ provenance = Provenance.now(
162
+ upstream_reported_total=reported_total,
163
+ rows_returned=rows,
164
+ records_incomplete=bool(reported_total is not None and reported_total > rows),
165
+ estimated_missing=(
166
+ max(0, reported_total - rows) if reported_total is not None else None
167
+ ),
168
+ ictrp_export_date=export_dates[0] if export_dates else None,
169
+ ictrp_last_refreshed=refreshed[0] if refreshed else None,
170
+ request_steps=list(steps),
171
+ cache_hit=False,
172
+ )
173
+
174
+ provenance.notes.append(f"Search keyword: {keyword!r}.")
175
+ if refreshed:
176
+ provenance.notes.append(
177
+ f"Record-level 'Last Refreshed on' values span {len(refreshed)} distinct "
178
+ f"dates in this set; the earliest is shown. Per-record values are on each trial."
179
+ )
180
+ if response_date:
181
+ provenance.notes.append(f"Upstream response Date header: {response_date}.")
182
+ return provenance