ictrp-mcp-server 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +65 -0
- package/LICENSE +37 -0
- package/README.md +208 -0
- package/README_ZH.md +189 -0
- package/dist/cli/setup-cli.d.ts +14 -0
- package/dist/cli/setup-cli.js +230 -0
- package/dist/cli/setup-cli.js.map +1 -0
- package/dist/index.d.ts +18 -0
- package/dist/index.js +477 -0
- package/dist/index.js.map +1 -0
- package/dist/runtime/bootstrap.d.ts +99 -0
- package/dist/runtime/bootstrap.js +350 -0
- package/dist/runtime/bootstrap.js.map +1 -0
- package/dist/runtime/env-probe.d.ts +108 -0
- package/dist/runtime/env-probe.js +479 -0
- package/dist/runtime/env-probe.js.map +1 -0
- package/dist/runtime/sidecar-client.d.ts +50 -0
- package/dist/runtime/sidecar-client.js +120 -0
- package/dist/runtime/sidecar-client.js.map +1 -0
- package/dist/runtime/supervisor.d.ts +47 -0
- package/dist/runtime/supervisor.js +248 -0
- package/dist/runtime/supervisor.js.map +1 -0
- package/package.json +60 -0
- package/sidecar/ictrp_sidecar.py +602 -0
- package/sidecar/vendor/ictrp_mcp/__init__.py +3 -0
- package/sidecar/vendor/ictrp_mcp/cache/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/cache/store.py +313 -0
- package/sidecar/vendor/ictrp_mcp/data/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/data/columns.py +108 -0
- package/sidecar/vendor/ictrp_mcp/data/jsonio.py +213 -0
- package/sidecar/vendor/ictrp_mcp/data/normalize.py +348 -0
- package/sidecar/vendor/ictrp_mcp/data/query.py +307 -0
- package/sidecar/vendor/ictrp_mcp/errors.py +123 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/__init__.py +0 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/export_guard.py +269 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/htmlstate.py +143 -0
- package/sidecar/vendor/ictrp_mcp/ictrp/session.py +245 -0
- package/sidecar/vendor/ictrp_mcp/offline.py +133 -0
- package/sidecar/vendor/ictrp_mcp/provenance.py +182 -0
- package/sidecar/vendor/ictrp_mcp/server.py +368 -0
- package/sidecar/vendor/ictrp_mcp/tools.py +712 -0
- package/sidecar/vendor/pyproject.toml +25 -0
|
@@ -0,0 +1,348 @@
|
|
|
1
|
+
"""CSV rows -> canonical trial dicts.
|
|
2
|
+
|
|
3
|
+
Every accessor here is total: a row of any length yields a dict with every
|
|
4
|
+
canonical key present. Missing values are `None`, never `""`, so that callers can
|
|
5
|
+
distinguish "we did not get this" from "the source said blank".
|
|
6
|
+
|
|
7
|
+
Type modelling reflects measured reality (docs/MEASUREMENTS.md section 4):
|
|
8
|
+
|
|
9
|
+
* `Target size` is polymorphic -- a bare integer, or per-arm breakdowns like
|
|
10
|
+
`'Drug A:49;Drug B:49;'`. A naive `int()` raises or loses rows.
|
|
11
|
+
* `Study design` is polymorphic -- a controlled term like `'Parallel'`, or a full
|
|
12
|
+
CT.gov sentence. It is NOT a closed enum.
|
|
13
|
+
* `Recruitment Status` casing is inconsistent upstream, so comparisons normalize.
|
|
14
|
+
* `Phase` uses a ChiCTR-specific vocabulary that shares no tokens with CT.gov.
|
|
15
|
+
"""
|
|
16
|
+
|
|
17
|
+
from __future__ import annotations
|
|
18
|
+
|
|
19
|
+
import re
|
|
20
|
+
from dataclasses import dataclass
|
|
21
|
+
from typing import Any
|
|
22
|
+
|
|
23
|
+
from .columns import CSV_COLUMNS
|
|
24
|
+
|
|
25
|
+
_NULLISH = {"", "n/a", "na", "not applicable", "none", "unknown", "null", "-"}
|
|
26
|
+
|
|
27
|
+
#: Date shapes measured in the export.
|
|
28
|
+
_DATE_DDMMYYYY = re.compile(r"^(\d{1,2})/(\d{1,2})/(\d{4})$")
|
|
29
|
+
_DATE_YYYYMMDD = re.compile(r"^(\d{4})(\d{2})(\d{2})$")
|
|
30
|
+
_DATE_DMONTHYYYY = re.compile(r"^(\d{1,2})\s+([A-Za-z]+)\s+(\d{4})$")
|
|
31
|
+
_DATETIME_DDMMYYYY = re.compile(
|
|
32
|
+
r"^(\d{1,2})/(\d{1,2})/(\d{4})\s+(\d{1,2}):(\d{2}):(\d{2})$"
|
|
33
|
+
)
|
|
34
|
+
|
|
35
|
+
_MONTHS = {
|
|
36
|
+
"january": 1, "february": 2, "march": 3, "april": 4, "may": 5, "june": 6,
|
|
37
|
+
"july": 7, "august": 8, "september": 9, "october": 10, "november": 11,
|
|
38
|
+
"december": 12,
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
#: Ordering for combined phases such as 'Phase 1/Phase 2'.
|
|
42
|
+
_ROMAN_ORDER = ["I", "II", "III", "IV"]
|
|
43
|
+
|
|
44
|
+
#: Roman numeral -> registry phase digit.
|
|
45
|
+
_ROMAN_NUMERAL = {"I": "1", "II": "2", "III": "3", "IV": "4"}
|
|
46
|
+
|
|
47
|
+
#: Registry phase digit -> roman numeral.
|
|
48
|
+
_NUMERAL_ROMAN = {"1": "I", "2": "II", "3": "III", "4": "IV"}
|
|
49
|
+
|
|
50
|
+
|
|
51
|
+
def _to_roman(token: str) -> str:
|
|
52
|
+
"""Normalize a captured phase token to a roman numeral (upper case)."""
|
|
53
|
+
token = token.upper()
|
|
54
|
+
return _NUMERAL_ROMAN.get(token, token)
|
|
55
|
+
|
|
56
|
+
|
|
57
|
+
def clean(value: str | None) -> str | None:
|
|
58
|
+
"""Strip and null-normalize a raw cell."""
|
|
59
|
+
if value is None:
|
|
60
|
+
return None
|
|
61
|
+
text = value.strip()
|
|
62
|
+
if text.lower() in _NULLISH:
|
|
63
|
+
return None
|
|
64
|
+
return text
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def parse_iso_date(value: str | None, *, day_first: bool = True) -> str | None:
|
|
68
|
+
"""Parse a measured export date shape into `YYYY-MM-DD`.
|
|
69
|
+
|
|
70
|
+
Day-first is verified, not assumed: cross-checked against the unambiguous
|
|
71
|
+
`Date registration3` twin across 6,262 rows with zero contradictions. Prefer
|
|
72
|
+
`parse_registration_date` which uses that unambiguous field directly.
|
|
73
|
+
"""
|
|
74
|
+
text = clean(value)
|
|
75
|
+
if text is None:
|
|
76
|
+
return None
|
|
77
|
+
|
|
78
|
+
if m := _DATE_YYYYMMDD.match(text):
|
|
79
|
+
return f"{m.group(1)}-{m.group(2)}-{m.group(3)}"
|
|
80
|
+
|
|
81
|
+
if m := _DATE_DDMMYYYY.match(text):
|
|
82
|
+
a, b, y = int(m.group(1)), int(m.group(2)), m.group(3)
|
|
83
|
+
day, month = (a, b) if day_first else (b, a)
|
|
84
|
+
return _safe_iso(y, month, day)
|
|
85
|
+
|
|
86
|
+
if m := _DATETIME_DDMMYYYY.match(text):
|
|
87
|
+
a, b, y = int(m.group(1)), int(m.group(2)), m.group(3)
|
|
88
|
+
day, month = (a, b) if day_first else (b, a)
|
|
89
|
+
return _safe_iso(y, month, day)
|
|
90
|
+
|
|
91
|
+
if m := _DATE_DMONTHYYYY.match(text):
|
|
92
|
+
month = _MONTHS.get(m.group(2).lower())
|
|
93
|
+
if month is None:
|
|
94
|
+
return None
|
|
95
|
+
return _safe_iso(m.group(3), month, int(m.group(1)))
|
|
96
|
+
|
|
97
|
+
return None
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def _safe_iso(year: str, month: int, day: int) -> str | None:
|
|
101
|
+
if not (1 <= month <= 12 and 1 <= day <= 31):
|
|
102
|
+
return None
|
|
103
|
+
return f"{int(year):04d}-{month:02d}-{day:02d}"
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def split_semicolon_list(value: str | None) -> list[str]:
|
|
107
|
+
"""Split a `;`-delimited field, dropping empties.
|
|
108
|
+
|
|
109
|
+
Measured: `Secondary ID` uses `;` (`'NCI-2026-06905;2026-1028'`), as do
|
|
110
|
+
outcome and intervention fields.
|
|
111
|
+
"""
|
|
112
|
+
text = clean(value)
|
|
113
|
+
if text is None:
|
|
114
|
+
return []
|
|
115
|
+
return [p.strip() for p in text.split(";") if p.strip()]
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
_TARGET_ARM_RE = re.compile(r"^(.+?):\s*(\d+)\s*$")
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
@dataclass(frozen=True)
|
|
122
|
+
class TargetSize:
|
|
123
|
+
"""Parsed `Target size`.
|
|
124
|
+
|
|
125
|
+
Either a single total, or per-arm counts. Measured examples:
|
|
126
|
+
`'200'` and `'Capecitabine maintenance group:49;S-1 maintenance group:49;'`.
|
|
127
|
+
"""
|
|
128
|
+
|
|
129
|
+
total: int | None
|
|
130
|
+
arms: dict[str, int]
|
|
131
|
+
raw: str | None
|
|
132
|
+
|
|
133
|
+
@property
|
|
134
|
+
def is_arm_split(self) -> bool:
|
|
135
|
+
return bool(self.arms)
|
|
136
|
+
|
|
137
|
+
@property
|
|
138
|
+
def summed(self) -> int | None:
|
|
139
|
+
if self.total is not None:
|
|
140
|
+
return self.total
|
|
141
|
+
if self.arms:
|
|
142
|
+
return sum(self.arms.values())
|
|
143
|
+
return None
|
|
144
|
+
|
|
145
|
+
|
|
146
|
+
def parse_target_size(value: str | None) -> TargetSize:
|
|
147
|
+
text = clean(value)
|
|
148
|
+
if text is None:
|
|
149
|
+
return TargetSize(total=None, arms={}, raw=None)
|
|
150
|
+
|
|
151
|
+
if text.isdigit():
|
|
152
|
+
return TargetSize(total=int(text), arms={}, raw=text)
|
|
153
|
+
|
|
154
|
+
arms: dict[str, int] = {}
|
|
155
|
+
for part in text.split(";"):
|
|
156
|
+
part = part.strip()
|
|
157
|
+
if not part:
|
|
158
|
+
continue
|
|
159
|
+
if m := _TARGET_ARM_RE.match(part):
|
|
160
|
+
arms[m.group(1).strip()] = int(m.group(2))
|
|
161
|
+
if arms:
|
|
162
|
+
return TargetSize(total=None, arms=arms, raw=text)
|
|
163
|
+
|
|
164
|
+
# Last resort: a leading integer inside free text.
|
|
165
|
+
m = re.search(r"\b(\d{1,7})\b", text)
|
|
166
|
+
return TargetSize(total=int(m.group(1)) if m else None, arms={}, raw=text)
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def parse_phase(value: str | None) -> tuple[str | None, str | None]:
|
|
170
|
+
"""Return `(registry_code, phase_roman)`.
|
|
171
|
+
|
|
172
|
+
The phase vocabulary is a genuine mixture and must be read defensively. Values
|
|
173
|
+
observed in a full export include `'Phase 1'`, `'Phase 2'`, `'Phase 1/Phase 2'`,
|
|
174
|
+
a bare `'2'`, `'N/A'`, `'Not selected'`, `'Not Applicable'`, plus ChiCTR-only
|
|
175
|
+
terms that contain no "phase" token at all (`'Other'`, `'Post-market'`,
|
|
176
|
+
`'Pilot study'`, `'Retrospective study'`,
|
|
177
|
+
`'New Treatment Measure Clinical Study'`).
|
|
178
|
+
|
|
179
|
+
`None` therefore means "no phase concept applies or none was recorded", which
|
|
180
|
+
is distinct from `'OTHER'` ("a phase concept exists that we cannot map"). The
|
|
181
|
+
raw string is always retained by the caller.
|
|
182
|
+
"""
|
|
183
|
+
# Read the raw cell, not `clean()`. `clean()` nulls `'N/A'` and `'Not
|
|
184
|
+
# Applicable'`, but here those are meaningful values that must be mapped to
|
|
185
|
+
# the NA code rather than collapsed into "absent".
|
|
186
|
+
if value is None:
|
|
187
|
+
return None, None
|
|
188
|
+
text = value.strip()
|
|
189
|
+
if not text:
|
|
190
|
+
return None, None
|
|
191
|
+
|
|
192
|
+
low = text.lower()
|
|
193
|
+
|
|
194
|
+
# Explicit non-phase markers.
|
|
195
|
+
if low in ("n/a", "na", "not applicable", "not selected", "none", "unknown", "null", "-"):
|
|
196
|
+
return "NA", None
|
|
197
|
+
|
|
198
|
+
# Phases appear as roman numerals ('Phase II', 'I (Phase I study)') and as
|
|
199
|
+
# bare digits ('Phase 2'). The alternation must try the longer roman numerals
|
|
200
|
+
# first, and must not use a trailing \b after a single 'i'.
|
|
201
|
+
combined = re.findall(r"phase\s*(iv|iii|ii|i|[1-4])(?![a-z0-9])", low)
|
|
202
|
+
if combined:
|
|
203
|
+
romans = sorted({_to_roman(tok) for tok in combined}, key=_ROMAN_ORDER.index)
|
|
204
|
+
if len(romans) == 1:
|
|
205
|
+
return f"PHASE{_ROMAN_NUMERAL[romans[0]]}", romans[0]
|
|
206
|
+
# Multi-phase values keep a single PHASE prefix and slash-joined numerals,
|
|
207
|
+
# e.g. 'PHASEI/II'. `phase_roman` is the human-readable pair.
|
|
208
|
+
return "PHASE" + "/".join(romans), "/".join(romans)
|
|
209
|
+
|
|
210
|
+
# Hyphenated ranges, e.g. '1-2', '2-3'.
|
|
211
|
+
if m := re.fullmatch(r"([1-4])\s*[-–]\s*([1-4])", low):
|
|
212
|
+
romans = [_NUMERAL_ROMAN[m.group(1)], _NUMERAL_ROMAN[m.group(2)]]
|
|
213
|
+
romans = sorted(set(romans), key=_ROMAN_ORDER.index)
|
|
214
|
+
return "PHASE" + "/".join(romans), "/".join(romans)
|
|
215
|
+
|
|
216
|
+
# A bare roman numeral, e.g. 'III'. Required to come after the range check so
|
|
217
|
+
# it cannot swallow a hyphenated value.
|
|
218
|
+
if low.upper() in _ROMAN_NUMERAL:
|
|
219
|
+
roman = low.upper()
|
|
220
|
+
return f"PHASE{_ROMAN_NUMERAL[roman]}", roman
|
|
221
|
+
|
|
222
|
+
# 'Phase 0' / exploratory. No PHASE0 registry code exists, so it is reported
|
|
223
|
+
# as its own concept rather than forced into PHASE1.
|
|
224
|
+
if low in ("0", "phase 0", "phase0"):
|
|
225
|
+
return "PHASE0", None
|
|
226
|
+
|
|
227
|
+
# Bare numerals, e.g. '2'. Only trusted when the cell is just a number, so a
|
|
228
|
+
# stray digit inside prose cannot be misread as a phase.
|
|
229
|
+
if low.isdigit() and low in {"1", "2", "3", "4"}:
|
|
230
|
+
return f"PHASE{low}", _NUMERAL_ROMAN[low]
|
|
231
|
+
|
|
232
|
+
if "post-market" in low or "post market" in low:
|
|
233
|
+
return "PHASE4", "IV"
|
|
234
|
+
if "pilot" in low or "retrospective" in low:
|
|
235
|
+
return "NA", None
|
|
236
|
+
|
|
237
|
+
return "OTHER", None
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
def _canonical_keys() -> tuple[str, ...]:
|
|
241
|
+
"""Stable snake_case keys for all 58 source columns."""
|
|
242
|
+
out: list[str] = []
|
|
243
|
+
for col in CSV_COLUMNS:
|
|
244
|
+
key = col.lower()
|
|
245
|
+
key = re.sub(r"[^a-z0-9]+", "_", key).strip("_")
|
|
246
|
+
out.append(key)
|
|
247
|
+
return tuple(out)
|
|
248
|
+
|
|
249
|
+
|
|
250
|
+
CANONICAL_KEYS: tuple[str, ...] = _canonical_keys()
|
|
251
|
+
|
|
252
|
+
#: source column -> canonical key
|
|
253
|
+
COLUMN_TO_KEY: dict[str, str] = dict(zip(CSV_COLUMNS, CANONICAL_KEYS))
|
|
254
|
+
|
|
255
|
+
#: canonical key -> source column
|
|
256
|
+
KEY_TO_COLUMN: dict[str, str] = dict(zip(CANONICAL_KEYS, CSV_COLUMNS))
|
|
257
|
+
|
|
258
|
+
|
|
259
|
+
def row_to_raw(row: list[str], header: tuple[str, ...]) -> dict[str, str | None]:
|
|
260
|
+
"""Map one CSV row onto canonical keys, preserving source column names."""
|
|
261
|
+
out: dict[str, str | None] = {}
|
|
262
|
+
for index, column in enumerate(header):
|
|
263
|
+
key = COLUMN_TO_KEY.get(column)
|
|
264
|
+
if key is None:
|
|
265
|
+
continue
|
|
266
|
+
out[key] = row[index] if index < len(row) else None
|
|
267
|
+
for column, key in COLUMN_TO_KEY.items():
|
|
268
|
+
out.setdefault(key, None)
|
|
269
|
+
return out
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def to_trial(row: list[str], header: tuple[str, ...]) -> dict[str, Any]:
|
|
273
|
+
"""Build a canonical trial dict from one CSV row.
|
|
274
|
+
|
|
275
|
+
Adds derived fields alongside the raw ones. Raw values are always retained;
|
|
276
|
+
derived values are additive and never overwrite the source.
|
|
277
|
+
"""
|
|
278
|
+
raw = row_to_raw(row, header)
|
|
279
|
+
trial: dict[str, Any] = dict(raw)
|
|
280
|
+
|
|
281
|
+
trial["trial_id"] = clean(raw.get("trialid"))
|
|
282
|
+
trial["source_register"] = clean(raw.get("source_register"))
|
|
283
|
+
trial["public_title"] = clean(raw.get("public_title"))
|
|
284
|
+
trial["scientific_title"] = clean(raw.get("scientific_title"))
|
|
285
|
+
trial["condition"] = clean(raw.get("condition"))
|
|
286
|
+
trial["countries"] = split_semicolon_list(raw.get("countries"))
|
|
287
|
+
trial["secondary_ids"] = split_semicolon_list(raw.get("secondary_id"))
|
|
288
|
+
trial["interventions"] = split_semicolon_list(raw.get("intervention"))
|
|
289
|
+
trial["primary_outcomes"] = split_semicolon_list(raw.get("primary_outcome"))
|
|
290
|
+
trial["secondary_outcomes"] = split_semicolon_list(raw.get("secondary_outcome"))
|
|
291
|
+
|
|
292
|
+
# Authoritative date: registration3 is unambiguous yyyymmdd. The dd/mm/yyyy
|
|
293
|
+
# twin is day-first (verified) but only used as a fallback.
|
|
294
|
+
authoritative = parse_iso_date(raw.get("date_registration3"))
|
|
295
|
+
fallback = parse_iso_date(raw.get("date_registration"))
|
|
296
|
+
trial["registration_date"] = authoritative or fallback
|
|
297
|
+
trial["registration_date_source"] = (
|
|
298
|
+
"date_registration3" if authoritative else ("date_registration" if fallback else None)
|
|
299
|
+
)
|
|
300
|
+
trial["registration_date_display"] = clean(raw.get("date_registration"))
|
|
301
|
+
# `Last Refreshed on` is reported as `d Month yyyy`; keep the raw string too
|
|
302
|
+
# because it is the visible data-currency marker users will compare against.
|
|
303
|
+
trial["last_refreshed_display"] = clean(raw.get("last_refreshed_on"))
|
|
304
|
+
trial["last_refreshed_date"] = parse_iso_date(raw.get("last_refreshed_on"))
|
|
305
|
+
trial["export_date_raw"] = clean(raw.get("export_date"))
|
|
306
|
+
trial["export_date"] = parse_iso_date(raw.get("export_date"))
|
|
307
|
+
|
|
308
|
+
phase_code, phase_roman = parse_phase(raw.get("phase"))
|
|
309
|
+
trial["phase_code"] = phase_code
|
|
310
|
+
trial["phase_roman"] = phase_roman
|
|
311
|
+
|
|
312
|
+
status = clean(raw.get("recruitment_status"))
|
|
313
|
+
trial["recruitment_status"] = status
|
|
314
|
+
trial["recruitment_status_normalized"] = status.lower() if status else None
|
|
315
|
+
|
|
316
|
+
target = parse_target_size(raw.get("target_size"))
|
|
317
|
+
trial["target_size_total"] = target.summed
|
|
318
|
+
trial["target_size_arms"] = target.arms or None
|
|
319
|
+
trial["target_size_is_arm_split"] = target.is_arm_split
|
|
320
|
+
|
|
321
|
+
age_min = clean(raw.get("inclusion_agemin"))
|
|
322
|
+
age_max = clean(raw.get("inclusion_agemax"))
|
|
323
|
+
trial["inclusion_age_min"] = _as_int(age_min)
|
|
324
|
+
trial["inclusion_age_max"] = _as_int(age_max)
|
|
325
|
+
|
|
326
|
+
# `results yes no` is a flag, not content, so it is excluded from the test.
|
|
327
|
+
# Measured: the substantive `results *` columns are ~0.0% populated for
|
|
328
|
+
# ChiCTR records, so this is a property of the registry rather than of an
|
|
329
|
+
# individual trial -- but it is computed per trial so other registries
|
|
330
|
+
# (which do post results) are represented correctly.
|
|
331
|
+
trial["carries_results_data"] = any(
|
|
332
|
+
clean(raw.get(f"results_{suffix}"))
|
|
333
|
+
for suffix in (
|
|
334
|
+
"date_posted", "url_link", "url_protocol", "date_completed",
|
|
335
|
+
"date_first_publication", "summary", "baseline_char", "adverse_events",
|
|
336
|
+
"outcome_measures", "ipd_plan", "ipd_description",
|
|
337
|
+
)
|
|
338
|
+
)
|
|
339
|
+
|
|
340
|
+
return trial
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def _as_int(value: str | None) -> int | None:
|
|
344
|
+
text = clean(value)
|
|
345
|
+
if text is None:
|
|
346
|
+
return None
|
|
347
|
+
m = re.search(r"-?\d+", text)
|
|
348
|
+
return int(m.group(0)) if m else None
|
|
@@ -0,0 +1,307 @@
|
|
|
1
|
+
"""Local querying over a materialized result set.
|
|
2
|
+
|
|
3
|
+
None of these functions touch the network. That is the point: once an export has
|
|
4
|
+
been materialized, filtering, faceting, summarizing and exporting are all local
|
|
5
|
+
operations, so a user can refine a search repeatedly without re-hitting the
|
|
6
|
+
portal.
|
|
7
|
+
|
|
8
|
+
Every count produced here is a count of what we actually hold. It is never a
|
|
9
|
+
claim about how many trials exist -- the export is provably incomplete
|
|
10
|
+
(docs/MEASUREMENTS.md section 2), so the accompanying total from the portal is
|
|
11
|
+
carried separately and never merged into these numbers.
|
|
12
|
+
"""
|
|
13
|
+
|
|
14
|
+
from __future__ import annotations
|
|
15
|
+
|
|
16
|
+
from collections import Counter
|
|
17
|
+
from typing import Any, Iterable
|
|
18
|
+
|
|
19
|
+
#: Operators supported by `apply_filters`.
|
|
20
|
+
_OPERATORS = {
|
|
21
|
+
"eq", "ne", "contains", "not_contains", "in", "not_in",
|
|
22
|
+
"gt", "gte", "lt", "lte", "exists", "not_exists", "is_null", "is_not_null",
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
|
|
26
|
+
def _coerce(value: Any) -> Any:
|
|
27
|
+
"""Lowercase strings for case-insensitive comparison; pass others through."""
|
|
28
|
+
if isinstance(value, str):
|
|
29
|
+
return value.strip().lower()
|
|
30
|
+
return value
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _field_value(trial: dict[str, Any], field: str) -> Any:
|
|
34
|
+
"""Resolve a field, preferring the curated alias over a raw source column.
|
|
35
|
+
|
|
36
|
+
Aliases must win. Several canonical raw keys (`phase`, `countries`,
|
|
37
|
+
`target_size`) hold unparsed source strings, while the derived key
|
|
38
|
+
(`phase_code`, `countries`, `target_size_total`) is what callers mean. If the
|
|
39
|
+
raw key were consulted first, `phase` would silently return `'Phase 2'`
|
|
40
|
+
instead of the normalized registry code.
|
|
41
|
+
"""
|
|
42
|
+
alias = _ALIASES.get(field)
|
|
43
|
+
if alias is not None and alias in trial:
|
|
44
|
+
return trial[alias]
|
|
45
|
+
if field in trial:
|
|
46
|
+
return trial[field]
|
|
47
|
+
return None
|
|
48
|
+
|
|
49
|
+
|
|
50
|
+
#: Convenience aliases so callers do not need to remember raw column spellings.
|
|
51
|
+
#: Keys that already exist as canonical trial fields (e.g. `trial_id`,
|
|
52
|
+
#: `source_register`) resolve directly and need no entry here; only renamed
|
|
53
|
+
#: fields are listed.
|
|
54
|
+
_ALIASES: dict[str, str] = {
|
|
55
|
+
"title": "public_title",
|
|
56
|
+
"status": "recruitment_status_normalized",
|
|
57
|
+
"register": "source_register",
|
|
58
|
+
"reg_date": "registration_date",
|
|
59
|
+
"phase": "phase_code",
|
|
60
|
+
"country": "countries",
|
|
61
|
+
"age_min": "inclusion_age_min",
|
|
62
|
+
"age_max": "inclusion_age_max",
|
|
63
|
+
"target_size": "target_size_total",
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _match(trial: dict[str, Any], field: str, operator: str, expected: Any) -> bool:
|
|
68
|
+
actual = _field_value(trial, field)
|
|
69
|
+
|
|
70
|
+
if operator == "exists" or operator == "is_not_null":
|
|
71
|
+
return actual is not None and actual != [] and actual != ""
|
|
72
|
+
if operator == "not_exists" or operator == "is_null":
|
|
73
|
+
return actual is None or actual == [] or actual == ""
|
|
74
|
+
|
|
75
|
+
if operator == "eq":
|
|
76
|
+
return _coerce(actual) == _coerce(expected)
|
|
77
|
+
if operator == "ne":
|
|
78
|
+
return _coerce(actual) != _coerce(expected)
|
|
79
|
+
|
|
80
|
+
if operator in ("contains", "not_contains"):
|
|
81
|
+
needle = _coerce(expected)
|
|
82
|
+
if isinstance(actual, list):
|
|
83
|
+
hit = any(_coerce(item) == needle or needle in str(_coerce(item)) for item in actual)
|
|
84
|
+
else:
|
|
85
|
+
hit = needle in str(_coerce(actual or ""))
|
|
86
|
+
return hit if operator == "contains" else not hit
|
|
87
|
+
|
|
88
|
+
if operator in ("in", "not_in"):
|
|
89
|
+
if not isinstance(expected, (list, tuple, set)):
|
|
90
|
+
expected = [expected]
|
|
91
|
+
options = {_coerce(e) for e in expected}
|
|
92
|
+
if isinstance(actual, list):
|
|
93
|
+
hit = any(_coerce(item) in options for item in actual)
|
|
94
|
+
else:
|
|
95
|
+
hit = _coerce(actual) in options
|
|
96
|
+
return hit if operator == "in" else not hit
|
|
97
|
+
|
|
98
|
+
if operator in ("gt", "gte", "lt", "lte"):
|
|
99
|
+
try:
|
|
100
|
+
left = float(actual) # type: ignore[arg-type]
|
|
101
|
+
right = float(expected) # type: ignore[arg-type]
|
|
102
|
+
except (TypeError, ValueError):
|
|
103
|
+
return False
|
|
104
|
+
return {
|
|
105
|
+
"gt": left > right, "gte": left >= right,
|
|
106
|
+
"lt": left < right, "lte": left <= right,
|
|
107
|
+
}[operator]
|
|
108
|
+
|
|
109
|
+
raise ValueError(f"unsupported operator: {operator!r}")
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def apply_filters(
|
|
113
|
+
trials: Iterable[dict[str, Any]],
|
|
114
|
+
filters: list[dict[str, Any]] | None,
|
|
115
|
+
) -> list[dict[str, Any]]:
|
|
116
|
+
"""Apply a conjunction of filters.
|
|
117
|
+
|
|
118
|
+
Each filter: `{"field": str, "op": str, "value": Any}`.
|
|
119
|
+
"""
|
|
120
|
+
rows = list(trials)
|
|
121
|
+
if not filters:
|
|
122
|
+
return rows
|
|
123
|
+
for spec in filters:
|
|
124
|
+
field = spec.get("field")
|
|
125
|
+
operator = spec.get("op", "eq")
|
|
126
|
+
if not field:
|
|
127
|
+
raise ValueError("each filter needs a 'field'")
|
|
128
|
+
if operator not in _OPERATORS:
|
|
129
|
+
raise ValueError(
|
|
130
|
+
f"unsupported operator {operator!r}; supported: {sorted(_OPERATORS)}"
|
|
131
|
+
)
|
|
132
|
+
expected = spec.get("value")
|
|
133
|
+
rows = [r for r in rows if _match(r, field, operator, expected)]
|
|
134
|
+
return rows
|
|
135
|
+
|
|
136
|
+
|
|
137
|
+
def sort_trials(
|
|
138
|
+
trials: list[dict[str, Any]],
|
|
139
|
+
sort_by: str | None,
|
|
140
|
+
descending: bool = False,
|
|
141
|
+
) -> list[dict[str, Any]]:
|
|
142
|
+
"""Sort, keeping null-ish values last in both directions.
|
|
143
|
+
|
|
144
|
+
Sorting is done on the populated subset only, then the missing ones are
|
|
145
|
+
appended, so a descending sort never floats blanks to the top.
|
|
146
|
+
"""
|
|
147
|
+
if not sort_by:
|
|
148
|
+
return trials
|
|
149
|
+
|
|
150
|
+
populated: list[dict[str, Any]] = []
|
|
151
|
+
missing: list[dict[str, Any]] = []
|
|
152
|
+
for trial in trials:
|
|
153
|
+
value = _field_value(trial, sort_by)
|
|
154
|
+
if value is None or value == "" or value == []:
|
|
155
|
+
missing.append(trial)
|
|
156
|
+
else:
|
|
157
|
+
populated.append(trial)
|
|
158
|
+
|
|
159
|
+
populated.sort(key=lambda t: _sortable(_field_value(t, sort_by)), reverse=descending)
|
|
160
|
+
return populated + missing
|
|
161
|
+
|
|
162
|
+
|
|
163
|
+
def _sortable(value: Any) -> Any:
|
|
164
|
+
if isinstance(value, list):
|
|
165
|
+
return ";".join(str(v) for v in value)
|
|
166
|
+
return value
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def facet(
|
|
170
|
+
trials: Iterable[dict[str, Any]],
|
|
171
|
+
field: str,
|
|
172
|
+
*,
|
|
173
|
+
limit: int = 50,
|
|
174
|
+
) -> list[dict[str, Any]]:
|
|
175
|
+
"""Value counts for a field, with the null bucket reported separately.
|
|
176
|
+
|
|
177
|
+
String values are case-folded before counting, because upstream casing is
|
|
178
|
+
inconsistent (measured: `'Not Recruiting'` alongside `'Not recruiting'`).
|
|
179
|
+
"""
|
|
180
|
+
counter: Counter[str] = Counter()
|
|
181
|
+
nulls = 0
|
|
182
|
+
|
|
183
|
+
for trial in trials:
|
|
184
|
+
value = _field_value(trial, field)
|
|
185
|
+
if value is None or value == "" or value == []:
|
|
186
|
+
nulls += 1
|
|
187
|
+
continue
|
|
188
|
+
if isinstance(value, list):
|
|
189
|
+
for item in value:
|
|
190
|
+
counter[str(item).strip()] += 1
|
|
191
|
+
else:
|
|
192
|
+
counter[str(value).strip()] += 1
|
|
193
|
+
|
|
194
|
+
folded: Counter[str] = Counter()
|
|
195
|
+
display: dict[str, str] = {}
|
|
196
|
+
for value, count in counter.items():
|
|
197
|
+
key = value.lower()
|
|
198
|
+
folded[key] += count
|
|
199
|
+
display.setdefault(key, value)
|
|
200
|
+
|
|
201
|
+
top = folded.most_common(limit)
|
|
202
|
+
return [
|
|
203
|
+
{"value": display[key], "count": count}
|
|
204
|
+
for key, count in top
|
|
205
|
+
] + ([{"value": None, "count": nulls}] if nulls else [])
|
|
206
|
+
|
|
207
|
+
|
|
208
|
+
def field_coverage(
|
|
209
|
+
trials: Iterable[dict[str, Any]],
|
|
210
|
+
fields: Iterable[str] | None = None,
|
|
211
|
+
) -> dict[str, dict[str, Any]]:
|
|
212
|
+
"""Population statistics per field.
|
|
213
|
+
|
|
214
|
+
This is the honest complement to a result count: it says how much of each
|
|
215
|
+
field we actually have, so a caller can see that e.g. ethics fields are
|
|
216
|
+
sparse for one registry rather than assuming blank means "none recorded".
|
|
217
|
+
"""
|
|
218
|
+
rows = list(trials)
|
|
219
|
+
total = len(rows)
|
|
220
|
+
names = list(fields) if fields else sorted({k for r in rows for k in r})
|
|
221
|
+
out: dict[str, dict[str, Any]] = {}
|
|
222
|
+
for name in names:
|
|
223
|
+
populated = 0
|
|
224
|
+
for row in rows:
|
|
225
|
+
value = _field_value(row, name)
|
|
226
|
+
if value is not None and value != "" and value != []:
|
|
227
|
+
populated += 1
|
|
228
|
+
out[name] = {
|
|
229
|
+
"populated": populated,
|
|
230
|
+
"total": total,
|
|
231
|
+
"coverage": round(populated / total, 4) if total else 0.0,
|
|
232
|
+
}
|
|
233
|
+
return out
|
|
234
|
+
|
|
235
|
+
|
|
236
|
+
def find_duplicates(
|
|
237
|
+
trials: Iterable[dict[str, Any]],
|
|
238
|
+
) -> list[dict[str, Any]]:
|
|
239
|
+
"""Group records that likely describe the same underlying trial.
|
|
240
|
+
|
|
241
|
+
Two distinct patterns are reported, because they need different handling:
|
|
242
|
+
|
|
243
|
+
* `shared_secondary_id` -- two or more held records name the same secondary
|
|
244
|
+
identifier. Both are in the set; a caller can compare them directly.
|
|
245
|
+
* `cross_reference` -- a held record points at an identifier that is itself
|
|
246
|
+
another held record's `TrialID`. The pair is a chain, not a shared value,
|
|
247
|
+
so matching on equality of secondary ids alone would miss it.
|
|
248
|
+
|
|
249
|
+
Only identifier evidence is used. Title matching is deliberately avoided: it
|
|
250
|
+
produces false positives on the multi-centre trials that legitimately share a
|
|
251
|
+
near-identical public title.
|
|
252
|
+
"""
|
|
253
|
+
rows = list(trials)
|
|
254
|
+
id_index: dict[str, str] = {}
|
|
255
|
+
for row in rows:
|
|
256
|
+
trial_id = row.get("trial_id")
|
|
257
|
+
if trial_id:
|
|
258
|
+
id_index.setdefault(trial_id.strip().upper(), trial_id)
|
|
259
|
+
|
|
260
|
+
by_secondary: dict[str, list[str]] = {}
|
|
261
|
+
for row in rows:
|
|
262
|
+
trial_id = row.get("trial_id")
|
|
263
|
+
if not trial_id:
|
|
264
|
+
continue
|
|
265
|
+
for secondary in row.get("secondary_ids") or []:
|
|
266
|
+
key = secondary.strip().upper()
|
|
267
|
+
if key and key != trial_id.strip().upper():
|
|
268
|
+
by_secondary.setdefault(key, []).append(trial_id)
|
|
269
|
+
|
|
270
|
+
groups: list[dict[str, Any]] = []
|
|
271
|
+
seen_pairs: set[tuple[str, str]] = set()
|
|
272
|
+
|
|
273
|
+
# Pattern 1: several held records name the same secondary identifier.
|
|
274
|
+
for secondary, holders in sorted(by_secondary.items()):
|
|
275
|
+
if len(holders) < 2:
|
|
276
|
+
continue
|
|
277
|
+
groups.append(
|
|
278
|
+
{
|
|
279
|
+
"match_type": "shared_secondary_id",
|
|
280
|
+
"matched_value": secondary,
|
|
281
|
+
"trial_ids": sorted(dict.fromkeys(holders)),
|
|
282
|
+
"resolvable": False,
|
|
283
|
+
}
|
|
284
|
+
)
|
|
285
|
+
|
|
286
|
+
# Pattern 2: a record's secondary id resolves to another held record.
|
|
287
|
+
for secondary, holders in sorted(by_secondary.items()):
|
|
288
|
+
target = id_index.get(secondary)
|
|
289
|
+
if not target:
|
|
290
|
+
continue
|
|
291
|
+
for holder in dict.fromkeys(holders):
|
|
292
|
+
if holder.strip().upper() == target.strip().upper():
|
|
293
|
+
continue
|
|
294
|
+
pair = tuple(sorted((holder, target)))
|
|
295
|
+
if pair in seen_pairs:
|
|
296
|
+
continue
|
|
297
|
+
seen_pairs.add(pair)
|
|
298
|
+
groups.append(
|
|
299
|
+
{
|
|
300
|
+
"match_type": "cross_reference",
|
|
301
|
+
"matched_value": secondary,
|
|
302
|
+
"trial_ids": list(pair),
|
|
303
|
+
"resolvable": True,
|
|
304
|
+
}
|
|
305
|
+
)
|
|
306
|
+
|
|
307
|
+
return groups
|