ictrp-mcp-server 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/CHANGELOG.md +65 -0
  2. package/LICENSE +37 -0
  3. package/README.md +208 -0
  4. package/README_ZH.md +189 -0
  5. package/dist/cli/setup-cli.d.ts +14 -0
  6. package/dist/cli/setup-cli.js +230 -0
  7. package/dist/cli/setup-cli.js.map +1 -0
  8. package/dist/index.d.ts +18 -0
  9. package/dist/index.js +477 -0
  10. package/dist/index.js.map +1 -0
  11. package/dist/runtime/bootstrap.d.ts +99 -0
  12. package/dist/runtime/bootstrap.js +350 -0
  13. package/dist/runtime/bootstrap.js.map +1 -0
  14. package/dist/runtime/env-probe.d.ts +108 -0
  15. package/dist/runtime/env-probe.js +479 -0
  16. package/dist/runtime/env-probe.js.map +1 -0
  17. package/dist/runtime/sidecar-client.d.ts +50 -0
  18. package/dist/runtime/sidecar-client.js +120 -0
  19. package/dist/runtime/sidecar-client.js.map +1 -0
  20. package/dist/runtime/supervisor.d.ts +47 -0
  21. package/dist/runtime/supervisor.js +248 -0
  22. package/dist/runtime/supervisor.js.map +1 -0
  23. package/package.json +60 -0
  24. package/sidecar/ictrp_sidecar.py +602 -0
  25. package/sidecar/vendor/ictrp_mcp/__init__.py +3 -0
  26. package/sidecar/vendor/ictrp_mcp/cache/__init__.py +0 -0
  27. package/sidecar/vendor/ictrp_mcp/cache/store.py +313 -0
  28. package/sidecar/vendor/ictrp_mcp/data/__init__.py +0 -0
  29. package/sidecar/vendor/ictrp_mcp/data/columns.py +108 -0
  30. package/sidecar/vendor/ictrp_mcp/data/jsonio.py +213 -0
  31. package/sidecar/vendor/ictrp_mcp/data/normalize.py +348 -0
  32. package/sidecar/vendor/ictrp_mcp/data/query.py +307 -0
  33. package/sidecar/vendor/ictrp_mcp/errors.py +123 -0
  34. package/sidecar/vendor/ictrp_mcp/ictrp/__init__.py +0 -0
  35. package/sidecar/vendor/ictrp_mcp/ictrp/export_guard.py +269 -0
  36. package/sidecar/vendor/ictrp_mcp/ictrp/htmlstate.py +143 -0
  37. package/sidecar/vendor/ictrp_mcp/ictrp/session.py +245 -0
  38. package/sidecar/vendor/ictrp_mcp/offline.py +133 -0
  39. package/sidecar/vendor/ictrp_mcp/provenance.py +182 -0
  40. package/sidecar/vendor/ictrp_mcp/server.py +368 -0
  41. package/sidecar/vendor/ictrp_mcp/tools.py +712 -0
  42. package/sidecar/vendor/pyproject.toml +25 -0
@@ -0,0 +1,348 @@
1
+ """CSV rows -> canonical trial dicts.
2
+
3
+ Every accessor here is total: a row of any length yields a dict with every
4
+ canonical key present. Missing values are `None`, never `""`, so that callers can
5
+ distinguish "we did not get this" from "the source said blank".
6
+
7
+ Type modelling reflects measured reality (docs/MEASUREMENTS.md section 4):
8
+
9
+ * `Target size` is polymorphic -- a bare integer, or per-arm breakdowns like
10
+ `'Drug A:49;Drug B:49;'`. A naive `int()` raises or loses rows.
11
+ * `Study design` is polymorphic -- a controlled term like `'Parallel'`, or a full
12
+ CT.gov sentence. It is NOT a closed enum.
13
+ * `Recruitment Status` casing is inconsistent upstream, so comparisons normalize.
14
+ * `Phase` uses a ChiCTR-specific vocabulary that shares no tokens with CT.gov.
15
+ """
16
+
17
+ from __future__ import annotations
18
+
19
+ import re
20
+ from dataclasses import dataclass
21
+ from typing import Any
22
+
23
+ from .columns import CSV_COLUMNS
24
+
25
+ _NULLISH = {"", "n/a", "na", "not applicable", "none", "unknown", "null", "-"}
26
+
27
+ #: Date shapes measured in the export.
28
+ _DATE_DDMMYYYY = re.compile(r"^(\d{1,2})/(\d{1,2})/(\d{4})$")
29
+ _DATE_YYYYMMDD = re.compile(r"^(\d{4})(\d{2})(\d{2})$")
30
+ _DATE_DMONTHYYYY = re.compile(r"^(\d{1,2})\s+([A-Za-z]+)\s+(\d{4})$")
31
+ _DATETIME_DDMMYYYY = re.compile(
32
+ r"^(\d{1,2})/(\d{1,2})/(\d{4})\s+(\d{1,2}):(\d{2}):(\d{2})$"
33
+ )
34
+
35
+ _MONTHS = {
36
+ "january": 1, "february": 2, "march": 3, "april": 4, "may": 5, "june": 6,
37
+ "july": 7, "august": 8, "september": 9, "october": 10, "november": 11,
38
+ "december": 12,
39
+ }
40
+
41
+ #: Ordering for combined phases such as 'Phase 1/Phase 2'.
42
+ _ROMAN_ORDER = ["I", "II", "III", "IV"]
43
+
44
+ #: Roman numeral -> registry phase digit.
45
+ _ROMAN_NUMERAL = {"I": "1", "II": "2", "III": "3", "IV": "4"}
46
+
47
+ #: Registry phase digit -> roman numeral.
48
+ _NUMERAL_ROMAN = {"1": "I", "2": "II", "3": "III", "4": "IV"}
49
+
50
+
51
+ def _to_roman(token: str) -> str:
52
+ """Normalize a captured phase token to a roman numeral (upper case)."""
53
+ token = token.upper()
54
+ return _NUMERAL_ROMAN.get(token, token)
55
+
56
+
57
+ def clean(value: str | None) -> str | None:
58
+ """Strip and null-normalize a raw cell."""
59
+ if value is None:
60
+ return None
61
+ text = value.strip()
62
+ if text.lower() in _NULLISH:
63
+ return None
64
+ return text
65
+
66
+
67
+ def parse_iso_date(value: str | None, *, day_first: bool = True) -> str | None:
68
+ """Parse a measured export date shape into `YYYY-MM-DD`.
69
+
70
+ Day-first is verified, not assumed: cross-checked against the unambiguous
71
+ `Date registration3` twin across 6,262 rows with zero contradictions. Prefer
72
+ `parse_registration_date` which uses that unambiguous field directly.
73
+ """
74
+ text = clean(value)
75
+ if text is None:
76
+ return None
77
+
78
+ if m := _DATE_YYYYMMDD.match(text):
79
+ return f"{m.group(1)}-{m.group(2)}-{m.group(3)}"
80
+
81
+ if m := _DATE_DDMMYYYY.match(text):
82
+ a, b, y = int(m.group(1)), int(m.group(2)), m.group(3)
83
+ day, month = (a, b) if day_first else (b, a)
84
+ return _safe_iso(y, month, day)
85
+
86
+ if m := _DATETIME_DDMMYYYY.match(text):
87
+ a, b, y = int(m.group(1)), int(m.group(2)), m.group(3)
88
+ day, month = (a, b) if day_first else (b, a)
89
+ return _safe_iso(y, month, day)
90
+
91
+ if m := _DATE_DMONTHYYYY.match(text):
92
+ month = _MONTHS.get(m.group(2).lower())
93
+ if month is None:
94
+ return None
95
+ return _safe_iso(m.group(3), month, int(m.group(1)))
96
+
97
+ return None
98
+
99
+
100
+ def _safe_iso(year: str, month: int, day: int) -> str | None:
101
+ if not (1 <= month <= 12 and 1 <= day <= 31):
102
+ return None
103
+ return f"{int(year):04d}-{month:02d}-{day:02d}"
104
+
105
+
106
+ def split_semicolon_list(value: str | None) -> list[str]:
107
+ """Split a `;`-delimited field, dropping empties.
108
+
109
+ Measured: `Secondary ID` uses `;` (`'NCI-2026-06905;2026-1028'`), as do
110
+ outcome and intervention fields.
111
+ """
112
+ text = clean(value)
113
+ if text is None:
114
+ return []
115
+ return [p.strip() for p in text.split(";") if p.strip()]
116
+
117
+
118
+ _TARGET_ARM_RE = re.compile(r"^(.+?):\s*(\d+)\s*$")
119
+
120
+
121
+ @dataclass(frozen=True)
122
+ class TargetSize:
123
+ """Parsed `Target size`.
124
+
125
+ Either a single total, or per-arm counts. Measured examples:
126
+ `'200'` and `'Capecitabine maintenance group:49;S-1 maintenance group:49;'`.
127
+ """
128
+
129
+ total: int | None
130
+ arms: dict[str, int]
131
+ raw: str | None
132
+
133
+ @property
134
+ def is_arm_split(self) -> bool:
135
+ return bool(self.arms)
136
+
137
+ @property
138
+ def summed(self) -> int | None:
139
+ if self.total is not None:
140
+ return self.total
141
+ if self.arms:
142
+ return sum(self.arms.values())
143
+ return None
144
+
145
+
146
+ def parse_target_size(value: str | None) -> TargetSize:
147
+ text = clean(value)
148
+ if text is None:
149
+ return TargetSize(total=None, arms={}, raw=None)
150
+
151
+ if text.isdigit():
152
+ return TargetSize(total=int(text), arms={}, raw=text)
153
+
154
+ arms: dict[str, int] = {}
155
+ for part in text.split(";"):
156
+ part = part.strip()
157
+ if not part:
158
+ continue
159
+ if m := _TARGET_ARM_RE.match(part):
160
+ arms[m.group(1).strip()] = int(m.group(2))
161
+ if arms:
162
+ return TargetSize(total=None, arms=arms, raw=text)
163
+
164
+ # Last resort: a leading integer inside free text.
165
+ m = re.search(r"\b(\d{1,7})\b", text)
166
+ return TargetSize(total=int(m.group(1)) if m else None, arms={}, raw=text)
167
+
168
+
169
+ def parse_phase(value: str | None) -> tuple[str | None, str | None]:
170
+ """Return `(registry_code, phase_roman)`.
171
+
172
+ The phase vocabulary is a genuine mixture and must be read defensively. Values
173
+ observed in a full export include `'Phase 1'`, `'Phase 2'`, `'Phase 1/Phase 2'`,
174
+ a bare `'2'`, `'N/A'`, `'Not selected'`, `'Not Applicable'`, plus ChiCTR-only
175
+ terms that contain no "phase" token at all (`'Other'`, `'Post-market'`,
176
+ `'Pilot study'`, `'Retrospective study'`,
177
+ `'New Treatment Measure Clinical Study'`).
178
+
179
+ `None` therefore means "no phase concept applies or none was recorded", which
180
+ is distinct from `'OTHER'` ("a phase concept exists that we cannot map"). The
181
+ raw string is always retained by the caller.
182
+ """
183
+ # Read the raw cell, not `clean()`. `clean()` nulls `'N/A'` and `'Not
184
+ # Applicable'`, but here those are meaningful values that must be mapped to
185
+ # the NA code rather than collapsed into "absent".
186
+ if value is None:
187
+ return None, None
188
+ text = value.strip()
189
+ if not text:
190
+ return None, None
191
+
192
+ low = text.lower()
193
+
194
+ # Explicit non-phase markers.
195
+ if low in ("n/a", "na", "not applicable", "not selected", "none", "unknown", "null", "-"):
196
+ return "NA", None
197
+
198
+ # Phases appear as roman numerals ('Phase II', 'I (Phase I study)') and as
199
+ # bare digits ('Phase 2'). The alternation must try the longer roman numerals
200
+ # first, and must not use a trailing \b after a single 'i'.
201
+ combined = re.findall(r"phase\s*(iv|iii|ii|i|[1-4])(?![a-z0-9])", low)
202
+ if combined:
203
+ romans = sorted({_to_roman(tok) for tok in combined}, key=_ROMAN_ORDER.index)
204
+ if len(romans) == 1:
205
+ return f"PHASE{_ROMAN_NUMERAL[romans[0]]}", romans[0]
206
+ # Multi-phase values keep a single PHASE prefix and slash-joined numerals,
207
+ # e.g. 'PHASEI/II'. `phase_roman` is the human-readable pair.
208
+ return "PHASE" + "/".join(romans), "/".join(romans)
209
+
210
+ # Hyphenated ranges, e.g. '1-2', '2-3'.
211
+ if m := re.fullmatch(r"([1-4])\s*[-–]\s*([1-4])", low):
212
+ romans = [_NUMERAL_ROMAN[m.group(1)], _NUMERAL_ROMAN[m.group(2)]]
213
+ romans = sorted(set(romans), key=_ROMAN_ORDER.index)
214
+ return "PHASE" + "/".join(romans), "/".join(romans)
215
+
216
+ # A bare roman numeral, e.g. 'III'. Required to come after the range check so
217
+ # it cannot swallow a hyphenated value.
218
+ if low.upper() in _ROMAN_NUMERAL:
219
+ roman = low.upper()
220
+ return f"PHASE{_ROMAN_NUMERAL[roman]}", roman
221
+
222
+ # 'Phase 0' / exploratory. No PHASE0 registry code exists, so it is reported
223
+ # as its own concept rather than forced into PHASE1.
224
+ if low in ("0", "phase 0", "phase0"):
225
+ return "PHASE0", None
226
+
227
+ # Bare numerals, e.g. '2'. Only trusted when the cell is just a number, so a
228
+ # stray digit inside prose cannot be misread as a phase.
229
+ if low.isdigit() and low in {"1", "2", "3", "4"}:
230
+ return f"PHASE{low}", _NUMERAL_ROMAN[low]
231
+
232
+ if "post-market" in low or "post market" in low:
233
+ return "PHASE4", "IV"
234
+ if "pilot" in low or "retrospective" in low:
235
+ return "NA", None
236
+
237
+ return "OTHER", None
238
+
239
+
240
+ def _canonical_keys() -> tuple[str, ...]:
241
+ """Stable snake_case keys for all 58 source columns."""
242
+ out: list[str] = []
243
+ for col in CSV_COLUMNS:
244
+ key = col.lower()
245
+ key = re.sub(r"[^a-z0-9]+", "_", key).strip("_")
246
+ out.append(key)
247
+ return tuple(out)
248
+
249
+
250
+ CANONICAL_KEYS: tuple[str, ...] = _canonical_keys()
251
+
252
+ #: source column -> canonical key
253
+ COLUMN_TO_KEY: dict[str, str] = dict(zip(CSV_COLUMNS, CANONICAL_KEYS))
254
+
255
+ #: canonical key -> source column
256
+ KEY_TO_COLUMN: dict[str, str] = dict(zip(CANONICAL_KEYS, CSV_COLUMNS))
257
+
258
+
259
+ def row_to_raw(row: list[str], header: tuple[str, ...]) -> dict[str, str | None]:
260
+ """Map one CSV row onto canonical keys, preserving source column names."""
261
+ out: dict[str, str | None] = {}
262
+ for index, column in enumerate(header):
263
+ key = COLUMN_TO_KEY.get(column)
264
+ if key is None:
265
+ continue
266
+ out[key] = row[index] if index < len(row) else None
267
+ for column, key in COLUMN_TO_KEY.items():
268
+ out.setdefault(key, None)
269
+ return out
270
+
271
+
272
+ def to_trial(row: list[str], header: tuple[str, ...]) -> dict[str, Any]:
273
+ """Build a canonical trial dict from one CSV row.
274
+
275
+ Adds derived fields alongside the raw ones. Raw values are always retained;
276
+ derived values are additive and never overwrite the source.
277
+ """
278
+ raw = row_to_raw(row, header)
279
+ trial: dict[str, Any] = dict(raw)
280
+
281
+ trial["trial_id"] = clean(raw.get("trialid"))
282
+ trial["source_register"] = clean(raw.get("source_register"))
283
+ trial["public_title"] = clean(raw.get("public_title"))
284
+ trial["scientific_title"] = clean(raw.get("scientific_title"))
285
+ trial["condition"] = clean(raw.get("condition"))
286
+ trial["countries"] = split_semicolon_list(raw.get("countries"))
287
+ trial["secondary_ids"] = split_semicolon_list(raw.get("secondary_id"))
288
+ trial["interventions"] = split_semicolon_list(raw.get("intervention"))
289
+ trial["primary_outcomes"] = split_semicolon_list(raw.get("primary_outcome"))
290
+ trial["secondary_outcomes"] = split_semicolon_list(raw.get("secondary_outcome"))
291
+
292
+ # Authoritative date: registration3 is unambiguous yyyymmdd. The dd/mm/yyyy
293
+ # twin is day-first (verified) but only used as a fallback.
294
+ authoritative = parse_iso_date(raw.get("date_registration3"))
295
+ fallback = parse_iso_date(raw.get("date_registration"))
296
+ trial["registration_date"] = authoritative or fallback
297
+ trial["registration_date_source"] = (
298
+ "date_registration3" if authoritative else ("date_registration" if fallback else None)
299
+ )
300
+ trial["registration_date_display"] = clean(raw.get("date_registration"))
301
+ # `Last Refreshed on` is reported as `d Month yyyy`; keep the raw string too
302
+ # because it is the visible data-currency marker users will compare against.
303
+ trial["last_refreshed_display"] = clean(raw.get("last_refreshed_on"))
304
+ trial["last_refreshed_date"] = parse_iso_date(raw.get("last_refreshed_on"))
305
+ trial["export_date_raw"] = clean(raw.get("export_date"))
306
+ trial["export_date"] = parse_iso_date(raw.get("export_date"))
307
+
308
+ phase_code, phase_roman = parse_phase(raw.get("phase"))
309
+ trial["phase_code"] = phase_code
310
+ trial["phase_roman"] = phase_roman
311
+
312
+ status = clean(raw.get("recruitment_status"))
313
+ trial["recruitment_status"] = status
314
+ trial["recruitment_status_normalized"] = status.lower() if status else None
315
+
316
+ target = parse_target_size(raw.get("target_size"))
317
+ trial["target_size_total"] = target.summed
318
+ trial["target_size_arms"] = target.arms or None
319
+ trial["target_size_is_arm_split"] = target.is_arm_split
320
+
321
+ age_min = clean(raw.get("inclusion_agemin"))
322
+ age_max = clean(raw.get("inclusion_agemax"))
323
+ trial["inclusion_age_min"] = _as_int(age_min)
324
+ trial["inclusion_age_max"] = _as_int(age_max)
325
+
326
+ # `results yes no` is a flag, not content, so it is excluded from the test.
327
+ # Measured: the substantive `results *` columns are ~0.0% populated for
328
+ # ChiCTR records, so this is a property of the registry rather than of an
329
+ # individual trial -- but it is computed per trial so other registries
330
+ # (which do post results) are represented correctly.
331
+ trial["carries_results_data"] = any(
332
+ clean(raw.get(f"results_{suffix}"))
333
+ for suffix in (
334
+ "date_posted", "url_link", "url_protocol", "date_completed",
335
+ "date_first_publication", "summary", "baseline_char", "adverse_events",
336
+ "outcome_measures", "ipd_plan", "ipd_description",
337
+ )
338
+ )
339
+
340
+ return trial
341
+
342
+
343
+ def _as_int(value: str | None) -> int | None:
344
+ text = clean(value)
345
+ if text is None:
346
+ return None
347
+ m = re.search(r"-?\d+", text)
348
+ return int(m.group(0)) if m else None
@@ -0,0 +1,307 @@
1
+ """Local querying over a materialized result set.
2
+
3
+ None of these functions touch the network. That is the point: once an export has
4
+ been materialized, filtering, faceting, summarizing and exporting are all local
5
+ operations, so a user can refine a search repeatedly without re-hitting the
6
+ portal.
7
+
8
+ Every count produced here is a count of what we actually hold. It is never a
9
+ claim about how many trials exist -- the export is provably incomplete
10
+ (docs/MEASUREMENTS.md section 2), so the accompanying total from the portal is
11
+ carried separately and never merged into these numbers.
12
+ """
13
+
14
+ from __future__ import annotations
15
+
16
+ from collections import Counter
17
+ from typing import Any, Iterable
18
+
19
+ #: Operators supported by `apply_filters`.
20
+ _OPERATORS = {
21
+ "eq", "ne", "contains", "not_contains", "in", "not_in",
22
+ "gt", "gte", "lt", "lte", "exists", "not_exists", "is_null", "is_not_null",
23
+ }
24
+
25
+
26
+ def _coerce(value: Any) -> Any:
27
+ """Lowercase strings for case-insensitive comparison; pass others through."""
28
+ if isinstance(value, str):
29
+ return value.strip().lower()
30
+ return value
31
+
32
+
33
+ def _field_value(trial: dict[str, Any], field: str) -> Any:
34
+ """Resolve a field, preferring the curated alias over a raw source column.
35
+
36
+ Aliases must win. Several canonical raw keys (`phase`, `countries`,
37
+ `target_size`) hold unparsed source strings, while the derived key
38
+ (`phase_code`, `countries`, `target_size_total`) is what callers mean. If the
39
+ raw key were consulted first, `phase` would silently return `'Phase 2'`
40
+ instead of the normalized registry code.
41
+ """
42
+ alias = _ALIASES.get(field)
43
+ if alias is not None and alias in trial:
44
+ return trial[alias]
45
+ if field in trial:
46
+ return trial[field]
47
+ return None
48
+
49
+
50
+ #: Convenience aliases so callers do not need to remember raw column spellings.
51
+ #: Keys that already exist as canonical trial fields (e.g. `trial_id`,
52
+ #: `source_register`) resolve directly and need no entry here; only renamed
53
+ #: fields are listed.
54
+ _ALIASES: dict[str, str] = {
55
+ "title": "public_title",
56
+ "status": "recruitment_status_normalized",
57
+ "register": "source_register",
58
+ "reg_date": "registration_date",
59
+ "phase": "phase_code",
60
+ "country": "countries",
61
+ "age_min": "inclusion_age_min",
62
+ "age_max": "inclusion_age_max",
63
+ "target_size": "target_size_total",
64
+ }
65
+
66
+
67
+ def _match(trial: dict[str, Any], field: str, operator: str, expected: Any) -> bool:
68
+ actual = _field_value(trial, field)
69
+
70
+ if operator == "exists" or operator == "is_not_null":
71
+ return actual is not None and actual != [] and actual != ""
72
+ if operator == "not_exists" or operator == "is_null":
73
+ return actual is None or actual == [] or actual == ""
74
+
75
+ if operator == "eq":
76
+ return _coerce(actual) == _coerce(expected)
77
+ if operator == "ne":
78
+ return _coerce(actual) != _coerce(expected)
79
+
80
+ if operator in ("contains", "not_contains"):
81
+ needle = _coerce(expected)
82
+ if isinstance(actual, list):
83
+ hit = any(_coerce(item) == needle or needle in str(_coerce(item)) for item in actual)
84
+ else:
85
+ hit = needle in str(_coerce(actual or ""))
86
+ return hit if operator == "contains" else not hit
87
+
88
+ if operator in ("in", "not_in"):
89
+ if not isinstance(expected, (list, tuple, set)):
90
+ expected = [expected]
91
+ options = {_coerce(e) for e in expected}
92
+ if isinstance(actual, list):
93
+ hit = any(_coerce(item) in options for item in actual)
94
+ else:
95
+ hit = _coerce(actual) in options
96
+ return hit if operator == "in" else not hit
97
+
98
+ if operator in ("gt", "gte", "lt", "lte"):
99
+ try:
100
+ left = float(actual) # type: ignore[arg-type]
101
+ right = float(expected) # type: ignore[arg-type]
102
+ except (TypeError, ValueError):
103
+ return False
104
+ return {
105
+ "gt": left > right, "gte": left >= right,
106
+ "lt": left < right, "lte": left <= right,
107
+ }[operator]
108
+
109
+ raise ValueError(f"unsupported operator: {operator!r}")
110
+
111
+
112
+ def apply_filters(
113
+ trials: Iterable[dict[str, Any]],
114
+ filters: list[dict[str, Any]] | None,
115
+ ) -> list[dict[str, Any]]:
116
+ """Apply a conjunction of filters.
117
+
118
+ Each filter: `{"field": str, "op": str, "value": Any}`.
119
+ """
120
+ rows = list(trials)
121
+ if not filters:
122
+ return rows
123
+ for spec in filters:
124
+ field = spec.get("field")
125
+ operator = spec.get("op", "eq")
126
+ if not field:
127
+ raise ValueError("each filter needs a 'field'")
128
+ if operator not in _OPERATORS:
129
+ raise ValueError(
130
+ f"unsupported operator {operator!r}; supported: {sorted(_OPERATORS)}"
131
+ )
132
+ expected = spec.get("value")
133
+ rows = [r for r in rows if _match(r, field, operator, expected)]
134
+ return rows
135
+
136
+
137
+ def sort_trials(
138
+ trials: list[dict[str, Any]],
139
+ sort_by: str | None,
140
+ descending: bool = False,
141
+ ) -> list[dict[str, Any]]:
142
+ """Sort, keeping null-ish values last in both directions.
143
+
144
+ Sorting is done on the populated subset only, then the missing ones are
145
+ appended, so a descending sort never floats blanks to the top.
146
+ """
147
+ if not sort_by:
148
+ return trials
149
+
150
+ populated: list[dict[str, Any]] = []
151
+ missing: list[dict[str, Any]] = []
152
+ for trial in trials:
153
+ value = _field_value(trial, sort_by)
154
+ if value is None or value == "" or value == []:
155
+ missing.append(trial)
156
+ else:
157
+ populated.append(trial)
158
+
159
+ populated.sort(key=lambda t: _sortable(_field_value(t, sort_by)), reverse=descending)
160
+ return populated + missing
161
+
162
+
163
+ def _sortable(value: Any) -> Any:
164
+ if isinstance(value, list):
165
+ return ";".join(str(v) for v in value)
166
+ return value
167
+
168
+
169
+ def facet(
170
+ trials: Iterable[dict[str, Any]],
171
+ field: str,
172
+ *,
173
+ limit: int = 50,
174
+ ) -> list[dict[str, Any]]:
175
+ """Value counts for a field, with the null bucket reported separately.
176
+
177
+ String values are case-folded before counting, because upstream casing is
178
+ inconsistent (measured: `'Not Recruiting'` alongside `'Not recruiting'`).
179
+ """
180
+ counter: Counter[str] = Counter()
181
+ nulls = 0
182
+
183
+ for trial in trials:
184
+ value = _field_value(trial, field)
185
+ if value is None or value == "" or value == []:
186
+ nulls += 1
187
+ continue
188
+ if isinstance(value, list):
189
+ for item in value:
190
+ counter[str(item).strip()] += 1
191
+ else:
192
+ counter[str(value).strip()] += 1
193
+
194
+ folded: Counter[str] = Counter()
195
+ display: dict[str, str] = {}
196
+ for value, count in counter.items():
197
+ key = value.lower()
198
+ folded[key] += count
199
+ display.setdefault(key, value)
200
+
201
+ top = folded.most_common(limit)
202
+ return [
203
+ {"value": display[key], "count": count}
204
+ for key, count in top
205
+ ] + ([{"value": None, "count": nulls}] if nulls else [])
206
+
207
+
208
+ def field_coverage(
209
+ trials: Iterable[dict[str, Any]],
210
+ fields: Iterable[str] | None = None,
211
+ ) -> dict[str, dict[str, Any]]:
212
+ """Population statistics per field.
213
+
214
+ This is the honest complement to a result count: it says how much of each
215
+ field we actually have, so a caller can see that e.g. ethics fields are
216
+ sparse for one registry rather than assuming blank means "none recorded".
217
+ """
218
+ rows = list(trials)
219
+ total = len(rows)
220
+ names = list(fields) if fields else sorted({k for r in rows for k in r})
221
+ out: dict[str, dict[str, Any]] = {}
222
+ for name in names:
223
+ populated = 0
224
+ for row in rows:
225
+ value = _field_value(row, name)
226
+ if value is not None and value != "" and value != []:
227
+ populated += 1
228
+ out[name] = {
229
+ "populated": populated,
230
+ "total": total,
231
+ "coverage": round(populated / total, 4) if total else 0.0,
232
+ }
233
+ return out
234
+
235
+
236
+ def find_duplicates(
237
+ trials: Iterable[dict[str, Any]],
238
+ ) -> list[dict[str, Any]]:
239
+ """Group records that likely describe the same underlying trial.
240
+
241
+ Two distinct patterns are reported, because they need different handling:
242
+
243
+ * `shared_secondary_id` -- two or more held records name the same secondary
244
+ identifier. Both are in the set; a caller can compare them directly.
245
+ * `cross_reference` -- a held record points at an identifier that is itself
246
+ another held record's `TrialID`. The pair is a chain, not a shared value,
247
+ so matching on equality of secondary ids alone would miss it.
248
+
249
+ Only identifier evidence is used. Title matching is deliberately avoided: it
250
+ produces false positives on the multi-centre trials that legitimately share a
251
+ near-identical public title.
252
+ """
253
+ rows = list(trials)
254
+ id_index: dict[str, str] = {}
255
+ for row in rows:
256
+ trial_id = row.get("trial_id")
257
+ if trial_id:
258
+ id_index.setdefault(trial_id.strip().upper(), trial_id)
259
+
260
+ by_secondary: dict[str, list[str]] = {}
261
+ for row in rows:
262
+ trial_id = row.get("trial_id")
263
+ if not trial_id:
264
+ continue
265
+ for secondary in row.get("secondary_ids") or []:
266
+ key = secondary.strip().upper()
267
+ if key and key != trial_id.strip().upper():
268
+ by_secondary.setdefault(key, []).append(trial_id)
269
+
270
+ groups: list[dict[str, Any]] = []
271
+ seen_pairs: set[tuple[str, str]] = set()
272
+
273
+ # Pattern 1: several held records name the same secondary identifier.
274
+ for secondary, holders in sorted(by_secondary.items()):
275
+ if len(holders) < 2:
276
+ continue
277
+ groups.append(
278
+ {
279
+ "match_type": "shared_secondary_id",
280
+ "matched_value": secondary,
281
+ "trial_ids": sorted(dict.fromkeys(holders)),
282
+ "resolvable": False,
283
+ }
284
+ )
285
+
286
+ # Pattern 2: a record's secondary id resolves to another held record.
287
+ for secondary, holders in sorted(by_secondary.items()):
288
+ target = id_index.get(secondary)
289
+ if not target:
290
+ continue
291
+ for holder in dict.fromkeys(holders):
292
+ if holder.strip().upper() == target.strip().upper():
293
+ continue
294
+ pair = tuple(sorted((holder, target)))
295
+ if pair in seen_pairs:
296
+ continue
297
+ seen_pairs.add(pair)
298
+ groups.append(
299
+ {
300
+ "match_type": "cross_reference",
301
+ "matched_value": secondary,
302
+ "trial_ids": list(pair),
303
+ "resolvable": True,
304
+ }
305
+ )
306
+
307
+ return groups