ictrp-mcp-server 0.1.1 → 0.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/CHANGELOG.md
CHANGED
|
@@ -3,6 +3,31 @@
|
|
|
3
3
|
Notable changes to the npm package `ictrp-mcp-server`. The Python package
|
|
4
4
|
`ictrp-mcp-service` is versioned separately in `pyproject.toml`.
|
|
5
5
|
|
|
6
|
+
## 0.1.2
|
|
7
|
+
|
|
8
|
+
**Default search view is now the clinician-triage view.** `ictrp_search` used to
|
|
9
|
+
return the 10 most common columns; it now returns the columns a clinician needs
|
|
10
|
+
to triage a target: title and scientific title, sponsor and secondary sponsor,
|
|
11
|
+
contact (PI) name/affiliation/email, recruitment status, phase, registration
|
|
12
|
+
date, condition, countries, age bounds and gender, target size, and a
|
|
13
|
+
`line_of_therapy_hint`. A new derived field, `line_of_therapy_hint`, is a
|
|
14
|
+
non-authoritative extraction of treatment line from the free-text
|
|
15
|
+
inclusion/exclusion criteria and the scientific title. It returns
|
|
16
|
+
`first-line`/`second-line`/`third-line`/`fourth-line`/`later-line`, or `null`
|
|
17
|
+
when nothing is stated -- `null` is documented as "not stated", never as
|
|
18
|
+
"first-line", because the export is incomplete and the column does not exist.
|
|
19
|
+
|
|
20
|
+
**Line-hint fix** — the hint missed trials whose line is stated only as a number
|
|
21
|
+
or only in the scientific title (e.g. "Received >=2 Prior Lines of Therapy"). It
|
|
22
|
+
now scans the scientific title and recognizes numeric line statements; a minimum
|
|
23
|
+
of N lines resolves to `first-line` only when N<=1, otherwise `later-line`.
|
|
24
|
+
|
|
25
|
+
Documented structural limits surfaced by the wider view: ICTRP has no "PI" and
|
|
26
|
+
no "participating-hospital list" column -- sponsor/PI come from
|
|
27
|
+
`primary_sponsor` + `contact_*`, and "participating hospitals" degrades to
|
|
28
|
+
`countries` + `contact_affiliation`. CT.gov `exclusion_criteria` is frequently
|
|
29
|
+
blank upstream, which is a source gap, not a parse failure.
|
|
30
|
+
|
|
6
31
|
## 0.1.1
|
|
7
32
|
|
|
8
33
|
**Retry on transient stalls** — the search POST intermittently stalls past a
|
package/dist/index.js
CHANGED
|
@@ -21,7 +21,7 @@ import { getEnvReport } from "./runtime/env-probe.js";
|
|
|
21
21
|
import { callSidecar, pingSidecar, SEARCH_TIMEOUT_MS, toErrorText, } from "./runtime/sidecar-client.js";
|
|
22
22
|
import { ensureSidecar, installShutdownHooks, sidecarLogTail } from "./runtime/supervisor.js";
|
|
23
23
|
const SERVER_NAME = "ictrp-mcp-server";
|
|
24
|
-
const SERVER_VERSION = "0.1.
|
|
24
|
+
const SERVER_VERSION = "0.1.2";
|
|
25
25
|
/**
|
|
26
26
|
* Attached to every payload that returns rows.
|
|
27
27
|
*
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "ictrp-mcp-server",
|
|
3
|
-
"version": "0.1.
|
|
3
|
+
"version": "0.1.2",
|
|
4
4
|
"description": "MCP server for the WHO International Clinical Trials Registry Platform (ICTRP). Plain HTTP, no browser. Every response is explicit about the export's known incompleteness.",
|
|
5
5
|
"type": "module",
|
|
6
6
|
"main": "dist/index.js",
|
|
@@ -323,6 +323,18 @@ def to_trial(row: list[str], header: tuple[str, ...]) -> dict[str, Any]:
|
|
|
323
323
|
trial["inclusion_age_min"] = _as_int(age_min)
|
|
324
324
|
trial["inclusion_age_max"] = _as_int(age_max)
|
|
325
325
|
|
|
326
|
+
# ICTRP has no structured "line of therapy" column. Treatment line is stated
|
|
327
|
+
# only inside the free-text inclusion/exclusion criteria (e.g. "first-line",
|
|
328
|
+
# "previously treated", "一线") and, frequently, the scientific title
|
|
329
|
+
# ("...Received >=2 Prior Lines of Therapy"). Derive a non-authoritative hint
|
|
330
|
+
# from that text so a caller can surface it without re-parsing. It is a hint,
|
|
331
|
+
# never a fact: absence here does NOT mean the trial is treatment-naive.
|
|
332
|
+
trial["line_of_therapy_hint"] = derive_line_of_therapy_hint(
|
|
333
|
+
raw.get("inclusion_criteria"),
|
|
334
|
+
raw.get("exclusion_criteria"),
|
|
335
|
+
raw.get("scientific_title"),
|
|
336
|
+
)
|
|
337
|
+
|
|
326
338
|
# `results yes no` is a flag, not content, so it is excluded from the test.
|
|
327
339
|
# Measured: the substantive `results *` columns are ~0.0% populated for
|
|
328
340
|
# ChiCTR records, so this is a property of the registry rather than of an
|
|
@@ -346,3 +358,66 @@ def _as_int(value: str | None) -> int | None:
|
|
|
346
358
|
return None
|
|
347
359
|
m = re.search(r"-?\d+", text)
|
|
348
360
|
return int(m.group(0)) if m else None
|
|
361
|
+
|
|
362
|
+
|
|
363
|
+
#: Treatment-line vocabulary, as it actually appears in live inclusion/exclusion
|
|
364
|
+
#: text. Both English and Chinese markers are matched; "refractory"/"relapsed"/
|
|
365
|
+
#: "previously treated" indicate a later line without naming a number.
|
|
366
|
+
_LINE_PATTERNS: tuple[tuple[str, str], ...] = (
|
|
367
|
+
(r"first[-\s]?line|1st[-\s]?line|一线|treatment[-\s]?naive|previously untreated", "first-line"),
|
|
368
|
+
(r"second[-\s]?line|2nd[-\s]?line|二线", "second-line"),
|
|
369
|
+
(r"third[-\s]?line|3rd[-\s]?line|三线", "third-line"),
|
|
370
|
+
(r"fourth[-\s]?line|4th[-\s]?line|四线", "fourth-line"),
|
|
371
|
+
(r"relapsed|refractory|previously treated|prior (therapy|treatment)|已接受.*治疗|经治", "later-line"),
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
#: Numeric "line" statements, e.g. ">=2 Prior Lines of Therapy" or "2-line",
|
|
375
|
+
#: where a word may sit between the number and "line" ("2 prior lines"). The
|
|
376
|
+
#: captured number is the *minimum* line; ">=2" means second-line or later, never
|
|
377
|
+
#: first-line, so it resolves to `later-line` rather than a specific line.
|
|
378
|
+
_LINE_NUMBER = re.compile(r"(\d+)\s*(?:[-\w.,]+\s)*?(?:line|lines|线)")
|
|
379
|
+
|
|
380
|
+
_WORD_NUMBERS = {
|
|
381
|
+
"one": 1, "two": 2, "three": 3, "four": 4, "five": 5,
|
|
382
|
+
"一": 1, "二": 2, "三": 3, "四": 4, "五": 5,
|
|
383
|
+
}
|
|
384
|
+
_WORD_NUMBER = re.compile(r"(one|two|three|four|five|一|二|三|四|五)\s*(?:line|lines|线)")
|
|
385
|
+
|
|
386
|
+
|
|
387
|
+
def derive_line_of_therapy_hint(
|
|
388
|
+
inclusion: str | None, exclusion: str | None, scientific_title: str | None = None
|
|
389
|
+
) -> str | None:
|
|
390
|
+
"""Best-effort treatment-line label from free-text criteria.
|
|
391
|
+
|
|
392
|
+
Returns the earliest explicit line if one is named (first > second > third >
|
|
393
|
+
fourth), else "later-line" when a numeric "N lines" (N>=2) or relapse/
|
|
394
|
+
refractory language is present, else `None`. `None` is deliberate: it means
|
|
395
|
+
"not stated in the criteria we hold", which must not be read as "first-line"
|
|
396
|
+
-- the export is incomplete and the column does not exist.
|
|
397
|
+
|
|
398
|
+
The scientific title is included because the line is often stated only there
|
|
399
|
+
(e.g. "...Received >=2 Prior Lines of Therapy"), not in the criteria body.
|
|
400
|
+
"""
|
|
401
|
+
text = f"{inclusion or ''} {exclusion or ''} {scientific_title or ''}"
|
|
402
|
+
if not text.strip():
|
|
403
|
+
return None
|
|
404
|
+
from .query import _coerce # local import avoids a cycle at module load
|
|
405
|
+
|
|
406
|
+
lowered = _coerce(text)
|
|
407
|
+
|
|
408
|
+
# Explicit worded lines take precedence -- they name a specific line.
|
|
409
|
+
for pattern, label in _LINE_PATTERNS:
|
|
410
|
+
if re.search(pattern, lowered):
|
|
411
|
+
return label
|
|
412
|
+
|
|
413
|
+
# Numeric "N line(s)": a minimum of N, so only the first-line case is exact;
|
|
414
|
+
# anything >=2 is "later-line", never a specific later number.
|
|
415
|
+
for match in _LINE_NUMBER.finditer(lowered):
|
|
416
|
+
if int(match.group(1)) <= 1:
|
|
417
|
+
return "first-line"
|
|
418
|
+
return "later-line"
|
|
419
|
+
for match in _WORD_NUMBER.finditer(lowered):
|
|
420
|
+
if _WORD_NUMBERS[match.group(1)] <= 1:
|
|
421
|
+
return "first-line"
|
|
422
|
+
return "later-line"
|
|
423
|
+
return None
|
|
@@ -35,16 +35,33 @@ from .provenance import INCOMPLETENESS_NOTICE, Provenance, derive_set_provenance
|
|
|
35
35
|
|
|
36
36
|
#: Fields shown by default when a caller does not ask for specific ones. Chosen to
|
|
37
37
|
#: be the high-coverage, decision-relevant columns rather than all 58.
|
|
38
|
+
#: The default view. Deliberately NOT just the 10 most common columns: it is the
|
|
39
|
+
#: view a clinician actually needs to triage a target -- name, sponsor/PI, where
|
|
40
|
+
#: it runs, eligibility shape, and a treatment-line hint. Sponsor/PI come from the
|
|
41
|
+
#: `primary_sponsor` + `contact_*` columns (ICTRP has no dedicated "PI" or
|
|
42
|
+
#: "participating-hospital" field; hospitals exist only as `countries` and the
|
|
43
|
+
#: contact's affiliation), so those are surfaced here. `line_of_therapy_hint` is a
|
|
44
|
+
#: non-authoritative extraction from the free-text criteria -- see normalize.py.
|
|
38
45
|
DEFAULT_FIELDS: tuple[str, ...] = (
|
|
39
46
|
"trial_id",
|
|
40
47
|
"source_register",
|
|
41
48
|
"public_title",
|
|
49
|
+
"scientific_title",
|
|
50
|
+
"primary_sponsor",
|
|
51
|
+
"secondary_sponsor",
|
|
52
|
+
"contact_firstname",
|
|
53
|
+
"contact_lastname",
|
|
54
|
+
"contact_affiliation",
|
|
42
55
|
"recruitment_status",
|
|
43
56
|
"phase_code",
|
|
44
57
|
"registration_date",
|
|
45
58
|
"condition",
|
|
46
59
|
"countries",
|
|
60
|
+
"inclusion_age_min",
|
|
61
|
+
"inclusion_age_max",
|
|
62
|
+
"inclusion_gender",
|
|
47
63
|
"target_size_total",
|
|
64
|
+
"line_of_therapy_hint",
|
|
48
65
|
"last_refreshed_display",
|
|
49
66
|
)
|
|
50
67
|
|