medsci-skills 5.1.0 → 5.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/metadata/distribution_files.json +35 -10
- package/metadata/distribution_manifest.json +1 -1
- package/package.json +1 -1
- package/skills/mllm-eval/SKILL.md +8 -0
- package/skills/mllm-eval/references/evaluation_axes.md +161 -0
- package/skills/model-evaluation/SKILL.md +12 -0
- package/skills/model-evaluation/references/metric_selection_grounding.md +139 -0
- package/skills/model-evaluation/scripts/check_metric_reporting.py +1 -1
- package/skills/model-evaluation/scripts/metric_reporting_challenge/fixture/det_good_wrapped.md +5 -0
- package/skills/model-evaluation/scripts/metric_reporting_challenge/fixture/det_no_iou.md +4 -0
- package/skills/model-evaluation/scripts/metric_reporting_challenge/verify.sh +6 -1
- package/skills/model-validation/SKILL.md +6 -0
- package/skills/model-validation/references/validation_design.md +150 -0
|
@@ -2533,8 +2533,13 @@
|
|
|
2533
2533
|
},
|
|
2534
2534
|
{
|
|
2535
2535
|
"path": "skills/mllm-eval/SKILL.md",
|
|
2536
|
-
"size":
|
|
2537
|
-
"sha256": "
|
|
2536
|
+
"size": 6720,
|
|
2537
|
+
"sha256": "afdc167bfccb6bad8a4019e0af5bd7c955129220d360511a89962ecd321bc44d"
|
|
2538
|
+
},
|
|
2539
|
+
{
|
|
2540
|
+
"path": "skills/mllm-eval/references/evaluation_axes.md",
|
|
2541
|
+
"size": 10857,
|
|
2542
|
+
"sha256": "49d77ab63feae5dba5cdae7e47d9de6ea97ee3b9eea4b9cba39ec8b0e185be72"
|
|
2538
2543
|
},
|
|
2539
2544
|
{
|
|
2540
2545
|
"path": "skills/mllm-eval/scripts/check_mllm_eval_completeness.py",
|
|
@@ -2623,18 +2628,23 @@
|
|
|
2623
2628
|
},
|
|
2624
2629
|
{
|
|
2625
2630
|
"path": "skills/model-evaluation/SKILL.md",
|
|
2626
|
-
"size":
|
|
2627
|
-
"sha256": "
|
|
2631
|
+
"size": 5780,
|
|
2632
|
+
"sha256": "334b4ca87a2f672446563f3fb6d78c0585386fac12aee29e10dc09465e892193"
|
|
2628
2633
|
},
|
|
2629
2634
|
{
|
|
2630
2635
|
"path": "skills/model-evaluation/references/metric_guide.md",
|
|
2631
2636
|
"size": 2454,
|
|
2632
2637
|
"sha256": "8d09ca7ce9fb9f66ee4942689294d9b12ae1d892ac67769cd68fdc38a4e220ee"
|
|
2633
2638
|
},
|
|
2639
|
+
{
|
|
2640
|
+
"path": "skills/model-evaluation/references/metric_selection_grounding.md",
|
|
2641
|
+
"size": 9588,
|
|
2642
|
+
"sha256": "56723e73b2b74299140d353921d1dba471bedf9aea0b8e769fc69995c4733995"
|
|
2643
|
+
},
|
|
2634
2644
|
{
|
|
2635
2645
|
"path": "skills/model-evaluation/scripts/check_metric_reporting.py",
|
|
2636
|
-
"size":
|
|
2637
|
-
"sha256": "
|
|
2646
|
+
"size": 9562,
|
|
2647
|
+
"sha256": "fd6c0205f652651a5a43910cb415ed8442100fc553ce04aa5d5905d97743a0bf"
|
|
2638
2648
|
},
|
|
2639
2649
|
{
|
|
2640
2650
|
"path": "skills/model-evaluation/scripts/metric_reporting_challenge/fixture/clf_bad.md",
|
|
@@ -2646,6 +2656,16 @@
|
|
|
2646
2656
|
"size": 186,
|
|
2647
2657
|
"sha256": "ba44a3b38b4128fa713555c8b332f211e9087e6bc016d80af4c3074c3ca6ef8e"
|
|
2648
2658
|
},
|
|
2659
|
+
{
|
|
2660
|
+
"path": "skills/model-evaluation/scripts/metric_reporting_challenge/fixture/det_good_wrapped.md",
|
|
2661
|
+
"size": 285,
|
|
2662
|
+
"sha256": "0c1f5ed5167969870602106e7a7c9a2ae53e3a3ddea5fa280c3beb6bf470c454"
|
|
2663
|
+
},
|
|
2664
|
+
{
|
|
2665
|
+
"path": "skills/model-evaluation/scripts/metric_reporting_challenge/fixture/det_no_iou.md",
|
|
2666
|
+
"size": 224,
|
|
2667
|
+
"sha256": "08227fbe46829b8eecdfe15680e1bb32087f1e6c5a353b3df6055c67d8300fc0"
|
|
2668
|
+
},
|
|
2649
2669
|
{
|
|
2650
2670
|
"path": "skills/model-evaluation/scripts/metric_reporting_challenge/fixture/seg_bad.md",
|
|
2651
2671
|
"size": 119,
|
|
@@ -2663,8 +2683,8 @@
|
|
|
2663
2683
|
},
|
|
2664
2684
|
{
|
|
2665
2685
|
"path": "skills/model-evaluation/scripts/metric_reporting_challenge/verify.sh",
|
|
2666
|
-
"size":
|
|
2667
|
-
"sha256": "
|
|
2686
|
+
"size": 1663,
|
|
2687
|
+
"sha256": "4aa6db49d3552f01a54ddb70484df5d214806d805b836b24beda81d9aca32bd6"
|
|
2668
2688
|
},
|
|
2669
2689
|
{
|
|
2670
2690
|
"path": "skills/model-evaluation/skill.yml",
|
|
@@ -2718,8 +2738,13 @@
|
|
|
2718
2738
|
},
|
|
2719
2739
|
{
|
|
2720
2740
|
"path": "skills/model-validation/SKILL.md",
|
|
2721
|
-
"size":
|
|
2722
|
-
"sha256": "
|
|
2741
|
+
"size": 9810,
|
|
2742
|
+
"sha256": "1a75a5a1f21f5a8778d0b77db2e99574bf37edda2a291d8dde6aafea4a207ff0"
|
|
2743
|
+
},
|
|
2744
|
+
{
|
|
2745
|
+
"path": "skills/model-validation/references/validation_design.md",
|
|
2746
|
+
"size": 11427,
|
|
2747
|
+
"sha256": "16d43b688ea63745174c7ca8fafd78a7342b26c34ad1e10e1fdbc117cceafb2e"
|
|
2723
2748
|
},
|
|
2724
2749
|
{
|
|
2725
2750
|
"path": "skills/model-validation/scripts/check_split_leakage.py",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "medsci-skills",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.2.0",
|
|
4
4
|
"description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE",
|
|
6
6
|
"homepage": "https://github.com/Aperivue/medsci-skills#readme",
|
|
@@ -106,3 +106,11 @@ mllm-eval (this skill: harness design + completeness gate, model-agnostic)
|
|
|
106
106
|
├─ write-paper + check-reporting (TRIPOD-LLM / MI-CLEAR-LLM)
|
|
107
107
|
└─ self-review / peer-review (ME0–ME8 reviewer probe)
|
|
108
108
|
```
|
|
109
|
+
|
|
110
|
+
## Reference Files
|
|
111
|
+
|
|
112
|
+
- `${CLAUDE_SKILL_DIR}/references/evaluation_axes.md` — the *why* behind the ME2–ME7 axes:
|
|
113
|
+
clinical-efficacy metrics beyond n-gram overlap (e.g. RadGraph-F1 / CheXbert-F1 vs BLEU/ROUGE),
|
|
114
|
+
faithfulness & hallucination, pretraining/benchmark contamination, prompt-sensitivity &
|
|
115
|
+
determinism, answer-matching, and the reader study — each mapped to its gate verdict. Load on
|
|
116
|
+
demand during Phases 2–4.
|
|
@@ -0,0 +1,161 @@
|
|
|
1
|
+
# Evaluation axes (mllm-eval)
|
|
2
|
+
|
|
3
|
+
Load-on-demand reference behind the ME2–ME7 axes: the clinical-efficacy metrics,
|
|
4
|
+
faithfulness, contamination, prompt-sensitivity, answer-matching, and reader-study
|
|
5
|
+
machinery an LLM/MLLM clinical evaluation must cover. Anchored to the radiology-NLP
|
|
6
|
+
metric literature — **RadGraph** (Jain et al., NeurIPS Datasets & Benchmarks 2021),
|
|
7
|
+
**CheXbert** (Smit et al., 2020), the rule-based **CheXpert** labeler (Irvin et al., 2019),
|
|
8
|
+
and **RadCliQ** (Yu et al., *Patterns* 2023) — and to the reporting standards CLAIM 2024,
|
|
9
|
+
**TRIPOD-LLM**, and **MI-CLEAR-LLM**. This skill **specifies and routes** these metrics to
|
|
10
|
+
their published extractors and to `/analyze-stats`; it does not run the model or compute the
|
|
11
|
+
scores itself. Never report n-gram overlap as clinical correctness, and never fabricate a
|
|
12
|
+
score.
|
|
13
|
+
|
|
14
|
+
## Clinical-efficacy metrics beyond n-gram overlap (ME2 → `NGRAM_ONLY`)
|
|
15
|
+
|
|
16
|
+
- **Why n-gram overlap fails.** BLEU / ROUGE / METEOR / CIDEr measure surface lexical
|
|
17
|
+
overlap. A report can score high while inverting laterality or omitting a pneumothorax, and
|
|
18
|
+
low while paraphrasing a correct finding. Yu et al. (*Patterns* 2023, RadCliQ) showed these
|
|
19
|
+
metrics correlate weakly with radiologist-assessed clinical error — so an n-gram score is a
|
|
20
|
+
**fluency proxy, not a correctness claim**.
|
|
21
|
+
- **RadGraph-F1.** Overlap of the (entity, relation) tuples the RadGraph schema (Jain et al.,
|
|
22
|
+
2021) extracts from the reference and the candidate report — it rewards getting the *same
|
|
23
|
+
findings and their relationships*, not the same words.
|
|
24
|
+
- **CheXbert-F1 / CheXpert labeler.** Agreement on the structured CheXpert observation labels
|
|
25
|
+
extracted by CheXbert (Smit et al., 2020) or by the rule-based CheXpert labeler (Irvin et
|
|
26
|
+
al., 2019) — a finding-label–level correctness signal.
|
|
27
|
+
- **RadCliQ.** A composite (Yu et al., 2023) that combines metrics to better predict the count
|
|
28
|
+
of radiologist-judged errors. Report it as a composite **alongside** its components, not as a
|
|
29
|
+
one-number replacement.
|
|
30
|
+
- **Advise:** report a clinical-efficacy metric (RadGraph-F1 / CheXbert-F1 / RadCliQ)
|
|
31
|
+
**with bootstrap CIs over reports** and a **per-finding / per-label breakdown**, and present
|
|
32
|
+
any BLEU/ROUGE explicitly labelled as a surface-overlap measure — never as the headline.
|
|
33
|
+
|
|
34
|
+
## Faithfulness & hallucination (ME3 → `FAITHFULNESS_MISSING`)
|
|
35
|
+
|
|
36
|
+
- **Fluency is not faithfulness.** A high-overlap, well-formed report can still assert findings
|
|
37
|
+
the image does not support. Measure faithfulness directly; do not infer it from an accuracy
|
|
38
|
+
number.
|
|
39
|
+
- **Atomic-fact decomposition.** Break the generated text into atomic clinical claims and check
|
|
40
|
+
each against the image / source, then report a **faithfulness (or hallucination) rate** — the
|
|
41
|
+
fraction of generated claims that are grounded.
|
|
42
|
+
- **Direction matters.** Separate **omission** (a true finding the model missed) from
|
|
43
|
+
**fabrication** (a false finding the model asserted); they carry different clinical risk and
|
|
44
|
+
should be reported separately, not folded into one error count.
|
|
45
|
+
- **False-premise / abstention probe.** Ask about an absent finding or an unanswerable
|
|
46
|
+
question; a faithful model abstains rather than confabulates. Named instruments: **MedVH**,
|
|
47
|
+
**Med-HALT**.
|
|
48
|
+
- **Advise:** a generation or VQA claim with no faithfulness and no false-premise/abstention
|
|
49
|
+
evaluation is the central MLLM gap — require both, with rates, before any clinical claim.
|
|
50
|
+
|
|
51
|
+
## Pretraining / benchmark contamination (ME4 → `CONTAMINATION_UNADDRESSED`)
|
|
52
|
+
|
|
53
|
+
- **Why public benchmarks are suspect.** VQA-RAD, SLAKE, MIMIC-CXR–derived sets, MedQA,
|
|
54
|
+
PMC-VQA, PathVQA, OpenI, and PubMedQA may sit inside the model's pretraining corpus, so a
|
|
55
|
+
high score can be **memorisation, not capability**. For a closed API the corpus is undisclosed,
|
|
56
|
+
so contamination cannot be excluded — only **bounded** and stated.
|
|
57
|
+
- **Three accepted checks (any one, stated explicitly):**
|
|
58
|
+
- **Cutoff vs release date** — compare the model's training cutoff against the benchmark's
|
|
59
|
+
release date; a benchmark that predates the cutoff is at risk.
|
|
60
|
+
- **Held-out / post-cutoff set** — evaluate on a private, institution-collected, or
|
|
61
|
+
after-cutoff set the model could not have seen.
|
|
62
|
+
- **Contamination probe** — canary strings, a perturbed-duplicate performance gap (score on
|
|
63
|
+
verbatim items vs lightly perturbed copies), or a membership/quiz test (ask the model to
|
|
64
|
+
reproduce held-out items).
|
|
65
|
+
- **Advise:** never write "no contamination" (or evaluate on a pre-cutoff public benchmark in
|
|
66
|
+
silence) without one of the checks above; an acknowledged-but-unmitigated risk is a stated
|
|
67
|
+
limitation, not a clean result.
|
|
68
|
+
|
|
69
|
+
## Prompt-sensitivity & determinism (ME5 → `PROMPT_PROVENANCE_MISSING`)
|
|
70
|
+
|
|
71
|
+
- **Outputs move with the prompt and the sampler.** Phrasing, format, the system prompt,
|
|
72
|
+
temperature, top-p/top-k, and run-to-run sampling all shift results. A single-prompt
|
|
73
|
+
single-run number overstates stability.
|
|
74
|
+
- **Closed APIs are non-deterministic even at temperature 0** — identical inputs can yield
|
|
75
|
+
different outputs across calls. Treat any single-run figure as a point estimate of a
|
|
76
|
+
distribution, not a fixed value.
|
|
77
|
+
- **Disclose (MI-CLEAR-LLM transparency):** the **exact prompt(s)** including the system prompt,
|
|
78
|
+
the **decoding settings** (temperature / top-p / seed), **≥ 3 runs** with reported variance
|
|
79
|
+
(e.g., mean ± SD), and a **prompt-robustness** check across **≥ 2 phrasings/formats** for the
|
|
80
|
+
headline result.
|
|
81
|
+
|
|
82
|
+
## Answer-matching for VQA / classification (ME6 → `ANSWER_MATCHING_MISSING`)
|
|
83
|
+
|
|
84
|
+
- **State the matching rule.** Free-text answers must be mapped to the key by a declared rule:
|
|
85
|
+
**exact** string match, **normalised** match (case/punctuation/synonym folding), or
|
|
86
|
+
**LLM-as-judge**. An unspecified rule makes the accuracy unreproducible.
|
|
87
|
+
- **An LLM judge is itself a model.** If a model adjudicates correctness, **validate the judge
|
|
88
|
+
against a human-labelled subset** and report its agreement; route judge validation to
|
|
89
|
+
`/design-ai-benchmarking`. An unvalidated LLM judge can launder the system's own errors.
|
|
90
|
+
- **Operating discipline.** Report accuracy **at the real clinical prevalence**, not on an
|
|
91
|
+
artificially balanced QA set, and state **how refusals/abstentions are scored** (counted
|
|
92
|
+
wrong, excluded, or credited) — the choice can move the headline.
|
|
93
|
+
|
|
94
|
+
## Reader study for generated reports (ME7 → `READER_STUDY_MISSING`)
|
|
95
|
+
|
|
96
|
+
- **Automated metrics do not establish clinical acceptability.** Even RadGraph-F1 / CheXbert-F1
|
|
97
|
+
measure agreement, not whether a clinician would act on the report safely. A deployment or
|
|
98
|
+
utility claim for generated text needs a **blinded clinical reader study**.
|
|
99
|
+
- **Design elements:** a pre-defined **error taxonomy** (clinically significant vs insignificant;
|
|
100
|
+
omission vs fabrication), a **severity scale**, and **inter-reader agreement**.
|
|
101
|
+
- **Route:** the rubric and the IRR design → `/design-ai-benchmarking`; the ICC/κ computation →
|
|
102
|
+
`/analyze-stats`; reader and case **sizing** → `/calc-sample-size`.
|
|
103
|
+
|
|
104
|
+
## Gate mapping
|
|
105
|
+
|
|
106
|
+
The deterministic gate (`scripts/check_mllm_eval_completeness.py`) is a presence check on the
|
|
107
|
+
plan text, task-aware. This reference is the *why* behind each verdict:
|
|
108
|
+
|
|
109
|
+
| Axis (this doc) | Gate verdict | Severity |
|
|
110
|
+
|---|---|---|
|
|
111
|
+
| n-gram only, no clinical metric (report-gen) | `NGRAM_ONLY` | Major |
|
|
112
|
+
| no adjudicated reference standard (report-gen) | `REFERENCE_STANDARD_MISSING` | Major |
|
|
113
|
+
| no faithfulness / false-premise (report-gen, vqa) | `FAITHFULNESS_MISSING` | Major |
|
|
114
|
+
| public benchmark, no contamination handling | `CONTAMINATION_UNADDRESSED` | Major |
|
|
115
|
+
| no blinded reader study (report-gen) | `READER_STUDY_MISSING` | Major (deploy) / Minor |
|
|
116
|
+
| prompt / decoding / multi-run incomplete | `PROMPT_PROVENANCE_MISSING` | Minor |
|
|
117
|
+
| no answer-matching rule (vqa, classification) | `ANSWER_MATCHING_MISSING` | Minor |
|
|
118
|
+
|
|
119
|
+
A Major verdict is a presence gap, not proof the work is wrong — resolve it by adding the axis
|
|
120
|
+
to the plan (or recording, with a stated reason, why it does not apply).
|
|
121
|
+
|
|
122
|
+
## Reporting fit & hand-off
|
|
123
|
+
|
|
124
|
+
Methods / Results stub → `/write-paper`. Item-level compliance with **TRIPOD-LLM**,
|
|
125
|
+
**MI-CLEAR-LLM**, **CLAIM 2024** (and STARD-AI / TRIPOD+AI where a diagnostic/prognostic claim
|
|
126
|
+
is made) → `/check-reporting`. Reviewer-side audit of a finished manuscript uses the
|
|
127
|
+
`mllm_evaluation.md` (ME0–ME8) probe via `/self-review` and `/peer-review`.
|
|
128
|
+
|
|
129
|
+
## Verification notes
|
|
130
|
+
|
|
131
|
+
Each claim here is grounded in a named public method/standard or described qualitatively; no
|
|
132
|
+
numbers, thresholds, or dataset contents are invented.
|
|
133
|
+
|
|
134
|
+
- **n-gram metrics correlate weakly with clinical error; clinical-efficacy metrics needed** —
|
|
135
|
+
Yu et al., *Patterns* 2023 (RadCliQ). Named public methods paper (matches the citation already
|
|
136
|
+
vendored in the `mllm_evaluation.md` probe).
|
|
137
|
+
- **RadGraph-F1 (entity-relation overlap)** — Jain et al., NeurIPS Datasets & Benchmarks 2021.
|
|
138
|
+
Named public methods paper.
|
|
139
|
+
- **CheXbert-F1 / CheXpert observation labels** — Smit et al., 2020 (CheXbert); Irvin et al.,
|
|
140
|
+
2019 (CheXpert labeler). Named public methods papers; the CheXpert label set is a factual
|
|
141
|
+
artifact, not invented.
|
|
142
|
+
- **RadCliQ as a composite** — Yu et al., *Patterns* 2023. Named public methods paper.
|
|
143
|
+
- **Atomic-fact faithfulness, omission-vs-fabrication, false-premise/abstention** — described as
|
|
144
|
+
established evaluation principles; **MedVH** and **Med-HALT** named as instruments only (as in
|
|
145
|
+
the probe), no scores invented.
|
|
146
|
+
- **Contamination of public benchmarks; closed-corpus unknowability; cutoff/held-out/probe
|
|
147
|
+
checks (canary, perturbed-duplicate gap, membership test)** — stated as accepted principles and
|
|
148
|
+
practices, qualitatively; benchmark **names** (VQA-RAD, SLAKE, MIMIC-CXR, MedQA, PMC-VQA,
|
|
149
|
+
PathVQA, OpenI, PubMedQA) are factual public-dataset names, no contents reproduced.
|
|
150
|
+
- **Closed-API non-determinism even at temperature 0; prompt/format/sampling sensitivity** —
|
|
151
|
+
described qualitatively as documented behavior; no figure attached.
|
|
152
|
+
- **Prompt + decoding + ≥3 runs + ≥2 phrasings disclosure** — MI-CLEAR-LLM transparency
|
|
153
|
+
(named standard); the ≥3 / ≥2 conventions are this skill's own house thresholds (carried from
|
|
154
|
+
SKILL.md / the ME-probe), not literature values.
|
|
155
|
+
- **LLM-as-judge must be validated against human labels** — stated as a methodological principle;
|
|
156
|
+
judge validation routed to `/design-ai-benchmarking`.
|
|
157
|
+
- **Reader study for deployment/utility claims; error taxonomy, severity, IRR** — consistent with
|
|
158
|
+
CLAIM 2024 / TRIPOD-LLM reporting expectations (named standards); sizing/IRR routed to
|
|
159
|
+
`/calc-sample-size` and `/analyze-stats`.
|
|
160
|
+
- **Metrics Reloaded / CLAIM 2024 / TRIPOD-LLM / MI-CLEAR-LLM / Model Cards (Mitchell 2019) /
|
|
161
|
+
Datasheets (Gebru 2021)** — named public standards, cited by name only.
|
|
@@ -79,6 +79,18 @@ figures → `/make-figures`; the numbers + subgroup performance → `/model-card
|
|
|
79
79
|
`scripts/check_metric_reporting.py` — flags a task-metric mismatch / missing uncertainty (stdlib,
|
|
80
80
|
network-free). Reproducible challenge: `bash ${CLAUDE_SKILL_DIR}/scripts/metric_reporting_challenge/verify.sh`.
|
|
81
81
|
|
|
82
|
+
## Reference Files
|
|
83
|
+
|
|
84
|
+
Load on demand (keep SKILL.md short):
|
|
85
|
+
- `${CLAUDE_SKILL_DIR}/references/metric_guide.md` — operational checklist: the task-correct metric
|
|
86
|
+
per task (segmentation Dice + HD95/NSD per structure; classification AUROC + AUPRC + sens/spec at
|
|
87
|
+
deployment prevalence; detection FROC/mAP with a stated IoU), plus calibration, subgroup slices,
|
|
88
|
+
run-variance, and the per-case CSV hand-off.
|
|
89
|
+
- `${CLAUDE_SKILL_DIR}/references/metric_selection_grounding.md` — the standards grounding behind
|
|
90
|
+
those choices: the Metrics Reloaded task-fingerprint principle, why each metric pairing is
|
|
91
|
+
required, calibration vs discrimination, disaggregated reporting, and the CLAIM 2024
|
|
92
|
+
reporting-fit map (`/check-reporting` owns the item audit).
|
|
93
|
+
|
|
82
94
|
## Boundaries
|
|
83
95
|
|
|
84
96
|
```
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
# Metric-selection grounding and CLAIM 2024 reporting fit (model-evaluation)
|
|
2
|
+
|
|
3
|
+
The *why* behind the operational checklist in `metric_guide.md`. Where `metric_guide.md` says
|
|
4
|
+
**what** to compute, this doc grounds **why** that pairing is required and where the outputs land
|
|
5
|
+
in the manuscript. Anchored to **Metrics Reloaded** (Maier-Hein & Reinke et al., *Nature Methods*
|
|
6
|
+
2024) and its pitfalls companion (Reinke et al., *Nature Methods* 2024), **CLAIM 2024** (Tejani et
|
|
7
|
+
al., *Radiology: Artificial Intelligence* 2024), **TRIPOD+AI** (the AI extension of TRIPOD; Collins
|
|
8
|
+
et al., *BMJ* 2024), calibration work (Guo et al., *ICML* 2017), and the Model Card **Factors**
|
|
9
|
+
(Mitchell et al., *FAT\** 2019). The deliverable is still the per-case CSV; the deterministic gate
|
|
10
|
+
is `scripts/check_metric_reporting.py`.
|
|
11
|
+
|
|
12
|
+
> Verify exact **CLAIM 2024 item numbers** and any **NSD tolerance** against the source before
|
|
13
|
+
> quoting them as formal values — the mapping and tolerances below are described qualitatively by
|
|
14
|
+
> design. Do not hand-type a metric value: every number comes from executed code.
|
|
15
|
+
|
|
16
|
+
## The Metrics Reloaded principle: task fingerprint → metric
|
|
17
|
+
|
|
18
|
+
- **The metric is derived from the problem, not from habit.** Metrics Reloaded selects metrics from
|
|
19
|
+
the problem fingerprint — task category (classification / segmentation / detection-localization),
|
|
20
|
+
structure size and shape, class prevalence, and whether *where* the model is right matters. A
|
|
21
|
+
metric that ignores a property that matters clinically is the wrong metric.
|
|
22
|
+
- **No single metric is sufficient.** Pair a counting/overlap metric with a complementary one (a
|
|
23
|
+
boundary metric, a calibration summary) so a blind spot in one is covered by the other. A lone
|
|
24
|
+
headline number is the recurring pitfall the companion paper warns about.
|
|
25
|
+
- **Report per-class / per-structure with a distribution**, not only a global mean — a mean hides
|
|
26
|
+
minority-class and small-structure failure.
|
|
27
|
+
- **Define edge-case behaviour explicitly** (empty reference, no positive cases): several metrics
|
|
28
|
+
are undefined there, and a silent convention changes the score.
|
|
29
|
+
|
|
30
|
+
## Segmentation — overlap **and** boundary, per structure
|
|
31
|
+
|
|
32
|
+
- Report an overlap metric (Dice or IoU) **with** a boundary metric (**HD95** or **NSD**). Dice is
|
|
33
|
+
a volume-overlap measure: it is insensitive to boundary error and unstable on small or thin
|
|
34
|
+
structures, so it can look high while the contour is clinically wrong.
|
|
35
|
+
- **HD95** = 95th-percentile Hausdorff distance — robust to a few outlier surface points relative to
|
|
36
|
+
the raw maximum Hausdorff. **NSD** (normalised surface distance / surface Dice) = the fraction of
|
|
37
|
+
the predicted surface within a **task-specific tolerance τ** of the reference surface; **τ must be
|
|
38
|
+
stated** and is chosen from clinical acceptability, not invented.
|
|
39
|
+
- Compute **per structure** with bootstrap 95% CIs obtained by resampling **patients, not pixels**
|
|
40
|
+
(Efron–Tibshirani bootstrap). State the rule for **empty-reference / false-positive-only** cases
|
|
41
|
+
(a Dice of 0/0 is undefined).
|
|
42
|
+
- Gate: `PIXEL_ACCURACY_SEG` (pixel/voxel accuracy is dominated by background — never the headline)
|
|
43
|
+
and `NO_BOUNDARY_METRIC`.
|
|
44
|
+
|
|
45
|
+
## Classification — discrimination, operating point at prevalence, then calibration
|
|
46
|
+
|
|
47
|
+
- **AUROC and AUPRC.** AUROC summarises ranking across thresholds; under class imbalance the
|
|
48
|
+
precision–recall view (AUPRC) is more informative, because the ROC's false-positive rate uses the
|
|
49
|
+
large negative denominator and can look optimistic when negatives dominate (Saito & Rehmsmeier,
|
|
50
|
+
*PLoS ONE* 2015).
|
|
51
|
+
- **Operating-point metrics at the deployment prevalence.** **PPV/NPV** (and accuracy) move with the
|
|
52
|
+
base rate, so values read off an artificially balanced test set mislead at deployment — report
|
|
53
|
+
them at the real prevalence. Sensitivity and specificity are prevalence-independent but depend on
|
|
54
|
+
the operating threshold, so **fix the threshold on the training/tuning folds, never the test set**
|
|
55
|
+
(choosing it on the test set is tuning-on-test and inflates the estimate).
|
|
56
|
+
- Bootstrap **95% CIs at the patient level**.
|
|
57
|
+
- Gate: `ACCURACY_ONLY` (a bare-accuracy headline under imbalance is flagged).
|
|
58
|
+
|
|
59
|
+
## Detection / localization — FROC or mAP with the IoU criterion stated
|
|
60
|
+
|
|
61
|
+
- Report **FROC** (sensitivity versus mean false positives per image) or **mAP**, and **state the
|
|
62
|
+
match criterion** — a prediction counts as a true positive only if its overlap with a
|
|
63
|
+
ground-truth object meets the stated IoU (or centroid/mask) threshold. Per Metrics Reloaded's
|
|
64
|
+
localization category, the metric is undefined without that criterion.
|
|
65
|
+
- **Patient-level accuracy is not a detection metric**, and a per-lesion result must not be reported
|
|
66
|
+
as per-patient — respect the analysis unit set in Phase 1.
|
|
67
|
+
- Gate: `DETECTION_METRIC_MISSING`.
|
|
68
|
+
|
|
69
|
+
## Calibration — a separate axis from discrimination
|
|
70
|
+
|
|
71
|
+
- A model can **discriminate well (high AUROC) and still be miscalibrated**; report calibration
|
|
72
|
+
separately (Guo et al., *ICML* 2017). Use a **reliability diagram** plus a summary (**ECE** or the
|
|
73
|
+
**Brier** score).
|
|
74
|
+
- **ECE is binning-sensitive** — state the binning scheme (or prefer a binning-robust summary) and
|
|
75
|
+
do not over-read a single ECE value.
|
|
76
|
+
- **TRIPOD+AI** requires reporting **both** discrimination and calibration for a clinical prediction
|
|
77
|
+
model — calibration is not optional reporting.
|
|
78
|
+
|
|
79
|
+
## Subgroup slices — disaggregated reporting
|
|
80
|
+
|
|
81
|
+
- Slice every headline metric by the Model Card **Factors** (scanner/vendor, site, age, sex,
|
|
82
|
+
disease severity), and report the **per-subgroup n** so the reader can see which estimates are
|
|
83
|
+
thin.
|
|
84
|
+
- Need enough events per subgroup to estimate the metric; otherwise **say so** rather than report a
|
|
85
|
+
noisy point estimate.
|
|
86
|
+
- This is disaggregated *reporting*. The formal fairness/equity audit lives in `/model-validation`
|
|
87
|
+
plus the equity probe — cross-reference, do not duplicate it here.
|
|
88
|
+
|
|
89
|
+
## CLAIM 2024 reporting fit — where the eval outputs land
|
|
90
|
+
|
|
91
|
+
CLAIM 2024 organises items under the manuscript sections. The model-evaluation deliverable feeds the
|
|
92
|
+
**Methods** (metric definitions, reference standard, data partition, threshold selection) and the
|
|
93
|
+
**Results** (metrics with uncertainty, calibration, subgroup/failure analysis). `/check-reporting`
|
|
94
|
+
owns the item-by-item CLAIM 2024 / TRIPOD+AI audit; this is the routing map.
|
|
95
|
+
|
|
96
|
+
| Eval output | CLAIM 2024 area (verify item #) | Note |
|
|
97
|
+
|---|---|---|
|
|
98
|
+
| Metric definitions + how each was computed | Methods | Name the metric and its formula/library; no undefined "accuracy" headline. |
|
|
99
|
+
| Reference / ground-truth standard + how derived | Methods | Reader count, blinding, adjudication — state it. |
|
|
100
|
+
| Held-out, patient-level data partition | Methods | Cross-link `/model-validation` (split leakage). |
|
|
101
|
+
| Performance metrics **with uncertainty (CIs)** | Results | Bootstrap CIs at the analysis unit. |
|
|
102
|
+
| Calibration (reliability diagram + ECE/Brier) | Results | Required alongside discrimination (TRIPOD+AI). |
|
|
103
|
+
| Subgroup + failure-case analysis | Results | Per-subgroup n; flag thin slices. |
|
|
104
|
+
| Threshold + operating point at prevalence | Methods / Results | Threshold fixed on tuning folds, reported prevalence. |
|
|
105
|
+
|
|
106
|
+
## What the skill checks / advises
|
|
107
|
+
|
|
108
|
+
1. Run the gate — `PIXEL_ACCURACY_SEG` / `NO_BOUNDARY_METRIC` / `ACCURACY_ONLY` /
|
|
109
|
+
`DETECTION_METRIC_MISSING` must all be zero.
|
|
110
|
+
2. Advise the author to **state τ** (NSD), **state the IoU criterion** (detection), and **state the
|
|
111
|
+
deployment prevalence** (operating-point metrics); report **per-structure / per-subgroup with n**;
|
|
112
|
+
give **patient-level bootstrap CIs**; and report **calibration alongside discrimination**.
|
|
113
|
+
3. Emit `eval/per_case_metrics.csv` for `/analyze-stats` (DeLong / NRI / IDI / decision curves /
|
|
114
|
+
MRMC) — numbers are never hand-typed, and an uncertain metric or CI method is flagged `[VERIFY]`.
|
|
115
|
+
|
|
116
|
+
## Verification notes
|
|
117
|
+
|
|
118
|
+
- **Metrics Reloaded** + pitfalls companion (Maier-Hein, Reinke et al., *Nature Methods* 2024):
|
|
119
|
+
named public standard — grounds the task-fingerprint principle, single-metric-insufficiency,
|
|
120
|
+
per-class reporting, edge-case definition, boundary metrics, and the detection/localization
|
|
121
|
+
category. Cited as a named method, not quoted.
|
|
122
|
+
- **CLAIM 2024** (Tejani et al., *Radiology: Artificial Intelligence* 2024): named reporting
|
|
123
|
+
checklist. Exact item numbers are **not** quoted — the table maps outputs to manuscript sections;
|
|
124
|
+
resolve item numbers against the source (`[VERIFY]`). `/check-reporting` owns the audit.
|
|
125
|
+
- **TRIPOD+AI** (Collins et al., *BMJ* 2024): named standard; grounds the calibration-and-
|
|
126
|
+
discrimination requirement. Written as base TRIPOD + AI extension.
|
|
127
|
+
- **AUPRC under imbalance** (Saito & Rehmsmeier, *PLoS ONE* 2015, **CC-BY**): principle only (ROC vs
|
|
128
|
+
PR on imbalanced data); no text copied.
|
|
129
|
+
- **Calibration / ECE** (Guo et al., *ICML* 2017): named methods paper; reliability diagram + ECE,
|
|
130
|
+
with the binning-sensitivity caveat stated qualitatively.
|
|
131
|
+
- **NSD / surface Dice with tolerance**: the tolerance-based surface metric (e.g., Nikolov et al.,
|
|
132
|
+
head-and-neck OAR segmentation) recommended for boundary error by Metrics Reloaded; **τ described
|
|
133
|
+
qualitatively, no value invented**.
|
|
134
|
+
- **Bootstrap CIs** (Efron & Tibshirani): canonical resampling method; patient-level resampling
|
|
135
|
+
principle.
|
|
136
|
+
- **Model Card Factors** (Mitchell et al., *FAT\** 2019): named documentation standard for
|
|
137
|
+
disaggregated reporting axes.
|
|
138
|
+
- No DOIs, dataset names, numeric thresholds, prevalences, NSD tolerances, or CLAIM item numbers are
|
|
139
|
+
fabricated; any uncertain specific is flagged `[VERIFY]`.
|
|
@@ -62,7 +62,7 @@ P = {
|
|
|
62
62
|
"spec": r"\b(specificit\w+|true[- ]?negative rate|tnr)\b",
|
|
63
63
|
"detection": r"\b(froc|map\b|mean average precision|sensitivity per (?:false positive|fp)|"
|
|
64
64
|
r"competition performance metric|cpm)\b",
|
|
65
|
-
"iou_crit": r"\b(?:iou|intersection over union|overlap)\b[
|
|
65
|
+
"iou_crit": r"\b(?:iou|intersection over union|overlap)\b[^.]{0,40}"
|
|
66
66
|
r"(?:threshold|criterion|>=|≥|>|\bof\b|above|exceed|\d\.\d)"
|
|
67
67
|
r"|match(?:ing)? criterion"
|
|
68
68
|
r"|(?:cent(?:er|re|roid)|distance)[- ]?based"
|
package/skills/model-evaluation/scripts/metric_reporting_challenge/fixture/det_good_wrapped.md
ADDED
|
@@ -0,0 +1,5 @@
|
|
|
1
|
+
# Results (detection)
|
|
2
|
+
Lesion detection was summarised with FROC. A predicted box was scored as a true
|
|
3
|
+
positive when its IoU
|
|
4
|
+
with the reference lesion exceeded 0.3. Sensitivity at 1 false positive per scan
|
|
5
|
+
was 0.84 (95% CI 0.80-0.88), with the operating point fixed on the tuning fold.
|
|
@@ -25,5 +25,10 @@ want seg_bad.md segmentation PIXEL_ACCURACY_SEG
|
|
|
25
25
|
clean seg_good.md segmentation
|
|
26
26
|
want clf_bad.md classification ACCURACY_ONLY
|
|
27
27
|
clean clf_good.md classification
|
|
28
|
+
# Detection branch: a stated IoU match criterion is required; a hard-wrapped criterion
|
|
29
|
+
# (IoU and its threshold on different physical lines) must still be detected — det_good_wrapped
|
|
30
|
+
# locks the iou_crit proximity window against newline-induced false fires.
|
|
31
|
+
want det_no_iou.md detection DETECTION_METRIC_MISSING
|
|
32
|
+
clean det_good_wrapped.md detection
|
|
28
33
|
|
|
29
|
-
echo "PASS: metric-reporting gate flags Dice-only/pixel-accuracy
|
|
34
|
+
echo "PASS: metric-reporting gate flags Dice-only/pixel-accuracy, accuracy-only, and detection without an IoU criterion; clears task-correct reports (including a line-wrapped IoU criterion)."
|
|
@@ -47,6 +47,12 @@ TorchIO — those produce the model, this validates and publishes it.
|
|
|
47
47
|
|
|
48
48
|
## Workflow
|
|
49
49
|
|
|
50
|
+
The design/audit rationale behind Phases 2–7 — the full data-leakage taxonomy, the
|
|
51
|
+
internal-vs-genuine-external validation ladder, comparator design, single-run vs multi-seed
|
|
52
|
+
variance, test-set sizing, and the CLAIM 2024 / TRIPOD+AI / STARD-AI reporting map — is in
|
|
53
|
+
`${CLAUDE_SKILL_DIR}/references/validation_design.md` (load on demand). The patient-disjointness
|
|
54
|
+
verdict itself is proven by `scripts/check_split_leakage.py` (Phase 2), not from that prose.
|
|
55
|
+
|
|
50
56
|
### Phase 1 — Reconstruct the task, the intended-use horizon, and the analysis unit
|
|
51
57
|
State the model's task (segmentation / classification / detection), its **intended-use horizon**
|
|
52
58
|
(screening, triage, pre-procedure, post-hoc), the **single headline metric** the conclusion leans on,
|
|
@@ -0,0 +1,150 @@
|
|
|
1
|
+
# Validation-design reference (model-validation)
|
|
2
|
+
|
|
3
|
+
Load-on-demand backbone for Phases 2–7 — the leakage taxonomy, the internal-vs-external
|
|
4
|
+
tier ladder, comparator design, run variance, test-set sizing, and the reporting map. Anchored
|
|
5
|
+
to **Kapoor & Narayanan** (*Patterns* 2023, leakage taxonomy), **Varoquaux & Cheplygina**
|
|
6
|
+
(*npj Digital Medicine* 2022, medical-imaging ML failure modes), **Metrics Reloaded**
|
|
7
|
+
(Maier-Hein & Reinke et al., *Nature Methods* 2024), **CLAIM 2024**, **TRIPOD+AI** (*BMJ* 2024),
|
|
8
|
+
and **STARD-AI** (*Nature Medicine* 2025). It explains *what to check and advise*; the patient-disjointness
|
|
9
|
+
verdict is proven by `scripts/check_split_leakage.py`, not by this prose.
|
|
10
|
+
|
|
11
|
+
## 1. The data-leakage taxonomy
|
|
12
|
+
|
|
13
|
+
Leakage = any information about a test case that could influence training. Organise the audit by
|
|
14
|
+
the three Kapoor-Narayanan (*Patterns* 2023) categories; the deterministic gate covers only the
|
|
15
|
+
first row.
|
|
16
|
+
|
|
17
|
+
| Category (Kapoor-Narayanan) | Imaging-specific leak | How it inflates the metric | How to catch |
|
|
18
|
+
|---|---|---|---|
|
|
19
|
+
| **No clean train/test separation** | **Patient-level overlap** — the same patient's images straddle train and test | Model memorises patient anatomy, not pathology | `check_split_leakage.py` → `PATIENT_OVERLAP` (set arithmetic on IDs) |
|
|
20
|
+
| ″ | **Near-duplicate / repeated-acquisition** — repeat scans, follow-ups, augmented copies, overlapping patches of one volume across splits | Test cases are not independent of training | Split on the **patient**, not the image/slice/series; dedup by patient before partitioning |
|
|
21
|
+
| ″ | **Preprocessing-before-split** — normalisation stats, intensity windowing, resampling, feature selection, ComBat harmonisation, or foundation-model embeddings fit on the **whole cohort** | Test statistics bleed into the training pipeline | Fit every transform on the **training fold only**; the test set is touched only at scoring time |
|
|
22
|
+
| ″ | **Temporal leakage** — a random split where future and past coexist, under a prognostic/surveillance claim | Model peeks at later-era data | Use a **temporal split** (train on earlier, test on later) when the claim is temporal |
|
|
23
|
+
| **Illegitimate features** | **Site / scanner / burned-in-label shortcut** — the model keys on acquisition site, vendor signature, a laterality token, or a body-part marker rather than the finding | Discrimination collapses off-site | Standalone confound check (shortcut-learning, Geirhos et al. *Nat Mach Intell* 2020; DeGrave et al. *Nat Mach Intell* 2021; Zech et al. *PLOS Med* 2018); subgroup-by-site slice |
|
|
24
|
+
| **Test set ≠ population of interest** | **Spectrum / selection bias** — test cases curated, enriched, or selected on an optional modality | Reported accuracy does not transfer to deployment | Confirm the test set reflects the intended-use population; see §2 / §6 |
|
|
25
|
+
|
|
26
|
+
The decisive question to ask of every preprocessing and selection step: **could any value used in
|
|
27
|
+
training have been computed only with knowledge of a test case?** If yes, it is leakage even when
|
|
28
|
+
the split table itself looks disjoint.
|
|
29
|
+
|
|
30
|
+
### Tuning-on-test (the test set must be touched once)
|
|
31
|
+
Architecture search, hyperparameter sweeps, early-stopping, **and operating-point / threshold
|
|
32
|
+
selection** that read the test set are all forms of the first category — the test set has become a
|
|
33
|
+
development set, and the headline metric is optimistic. Fix the threshold and select the model on
|
|
34
|
+
the **training/validation folds**, then evaluate the frozen model on the test set exactly once.
|
|
35
|
+
"Developed with external validation" where the single external set was also used for tuning is no
|
|
36
|
+
longer external validation (§2).
|
|
37
|
+
|
|
38
|
+
## 2. Internal vs genuine external validation
|
|
39
|
+
|
|
40
|
+
Classify the evidence honestly and let the tier cap the claim. Cross-validation and bootstrap are
|
|
41
|
+
development-time **optimism corrections** (apparent-performance debiasing), **not** external
|
|
42
|
+
validation — this is the long-standing TRIPOD / prediction-model distinction (Collins et al.,
|
|
43
|
+
TRIPOD 2015; TRIPOD+AI, *BMJ* 2024).
|
|
44
|
+
|
|
45
|
+
```
|
|
46
|
+
apparent (train=test, never sufficient)
|
|
47
|
+
→ internal random split
|
|
48
|
+
→ k-fold cross-validation / bootstrap ← still internal (optimism correction)
|
|
49
|
+
→ temporal split (later era held out)
|
|
50
|
+
→ geographic / external (different site, scanner, vendor)
|
|
51
|
+
→ multi-site / prospective external
|
|
52
|
+
```
|
|
53
|
+
|
|
54
|
+
- A **generalisability or deployment-readiness** claim needs at least a genuine external tier
|
|
55
|
+
(different site/scanner/vendor), not an internal split.
|
|
56
|
+
- Single-centre external validation supports a narrower claim than multi-site; say which.
|
|
57
|
+
- Reusing the external set for any tuning demotes it back to internal — flag the contradiction.
|
|
58
|
+
|
|
59
|
+
## 3. Comparator design
|
|
60
|
+
|
|
61
|
+
A standalone metric rarely answers the clinical question; decide what the model is measured
|
|
62
|
+
*against*, evaluated on the **same** test set. CLAIM 2024 and TRIPOD+AI both ask for comparison to
|
|
63
|
+
current practice / an existing model.
|
|
64
|
+
|
|
65
|
+
| Comparator | When | Hand-off |
|
|
66
|
+
|---|---|---|
|
|
67
|
+
| **Clinical / no-model baseline** | "does the model beat current standard of care?" | — |
|
|
68
|
+
| **Incremental value over an existing score** | model added on top of an established risk score / radiologist read | added-value statistics (NRI / IDI / decision curve) → `/analyze-stats` |
|
|
69
|
+
| **Reader comparison (standalone or AI-assisted)** | model vs / with radiologists | rubric, reader panel, inter-rater design → `/design-ai-benchmarking` |
|
|
70
|
+
|
|
71
|
+
Name whether the claim is **standalone** (model alone) or **assistive** (clinician + model); they
|
|
72
|
+
need different comparators and different reporting.
|
|
73
|
+
|
|
74
|
+
## 4. Single-run vs multi-seed variance
|
|
75
|
+
|
|
76
|
+
A single training run overstates precision: deep-model metrics move with the random seed
|
|
77
|
+
(initialisation, data order, augmentation), and some GPU ops are non-deterministic even with cuDNN
|
|
78
|
+
deterministic flags set (Varoquaux & Cheplygina, *npj Digit Med* 2022; reproducibility crisis,
|
|
79
|
+
Kapoor & Narayanan 2023). Require the headline metric as **mean ± SD over ≥ 3 seeds / runs**, or a
|
|
80
|
+
**single fixed reported seed with the determinism caveat stated**. A point estimate from one run,
|
|
81
|
+
presented as if exact, is a reporting defect.
|
|
82
|
+
|
|
83
|
+
## 5. Test-set sizing
|
|
84
|
+
|
|
85
|
+
Check **events per class in the test set**, not the cohort total. A metric computed on a sparse
|
|
86
|
+
positive set has a confidence interval spanning much of the usable range, so a headline AUROC /
|
|
87
|
+
sensitivity can be statistically uninformative even when the cohort is large.
|
|
88
|
+
|
|
89
|
+
- Size the test set for the **CI width** of the headline metric and for **per-subgroup** estimates
|
|
90
|
+
you intend to report.
|
|
91
|
+
- **Calibration** in particular is data-hungry — prediction-model validation guidance uses a rule
|
|
92
|
+
of thumb of roughly ≥ 100 events and ≥ 100 non-events before a reliability assessment is stable
|
|
93
|
+
(treat as a rule of thumb, not a hard cutoff; confirm for the specific design).
|
|
94
|
+
- Hand the formal calculation (diagnostic-accuracy precision, AUC precision, agreement, calibration
|
|
95
|
+
sample size) to `/calc-sample-size`.
|
|
96
|
+
|
|
97
|
+
Metric **selection** (Dice + boundary metric; AUROC + AUPRC under imbalance; FROC/mAP with the IoU
|
|
98
|
+
criterion) is owned by `/model-evaluation` (`references/metric_guide.md`, anchored to Metrics
|
|
99
|
+
Reloaded); this skill only checks that the chosen metric is task- and prevalence-correct.
|
|
100
|
+
|
|
101
|
+
## 6. Reporting-guideline fit
|
|
102
|
+
|
|
103
|
+
Map the study to its standard via `/check-reporting`, and **name both the base instrument and the
|
|
104
|
+
AI extension**, citing each at its actual maturity (published guideline vs protocol-stage), never
|
|
105
|
+
beyond it. The four standards below are cross-checked against the repository's `check-reporting`
|
|
106
|
+
verified checklists.
|
|
107
|
+
|
|
108
|
+
| Study framing | Primary standard (extension) | Base instrument | Risk of bias |
|
|
109
|
+
|---|---|---|---|
|
|
110
|
+
| Diagnostic / triage **imaging-AI** study (standalone or assistive) | **CLAIM 2024** update (Tejani et al., *Radiology: AI* 2024) | CLAIM 2020 (Mongan, Moy, Kahn, *Radiology: AI* 2020) | PROBAST+AI |
|
|
111
|
+
| **Prediction model** (diagnostic or prognostic; regression or ML) | **TRIPOD+AI** (Collins, Moons et al., *BMJ* 2024) | TRIPOD 2015 (Collins, Reitsma, Altman, Moons) | **PROBAST+AI** (Moons et al., *BMJ* 2025; replaces PROBAST-2019) on base PROBAST (Wolff et al., *Ann Intern Med* 2019) |
|
|
112
|
+
| **Diagnostic accuracy** study (index test vs reference standard; sens/spec) | **STARD-AI** (Sounderajah et al., *Nature Medicine* 2025) | STARD 2015 (Bossuyt et al., *BMJ* 2015) | QUADAS-2 / QUADAS-C |
|
|
113
|
+
|
|
114
|
+
Tie the partition, leakage controls, validation tier, comparator, run variance, and test-set sizing
|
|
115
|
+
above to the specific items these standards request (data partition, sample size, model evaluation,
|
|
116
|
+
comparison to current practice, reproducibility).
|
|
117
|
+
|
|
118
|
+
## Hand-offs
|
|
119
|
+
- Patient-disjointness proof → `scripts/check_split_leakage.py` (Phase 2, run first).
|
|
120
|
+
- Test-set / event sizing → `/calc-sample-size`.
|
|
121
|
+
- Reader-comparison rubric + inter-rater design → `/design-ai-benchmarking`.
|
|
122
|
+
- Per-case metric computation + reporting gate → `/model-evaluation` → `/analyze-stats`.
|
|
123
|
+
- Item-by-item compliance → `/check-reporting`; Methods write-up → `/write-paper`; reviewer-side
|
|
124
|
+
audit of the finished draft → `/self-review` (MD0–MD8 `model_development.md` probe).
|
|
125
|
+
|
|
126
|
+
## Verification notes (what each claim is grounded on)
|
|
127
|
+
- **Leakage taxonomy / three categories, reproducibility crisis** — Kapoor & Narayanan, "Leakage and
|
|
128
|
+
the reproducibility crisis in machine-learning-based science," *Patterns* 2023. Imaging-specific
|
|
129
|
+
failure modes (patient-level split, preprocessing-before-split) — Varoquaux & Cheplygina,
|
|
130
|
+
*npj Digital Medicine* 2022. Both already cited in `check_split_leakage.py`.
|
|
131
|
+
- **Site/scanner/shortcut leakage** — shortcut learning, Geirhos et al., *Nature Machine
|
|
132
|
+
Intelligence* 2020; radiographic shortcut example, DeGrave et al., *Nature Machine Intelligence*
|
|
133
|
+
2021; cross-site generalisation failure, Zech et al., *PLOS Medicine* 2018. Used as named
|
|
134
|
+
examples of the "illegitimate features" / spectrum-bias rows, not as numeric claims.
|
|
135
|
+
- **Internal vs external, optimism correction, CV ≠ external** — TRIPOD 2015 (Collins, Reitsma,
|
|
136
|
+
Altman, Moons) and TRIPOD+AI (*BMJ* 2024). The tier ladder mirrors the skill's Phase 3.
|
|
137
|
+
- **Metric selection deferral** — Metrics Reloaded (Maier-Hein & Reinke et al., *Nature Methods*
|
|
138
|
+
2024); detail lives in `/model-evaluation`.
|
|
139
|
+
- **Reporting map** — CLAIM 2024 update (Tejani et al., *Radiology: AI* 2024) on base CLAIM 2020
|
|
140
|
+
(Mongan, Moy, Kahn); TRIPOD+AI (*BMJ* 2024); STARD-AI (Sounderajah et al., *Nature Medicine*
|
|
141
|
+
2025) on base STARD 2015 (Bossuyt et al., *BMJ* 2015); PROBAST+AI (Moons et al., *BMJ* 2025) on
|
|
142
|
+
base PROBAST (Wolff et al., *Ann Intern Med* 2019). All four are cross-checked against this
|
|
143
|
+
repository's `check-reporting` verified checklists (CLAIM 2024 = e240300; TRIPOD+AI = e078378;
|
|
144
|
+
STARD-AI = DOI 10.1038/s41591-025-03953-8, PMID 40954311; PROBAST+AI = e082505).
|
|
145
|
+
- **Numbers deliberately not asserted**: the only quantitative figure is the ~100 events/non-events
|
|
146
|
+
calibration rule of thumb, flagged as a rule of thumb to confirm per design — no dataset names,
|
|
147
|
+
thresholds, or performance numbers are invented here. The reporting-standard DOIs/PMIDs above are
|
|
148
|
+
carried from the repository's verified `check-reporting` checklists; still re-confirm the exact
|
|
149
|
+
identifier via `/search-lit` before quoting any of them in a manuscript, and mark any uncertain
|
|
150
|
+
item `[VERIFY]`.
|