medsci-skills 5.1.0 → 5.2.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -2533,8 +2533,13 @@
2533
2533
  },
2534
2534
  {
2535
2535
  "path": "skills/mllm-eval/SKILL.md",
2536
- "size": 6288,
2537
- "sha256": "ccf3da2f70b356d432b3d33500f667f9d7858500a9ff325631a9a16f5f300a8b"
2536
+ "size": 6720,
2537
+ "sha256": "afdc167bfccb6bad8a4019e0af5bd7c955129220d360511a89962ecd321bc44d"
2538
+ },
2539
+ {
2540
+ "path": "skills/mllm-eval/references/evaluation_axes.md",
2541
+ "size": 10857,
2542
+ "sha256": "49d77ab63feae5dba5cdae7e47d9de6ea97ee3b9eea4b9cba39ec8b0e185be72"
2538
2543
  },
2539
2544
  {
2540
2545
  "path": "skills/mllm-eval/scripts/check_mllm_eval_completeness.py",
@@ -2623,18 +2628,23 @@
2623
2628
  },
2624
2629
  {
2625
2630
  "path": "skills/model-evaluation/SKILL.md",
2626
- "size": 5031,
2627
- "sha256": "17ffde905359e4cffdf747422b7d41c214d2abfcc71e50e1b1e4a689d87fa695"
2631
+ "size": 5780,
2632
+ "sha256": "334b4ca87a2f672446563f3fb6d78c0585386fac12aee29e10dc09465e892193"
2628
2633
  },
2629
2634
  {
2630
2635
  "path": "skills/model-evaluation/references/metric_guide.md",
2631
2636
  "size": 2454,
2632
2637
  "sha256": "8d09ca7ce9fb9f66ee4942689294d9b12ae1d892ac67769cd68fdc38a4e220ee"
2633
2638
  },
2639
+ {
2640
+ "path": "skills/model-evaluation/references/metric_selection_grounding.md",
2641
+ "size": 9588,
2642
+ "sha256": "56723e73b2b74299140d353921d1dba471bedf9aea0b8e769fc69995c4733995"
2643
+ },
2634
2644
  {
2635
2645
  "path": "skills/model-evaluation/scripts/check_metric_reporting.py",
2636
- "size": 9564,
2637
- "sha256": "c33f52ee62ae93417027d0bb0b6cf2a95f5747d52403ab99c075bc64a5e2c593"
2646
+ "size": 9562,
2647
+ "sha256": "fd6c0205f652651a5a43910cb415ed8442100fc553ce04aa5d5905d97743a0bf"
2638
2648
  },
2639
2649
  {
2640
2650
  "path": "skills/model-evaluation/scripts/metric_reporting_challenge/fixture/clf_bad.md",
@@ -2646,6 +2656,16 @@
2646
2656
  "size": 186,
2647
2657
  "sha256": "ba44a3b38b4128fa713555c8b332f211e9087e6bc016d80af4c3074c3ca6ef8e"
2648
2658
  },
2659
+ {
2660
+ "path": "skills/model-evaluation/scripts/metric_reporting_challenge/fixture/det_good_wrapped.md",
2661
+ "size": 285,
2662
+ "sha256": "0c1f5ed5167969870602106e7a7c9a2ae53e3a3ddea5fa280c3beb6bf470c454"
2663
+ },
2664
+ {
2665
+ "path": "skills/model-evaluation/scripts/metric_reporting_challenge/fixture/det_no_iou.md",
2666
+ "size": 224,
2667
+ "sha256": "08227fbe46829b8eecdfe15680e1bb32087f1e6c5a353b3df6055c67d8300fc0"
2668
+ },
2649
2669
  {
2650
2670
  "path": "skills/model-evaluation/scripts/metric_reporting_challenge/fixture/seg_bad.md",
2651
2671
  "size": 119,
@@ -2663,8 +2683,8 @@
2663
2683
  },
2664
2684
  {
2665
2685
  "path": "skills/model-evaluation/scripts/metric_reporting_challenge/verify.sh",
2666
- "size": 1235,
2667
- "sha256": "f819d1333a7206db6383ccbd41c6eadae015004e61ae335e67a3334936c58cc8"
2686
+ "size": 1663,
2687
+ "sha256": "4aa6db49d3552f01a54ddb70484df5d214806d805b836b24beda81d9aca32bd6"
2668
2688
  },
2669
2689
  {
2670
2690
  "path": "skills/model-evaluation/skill.yml",
@@ -2718,8 +2738,13 @@
2718
2738
  },
2719
2739
  {
2720
2740
  "path": "skills/model-validation/SKILL.md",
2721
- "size": 9347,
2722
- "sha256": "ecd48672a03923bf1ace63528fd2dbcf138cd880103cc8c40345b3857d66ad1c"
2741
+ "size": 9810,
2742
+ "sha256": "1a75a5a1f21f5a8778d0b77db2e99574bf37edda2a291d8dde6aafea4a207ff0"
2743
+ },
2744
+ {
2745
+ "path": "skills/model-validation/references/validation_design.md",
2746
+ "size": 11427,
2747
+ "sha256": "16d43b688ea63745174c7ca8fafd78a7342b26c34ad1e10e1fdbc117cceafb2e"
2723
2748
  },
2724
2749
  {
2725
2750
  "path": "skills/model-validation/scripts/check_split_leakage.py",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schema_version": 1,
3
- "version": "5.1.0",
3
+ "version": "5.2.0",
4
4
  "owned_skills": [
5
5
  "academic-aio",
6
6
  "add-journal",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "medsci-skills",
3
- "version": "5.1.0",
3
+ "version": "5.2.0",
4
4
  "description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
5
5
  "license": "SEE LICENSE IN LICENSE",
6
6
  "homepage": "https://github.com/Aperivue/medsci-skills#readme",
@@ -106,3 +106,11 @@ mllm-eval (this skill: harness design + completeness gate, model-agnostic)
106
106
  ├─ write-paper + check-reporting (TRIPOD-LLM / MI-CLEAR-LLM)
107
107
  └─ self-review / peer-review (ME0–ME8 reviewer probe)
108
108
  ```
109
+
110
+ ## Reference Files
111
+
112
+ - `${CLAUDE_SKILL_DIR}/references/evaluation_axes.md` — the *why* behind the ME2–ME7 axes:
113
+ clinical-efficacy metrics beyond n-gram overlap (e.g. RadGraph-F1 / CheXbert-F1 vs BLEU/ROUGE),
114
+ faithfulness & hallucination, pretraining/benchmark contamination, prompt-sensitivity &
115
+ determinism, answer-matching, and the reader study — each mapped to its gate verdict. Load on
116
+ demand during Phases 2–4.
@@ -0,0 +1,161 @@
1
+ # Evaluation axes (mllm-eval)
2
+
3
+ Load-on-demand reference behind the ME2–ME7 axes: the clinical-efficacy metrics,
4
+ faithfulness, contamination, prompt-sensitivity, answer-matching, and reader-study
5
+ machinery an LLM/MLLM clinical evaluation must cover. Anchored to the radiology-NLP
6
+ metric literature — **RadGraph** (Jain et al., NeurIPS Datasets & Benchmarks 2021),
7
+ **CheXbert** (Smit et al., 2020), the rule-based **CheXpert** labeler (Irvin et al., 2019),
8
+ and **RadCliQ** (Yu et al., *Patterns* 2023) — and to the reporting standards CLAIM 2024,
9
+ **TRIPOD-LLM**, and **MI-CLEAR-LLM**. This skill **specifies and routes** these metrics to
10
+ their published extractors and to `/analyze-stats`; it does not run the model or compute the
11
+ scores itself. Never report n-gram overlap as clinical correctness, and never fabricate a
12
+ score.
13
+
14
+ ## Clinical-efficacy metrics beyond n-gram overlap (ME2 → `NGRAM_ONLY`)
15
+
16
+ - **Why n-gram overlap fails.** BLEU / ROUGE / METEOR / CIDEr measure surface lexical
17
+ overlap. A report can score high while inverting laterality or omitting a pneumothorax, and
18
+ low while paraphrasing a correct finding. Yu et al. (*Patterns* 2023, RadCliQ) showed these
19
+ metrics correlate weakly with radiologist-assessed clinical error — so an n-gram score is a
20
+ **fluency proxy, not a correctness claim**.
21
+ - **RadGraph-F1.** Overlap of the (entity, relation) tuples the RadGraph schema (Jain et al.,
22
+ 2021) extracts from the reference and the candidate report — it rewards getting the *same
23
+ findings and their relationships*, not the same words.
24
+ - **CheXbert-F1 / CheXpert labeler.** Agreement on the structured CheXpert observation labels
25
+ extracted by CheXbert (Smit et al., 2020) or by the rule-based CheXpert labeler (Irvin et
26
+ al., 2019) — a finding-label–level correctness signal.
27
+ - **RadCliQ.** A composite (Yu et al., 2023) that combines metrics to better predict the count
28
+ of radiologist-judged errors. Report it as a composite **alongside** its components, not as a
29
+ one-number replacement.
30
+ - **Advise:** report a clinical-efficacy metric (RadGraph-F1 / CheXbert-F1 / RadCliQ)
31
+ **with bootstrap CIs over reports** and a **per-finding / per-label breakdown**, and present
32
+ any BLEU/ROUGE explicitly labelled as a surface-overlap measure — never as the headline.
33
+
34
+ ## Faithfulness & hallucination (ME3 → `FAITHFULNESS_MISSING`)
35
+
36
+ - **Fluency is not faithfulness.** A high-overlap, well-formed report can still assert findings
37
+ the image does not support. Measure faithfulness directly; do not infer it from an accuracy
38
+ number.
39
+ - **Atomic-fact decomposition.** Break the generated text into atomic clinical claims and check
40
+ each against the image / source, then report a **faithfulness (or hallucination) rate** — the
41
+ fraction of generated claims that are grounded.
42
+ - **Direction matters.** Separate **omission** (a true finding the model missed) from
43
+ **fabrication** (a false finding the model asserted); they carry different clinical risk and
44
+ should be reported separately, not folded into one error count.
45
+ - **False-premise / abstention probe.** Ask about an absent finding or an unanswerable
46
+ question; a faithful model abstains rather than confabulates. Named instruments: **MedVH**,
47
+ **Med-HALT**.
48
+ - **Advise:** a generation or VQA claim with no faithfulness and no false-premise/abstention
49
+ evaluation is the central MLLM gap — require both, with rates, before any clinical claim.
50
+
51
+ ## Pretraining / benchmark contamination (ME4 → `CONTAMINATION_UNADDRESSED`)
52
+
53
+ - **Why public benchmarks are suspect.** VQA-RAD, SLAKE, MIMIC-CXR–derived sets, MedQA,
54
+ PMC-VQA, PathVQA, OpenI, and PubMedQA may sit inside the model's pretraining corpus, so a
55
+ high score can be **memorisation, not capability**. For a closed API the corpus is undisclosed,
56
+ so contamination cannot be excluded — only **bounded** and stated.
57
+ - **Three accepted checks (any one, stated explicitly):**
58
+ - **Cutoff vs release date** — compare the model's training cutoff against the benchmark's
59
+ release date; a benchmark that predates the cutoff is at risk.
60
+ - **Held-out / post-cutoff set** — evaluate on a private, institution-collected, or
61
+ after-cutoff set the model could not have seen.
62
+ - **Contamination probe** — canary strings, a perturbed-duplicate performance gap (score on
63
+ verbatim items vs lightly perturbed copies), or a membership/quiz test (ask the model to
64
+ reproduce held-out items).
65
+ - **Advise:** never write "no contamination" (or evaluate on a pre-cutoff public benchmark in
66
+ silence) without one of the checks above; an acknowledged-but-unmitigated risk is a stated
67
+ limitation, not a clean result.
68
+
69
+ ## Prompt-sensitivity & determinism (ME5 → `PROMPT_PROVENANCE_MISSING`)
70
+
71
+ - **Outputs move with the prompt and the sampler.** Phrasing, format, the system prompt,
72
+ temperature, top-p/top-k, and run-to-run sampling all shift results. A single-prompt
73
+ single-run number overstates stability.
74
+ - **Closed APIs are non-deterministic even at temperature 0** — identical inputs can yield
75
+ different outputs across calls. Treat any single-run figure as a point estimate of a
76
+ distribution, not a fixed value.
77
+ - **Disclose (MI-CLEAR-LLM transparency):** the **exact prompt(s)** including the system prompt,
78
+ the **decoding settings** (temperature / top-p / seed), **≥ 3 runs** with reported variance
79
+ (e.g., mean ± SD), and a **prompt-robustness** check across **≥ 2 phrasings/formats** for the
80
+ headline result.
81
+
82
+ ## Answer-matching for VQA / classification (ME6 → `ANSWER_MATCHING_MISSING`)
83
+
84
+ - **State the matching rule.** Free-text answers must be mapped to the key by a declared rule:
85
+ **exact** string match, **normalised** match (case/punctuation/synonym folding), or
86
+ **LLM-as-judge**. An unspecified rule makes the accuracy unreproducible.
87
+ - **An LLM judge is itself a model.** If a model adjudicates correctness, **validate the judge
88
+ against a human-labelled subset** and report its agreement; route judge validation to
89
+ `/design-ai-benchmarking`. An unvalidated LLM judge can launder the system's own errors.
90
+ - **Operating discipline.** Report accuracy **at the real clinical prevalence**, not on an
91
+ artificially balanced QA set, and state **how refusals/abstentions are scored** (counted
92
+ wrong, excluded, or credited) — the choice can move the headline.
93
+
94
+ ## Reader study for generated reports (ME7 → `READER_STUDY_MISSING`)
95
+
96
+ - **Automated metrics do not establish clinical acceptability.** Even RadGraph-F1 / CheXbert-F1
97
+ measure agreement, not whether a clinician would act on the report safely. A deployment or
98
+ utility claim for generated text needs a **blinded clinical reader study**.
99
+ - **Design elements:** a pre-defined **error taxonomy** (clinically significant vs insignificant;
100
+ omission vs fabrication), a **severity scale**, and **inter-reader agreement**.
101
+ - **Route:** the rubric and the IRR design → `/design-ai-benchmarking`; the ICC/κ computation →
102
+ `/analyze-stats`; reader and case **sizing** → `/calc-sample-size`.
103
+
104
+ ## Gate mapping
105
+
106
+ The deterministic gate (`scripts/check_mllm_eval_completeness.py`) is a presence check on the
107
+ plan text, task-aware. This reference is the *why* behind each verdict:
108
+
109
+ | Axis (this doc) | Gate verdict | Severity |
110
+ |---|---|---|
111
+ | n-gram only, no clinical metric (report-gen) | `NGRAM_ONLY` | Major |
112
+ | no adjudicated reference standard (report-gen) | `REFERENCE_STANDARD_MISSING` | Major |
113
+ | no faithfulness / false-premise (report-gen, vqa) | `FAITHFULNESS_MISSING` | Major |
114
+ | public benchmark, no contamination handling | `CONTAMINATION_UNADDRESSED` | Major |
115
+ | no blinded reader study (report-gen) | `READER_STUDY_MISSING` | Major (deploy) / Minor |
116
+ | prompt / decoding / multi-run incomplete | `PROMPT_PROVENANCE_MISSING` | Minor |
117
+ | no answer-matching rule (vqa, classification) | `ANSWER_MATCHING_MISSING` | Minor |
118
+
119
+ A Major verdict is a presence gap, not proof the work is wrong — resolve it by adding the axis
120
+ to the plan (or recording, with a stated reason, why it does not apply).
121
+
122
+ ## Reporting fit & hand-off
123
+
124
+ Methods / Results stub → `/write-paper`. Item-level compliance with **TRIPOD-LLM**,
125
+ **MI-CLEAR-LLM**, **CLAIM 2024** (and STARD-AI / TRIPOD+AI where a diagnostic/prognostic claim
126
+ is made) → `/check-reporting`. Reviewer-side audit of a finished manuscript uses the
127
+ `mllm_evaluation.md` (ME0–ME8) probe via `/self-review` and `/peer-review`.
128
+
129
+ ## Verification notes
130
+
131
+ Each claim here is grounded in a named public method/standard or described qualitatively; no
132
+ numbers, thresholds, or dataset contents are invented.
133
+
134
+ - **n-gram metrics correlate weakly with clinical error; clinical-efficacy metrics needed** —
135
+ Yu et al., *Patterns* 2023 (RadCliQ). Named public methods paper (matches the citation already
136
+ vendored in the `mllm_evaluation.md` probe).
137
+ - **RadGraph-F1 (entity-relation overlap)** — Jain et al., NeurIPS Datasets & Benchmarks 2021.
138
+ Named public methods paper.
139
+ - **CheXbert-F1 / CheXpert observation labels** — Smit et al., 2020 (CheXbert); Irvin et al.,
140
+ 2019 (CheXpert labeler). Named public methods papers; the CheXpert label set is a factual
141
+ artifact, not invented.
142
+ - **RadCliQ as a composite** — Yu et al., *Patterns* 2023. Named public methods paper.
143
+ - **Atomic-fact faithfulness, omission-vs-fabrication, false-premise/abstention** — described as
144
+ established evaluation principles; **MedVH** and **Med-HALT** named as instruments only (as in
145
+ the probe), no scores invented.
146
+ - **Contamination of public benchmarks; closed-corpus unknowability; cutoff/held-out/probe
147
+ checks (canary, perturbed-duplicate gap, membership test)** — stated as accepted principles and
148
+ practices, qualitatively; benchmark **names** (VQA-RAD, SLAKE, MIMIC-CXR, MedQA, PMC-VQA,
149
+ PathVQA, OpenI, PubMedQA) are factual public-dataset names, no contents reproduced.
150
+ - **Closed-API non-determinism even at temperature 0; prompt/format/sampling sensitivity** —
151
+ described qualitatively as documented behavior; no figure attached.
152
+ - **Prompt + decoding + ≥3 runs + ≥2 phrasings disclosure** — MI-CLEAR-LLM transparency
153
+ (named standard); the ≥3 / ≥2 conventions are this skill's own house thresholds (carried from
154
+ SKILL.md / the ME-probe), not literature values.
155
+ - **LLM-as-judge must be validated against human labels** — stated as a methodological principle;
156
+ judge validation routed to `/design-ai-benchmarking`.
157
+ - **Reader study for deployment/utility claims; error taxonomy, severity, IRR** — consistent with
158
+ CLAIM 2024 / TRIPOD-LLM reporting expectations (named standards); sizing/IRR routed to
159
+ `/calc-sample-size` and `/analyze-stats`.
160
+ - **Metrics Reloaded / CLAIM 2024 / TRIPOD-LLM / MI-CLEAR-LLM / Model Cards (Mitchell 2019) /
161
+ Datasheets (Gebru 2021)** — named public standards, cited by name only.
@@ -79,6 +79,18 @@ figures → `/make-figures`; the numbers + subgroup performance → `/model-card
79
79
  `scripts/check_metric_reporting.py` — flags a task-metric mismatch / missing uncertainty (stdlib,
80
80
  network-free). Reproducible challenge: `bash ${CLAUDE_SKILL_DIR}/scripts/metric_reporting_challenge/verify.sh`.
81
81
 
82
+ ## Reference Files
83
+
84
+ Load on demand (keep SKILL.md short):
85
+ - `${CLAUDE_SKILL_DIR}/references/metric_guide.md` — operational checklist: the task-correct metric
86
+ per task (segmentation Dice + HD95/NSD per structure; classification AUROC + AUPRC + sens/spec at
87
+ deployment prevalence; detection FROC/mAP with a stated IoU), plus calibration, subgroup slices,
88
+ run-variance, and the per-case CSV hand-off.
89
+ - `${CLAUDE_SKILL_DIR}/references/metric_selection_grounding.md` — the standards grounding behind
90
+ those choices: the Metrics Reloaded task-fingerprint principle, why each metric pairing is
91
+ required, calibration vs discrimination, disaggregated reporting, and the CLAIM 2024
92
+ reporting-fit map (`/check-reporting` owns the item audit).
93
+
82
94
  ## Boundaries
83
95
 
84
96
  ```
@@ -0,0 +1,139 @@
1
+ # Metric-selection grounding and CLAIM 2024 reporting fit (model-evaluation)
2
+
3
+ The *why* behind the operational checklist in `metric_guide.md`. Where `metric_guide.md` says
4
+ **what** to compute, this doc grounds **why** that pairing is required and where the outputs land
5
+ in the manuscript. Anchored to **Metrics Reloaded** (Maier-Hein & Reinke et al., *Nature Methods*
6
+ 2024) and its pitfalls companion (Reinke et al., *Nature Methods* 2024), **CLAIM 2024** (Tejani et
7
+ al., *Radiology: Artificial Intelligence* 2024), **TRIPOD+AI** (the AI extension of TRIPOD; Collins
8
+ et al., *BMJ* 2024), calibration work (Guo et al., *ICML* 2017), and the Model Card **Factors**
9
+ (Mitchell et al., *FAT\** 2019). The deliverable is still the per-case CSV; the deterministic gate
10
+ is `scripts/check_metric_reporting.py`.
11
+
12
+ > Verify exact **CLAIM 2024 item numbers** and any **NSD tolerance** against the source before
13
+ > quoting them as formal values — the mapping and tolerances below are described qualitatively by
14
+ > design. Do not hand-type a metric value: every number comes from executed code.
15
+
16
+ ## The Metrics Reloaded principle: task fingerprint → metric
17
+
18
+ - **The metric is derived from the problem, not from habit.** Metrics Reloaded selects metrics from
19
+ the problem fingerprint — task category (classification / segmentation / detection-localization),
20
+ structure size and shape, class prevalence, and whether *where* the model is right matters. A
21
+ metric that ignores a property that matters clinically is the wrong metric.
22
+ - **No single metric is sufficient.** Pair a counting/overlap metric with a complementary one (a
23
+ boundary metric, a calibration summary) so a blind spot in one is covered by the other. A lone
24
+ headline number is the recurring pitfall the companion paper warns about.
25
+ - **Report per-class / per-structure with a distribution**, not only a global mean — a mean hides
26
+ minority-class and small-structure failure.
27
+ - **Define edge-case behaviour explicitly** (empty reference, no positive cases): several metrics
28
+ are undefined there, and a silent convention changes the score.
29
+
30
+ ## Segmentation — overlap **and** boundary, per structure
31
+
32
+ - Report an overlap metric (Dice or IoU) **with** a boundary metric (**HD95** or **NSD**). Dice is
33
+ a volume-overlap measure: it is insensitive to boundary error and unstable on small or thin
34
+ structures, so it can look high while the contour is clinically wrong.
35
+ - **HD95** = 95th-percentile Hausdorff distance — robust to a few outlier surface points relative to
36
+ the raw maximum Hausdorff. **NSD** (normalised surface distance / surface Dice) = the fraction of
37
+ the predicted surface within a **task-specific tolerance τ** of the reference surface; **τ must be
38
+ stated** and is chosen from clinical acceptability, not invented.
39
+ - Compute **per structure** with bootstrap 95% CIs obtained by resampling **patients, not pixels**
40
+ (Efron–Tibshirani bootstrap). State the rule for **empty-reference / false-positive-only** cases
41
+ (a Dice of 0/0 is undefined).
42
+ - Gate: `PIXEL_ACCURACY_SEG` (pixel/voxel accuracy is dominated by background — never the headline)
43
+ and `NO_BOUNDARY_METRIC`.
44
+
45
+ ## Classification — discrimination, operating point at prevalence, then calibration
46
+
47
+ - **AUROC and AUPRC.** AUROC summarises ranking across thresholds; under class imbalance the
48
+ precision–recall view (AUPRC) is more informative, because the ROC's false-positive rate uses the
49
+ large negative denominator and can look optimistic when negatives dominate (Saito & Rehmsmeier,
50
+ *PLoS ONE* 2015).
51
+ - **Operating-point metrics at the deployment prevalence.** **PPV/NPV** (and accuracy) move with the
52
+ base rate, so values read off an artificially balanced test set mislead at deployment — report
53
+ them at the real prevalence. Sensitivity and specificity are prevalence-independent but depend on
54
+ the operating threshold, so **fix the threshold on the training/tuning folds, never the test set**
55
+ (choosing it on the test set is tuning-on-test and inflates the estimate).
56
+ - Bootstrap **95% CIs at the patient level**.
57
+ - Gate: `ACCURACY_ONLY` (a bare-accuracy headline under imbalance is flagged).
58
+
59
+ ## Detection / localization — FROC or mAP with the IoU criterion stated
60
+
61
+ - Report **FROC** (sensitivity versus mean false positives per image) or **mAP**, and **state the
62
+ match criterion** — a prediction counts as a true positive only if its overlap with a
63
+ ground-truth object meets the stated IoU (or centroid/mask) threshold. Per Metrics Reloaded's
64
+ localization category, the metric is undefined without that criterion.
65
+ - **Patient-level accuracy is not a detection metric**, and a per-lesion result must not be reported
66
+ as per-patient — respect the analysis unit set in Phase 1.
67
+ - Gate: `DETECTION_METRIC_MISSING`.
68
+
69
+ ## Calibration — a separate axis from discrimination
70
+
71
+ - A model can **discriminate well (high AUROC) and still be miscalibrated**; report calibration
72
+ separately (Guo et al., *ICML* 2017). Use a **reliability diagram** plus a summary (**ECE** or the
73
+ **Brier** score).
74
+ - **ECE is binning-sensitive** — state the binning scheme (or prefer a binning-robust summary) and
75
+ do not over-read a single ECE value.
76
+ - **TRIPOD+AI** requires reporting **both** discrimination and calibration for a clinical prediction
77
+ model — calibration is not optional reporting.
78
+
79
+ ## Subgroup slices — disaggregated reporting
80
+
81
+ - Slice every headline metric by the Model Card **Factors** (scanner/vendor, site, age, sex,
82
+ disease severity), and report the **per-subgroup n** so the reader can see which estimates are
83
+ thin.
84
+ - Need enough events per subgroup to estimate the metric; otherwise **say so** rather than report a
85
+ noisy point estimate.
86
+ - This is disaggregated *reporting*. The formal fairness/equity audit lives in `/model-validation`
87
+ plus the equity probe — cross-reference, do not duplicate it here.
88
+
89
+ ## CLAIM 2024 reporting fit — where the eval outputs land
90
+
91
+ CLAIM 2024 organises items under the manuscript sections. The model-evaluation deliverable feeds the
92
+ **Methods** (metric definitions, reference standard, data partition, threshold selection) and the
93
+ **Results** (metrics with uncertainty, calibration, subgroup/failure analysis). `/check-reporting`
94
+ owns the item-by-item CLAIM 2024 / TRIPOD+AI audit; this is the routing map.
95
+
96
+ | Eval output | CLAIM 2024 area (verify item #) | Note |
97
+ |---|---|---|
98
+ | Metric definitions + how each was computed | Methods | Name the metric and its formula/library; no undefined "accuracy" headline. |
99
+ | Reference / ground-truth standard + how derived | Methods | Reader count, blinding, adjudication — state it. |
100
+ | Held-out, patient-level data partition | Methods | Cross-link `/model-validation` (split leakage). |
101
+ | Performance metrics **with uncertainty (CIs)** | Results | Bootstrap CIs at the analysis unit. |
102
+ | Calibration (reliability diagram + ECE/Brier) | Results | Required alongside discrimination (TRIPOD+AI). |
103
+ | Subgroup + failure-case analysis | Results | Per-subgroup n; flag thin slices. |
104
+ | Threshold + operating point at prevalence | Methods / Results | Threshold fixed on tuning folds, reported prevalence. |
105
+
106
+ ## What the skill checks / advises
107
+
108
+ 1. Run the gate — `PIXEL_ACCURACY_SEG` / `NO_BOUNDARY_METRIC` / `ACCURACY_ONLY` /
109
+ `DETECTION_METRIC_MISSING` must all be zero.
110
+ 2. Advise the author to **state τ** (NSD), **state the IoU criterion** (detection), and **state the
111
+ deployment prevalence** (operating-point metrics); report **per-structure / per-subgroup with n**;
112
+ give **patient-level bootstrap CIs**; and report **calibration alongside discrimination**.
113
+ 3. Emit `eval/per_case_metrics.csv` for `/analyze-stats` (DeLong / NRI / IDI / decision curves /
114
+ MRMC) — numbers are never hand-typed, and an uncertain metric or CI method is flagged `[VERIFY]`.
115
+
116
+ ## Verification notes
117
+
118
+ - **Metrics Reloaded** + pitfalls companion (Maier-Hein, Reinke et al., *Nature Methods* 2024):
119
+ named public standard — grounds the task-fingerprint principle, single-metric-insufficiency,
120
+ per-class reporting, edge-case definition, boundary metrics, and the detection/localization
121
+ category. Cited as a named method, not quoted.
122
+ - **CLAIM 2024** (Tejani et al., *Radiology: Artificial Intelligence* 2024): named reporting
123
+ checklist. Exact item numbers are **not** quoted — the table maps outputs to manuscript sections;
124
+ resolve item numbers against the source (`[VERIFY]`). `/check-reporting` owns the audit.
125
+ - **TRIPOD+AI** (Collins et al., *BMJ* 2024): named standard; grounds the calibration-and-
126
+ discrimination requirement. Written as base TRIPOD + AI extension.
127
+ - **AUPRC under imbalance** (Saito & Rehmsmeier, *PLoS ONE* 2015, **CC-BY**): principle only (ROC vs
128
+ PR on imbalanced data); no text copied.
129
+ - **Calibration / ECE** (Guo et al., *ICML* 2017): named methods paper; reliability diagram + ECE,
130
+ with the binning-sensitivity caveat stated qualitatively.
131
+ - **NSD / surface Dice with tolerance**: the tolerance-based surface metric (e.g., Nikolov et al.,
132
+ head-and-neck OAR segmentation) recommended for boundary error by Metrics Reloaded; **τ described
133
+ qualitatively, no value invented**.
134
+ - **Bootstrap CIs** (Efron & Tibshirani): canonical resampling method; patient-level resampling
135
+ principle.
136
+ - **Model Card Factors** (Mitchell et al., *FAT\** 2019): named documentation standard for
137
+ disaggregated reporting axes.
138
+ - No DOIs, dataset names, numeric thresholds, prevalences, NSD tolerances, or CLAIM item numbers are
139
+ fabricated; any uncertain specific is flagged `[VERIFY]`.
@@ -62,7 +62,7 @@ P = {
62
62
  "spec": r"\b(specificit\w+|true[- ]?negative rate|tnr)\b",
63
63
  "detection": r"\b(froc|map\b|mean average precision|sensitivity per (?:false positive|fp)|"
64
64
  r"competition performance metric|cpm)\b",
65
- "iou_crit": r"\b(?:iou|intersection over union|overlap)\b[^.\n]{0,40}"
65
+ "iou_crit": r"\b(?:iou|intersection over union|overlap)\b[^.]{0,40}"
66
66
  r"(?:threshold|criterion|>=|≥|>|\bof\b|above|exceed|\d\.\d)"
67
67
  r"|match(?:ing)? criterion"
68
68
  r"|(?:cent(?:er|re|roid)|distance)[- ]?based"
@@ -0,0 +1,5 @@
1
+ # Results (detection)
2
+ Lesion detection was summarised with FROC. A predicted box was scored as a true
3
+ positive when its IoU
4
+ with the reference lesion exceeded 0.3. Sensitivity at 1 false positive per scan
5
+ was 0.84 (95% CI 0.80-0.88), with the operating point fixed on the tuning fold.
@@ -0,0 +1,4 @@
1
+ # Results (detection)
2
+ Lesion detection performance was summarised with FROC and mean average precision.
3
+ Sensitivity at 1 false positive per scan was 0.84 (95% CI 0.80-0.88). The operating
4
+ point was fixed on the tuning fold.
@@ -25,5 +25,10 @@ want seg_bad.md segmentation PIXEL_ACCURACY_SEG
25
25
  clean seg_good.md segmentation
26
26
  want clf_bad.md classification ACCURACY_ONLY
27
27
  clean clf_good.md classification
28
+ # Detection branch: a stated IoU match criterion is required; a hard-wrapped criterion
29
+ # (IoU and its threshold on different physical lines) must still be detected — det_good_wrapped
30
+ # locks the iou_crit proximity window against newline-induced false fires.
31
+ want det_no_iou.md detection DETECTION_METRIC_MISSING
32
+ clean det_good_wrapped.md detection
28
33
 
29
- echo "PASS: metric-reporting gate flags Dice-only/pixel-accuracy and accuracy-only, clears task-correct reports."
34
+ echo "PASS: metric-reporting gate flags Dice-only/pixel-accuracy, accuracy-only, and detection without an IoU criterion; clears task-correct reports (including a line-wrapped IoU criterion)."
@@ -47,6 +47,12 @@ TorchIO — those produce the model, this validates and publishes it.
47
47
 
48
48
  ## Workflow
49
49
 
50
+ The design/audit rationale behind Phases 2–7 — the full data-leakage taxonomy, the
51
+ internal-vs-genuine-external validation ladder, comparator design, single-run vs multi-seed
52
+ variance, test-set sizing, and the CLAIM 2024 / TRIPOD+AI / STARD-AI reporting map — is in
53
+ `${CLAUDE_SKILL_DIR}/references/validation_design.md` (load on demand). The patient-disjointness
54
+ verdict itself is proven by `scripts/check_split_leakage.py` (Phase 2), not from that prose.
55
+
50
56
  ### Phase 1 — Reconstruct the task, the intended-use horizon, and the analysis unit
51
57
  State the model's task (segmentation / classification / detection), its **intended-use horizon**
52
58
  (screening, triage, pre-procedure, post-hoc), the **single headline metric** the conclusion leans on,
@@ -0,0 +1,150 @@
1
+ # Validation-design reference (model-validation)
2
+
3
+ Load-on-demand backbone for Phases 2–7 — the leakage taxonomy, the internal-vs-external
4
+ tier ladder, comparator design, run variance, test-set sizing, and the reporting map. Anchored
5
+ to **Kapoor & Narayanan** (*Patterns* 2023, leakage taxonomy), **Varoquaux & Cheplygina**
6
+ (*npj Digital Medicine* 2022, medical-imaging ML failure modes), **Metrics Reloaded**
7
+ (Maier-Hein & Reinke et al., *Nature Methods* 2024), **CLAIM 2024**, **TRIPOD+AI** (*BMJ* 2024),
8
+ and **STARD-AI** (*Nature Medicine* 2025). It explains *what to check and advise*; the patient-disjointness
9
+ verdict is proven by `scripts/check_split_leakage.py`, not by this prose.
10
+
11
+ ## 1. The data-leakage taxonomy
12
+
13
+ Leakage = any information about a test case that could influence training. Organise the audit by
14
+ the three Kapoor-Narayanan (*Patterns* 2023) categories; the deterministic gate covers only the
15
+ first row.
16
+
17
+ | Category (Kapoor-Narayanan) | Imaging-specific leak | How it inflates the metric | How to catch |
18
+ |---|---|---|---|
19
+ | **No clean train/test separation** | **Patient-level overlap** — the same patient's images straddle train and test | Model memorises patient anatomy, not pathology | `check_split_leakage.py` → `PATIENT_OVERLAP` (set arithmetic on IDs) |
20
+ | ″ | **Near-duplicate / repeated-acquisition** — repeat scans, follow-ups, augmented copies, overlapping patches of one volume across splits | Test cases are not independent of training | Split on the **patient**, not the image/slice/series; dedup by patient before partitioning |
21
+ | ″ | **Preprocessing-before-split** — normalisation stats, intensity windowing, resampling, feature selection, ComBat harmonisation, or foundation-model embeddings fit on the **whole cohort** | Test statistics bleed into the training pipeline | Fit every transform on the **training fold only**; the test set is touched only at scoring time |
22
+ | ″ | **Temporal leakage** — a random split where future and past coexist, under a prognostic/surveillance claim | Model peeks at later-era data | Use a **temporal split** (train on earlier, test on later) when the claim is temporal |
23
+ | **Illegitimate features** | **Site / scanner / burned-in-label shortcut** — the model keys on acquisition site, vendor signature, a laterality token, or a body-part marker rather than the finding | Discrimination collapses off-site | Standalone confound check (shortcut-learning, Geirhos et al. *Nat Mach Intell* 2020; DeGrave et al. *Nat Mach Intell* 2021; Zech et al. *PLOS Med* 2018); subgroup-by-site slice |
24
+ | **Test set ≠ population of interest** | **Spectrum / selection bias** — test cases curated, enriched, or selected on an optional modality | Reported accuracy does not transfer to deployment | Confirm the test set reflects the intended-use population; see §2 / §6 |
25
+
26
+ The decisive question to ask of every preprocessing and selection step: **could any value used in
27
+ training have been computed only with knowledge of a test case?** If yes, it is leakage even when
28
+ the split table itself looks disjoint.
29
+
30
+ ### Tuning-on-test (the test set must be touched once)
31
+ Architecture search, hyperparameter sweeps, early-stopping, **and operating-point / threshold
32
+ selection** that read the test set are all forms of the first category — the test set has become a
33
+ development set, and the headline metric is optimistic. Fix the threshold and select the model on
34
+ the **training/validation folds**, then evaluate the frozen model on the test set exactly once.
35
+ "Developed with external validation" where the single external set was also used for tuning is no
36
+ longer external validation (§2).
37
+
38
+ ## 2. Internal vs genuine external validation
39
+
40
+ Classify the evidence honestly and let the tier cap the claim. Cross-validation and bootstrap are
41
+ development-time **optimism corrections** (apparent-performance debiasing), **not** external
42
+ validation — this is the long-standing TRIPOD / prediction-model distinction (Collins et al.,
43
+ TRIPOD 2015; TRIPOD+AI, *BMJ* 2024).
44
+
45
+ ```
46
+ apparent (train=test, never sufficient)
47
+ → internal random split
48
+ → k-fold cross-validation / bootstrap ← still internal (optimism correction)
49
+ → temporal split (later era held out)
50
+ → geographic / external (different site, scanner, vendor)
51
+ → multi-site / prospective external
52
+ ```
53
+
54
+ - A **generalisability or deployment-readiness** claim needs at least a genuine external tier
55
+ (different site/scanner/vendor), not an internal split.
56
+ - Single-centre external validation supports a narrower claim than multi-site; say which.
57
+ - Reusing the external set for any tuning demotes it back to internal — flag the contradiction.
58
+
59
+ ## 3. Comparator design
60
+
61
+ A standalone metric rarely answers the clinical question; decide what the model is measured
62
+ *against*, evaluated on the **same** test set. CLAIM 2024 and TRIPOD+AI both ask for comparison to
63
+ current practice / an existing model.
64
+
65
+ | Comparator | When | Hand-off |
66
+ |---|---|---|
67
+ | **Clinical / no-model baseline** | "does the model beat current standard of care?" | — |
68
+ | **Incremental value over an existing score** | model added on top of an established risk score / radiologist read | added-value statistics (NRI / IDI / decision curve) → `/analyze-stats` |
69
+ | **Reader comparison (standalone or AI-assisted)** | model vs / with radiologists | rubric, reader panel, inter-rater design → `/design-ai-benchmarking` |
70
+
71
+ Name whether the claim is **standalone** (model alone) or **assistive** (clinician + model); they
72
+ need different comparators and different reporting.
73
+
74
+ ## 4. Single-run vs multi-seed variance
75
+
76
+ A single training run overstates precision: deep-model metrics move with the random seed
77
+ (initialisation, data order, augmentation), and some GPU ops are non-deterministic even with cuDNN
78
+ deterministic flags set (Varoquaux & Cheplygina, *npj Digit Med* 2022; reproducibility crisis,
79
+ Kapoor & Narayanan 2023). Require the headline metric as **mean ± SD over ≥ 3 seeds / runs**, or a
80
+ **single fixed reported seed with the determinism caveat stated**. A point estimate from one run,
81
+ presented as if exact, is a reporting defect.
82
+
83
+ ## 5. Test-set sizing
84
+
85
+ Check **events per class in the test set**, not the cohort total. A metric computed on a sparse
86
+ positive set has a confidence interval spanning much of the usable range, so a headline AUROC /
87
+ sensitivity can be statistically uninformative even when the cohort is large.
88
+
89
+ - Size the test set for the **CI width** of the headline metric and for **per-subgroup** estimates
90
+ you intend to report.
91
+ - **Calibration** in particular is data-hungry — prediction-model validation guidance uses a rule
92
+ of thumb of roughly ≥ 100 events and ≥ 100 non-events before a reliability assessment is stable
93
+ (treat as a rule of thumb, not a hard cutoff; confirm for the specific design).
94
+ - Hand the formal calculation (diagnostic-accuracy precision, AUC precision, agreement, calibration
95
+ sample size) to `/calc-sample-size`.
96
+
97
+ Metric **selection** (Dice + boundary metric; AUROC + AUPRC under imbalance; FROC/mAP with the IoU
98
+ criterion) is owned by `/model-evaluation` (`references/metric_guide.md`, anchored to Metrics
99
+ Reloaded); this skill only checks that the chosen metric is task- and prevalence-correct.
100
+
101
+ ## 6. Reporting-guideline fit
102
+
103
+ Map the study to its standard via `/check-reporting`, and **name both the base instrument and the
104
+ AI extension**, citing each at its actual maturity (published guideline vs protocol-stage), never
105
+ beyond it. The four standards below are cross-checked against the repository's `check-reporting`
106
+ verified checklists.
107
+
108
+ | Study framing | Primary standard (extension) | Base instrument | Risk of bias |
109
+ |---|---|---|---|
110
+ | Diagnostic / triage **imaging-AI** study (standalone or assistive) | **CLAIM 2024** update (Tejani et al., *Radiology: AI* 2024) | CLAIM 2020 (Mongan, Moy, Kahn, *Radiology: AI* 2020) | PROBAST+AI |
111
+ | **Prediction model** (diagnostic or prognostic; regression or ML) | **TRIPOD+AI** (Collins, Moons et al., *BMJ* 2024) | TRIPOD 2015 (Collins, Reitsma, Altman, Moons) | **PROBAST+AI** (Moons et al., *BMJ* 2025; replaces PROBAST-2019) on base PROBAST (Wolff et al., *Ann Intern Med* 2019) |
112
+ | **Diagnostic accuracy** study (index test vs reference standard; sens/spec) | **STARD-AI** (Sounderajah et al., *Nature Medicine* 2025) | STARD 2015 (Bossuyt et al., *BMJ* 2015) | QUADAS-2 / QUADAS-C |
113
+
114
+ Tie the partition, leakage controls, validation tier, comparator, run variance, and test-set sizing
115
+ above to the specific items these standards request (data partition, sample size, model evaluation,
116
+ comparison to current practice, reproducibility).
117
+
118
+ ## Hand-offs
119
+ - Patient-disjointness proof → `scripts/check_split_leakage.py` (Phase 2, run first).
120
+ - Test-set / event sizing → `/calc-sample-size`.
121
+ - Reader-comparison rubric + inter-rater design → `/design-ai-benchmarking`.
122
+ - Per-case metric computation + reporting gate → `/model-evaluation` → `/analyze-stats`.
123
+ - Item-by-item compliance → `/check-reporting`; Methods write-up → `/write-paper`; reviewer-side
124
+ audit of the finished draft → `/self-review` (MD0–MD8 `model_development.md` probe).
125
+
126
+ ## Verification notes (what each claim is grounded on)
127
+ - **Leakage taxonomy / three categories, reproducibility crisis** — Kapoor & Narayanan, "Leakage and
128
+ the reproducibility crisis in machine-learning-based science," *Patterns* 2023. Imaging-specific
129
+ failure modes (patient-level split, preprocessing-before-split) — Varoquaux & Cheplygina,
130
+ *npj Digital Medicine* 2022. Both already cited in `check_split_leakage.py`.
131
+ - **Site/scanner/shortcut leakage** — shortcut learning, Geirhos et al., *Nature Machine
132
+ Intelligence* 2020; radiographic shortcut example, DeGrave et al., *Nature Machine Intelligence*
133
+ 2021; cross-site generalisation failure, Zech et al., *PLOS Medicine* 2018. Used as named
134
+ examples of the "illegitimate features" / spectrum-bias rows, not as numeric claims.
135
+ - **Internal vs external, optimism correction, CV ≠ external** — TRIPOD 2015 (Collins, Reitsma,
136
+ Altman, Moons) and TRIPOD+AI (*BMJ* 2024). The tier ladder mirrors the skill's Phase 3.
137
+ - **Metric selection deferral** — Metrics Reloaded (Maier-Hein & Reinke et al., *Nature Methods*
138
+ 2024); detail lives in `/model-evaluation`.
139
+ - **Reporting map** — CLAIM 2024 update (Tejani et al., *Radiology: AI* 2024) on base CLAIM 2020
140
+ (Mongan, Moy, Kahn); TRIPOD+AI (*BMJ* 2024); STARD-AI (Sounderajah et al., *Nature Medicine*
141
+ 2025) on base STARD 2015 (Bossuyt et al., *BMJ* 2015); PROBAST+AI (Moons et al., *BMJ* 2025) on
142
+ base PROBAST (Wolff et al., *Ann Intern Med* 2019). All four are cross-checked against this
143
+ repository's `check-reporting` verified checklists (CLAIM 2024 = e240300; TRIPOD+AI = e078378;
144
+ STARD-AI = DOI 10.1038/s41591-025-03953-8, PMID 40954311; PROBAST+AI = e082505).
145
+ - **Numbers deliberately not asserted**: the only quantitative figure is the ~100 events/non-events
146
+ calibration rule of thumb, flagged as a rule of thumb to confirm per design — no dataset names,
147
+ thresholds, or performance numbers are invented here. The reporting-standard DOIs/PMIDs above are
148
+ carried from the repository's verified `check-reporting` checklists; still re-confirm the exact
149
+ identifier via `/search-lit` before quoting any of them in a manuscript, and mark any uncertain
150
+ item `[VERIFY]`.