medsci-skills 5.12.0 → 5.14.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/metadata/distribution_files.json +48 -18
- package/metadata/distribution_manifest.json +1 -1
- package/package.json +1 -1
- package/skills/analyze-stats/SKILL.md +4 -1
- package/skills/analyze-stats/references/analysis_guides/calibration.md +132 -0
- package/skills/analyze-stats/references/analysis_guides/diagnostic_accuracy.md +184 -0
- package/skills/analyze-stats/references/analysis_guides/survival.md +149 -0
- package/skills/peer-review/references/domain-probes/diagnostic_accuracy.md +1 -0
- package/skills/peer-review/references/domain-probes/survival_prognostic.md +2 -0
- package/skills/self-review/references/domain-probes/diagnostic_accuracy.md +1 -0
- package/skills/self-review/references/domain-probes/survival_prognostic.md +2 -0
- package/skills/write-paper/SKILL.md +3 -3
- package/skills/write-paper/references/exemplar_discussion/README.md +2 -0
- package/skills/write-paper/references/exemplar_discussion/rct_consort.md +42 -0
- package/skills/write-paper/references/exemplar_methods/README.md +1 -0
- package/skills/write-paper/references/exemplar_methods/rct_consort.md +55 -0
- package/skills/write-paper/references/exemplar_results/README.md +2 -0
- package/skills/write-paper/references/exemplar_results/rct_consort.md +38 -0
|
@@ -153,14 +153,24 @@
|
|
|
153
153
|
},
|
|
154
154
|
{
|
|
155
155
|
"path": "skills/analyze-stats/SKILL.md",
|
|
156
|
-
"size":
|
|
157
|
-
"sha256": "
|
|
156
|
+
"size": 56948,
|
|
157
|
+
"sha256": "e7f01164e2fc9db924e49ee226b13bcf78d4d7c0463099c0f666985152a91d2e"
|
|
158
158
|
},
|
|
159
159
|
{
|
|
160
160
|
"path": "skills/analyze-stats/references/analysis_guides/agreement_reliability.md",
|
|
161
161
|
"size": 6674,
|
|
162
162
|
"sha256": "599ad2551f547043b2bec79e3139d9c2c3ae33c5025784359b22022f882504c2"
|
|
163
163
|
},
|
|
164
|
+
{
|
|
165
|
+
"path": "skills/analyze-stats/references/analysis_guides/calibration.md",
|
|
166
|
+
"size": 6470,
|
|
167
|
+
"sha256": "ac33862c6c707997be5a2511ed6a6d546c1e829fc9fc033d1c85eca9afa3ca9a"
|
|
168
|
+
},
|
|
169
|
+
{
|
|
170
|
+
"path": "skills/analyze-stats/references/analysis_guides/diagnostic_accuracy.md",
|
|
171
|
+
"size": 9303,
|
|
172
|
+
"sha256": "e788d612aa4227500e997ec405ab15a300d28c3b6fb2a1e1df33823764cb9537"
|
|
173
|
+
},
|
|
164
174
|
{
|
|
165
175
|
"path": "skills/analyze-stats/references/analysis_guides/health_economic_evaluation.md",
|
|
166
176
|
"size": 5362,
|
|
@@ -221,6 +231,11 @@
|
|
|
221
231
|
"size": 12803,
|
|
222
232
|
"sha256": "1639f4810fb5946b24e782b3ce463c000222dcac75709e1e3dac78ad4a2ad72f"
|
|
223
233
|
},
|
|
234
|
+
{
|
|
235
|
+
"path": "skills/analyze-stats/references/analysis_guides/survival.md",
|
|
236
|
+
"size": 7191,
|
|
237
|
+
"sha256": "6b7f220f2aa165fd16eda58b0b40ac743c122c03757bc00f61efc884b203f337"
|
|
238
|
+
},
|
|
224
239
|
{
|
|
225
240
|
"path": "skills/analyze-stats/references/analysis_guides/test_selection.md",
|
|
226
241
|
"size": 3439,
|
|
@@ -2953,8 +2968,8 @@
|
|
|
2953
2968
|
},
|
|
2954
2969
|
{
|
|
2955
2970
|
"path": "skills/peer-review/references/domain-probes/diagnostic_accuracy.md",
|
|
2956
|
-
"size":
|
|
2957
|
-
"sha256": "
|
|
2971
|
+
"size": 11967,
|
|
2972
|
+
"sha256": "8c0a8bd30fe16507e2997ddba1bbcfb4ee7eade7df44f1981e1427cd3a28d233"
|
|
2958
2973
|
},
|
|
2959
2974
|
{
|
|
2960
2975
|
"path": "skills/peer-review/references/domain-probes/equity_fairness.md",
|
|
@@ -3043,8 +3058,8 @@
|
|
|
3043
3058
|
},
|
|
3044
3059
|
{
|
|
3045
3060
|
"path": "skills/peer-review/references/domain-probes/survival_prognostic.md",
|
|
3046
|
-
"size":
|
|
3047
|
-
"sha256": "
|
|
3061
|
+
"size": 14297,
|
|
3062
|
+
"sha256": "32cd929ff2e151f33bc5b106eddef0d38d5bd15672054e690c603ee82a22a817"
|
|
3048
3063
|
},
|
|
3049
3064
|
{
|
|
3050
3065
|
"path": "skills/peer-review/references/exemplar_reviews/README.md",
|
|
@@ -3513,8 +3528,8 @@
|
|
|
3513
3528
|
},
|
|
3514
3529
|
{
|
|
3515
3530
|
"path": "skills/self-review/references/domain-probes/diagnostic_accuracy.md",
|
|
3516
|
-
"size":
|
|
3517
|
-
"sha256": "
|
|
3531
|
+
"size": 11967,
|
|
3532
|
+
"sha256": "8c0a8bd30fe16507e2997ddba1bbcfb4ee7eade7df44f1981e1427cd3a28d233"
|
|
3518
3533
|
},
|
|
3519
3534
|
{
|
|
3520
3535
|
"path": "skills/self-review/references/domain-probes/equity_fairness.md",
|
|
@@ -3603,8 +3618,8 @@
|
|
|
3603
3618
|
},
|
|
3604
3619
|
{
|
|
3605
3620
|
"path": "skills/self-review/references/domain-probes/survival_prognostic.md",
|
|
3606
|
-
"size":
|
|
3607
|
-
"sha256": "
|
|
3621
|
+
"size": 14297,
|
|
3622
|
+
"sha256": "32cd929ff2e151f33bc5b106eddef0d38d5bd15672054e690c603ee82a22a817"
|
|
3608
3623
|
},
|
|
3609
3624
|
{
|
|
3610
3625
|
"path": "skills/self-review/references/exemplar_findings/README.md",
|
|
@@ -3908,8 +3923,8 @@
|
|
|
3908
3923
|
},
|
|
3909
3924
|
{
|
|
3910
3925
|
"path": "skills/write-paper/SKILL.md",
|
|
3911
|
-
"size":
|
|
3912
|
-
"sha256": "
|
|
3926
|
+
"size": 67054,
|
|
3927
|
+
"sha256": "62b3cbff56a139ed20a783c462fb6dde6b7a932bb057762a2d1c6382c8d044c1"
|
|
3913
3928
|
},
|
|
3914
3929
|
{
|
|
3915
3930
|
"path": "skills/write-paper/references/exemplar_abstract.md",
|
|
@@ -3928,8 +3943,8 @@
|
|
|
3928
3943
|
},
|
|
3929
3944
|
{
|
|
3930
3945
|
"path": "skills/write-paper/references/exemplar_discussion/README.md",
|
|
3931
|
-
"size":
|
|
3932
|
-
"sha256": "
|
|
3946
|
+
"size": 2902,
|
|
3947
|
+
"sha256": "1db58e8d3801a7fd2bd6a6bb9c271f22436a71e82ba034a8c888e9468945a026"
|
|
3933
3948
|
},
|
|
3934
3949
|
{
|
|
3935
3950
|
"path": "skills/write-paper/references/exemplar_discussion/ai_validation_tripod_claim.md",
|
|
@@ -3951,6 +3966,11 @@
|
|
|
3951
3966
|
"size": 3652,
|
|
3952
3967
|
"sha256": "21ec3cbfcb626cc00f09e66aa27d8f2f835a202ed083bcebcba57b787ddff866"
|
|
3953
3968
|
},
|
|
3969
|
+
{
|
|
3970
|
+
"path": "skills/write-paper/references/exemplar_discussion/rct_consort.md",
|
|
3971
|
+
"size": 2613,
|
|
3972
|
+
"sha256": "2fa6e9f621219974f07294e8637da83d923327207e38e1692b62bb45c7d20ab5"
|
|
3973
|
+
},
|
|
3954
3974
|
{
|
|
3955
3975
|
"path": "skills/write-paper/references/exemplar_introduction.md",
|
|
3956
3976
|
"size": 2636,
|
|
@@ -3958,8 +3978,8 @@
|
|
|
3958
3978
|
},
|
|
3959
3979
|
{
|
|
3960
3980
|
"path": "skills/write-paper/references/exemplar_methods/README.md",
|
|
3961
|
-
"size":
|
|
3962
|
-
"sha256": "
|
|
3981
|
+
"size": 2426,
|
|
3982
|
+
"sha256": "b55addd809841fdb8ae8e834693c7a03602cb1d8aeb00131ed86ecb8fbc713f6"
|
|
3963
3983
|
},
|
|
3964
3984
|
{
|
|
3965
3985
|
"path": "skills/write-paper/references/exemplar_methods/ai_validation_tripod_claim.md",
|
|
@@ -3981,10 +4001,15 @@
|
|
|
3981
4001
|
"size": 2364,
|
|
3982
4002
|
"sha256": "e2ea3f64be321d2fb8249d4d528cfd24231c0b69d2562e4290cf3dcddd29d419"
|
|
3983
4003
|
},
|
|
4004
|
+
{
|
|
4005
|
+
"path": "skills/write-paper/references/exemplar_methods/rct_consort.md",
|
|
4006
|
+
"size": 3268,
|
|
4007
|
+
"sha256": "742e343b163336a6dda2fa2054e0f486df489e1d6cb2cf3eb782e05d25d9710d"
|
|
4008
|
+
},
|
|
3984
4009
|
{
|
|
3985
4010
|
"path": "skills/write-paper/references/exemplar_results/README.md",
|
|
3986
|
-
"size":
|
|
3987
|
-
"sha256": "
|
|
4011
|
+
"size": 2989,
|
|
4012
|
+
"sha256": "b67ec682f899a1774af10913eaeb8f33956859da2ef787250c01ef3b437f1155"
|
|
3988
4013
|
},
|
|
3989
4014
|
{
|
|
3990
4015
|
"path": "skills/write-paper/references/exemplar_results/ai_validation_tripod_claim.md",
|
|
@@ -4006,6 +4031,11 @@
|
|
|
4006
4031
|
"size": 2452,
|
|
4007
4032
|
"sha256": "4c8e1fc5c5d00a3c0dc26ea0d94dc80e3077a96503ada653ccdcee8eeb8b14da"
|
|
4008
4033
|
},
|
|
4034
|
+
{
|
|
4035
|
+
"path": "skills/write-paper/references/exemplar_results/rct_consort.md",
|
|
4036
|
+
"size": 2356,
|
|
4037
|
+
"sha256": "6a0becfe8f24db7f686bd23f26c9caec749dd996914705679ea915398fc7c3dd"
|
|
4038
|
+
},
|
|
4009
4039
|
{
|
|
4010
4040
|
"path": "skills/write-paper/references/journal_profiles/AJNR.md",
|
|
4011
4041
|
"size": 6381,
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "medsci-skills",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.14.0",
|
|
4
4
|
"description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE",
|
|
6
6
|
"homepage": "https://github.com/Aperivue/medsci-skills#readme",
|
|
@@ -393,6 +393,7 @@ tbl %>% as_flex_table() %>% flextable::save_as_docx(path = "table.docx")
|
|
|
393
393
|
|
|
394
394
|
### Diagnostic Accuracy
|
|
395
395
|
|
|
396
|
+
- **Methodology guide**: `references/analysis_guides/diagnostic_accuracy.md` (**load before generating code** — every metric with a CI on a stated analysis unit; the confidence-weighted trap [unweighted-baseline AUC + monotonic-encoding check, produce-side of probe D9]; paired DeLong vs MRMC for reader-generalising claims; per-stratum admissibility [D10]; one-scale-per-comparison [D11])
|
|
396
397
|
- Template: `references/templates/diagnostic_accuracy.py`
|
|
397
398
|
- Always report: sensitivity, specificity, PPV, NPV, accuracy, AUC
|
|
398
399
|
- CIs: Wilson score for proportions, DeLong for AUC
|
|
@@ -487,6 +488,7 @@ tbl %>% as_flex_table() %>% flextable::save_as_docx(path = "table.docx")
|
|
|
487
488
|
|
|
488
489
|
### Survival Analysis
|
|
489
490
|
|
|
491
|
+
- **Methodology guide**: `references/analysis_guides/survival.md` (**load before generating code** — competing risks first [naive 1−KM overestimates → produce the Aalen–Johansen/Fine–Gray CIF; cause-specific vs subdistribution for which question, produce-side of probe S3]; PH check → RMST when violated; reverse-KM follow-up + C-index variant [S6]; estimand provenance [S8])
|
|
490
492
|
- Table type guide: `references/table-standards/table-types/survival_results.md` (Cox results table: events/person-time, reverse-KM median follow-up, univariable + adjusted HR with CI, PH-assumption footnote, EPV/sparse-stratum and RMST-when-PH-violated rules)
|
|
491
493
|
- Kaplan-Meier curves with number-at-risk table
|
|
492
494
|
- Log-rank test for group comparison
|
|
@@ -543,7 +545,8 @@ When death or other events preclude the outcome of interest, standard KM overest
|
|
|
543
545
|
- **Guide**: Load `analysis_guides/regression.md` before generating code
|
|
544
546
|
- **Template**: `references/templates/regression.py` (set `regression_type = "logistic"`)
|
|
545
547
|
- Run univariable analysis first, then multivariable with clinically selected variables
|
|
546
|
-
- Required outputs: OR table (univariable + multivariable), C-statistic (95% CI), Hosmer
|
|
548
|
+
- Required outputs: OR table (univariable + multivariable), C-statistic (95% CI), and **calibration** (intercept + slope + flexible plot — **not** Hosmer–Lemeshow, which is deprecated; see the calibration guide)
|
|
549
|
+
- **Prediction-model calibration guide**: `references/analysis_guides/calibration.md` (**load before generating code** for any model that outputs a risk used for a decision — the apparent slope of exactly 1.00 is the in-sample tell, so produce the **bootstrap optimism-corrected** slope/intercept; Van Calster's calibration levels; scaled Brier; why Hosmer–Lemeshow is dropped; produce-side of probe S7)
|
|
547
550
|
- Check VIF < 5, EPV >= 10 (warn if violated)
|
|
548
551
|
- **Nested observation units**: when rows are clustered within subjects (multiple lesions/visits per patient), use cluster-robust standard errors (`cov_type="cluster"`, `cov_kwds={"groups": id}` in statsmodels) or a mixed-effects logistic model — a naive logit CI assumes independent rows and is too narrow
|
|
549
552
|
- Box-Tidwell test for continuous predictor linearity
|
|
@@ -0,0 +1,132 @@
|
|
|
1
|
+
# Prediction-Model Calibration Guide
|
|
2
|
+
|
|
3
|
+
For a clinical prediction model (a logistic risk score or a Cox/survival model at a fixed
|
|
4
|
+
horizon), **discrimination (AUC / C-index) is not enough** — a model that ranks well can still
|
|
5
|
+
output probabilities that are systematically too high or too low. The ways calibration fails
|
|
6
|
+
review are (1) reporting **apparent (in-sample)** calibration — a slope of exactly 1.00 is the
|
|
7
|
+
fingerprint — with no internal-validation correction, (2) leaning on the **Hosmer–Lemeshow**
|
|
8
|
+
test (deprecated), and (3) omitting calibration entirely for a model meant to guide care. This
|
|
9
|
+
guide produces the corrected estimand; it is the produce-side of probe **S7**.
|
|
10
|
+
|
|
11
|
+
---
|
|
12
|
+
|
|
13
|
+
## When to Use
|
|
14
|
+
|
|
15
|
+
- Any model that outputs a **risk / probability** used for a decision (surveillance intensity,
|
|
16
|
+
treatment eligibility, triage) — logistic or survival-at-a-horizon.
|
|
17
|
+
- Reported **alongside** discrimination and, for a utility claim, decision-curve net benefit —
|
|
18
|
+
never discrimination alone.
|
|
19
|
+
- NOT a substitute for external validation: internal (bootstrap/CV) calibration corrects
|
|
20
|
+
optimism but does not establish transportability.
|
|
21
|
+
|
|
22
|
+
---
|
|
23
|
+
|
|
24
|
+
## Apparent calibration is optimistic — correct it (this is the S7 issue)
|
|
25
|
+
|
|
26
|
+
Fit a logistic model by maximum likelihood and its **apparent** calibration slope on the same
|
|
27
|
+
data is **exactly 1.00** and its calibration-in-the-large intercept **exactly 0** — by
|
|
28
|
+
construction, not because the model is well-calibrated. A slope printed as `1.00` (or metrics
|
|
29
|
+
with no `bootstrap` / `cross-valid` / `optimism` / `held-out` token nearby) is presumptively
|
|
30
|
+
in-sample. Produce the **bootstrap optimism-corrected** slope instead (Harrell/Steyerberg).
|
|
31
|
+
|
|
32
|
+
```python
|
|
33
|
+
import numpy as np, statsmodels.api as sm
|
|
34
|
+
X = ... # design matrix (add_constant), y = 0/1 outcome
|
|
35
|
+
def cal_slope(y, lp):
|
|
36
|
+
m = sm.GLM(y, sm.add_constant(lp), family=sm.families.Binomial()).fit()
|
|
37
|
+
return m.params[1], m.params[0] # slope, calibration-in-the-large intercept
|
|
38
|
+
|
|
39
|
+
full = sm.GLM(y, X, family=sm.families.Binomial()).fit()
|
|
40
|
+
app_slope, app_int = cal_slope(y, X @ full.params) # apparent: slope ~1.00, intercept ~0
|
|
41
|
+
|
|
42
|
+
rng = np.random.default_rng(42); n = len(y); opt = []
|
|
43
|
+
for _ in range(500): # bootstrap optimism (Harrell)
|
|
44
|
+
idx = rng.integers(0, n, n)
|
|
45
|
+
bm = sm.GLM(y[idx], X[idx], family=sm.families.Binomial()).fit()
|
|
46
|
+
s_boot, _ = cal_slope(y[idx], X[idx] @ bm.params) # boot model on boot data (apparent)
|
|
47
|
+
s_orig, _ = cal_slope(y, X @ bm.params) # boot model on original data (test)
|
|
48
|
+
opt.append(s_boot - s_orig)
|
|
49
|
+
corrected_slope = app_slope - float(np.mean(opt)) # < 1.00 when the model overfits
|
|
50
|
+
print(f"apparent slope {app_slope:.3f} -> optimism-corrected {corrected_slope:.3f}")
|
|
51
|
+
```
|
|
52
|
+
|
|
53
|
+
A corrected slope **< 1** means predictions are too extreme (overfit) and should be shrunk
|
|
54
|
+
(a penalized/uniform-shrinkage refit); a slope **> 1** means they are too moderate.
|
|
55
|
+
|
|
56
|
+
---
|
|
57
|
+
|
|
58
|
+
## The four levels of calibration (report weak calibration at least)
|
|
59
|
+
|
|
60
|
+
Van Calster's hierarchy — report at minimum **weak calibration** (intercept + slope):
|
|
61
|
+
|
|
62
|
+
- **Mean** (calibration-in-the-large): mean predicted = observed event rate (the intercept).
|
|
63
|
+
- **Weak**: intercept ≈ 0 **and** slope ≈ 1.
|
|
64
|
+
- **Moderate**: a **flexible calibration curve** (loess / spline of observed on predicted),
|
|
65
|
+
not decile bins — the plot most reviewers now expect.
|
|
66
|
+
- **Strong**: correct per-covariate (rarely achievable; not required).
|
|
67
|
+
|
|
68
|
+
```python
|
|
69
|
+
import matplotlib.pyplot as plt
|
|
70
|
+
from sklearn.calibration import calibration_curve
|
|
71
|
+
phat = full.predict(X)
|
|
72
|
+
frac_pos, mean_pred = calibration_curve(y, phat, n_bins=10, strategy="quantile")
|
|
73
|
+
plt.plot([0, 1], [0, 1], "--"); plt.plot(mean_pred, frac_pos, "o-") # add a loess curve for moderate
|
|
74
|
+
plt.xlabel("Predicted probability"); plt.ylabel("Observed frequency")
|
|
75
|
+
```
|
|
76
|
+
|
|
77
|
+
---
|
|
78
|
+
|
|
79
|
+
## Brier score; do NOT rely on Hosmer–Lemeshow
|
|
80
|
+
|
|
81
|
+
- **Brier score** = mean squared error of the probabilities; report the **scaled Brier**
|
|
82
|
+
(1 − Brier / Brier_null) so it is interpretable against the event rate.
|
|
83
|
+
- **Hosmer–Lemeshow is deprecated** (Van Calster 2016; Austin & Steyerberg): its p-value depends
|
|
84
|
+
on an arbitrary number of bins, it is underpowered in small samples and rejects trivially in
|
|
85
|
+
large ones, and it gives no direction or magnitude. Report the **calibration slope + intercept
|
|
86
|
+
+ a flexible calibration plot** instead of an H–L p-value.
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
from sklearn.metrics import brier_score_loss
|
|
90
|
+
brier = brier_score_loss(y, phat); scaled = 1 - brier / (y.mean() * (1 - y.mean()))
|
|
91
|
+
```
|
|
92
|
+
|
|
93
|
+
## Survival models
|
|
94
|
+
|
|
95
|
+
For a Cox/survival model, calibrate the **predicted vs observed risk at a fixed horizon**
|
|
96
|
+
(e.g. 3-year): group by predicted-risk decile and compare to a Kaplan–Meier / pseudo-value
|
|
97
|
+
estimate at that time, or use `rms::calibrate` / `pec` in R with bootstrap optimism correction.
|
|
98
|
+
State the horizon; a model can be well-calibrated at 1 year and not at 5.
|
|
99
|
+
|
|
100
|
+
---
|
|
101
|
+
|
|
102
|
+
## Reporting
|
|
103
|
+
|
|
104
|
+
- Calibration **intercept and slope** with the **internal-validation method named**
|
|
105
|
+
(bootstrap/CV), not the apparent slope of 1.00; the flexible calibration plot.
|
|
106
|
+
- Scaled Brier; the horizon (survival); the cohort each metric was computed on (development vs
|
|
107
|
+
held-out vs external) stated explicitly.
|
|
108
|
+
- Discrimination **and** calibration together; add decision-curve net benefit for a utility claim
|
|
109
|
+
(see `table-standards/table-types/incremental_value.md` and the `make-figures` decision-curve
|
|
110
|
+
exemplar).
|
|
111
|
+
|
|
112
|
+
---
|
|
113
|
+
|
|
114
|
+
## Common failures (flag at review)
|
|
115
|
+
|
|
116
|
+
- **Apparent calibration slope of exactly 1.00** (and intercept 0) with no bootstrap/CV/external
|
|
117
|
+
token — in-sample fit presented as calibration (S7).
|
|
118
|
+
- **Hosmer–Lemeshow p-value** offered as the calibration evidence (deprecated).
|
|
119
|
+
- **Discrimination reported without calibration** for a model meant to guide care (S7 → MAJOR).
|
|
120
|
+
- **Decile-bin calibration only**, no flexible curve; or a survival calibration with no stated
|
|
121
|
+
horizon.
|
|
122
|
+
|
|
123
|
+
---
|
|
124
|
+
|
|
125
|
+
## Anti-Hallucination
|
|
126
|
+
|
|
127
|
+
- Never hand-type a calibration slope/intercept, Brier, or CI — compute it from predictions with
|
|
128
|
+
a seeded script.
|
|
129
|
+
- Do not report a calibration slope of 1.00 as evidence of good calibration — it is the
|
|
130
|
+
apparent-fit artifact; report the optimism-corrected value the bootstrap produced.
|
|
131
|
+
- Name the validation source of every calibration number (development / bootstrap-corrected /
|
|
132
|
+
external); do not present development-sample calibration as validated performance.
|
|
@@ -0,0 +1,184 @@
|
|
|
1
|
+
# Diagnostic Accuracy & Reader-Study Guide
|
|
2
|
+
|
|
3
|
+
Estimating how well an index test (or an AI model / reader) separates disease from
|
|
4
|
+
no-disease against a reference standard. The point estimates are easy to compute; the
|
|
5
|
+
ways these analyses fail review are (1) reporting AUC / sensitivity / specificity **without
|
|
6
|
+
CIs or on the wrong analysis unit**, (2) a **confidence-weighted / rating** score whose
|
|
7
|
+
novelty is never tested against the simpler **unweighted** baseline, and (3) comparing two
|
|
8
|
+
AUCs with a **fixed-reader** test when the claim is meant to **generalise to readers**.
|
|
9
|
+
|
|
10
|
+
---
|
|
11
|
+
|
|
12
|
+
## When to Use
|
|
13
|
+
|
|
14
|
+
- **Single index test** vs a reference standard → sensitivity, specificity, PPV, NPV, accuracy
|
|
15
|
+
(each with a 95% CI), and AUC (with a DeLong or bootstrap CI).
|
|
16
|
+
- **Two tests / models on the same cases** → a **paired** AUC comparison (DeLong `roc.test`,
|
|
17
|
+
`paired = TRUE`) — never two independent CIs eyeballed for overlap.
|
|
18
|
+
- **Multi-reader multi-case (MRMC)** reader study (AI-vs-reader, modality comparison) → an
|
|
19
|
+
MRMC method (Obuchowski–Rockette / DBM) that carries **reader + case** variance; see
|
|
20
|
+
`table-standards/table-types/reader_study.md` and `make-figures` `exemplar_plots/mrmc_roc.md`.
|
|
21
|
+
- NOT for: pooling accuracy across studies (that is a DTA meta-analysis — bivariate / HSROC via
|
|
22
|
+
`mada`; see the `/meta-analysis` DTA path); not for rater **agreement** (that is
|
|
23
|
+
`analysis_guides/agreement_reliability.md`).
|
|
24
|
+
|
|
25
|
+
---
|
|
26
|
+
|
|
27
|
+
## Every metric with a CI, on a stated analysis unit (the floor)
|
|
28
|
+
|
|
29
|
+
A bare AUC / sensitivity / specificity is not reportable. Compute the CI, and state the unit
|
|
30
|
+
(**per-patient vs per-lesion** — multiple lesions per patient are clustered, exactly as in
|
|
31
|
+
`agreement_reliability.md`; a per-lesion count inflates *n*).
|
|
32
|
+
|
|
33
|
+
```python
|
|
34
|
+
import numpy as np, pandas as pd
|
|
35
|
+
from sklearn.metrics import roc_auc_score
|
|
36
|
+
from statsmodels.stats.proportion import proportion_confint
|
|
37
|
+
|
|
38
|
+
df = pd.read_csv("reader_calls.csv") # truth (0/1), call (0/1), confidence (1..K), stratum
|
|
39
|
+
|
|
40
|
+
# sensitivity / specificity / PPV / NPV at the operating point, each with a Wilson CI
|
|
41
|
+
tp = int(((df.truth == 1) & (df.call == 1)).sum()); fn = int(((df.truth == 1) & (df.call == 0)).sum())
|
|
42
|
+
tn = int(((df.truth == 0) & (df.call == 0)).sum()); fp = int(((df.truth == 0) & (df.call == 1)).sum())
|
|
43
|
+
for name, num, den in [("sensitivity", tp, tp+fn), ("specificity", tn, tn+fp),
|
|
44
|
+
("PPV", tp, tp+fp), ("NPV", tn, tn+fn)]:
|
|
45
|
+
lo, hi = proportion_confint(num, den, method="wilson")
|
|
46
|
+
print(f"{name} = {num/den:.3f} (95% CI {lo:.3f}-{hi:.3f}), n={den}")
|
|
47
|
+
```
|
|
48
|
+
|
|
49
|
+
PPV and NPV are **prevalence-dependent** — do not transport them to a population with a
|
|
50
|
+
different base rate; report sensitivity/specificity (prevalence-invariant) plus the study
|
|
51
|
+
prevalence, and recompute predictive values at the target prevalence with Bayes' rule.
|
|
52
|
+
|
|
53
|
+
---
|
|
54
|
+
|
|
55
|
+
## The confidence-weighted trap comes first (this, not the AUC, is the issue)
|
|
56
|
+
|
|
57
|
+
When the novelty is a **confidence-weighted / rating-collapsed** score used as the ROC
|
|
58
|
+
predictor, you must (a) confirm the (call × confidence) encoding is **strictly monotone** — a
|
|
59
|
+
folded encoding silently collides the most-confident-positive with a negative call — and
|
|
60
|
+
(b) report the **unweighted binary-call AUC** beside the weighted one, so the weighting earns
|
|
61
|
+
its place. This is the produce-side of self-review / peer-review probe **D9**.
|
|
62
|
+
|
|
63
|
+
```python
|
|
64
|
+
K = int(df.confidence.max())
|
|
65
|
+
|
|
66
|
+
# intended strictly-monotone composite: negative calls rank below positive calls, and within
|
|
67
|
+
# a call higher confidence pushes further from the decision boundary.
|
|
68
|
+
def composite(call, conf, K):
|
|
69
|
+
return np.where(call == 1, K + conf, K + 1 - conf) # neg: 1..K, pos: K+1..2K
|
|
70
|
+
|
|
71
|
+
# (1) MONOTONIC-ENCODING CHECK — every distinct (call, confidence) must map to a distinct,
|
|
72
|
+
# correctly-ordered score. A collision is the folded-score bug (e.g. real/1 == ai/1).
|
|
73
|
+
combos = df[["call", "confidence"]].drop_duplicates().copy()
|
|
74
|
+
combos["score"] = composite(combos.call.values, combos.confidence.values, K)
|
|
75
|
+
if combos.score.nunique() != len(combos):
|
|
76
|
+
raise ValueError("ENCODING COLLISION — the confidence weighting is not strictly monotone "
|
|
77
|
+
"(folded-score bug); fix the encoding before computing AUC.")
|
|
78
|
+
|
|
79
|
+
# (2) UNWEIGHTED BASELINE beside the weighted primary
|
|
80
|
+
df["score"] = composite(df.call.values, df.confidence.values, K)
|
|
81
|
+
print(f"AUC (confidence-weighted) = {roc_auc_score(df.truth, df.score):.3f}")
|
|
82
|
+
print(f"AUC (unweighted binary call) = {roc_auc_score(df.truth, df.call):.3f}")
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
If the weighted AUC materially exceeds the unweighted one, the weighting must be justified;
|
|
86
|
+
if it does not, report the simpler estimator as primary. A weighted score that changed a
|
|
87
|
+
hypothesis's direction versus its unweighted baseline is a **post-lock change to disclose**,
|
|
88
|
+
not a silent primary.
|
|
89
|
+
|
|
90
|
+
---
|
|
91
|
+
|
|
92
|
+
## AUC confidence intervals and comparing two AUCs
|
|
93
|
+
|
|
94
|
+
DeLong is the field-standard analytic AUC CI and paired comparison; `pROC` (R) is canonical.
|
|
95
|
+
|
|
96
|
+
```r
|
|
97
|
+
library(pROC)
|
|
98
|
+
d <- read.csv("reader_calls.csv")
|
|
99
|
+
rw <- roc(d$truth, d$score, quiet = TRUE)
|
|
100
|
+
ci.auc(rw) # DeLong 95% CI for the weighted AUC
|
|
101
|
+
ru <- roc(d$truth, d$call, quiet = TRUE)
|
|
102
|
+
roc.test(rw, ru, method = "delong", paired = TRUE) # paired, SAME cases (fixed-reader)
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
A portable Python bootstrap CI (use when a DeLong implementation is unavailable):
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
def auc_ci_bootstrap(y, s, n_boot=2000, seed=42):
|
|
109
|
+
rng = np.random.default_rng(seed); y = np.asarray(y); s = np.asarray(s); n = len(y)
|
|
110
|
+
boots = [roc_auc_score(y[i], s[i]) for i in (rng.integers(0, n, n) for _ in range(n_boot))
|
|
111
|
+
if len(np.unique(y[i])) == 2]
|
|
112
|
+
return tuple(np.percentile(boots, [2.5, 97.5]))
|
|
113
|
+
```
|
|
114
|
+
|
|
115
|
+
**Fixed-reader vs MRMC.** A paired DeLong test treats the readers/cases as fixed. If the claim
|
|
116
|
+
is that the result **generalises to readers** (a reader sample), a fixed-reader CI understates
|
|
117
|
+
uncertainty — use an MRMC method (Obuchowski–Rockette / DBM) that carries reader + case
|
|
118
|
+
variance, report per-reader and reader-averaged AUC, and state the unit (per-patient vs
|
|
119
|
+
per-lesion) and the superiority/non-inferiority margin. This is probe D9's MRMC lead.
|
|
120
|
+
|
|
121
|
+
---
|
|
122
|
+
|
|
123
|
+
## Per-stratum admissibility: produce the table the claim is checked against (D10)
|
|
124
|
+
|
|
125
|
+
A blanket "no subgroup met AUC ≥ 0.75 with lower bound ≥ 0.70" is falsified by a single tabled
|
|
126
|
+
stratum that meets it. Produce the per-stratum AUC + CI and test each row against the rule
|
|
127
|
+
rather than asserting a global negative.
|
|
128
|
+
|
|
129
|
+
```python
|
|
130
|
+
rule = lambda auc, lo: (auc >= 0.75) and (lo >= 0.70)
|
|
131
|
+
for name, g in df.groupby("stratum"):
|
|
132
|
+
if g.truth.nunique() < 2: # single-class stratum: not estimable
|
|
133
|
+
print(f"{name}: not estimable (one class)"); continue
|
|
134
|
+
auc = roc_auc_score(g.truth, g.score); lo, hi = auc_ci_bootstrap(g.truth, g.score)
|
|
135
|
+
print(f"{name}: AUC {auc:.3f} (95% CI {lo:.3f}-{hi:.3f}) "
|
|
136
|
+
f"{'QUALIFIES' if rule(auc, lo) else 'below'}")
|
|
137
|
+
```
|
|
138
|
+
|
|
139
|
+
With *k* strata some crossings are expected — frame qualifying strata as hypothesis-generating
|
|
140
|
+
(note the multiplicity), but do **not** deny a row the paper's own table satisfies.
|
|
141
|
+
|
|
142
|
+
---
|
|
143
|
+
|
|
144
|
+
## One scale per comparison (D11)
|
|
145
|
+
|
|
146
|
+
Two values sharing a comparison column (or a row-wise "A vs B") must be computed under the
|
|
147
|
+
**same normalisation / definition** — e.g. both volume errors as a standard relative error, not
|
|
148
|
+
one relative and one range-normalised. If they are not identical, recompute on a common scale
|
|
149
|
+
before any superiority/comparability claim, or footnote "not on the same scale".
|
|
150
|
+
|
|
151
|
+
---
|
|
152
|
+
|
|
153
|
+
## Reporting
|
|
154
|
+
|
|
155
|
+
- Sensitivity, specificity, PPV, NPV, accuracy, and AUC each **with a 95% CI**; the operating
|
|
156
|
+
point and how it was chosen (Youden vs a prespecified clinical threshold — a threshold picked
|
|
157
|
+
on the same data is optimistic and needs a held-out or cross-validated estimate).
|
|
158
|
+
- The **analysis unit** (per-patient vs per-lesion) and clustering handling; study prevalence.
|
|
159
|
+
- For a weighted-score primary: the **unweighted-baseline AUC** beside it (D9).
|
|
160
|
+
- AUC alone is insufficient for a clinical claim — pair discrimination with **calibration** and a
|
|
161
|
+
**decision-curve / net-benefit** pass at the relevant threshold (see the incremental-value
|
|
162
|
+
table type and the `make-figures` decision-curve exemplar).
|
|
163
|
+
|
|
164
|
+
---
|
|
165
|
+
|
|
166
|
+
## Common failures (flag at review)
|
|
167
|
+
|
|
168
|
+
- **AUC / sensitivity / specificity with no CI**, or a per-lesion count reported as if per-patient.
|
|
169
|
+
- **Confidence-weighted AUC with no unweighted baseline** and no monotonic-encoding check (D9) —
|
|
170
|
+
the weighting may have created the result or hidden a folded-score bug.
|
|
171
|
+
- **Two AUCs compared by CI overlap** instead of a paired DeLong test; a **fixed-reader** CI used
|
|
172
|
+
for a claim that generalises to readers (needs MRMC).
|
|
173
|
+
- **"No stratum met the rule" contradicted by the per-stratum table** (D10).
|
|
174
|
+
- **Mixed-normalisation head-to-head** in one column (D11).
|
|
175
|
+
- **PPV/NPV transported** across a prevalence change.
|
|
176
|
+
|
|
177
|
+
---
|
|
178
|
+
|
|
179
|
+
## Anti-Hallucination
|
|
180
|
+
|
|
181
|
+
- Never hand-type an AUC, sensitivity/specificity, or CI — compute it from the calls CSV with a
|
|
182
|
+
seeded script; carry each estimate together with its CI (never a bare AUC).
|
|
183
|
+
- Do not report a weighted-score AUC without the unweighted baseline the code produced.
|
|
184
|
+
- Do not quote a DeLong comparison p-value that a paired test on the same cases did not produce.
|
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
# Survival / Time-to-Event Guide
|
|
2
|
+
|
|
3
|
+
Estimating time-to-event outcomes and prognostic effects. The estimator is easy to
|
|
4
|
+
call; the ways these analyses fail review are (1) ignoring **competing risks** so a naive
|
|
5
|
+
1−KM **overestimates** the cumulative incidence, (2) reporting a **single time-averaged
|
|
6
|
+
hazard ratio** when the proportional-hazards assumption is violated, and (3) **estimand
|
|
7
|
+
drift** — quoting a subdistribution hazard for an etiologic claim or a cause-specific hazard
|
|
8
|
+
for an absolute-risk claim. This guide produces the right estimand; the operational caveats
|
|
9
|
+
(EPV gate, cluster-robust CIs, interval-censoring) live in the SKILL.md `### Survival
|
|
10
|
+
Analysis` section.
|
|
11
|
+
|
|
12
|
+
---
|
|
13
|
+
|
|
14
|
+
## When to Use
|
|
15
|
+
|
|
16
|
+
- **Single event type, right-censored** → Kaplan–Meier (unadjusted) + Cox proportional-hazards
|
|
17
|
+
(adjusted HR with 95% CI); the log-rank test for a KM group comparison.
|
|
18
|
+
- **≥2 competing event types** (recurrence + competing death, or cause-specific mortality) →
|
|
19
|
+
cumulative incidence functions + **cause-specific Cox** or **Fine–Gray** — see below.
|
|
20
|
+
- **Prognostic model discrimination** → a C-index variant matched to the censoring + a
|
|
21
|
+
time-dependent AUC at a clinical horizon (S6).
|
|
22
|
+
- NOT for: events detected only at scheduled visits → interval-censored methods (SKILL.md);
|
|
23
|
+
recurrent events per subject → a cluster-robust or frailty model (SKILL.md cluster rule);
|
|
24
|
+
rater agreement → `analysis_guides/agreement_reliability.md`.
|
|
25
|
+
|
|
26
|
+
---
|
|
27
|
+
|
|
28
|
+
## Competing risks come first (this, not the HR, is the issue)
|
|
29
|
+
|
|
30
|
+
If a subject can experience an event that **precludes** the event of interest (death before
|
|
31
|
+
recurrence), treating the competing event as ordinary censoring is *informative* censoring:
|
|
32
|
+
the naive **1−KM overestimates** the cumulative incidence of the event of interest. Produce the
|
|
33
|
+
**cumulative incidence function (CIF)** with an Aalen–Johansen / Fine–Gray estimator instead.
|
|
34
|
+
This is the produce-side of probe **S3**.
|
|
35
|
+
|
|
36
|
+
```python
|
|
37
|
+
# lifelines: cumulative incidence for a competing-risks event of interest
|
|
38
|
+
from lifelines import AalenJohansenFitter, KaplanMeierFitter
|
|
39
|
+
import pandas as pd
|
|
40
|
+
df = pd.read_csv("survival.csv") # time, event_type (0=censored, 1=interest, 2=competing)
|
|
41
|
+
|
|
42
|
+
ajf = AalenJohansenFitter()
|
|
43
|
+
ajf.fit(df["time"], df["event_type"], event_of_interest=1)
|
|
44
|
+
print(ajf.cumulative_density_.tail(1)) # correct CIF for cause 1
|
|
45
|
+
|
|
46
|
+
kmf = KaplanMeierFitter() # NAIVE (cause 2 censored) — overestimates
|
|
47
|
+
kmf.fit(df["time"], (df["event_type"] == 1).astype(int))
|
|
48
|
+
print("naive 1-KM:", float(1 - kmf.survival_function_.iloc[-1, 0]))
|
|
49
|
+
# the naive value will exceed the Aalen-Johansen CIF whenever the competing event is common.
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
**Which model for which question (S8 estimand):**
|
|
53
|
+
|
|
54
|
+
- **Cause-specific hazard** (a Cox model that censors the competing event) answers an
|
|
55
|
+
**etiologic** question — "does X change the rate of recurrence among those still at risk?"
|
|
56
|
+
- **Fine–Gray subdistribution hazard** (`cmprsk::crr` / `survival::finegray` in R) answers a
|
|
57
|
+
**prognostic / absolute-risk** question — "does X change the cumulative *incidence*?" Quote an
|
|
58
|
+
sHR for an incidence claim and a cause-specific HR for an etiologic claim — not the reverse.
|
|
59
|
+
|
|
60
|
+
```r
|
|
61
|
+
library(survival) # Fine-Gray via a weighted Cox on the finegray split
|
|
62
|
+
fg <- finegray(Surv(time, factor(event_type)) ~ ., data = d, etype = 1)
|
|
63
|
+
crr <- coxph(Surv(fgstart, fgstop, fgstatus) ~ x, weight = fgwt, data = fg) # subdistribution HR
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
---
|
|
67
|
+
|
|
68
|
+
## Proportional hazards, and RMST when it fails
|
|
69
|
+
|
|
70
|
+
A single Cox HR is a time-average; if PH is violated it averages a changing effect. Test PH,
|
|
71
|
+
and when it fails report a **restricted mean survival time (RMST) difference** at a fixed
|
|
72
|
+
horizon (an estimand that stays interpretable under non-PH) rather than one HR.
|
|
73
|
+
|
|
74
|
+
```python
|
|
75
|
+
from lifelines import CoxPHFitter
|
|
76
|
+
cph = CoxPHFitter().fit(df, "time", "event")
|
|
77
|
+
cph.check_assumptions(df, p_value_threshold=0.05) # Schoenfeld; significant -> PH violated
|
|
78
|
+
```
|
|
79
|
+
|
|
80
|
+
```r
|
|
81
|
+
library(survRM2)
|
|
82
|
+
rmst2(time, status, arm, tau = 3) # RMST difference at tau=3 years, valid under non-PH
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Do not report a single time-averaged HR alongside a significant Schoenfeld test without also
|
|
86
|
+
giving a piecewise/time-stratified HR or an RMST difference (SKILL.md PH-violation rule).
|
|
87
|
+
|
|
88
|
+
---
|
|
89
|
+
|
|
90
|
+
## Follow-up and discrimination (S6)
|
|
91
|
+
|
|
92
|
+
- **Reverse Kaplan–Meier median follow-up** — the honest "how long were people followed"
|
|
93
|
+
(median event time answers a different question). Compute it by **swapping the event
|
|
94
|
+
indicator** (censored observations become the "events"); report it per cohort and per
|
|
95
|
+
outcome, with the censoring date.
|
|
96
|
+
|
|
97
|
+
```python
|
|
98
|
+
kmf.fit(df["time"], 1 - df["event"]) # swap: censored -> event
|
|
99
|
+
print("reverse-KM median follow-up:", kmf.median_survival_time_)
|
|
100
|
+
```
|
|
101
|
+
|
|
102
|
+
- **C-index variant** — Harrell's C is biased under heavy or non-random censoring; report
|
|
103
|
+
**Uno's IPCW C** (`survC1::Est.Cval` / `timeROC`) and a **time-dependent AUC at a clinical
|
|
104
|
+
horizon** (2-/3-year) beside it, and state which variant and horizon (S6).
|
|
105
|
+
|
|
106
|
+
---
|
|
107
|
+
|
|
108
|
+
## Estimand provenance (S8)
|
|
109
|
+
|
|
110
|
+
State the survival estimand explicitly and hold it consistent across Abstract / Methods /
|
|
111
|
+
Results — event-free survival vs cause-specific cumulative incidence vs all-cause mortality,
|
|
112
|
+
and subject vs population level — with the evaluation horizon fixed in advance. Do not
|
|
113
|
+
re-designate the primary endpoint, model, or horizon after seeing results, and make every
|
|
114
|
+
derived statistic (an E-value, an sHR-vs-cause-specific contrast) trace to the *declared
|
|
115
|
+
primary* estimand. The self-review skill automates the registration ↔ manuscript and E-value
|
|
116
|
+
arithmetic checks (Phase 2.5f); see also `estimand-provenance-lock`.
|
|
117
|
+
|
|
118
|
+
---
|
|
119
|
+
|
|
120
|
+
## Reporting
|
|
121
|
+
|
|
122
|
+
- KM curves **with a number-at-risk table**; median survival with 95% CI (or the reason it is
|
|
123
|
+
not reached).
|
|
124
|
+
- Cox HR (95% CI) with the **PH check stated**; for competing risks, the CIF and whether the
|
|
125
|
+
HR is cause-specific or subdistribution.
|
|
126
|
+
- Events and person-time per group; the **reverse-KM median follow-up**; EPV for the model.
|
|
127
|
+
- The estimand and horizon, stated once and consistent everywhere.
|
|
128
|
+
|
|
129
|
+
---
|
|
130
|
+
|
|
131
|
+
## Common failures (flag at review)
|
|
132
|
+
|
|
133
|
+
- **Competing risks ignored** — naive 1−KM (or a cause-specific model presented as absolute
|
|
134
|
+
risk) overestimates incidence; report a CIF and name cause-specific vs subdistribution (S3/S8).
|
|
135
|
+
- **A single time-averaged HR under a violated PH assumption** — needs a piecewise HR or RMST.
|
|
136
|
+
- **Harrell's C under heavy censoring** with no Uno/IPCW variant and no horizon (S6).
|
|
137
|
+
- **Median *survival* reported as "follow-up"** instead of the reverse-KM follow-up.
|
|
138
|
+
- **Estimand drift** — a primary endpoint/model/horizon re-designated post-hoc, or a derived
|
|
139
|
+
statistic quoted off a non-primary estimate (S8).
|
|
140
|
+
|
|
141
|
+
---
|
|
142
|
+
|
|
143
|
+
## Anti-Hallucination
|
|
144
|
+
|
|
145
|
+
- Never hand-type an HR, median, CIF, or CI — compute it from the time-to-event CSV with a
|
|
146
|
+
seeded script, and carry each estimate together with its CI.
|
|
147
|
+
- Do not report a Cox HR without the proportional-hazards check the code actually ran.
|
|
148
|
+
- Under competing risks, do not quote a 1−KM cumulative incidence — report the Aalen–Johansen /
|
|
149
|
+
Fine–Gray CIF the code produced.
|
|
@@ -47,6 +47,7 @@ A checklist for **diagnostic test accuracy (DTA) primary studies** — an index
|
|
|
47
47
|
- When a reader study's novelty is a **confidence-weighted** (or rating-collapsed) score used as the ROC/AUC predictor, the **unweighted binary-call AUC** must be reported side-by-side (as a sensitivity analysis). Without it, you cannot tell whether the weighting *created* the result or hid an estimator fragility (e.g. a folded/non-monotonic encoding that collapses `real/5` with `ai/1`).
|
|
48
48
|
- Lead: if the primary predictor is a confidence/rating→single-score collapse and no unweighted binary-call AUC appears, ask for it; also confirm the (call × confidence) encoding is strictly monotonic (no boundary collision) — the folded-score bug understates one hypothesis and can flip another.
|
|
49
49
|
- Severity: MAJOR when the weighted score is the headline and no unweighted baseline is shown; the weighting must "earn its place" against the simpler estimator.
|
|
50
|
+
- Produce the fix: `analyze-stats` `references/analysis_guides/diagnostic_accuracy.md` has the monotonic-encoding check + the unweighted-baseline AUC beside the weighted primary (and the per-stratum admissibility table for D10 and the one-scale-per-comparison rule for D11).
|
|
50
51
|
|
|
51
52
|
**D10 — "No stratum met threshold X" vs a per-stratum table that does meet X**:
|
|
52
53
|
- When the manuscript states a numeric admissibility/deployability rule (e.g. "AUC ≥ 0.75 **and** lower 95% bound ≥ 0.70") and concludes "**no stratum met** the rule" / "all strata were below", cross-check that claim against the **per-stratum AUC + CI table**. A blanket negative-stratum claim contradicted by a tabled stratum that literally satisfies the rule (e.g. ultrasound 0.789, CI 0.742–0.834) is a self-contradiction a reviewer verifies with arithmetic.
|
|
@@ -28,6 +28,7 @@ A 9-probe checklist for time-to-event outcomes and prognostic model development.
|
|
|
28
28
|
- Cause-specific hazards or Fine-Gray subdistribution hazards used?
|
|
29
29
|
- Patient developing one event still at risk for the other (informative censoring by death)?
|
|
30
30
|
- If competing-risks structure is ignored and outcomes are treated as independent right-censored events → MAJOR.
|
|
31
|
+
- Produce the fix: `analyze-stats` `references/analysis_guides/survival.md` has the Aalen–Johansen/Fine–Gray cumulative-incidence code (naive 1−KM overestimates) + the cause-specific-vs-subdistribution estimand choice (S8) and the PH→RMST fallback.
|
|
31
32
|
|
|
32
33
|
**S4 — Cutoff derivation optimism**:
|
|
33
34
|
- Cutoffs derived via maximally selected log-rank statistics, AUC-based Youden's J, or similar data-driven methods?
|
|
@@ -55,6 +56,7 @@ A 9-probe checklist for time-to-event outcomes and prognostic model development.
|
|
|
55
56
|
- Decision-curve analysis at clinically relevant probability thresholds?
|
|
56
57
|
- For a prognostic model intended to guide surveillance intensity, treatment intensification, or eligibility for adjuvant therapy, discrimination alone is insufficient. If Methods mention calibration but Results/supplement contain no calibration plot or numeric metrics → MAJOR.
|
|
57
58
|
- **Apparent-vs-optimism-corrected deterministic tell**: discrimination/calibration reported with no internal-validation token nearby (`optimism`, `bootstrap`, `cross-valid`, `held-out`, `external`) is presumptively **apparent (in-sample) performance**. A **calibration slope reported as exactly 1.00** (and/or Uno's C, Brier, net-benefit all on the development sample) is the in-sample-fit signature — the model was evaluated on the data it was fit to. Flag MINOR (a bootstrap optimism correction is cheap and usually reproduces the estimate); escalate to MAJOR when the model is assessed for clinical utility / net-benefit, where optimistic metrics directly inflate the deployment claim.
|
|
59
|
+
- Produce the fix: `analyze-stats` `references/analysis_guides/calibration.md` has the bootstrap optimism-corrected calibration slope/intercept (the apparent slope is 1.00 by construction), the flexible calibration curve + scaled Brier, and why Hosmer–Lemeshow is dropped.
|
|
58
60
|
|
|
59
61
|
**S8 — Estimand provenance**:
|
|
60
62
|
- Is the survival estimand stated explicitly and held consistent across Abstract / Methods / Results — event-free survival, cause-specific cumulative incidence, all-cause mortality — and at the subject vs population level? A subdistribution hazard (Fine-Gray) answers a different question than a cause-specific hazard; quoting an sHR for an etiologic claim, or a cause-specific HR for an absolute-risk claim, is an estimand mismatch.
|
|
@@ -47,6 +47,7 @@ A checklist for **diagnostic test accuracy (DTA) primary studies** — an index
|
|
|
47
47
|
- When a reader study's novelty is a **confidence-weighted** (or rating-collapsed) score used as the ROC/AUC predictor, the **unweighted binary-call AUC** must be reported side-by-side (as a sensitivity analysis). Without it, you cannot tell whether the weighting *created* the result or hid an estimator fragility (e.g. a folded/non-monotonic encoding that collapses `real/5` with `ai/1`).
|
|
48
48
|
- Lead: if the primary predictor is a confidence/rating→single-score collapse and no unweighted binary-call AUC appears, ask for it; also confirm the (call × confidence) encoding is strictly monotonic (no boundary collision) — the folded-score bug understates one hypothesis and can flip another.
|
|
49
49
|
- Severity: MAJOR when the weighted score is the headline and no unweighted baseline is shown; the weighting must "earn its place" against the simpler estimator.
|
|
50
|
+
- Produce the fix: `analyze-stats` `references/analysis_guides/diagnostic_accuracy.md` has the monotonic-encoding check + the unweighted-baseline AUC beside the weighted primary (and the per-stratum admissibility table for D10 and the one-scale-per-comparison rule for D11).
|
|
50
51
|
|
|
51
52
|
**D10 — "No stratum met threshold X" vs a per-stratum table that does meet X**:
|
|
52
53
|
- When the manuscript states a numeric admissibility/deployability rule (e.g. "AUC ≥ 0.75 **and** lower 95% bound ≥ 0.70") and concludes "**no stratum met** the rule" / "all strata were below", cross-check that claim against the **per-stratum AUC + CI table**. A blanket negative-stratum claim contradicted by a tabled stratum that literally satisfies the rule (e.g. ultrasound 0.789, CI 0.742–0.834) is a self-contradiction a reviewer verifies with arithmetic.
|
|
@@ -28,6 +28,7 @@ A 9-probe checklist for time-to-event outcomes and prognostic model development.
|
|
|
28
28
|
- Cause-specific hazards or Fine-Gray subdistribution hazards used?
|
|
29
29
|
- Patient developing one event still at risk for the other (informative censoring by death)?
|
|
30
30
|
- If competing-risks structure is ignored and outcomes are treated as independent right-censored events → MAJOR.
|
|
31
|
+
- Produce the fix: `analyze-stats` `references/analysis_guides/survival.md` has the Aalen–Johansen/Fine–Gray cumulative-incidence code (naive 1−KM overestimates) + the cause-specific-vs-subdistribution estimand choice (S8) and the PH→RMST fallback.
|
|
31
32
|
|
|
32
33
|
**S4 — Cutoff derivation optimism**:
|
|
33
34
|
- Cutoffs derived via maximally selected log-rank statistics, AUC-based Youden's J, or similar data-driven methods?
|
|
@@ -55,6 +56,7 @@ A 9-probe checklist for time-to-event outcomes and prognostic model development.
|
|
|
55
56
|
- Decision-curve analysis at clinically relevant probability thresholds?
|
|
56
57
|
- For a prognostic model intended to guide surveillance intensity, treatment intensification, or eligibility for adjuvant therapy, discrimination alone is insufficient. If Methods mention calibration but Results/supplement contain no calibration plot or numeric metrics → MAJOR.
|
|
57
58
|
- **Apparent-vs-optimism-corrected deterministic tell**: discrimination/calibration reported with no internal-validation token nearby (`optimism`, `bootstrap`, `cross-valid`, `held-out`, `external`) is presumptively **apparent (in-sample) performance**. A **calibration slope reported as exactly 1.00** (and/or Uno's C, Brier, net-benefit all on the development sample) is the in-sample-fit signature — the model was evaluated on the data it was fit to. Flag MINOR (a bootstrap optimism correction is cheap and usually reproduces the estimate); escalate to MAJOR when the model is assessed for clinical utility / net-benefit, where optimistic metrics directly inflate the deployment claim.
|
|
59
|
+
- Produce the fix: `analyze-stats` `references/analysis_guides/calibration.md` has the bootstrap optimism-corrected calibration slope/intercept (the apparent slope is 1.00 by construction), the flexible calibration curve + scaled Brier, and why Hosmer–Lemeshow is dropped.
|
|
58
60
|
|
|
59
61
|
**S8 — Estimand provenance**:
|
|
60
62
|
- Is the survival estimand stated explicitly and held consistent across Abstract / Methods / Results — event-free survival, cause-specific cumulative incidence, all-cause mortality — and at the subject vs population level? A subdistribution hazard (Fine-Gray) answers a different question than a cause-specific hazard; quoting an sHR for an etiologic claim, or a cause-specific HR for an absolute-risk claim, is an estimand mismatch.
|
|
@@ -221,7 +221,7 @@ Design all tables and figures BEFORE writing prose. This ensures the narrative s
|
|
|
221
221
|
|
|
222
222
|
Write the Methods section first -- it is the most objective and anchors the rest of the paper.
|
|
223
223
|
|
|
224
|
-
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/methods.md` for PICO structure, backbone article usage, checklist cross-reference, and terminology conventions. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_methods/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — it lists, paragraph by paragraph, what each Methods paragraph must establish plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
|
|
224
|
+
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/methods.md` for PICO structure, backbone article usage, checklist cross-reference, and terminology conventions. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_methods/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020, RCT/CONSORT 2010) — it lists, paragraph by paragraph, what each Methods paragraph must establish plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
|
|
225
225
|
|
|
226
226
|
**Writing order within Methods:**
|
|
227
227
|
1. Study Design and Setting
|
|
@@ -255,7 +255,7 @@ Write the Methods section first -- it is the most objective and anchors the rest
|
|
|
255
255
|
Write Results aligned to the approved tables and figures. **Results = "What did we find?"
|
|
256
256
|
— nothing more.** Every sentence must be a factual statement backed by a number.
|
|
257
257
|
|
|
258
|
-
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/results.md` for mirror-symmetry rules, flowchart requirements, missing data handling, and the anti-interpretation self-check. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_results/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — each follows its `exemplar_methods/` sibling in Methods order, listing what each Results paragraph must establish (flow → baseline/prevalence → primary estimate with CIs → calibration/agreement → subgroups → sensitivity; for meta-analysis, PRISMA flow → characteristics+provenance → RoB → pooled estimate with I²/τ²/prediction interval → subgroup interaction → publication bias) plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
|
|
258
|
+
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/results.md` for mirror-symmetry rules, flowchart requirements, missing data handling, and the anti-interpretation self-check. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_results/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020, RCT/CONSORT 2010) — each follows its `exemplar_methods/` sibling in Methods order, listing what each Results paragraph must establish (flow → baseline/prevalence → primary estimate with CIs → calibration/agreement → subgroups → sensitivity; for meta-analysis, PRISMA flow → characteristics+provenance → RoB → pooled estimate with I²/τ²/prediction interval → subgroup interaction → publication bias; for an RCT, CONSORT flow → baseline-by-arm with no p-values → ITT primary with CI → secondary+harms → per-protocol beside ITT) plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
|
|
259
259
|
|
|
260
260
|
**Rules:**
|
|
261
261
|
- Every number in the text must match the corresponding table cell exactly.
|
|
@@ -296,7 +296,7 @@ Write Results aligned to the approved tables and figures. **Results = "What did
|
|
|
296
296
|
|
|
297
297
|
### Phase 5: Discussion
|
|
298
298
|
|
|
299
|
-
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/discussion.md` for the 4-paragraph structure, word limits, limitation writing guidelines, and Table/Figure citation rules. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_discussion/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — completing the exemplar trio, each lists what every Discussion paragraph must establish (key finding → interpretation/comparison → limitations → generalizability → conclusion matched to the evidence) plus the element that type most often omits (spectrum/verification bias; evidence-tier separation and optimism caveats; mandatory causal caution; for meta-analysis, GRADE certainty + heterogeneity source + non-independence/overlap caveat). For case reports, use `${CLAUDE_SKILL_DIR}/references/exemplar_case_report.md` instead: it controls literature-boundary wording, n=1 causal caution, and bedside teaching-point framing. Model the structure; the exemplars are synthetic, introduce no new results, and are not prose to copy.
|
|
299
|
+
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/discussion.md` for the 4-paragraph structure, word limits, limitation writing guidelines, and Table/Figure citation rules. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_discussion/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020, RCT/CONSORT 2010) — completing the exemplar trio, each lists what every Discussion paragraph must establish (key finding → interpretation/comparison → limitations → generalizability → conclusion matched to the evidence) plus the element that type most often omits (spectrum/verification bias; evidence-tier separation and optimism caveats; mandatory causal caution; for meta-analysis, GRADE certainty + heterogeneity source + non-independence/overlap caveat; for an RCT, blinding/attrition limitation + clinical-vs-statistical significance vs the MCID). For case reports, use `${CLAUDE_SKILL_DIR}/references/exemplar_case_report.md` instead: it controls literature-boundary wording, n=1 causal caution, and bedside teaching-point framing. Model the structure; the exemplars are synthetic, introduce no new results, and are not prose to copy.
|
|
300
300
|
|
|
301
301
|
**Before drafting, collect user input (Discussion Planning Gate).**
|
|
302
302
|
|
|
@@ -30,6 +30,8 @@ closes with a conclusion matched to the evidence — introducing no new results.
|
|
|
30
30
|
- `meta_analysis_prisma.md` — summary-of-evidence with certainty framing, heterogeneity-source
|
|
31
31
|
and non-independence/overlap caveats, GRADE limitations, no guideline-grade conclusion
|
|
32
32
|
(PRISMA 2020).
|
|
33
|
+
- `rct_consort.md` — primary-on-ITT key finding, clinical-vs-statistical significance vs the MCID,
|
|
34
|
+
blinding/attrition limitations, single-trial conclusion (CONSORT 2010).
|
|
33
35
|
|
|
34
36
|
## Curator guidelines (for adding more)
|
|
35
37
|
|
|
@@ -0,0 +1,42 @@
|
|
|
1
|
+
# Discussion structure — randomized controlled trial (CONSORT 2010)
|
|
2
|
+
|
|
3
|
+
A structure model for the Discussion of a parallel-group RCT, completing the trio with the
|
|
4
|
+
`exemplar_methods/` and `exemplar_results/` siblings. Each heading is a paragraph; each bullet is
|
|
5
|
+
*what it must establish*. Fill the `[brackets]`; do not copy this text. Follows
|
|
6
|
+
`section_guides/discussion.md`. Introduces **no new results** and cites **no tables/figures**.
|
|
7
|
+
|
|
8
|
+
## Paragraph 1 — Key finding
|
|
9
|
+
- Restate the **primary** result (the effect estimate + CI on the ITT population) in 2–3
|
|
10
|
+
sentences, matched to Results — no new estimate, and no elevation of a secondary outcome.
|
|
11
|
+
- One sentence on clinical meaning, tied to the **pre-stated MCID** (a statistically significant
|
|
12
|
+
effect smaller than the MCID is not clinically meaningful).
|
|
13
|
+
- **If the primary is null or non-inferiority is not shown, frame it by the CI vs the margin**,
|
|
14
|
+
not as "no difference" — state what the interval excludes.
|
|
15
|
+
|
|
16
|
+
## Paragraphs 2–3 — Interpretation and comparison
|
|
17
|
+
- Compare to prior trials and meta-analyses (agree / differ and why: population, comparator,
|
|
18
|
+
dose, endpoint); avoid over-generalizing beyond the enrolled population.
|
|
19
|
+
- Discuss mechanism / plausibility as hypothesis; keep secondary and subgroup findings explicitly
|
|
20
|
+
exploratory (do not narrate a subgroup as if it were a confirmatory result).
|
|
21
|
+
|
|
22
|
+
## Limitations
|
|
23
|
+
- **Risk-of-bias honesty**: open-label / unblinded assessment, attrition and how missing primary
|
|
24
|
+
data were handled, any post-randomization exclusions, early stopping and its effect on the
|
|
25
|
+
estimate, single-centre / narrow eligibility limiting external validity.
|
|
26
|
+
- Whether the trial was powered for the primary only (secondary/subgroup analyses under-powered).
|
|
27
|
+
|
|
28
|
+
## Generalizability
|
|
29
|
+
- The population and setting the result applies to, and where it may not transfer (eligibility
|
|
30
|
+
restrictions, standard-of-care comparator specific to the setting, adherence in a trial vs
|
|
31
|
+
routine care).
|
|
32
|
+
|
|
33
|
+
## Conclusion
|
|
34
|
+
- Matched to the evidence — a single trial supports the primary contrast in the studied
|
|
35
|
+
population; use "in this trial, [intervention] reduced/did not reduce [outcome]", not a
|
|
36
|
+
guideline-grade recommendation the evidence tier does not license.
|
|
37
|
+
|
|
38
|
+
## Common omission
|
|
39
|
+
- An explicit **blinding / attrition limitation and a clinical-vs-statistical-significance caveat**
|
|
40
|
+
tied to the MCID — the Discussion elements RCT drafts most often soften. Cross-reference
|
|
41
|
+
`section_guides/discussion.md`, the `peer-review/references/domain-probes/rct_trial.md` probes,
|
|
42
|
+
and, for an AI intervention, the CONSORT-AI reporting items.
|
|
@@ -24,6 +24,7 @@ fit). A missing element is a gap to fill before drafting Results.
|
|
|
24
24
|
- `ai_validation_tripod_claim.md` — an AI/ML model development + validation study (TRIPOD+AI / CLAIM).
|
|
25
25
|
- `observational_cohort_strobe.md` — an exposure→outcome observational cohort (STROBE).
|
|
26
26
|
- `meta_analysis_prisma.md` — a systematic review with quantitative synthesis (PRISMA 2020).
|
|
27
|
+
- `rct_consort.md` — a parallel-group randomized controlled trial (CONSORT 2010).
|
|
27
28
|
|
|
28
29
|
## Curator guidelines (for adding more)
|
|
29
30
|
|
|
@@ -0,0 +1,55 @@
|
|
|
1
|
+
# Methods structure — randomized controlled trial (CONSORT 2010)
|
|
2
|
+
|
|
3
|
+
A structure model for a parallel-group RCT. Each heading is a paragraph; each bullet is *what it
|
|
4
|
+
must establish*. Fill the `[brackets]`; do not copy this text. Anchors to CONSORT 2010; pairs the
|
|
5
|
+
`peer-review/references/domain-probes/rct_trial.md` probes (and CONSORT-AI/SPIRIT-AI for an AI
|
|
6
|
+
intervention).
|
|
7
|
+
|
|
8
|
+
## Trial design and registration
|
|
9
|
+
- Design (parallel-group / crossover / cluster), allocation ratio (e.g. 1:1), and framework
|
|
10
|
+
(superiority / non-inferiority / equivalence) — with the **margin** stated up front for the
|
|
11
|
+
latter two.
|
|
12
|
+
- **Prospective trial registration** (ClinicalTrials.gov / a WHO ICTRP registry) with the number,
|
|
13
|
+
and the protocol reference; ethics approval and informed consent.
|
|
14
|
+
|
|
15
|
+
## Participants, setting, interventions
|
|
16
|
+
- Eligibility as a numbered inclusion / exclusion list; the settings and the recruitment and
|
|
17
|
+
follow-up dates.
|
|
18
|
+
- The experimental and comparator interventions in enough detail to replicate (dose, schedule,
|
|
19
|
+
who delivered them); the comparator is a real, defined control (placebo / standard-of-care),
|
|
20
|
+
not a strawman.
|
|
21
|
+
|
|
22
|
+
## Outcomes
|
|
23
|
+
- The **single pre-specified primary outcome** with its metric and time point (do not re-designate
|
|
24
|
+
the primary after seeing data); secondary outcomes listed; any changes to outcomes after trial
|
|
25
|
+
commencement disclosed with dates.
|
|
26
|
+
|
|
27
|
+
## Sample size
|
|
28
|
+
- The power calculation: assumed effect (the MCID, not the hoped-for effect), α, power, and the
|
|
29
|
+
resulting N with attrition inflation; for non-inferiority, the margin and its justification.
|
|
30
|
+
|
|
31
|
+
## Randomization, allocation concealment, blinding
|
|
32
|
+
- **Sequence generation** (how the random sequence was produced), **allocation concealment**
|
|
33
|
+
(the mechanism that hid the next assignment — central/pharmacy/sealed opaque envelopes), and
|
|
34
|
+
**implementation** (who enrolled, who assigned) — three distinct items, all required.
|
|
35
|
+
- **Blinding**: who was blinded (participants, clinicians, outcome assessors, analysts) and how;
|
|
36
|
+
if open-label, state it and how assessor blinding was preserved.
|
|
37
|
+
|
|
38
|
+
## Statistical methods
|
|
39
|
+
- The primary analysis and estimand: the contrast, the population (**intention-to-treat** as
|
|
40
|
+
primary — everyone as randomized; per-protocol as a secondary, and the primary population for a
|
|
41
|
+
non-inferiority trial stated), and how missing data were handled (not naive complete-case for a
|
|
42
|
+
dropout-prone endpoint).
|
|
43
|
+
- Pre-specified subgroups and interim analyses / stopping rules; multiplicity handling; software.
|
|
44
|
+
|
|
45
|
+
## Reporting-guideline fit
|
|
46
|
+
- CONSORT 2010 (+ CONSORT-AI for an AI intervention). Critical items: registration, sequence
|
|
47
|
+
generation + allocation concealment + blinding, the pre-specified primary, ITT, and a
|
|
48
|
+
participant flow diagram that reconciles (randomized → received → analyzed).
|
|
49
|
+
|
|
50
|
+
## Common omission
|
|
51
|
+
- **Allocation concealment described as blinding** (they are different — concealment protects
|
|
52
|
+
*assignment*, blinding protects *ascertainment*), and an unstated **ITT vs per-protocol**
|
|
53
|
+
primary population — the Methods elements RCT drafts most often conflate or skip, and the first
|
|
54
|
+
a trials reviewer checks. Cross-reference `section_guides/methods.md` and the
|
|
55
|
+
`peer-review/references/domain-probes/rct_trial.md` probes.
|
|
@@ -30,6 +30,8 @@ element that interprets rather than reports belongs in the Discussion.
|
|
|
30
30
|
- `meta_analysis_prisma.md` — PRISMA flow, study/patient characteristics + provenance, pooled
|
|
31
31
|
estimate with I²/τ²/prediction interval, subgroup interaction, sensitivity, publication bias
|
|
32
32
|
(PRISMA 2020).
|
|
33
|
+
- `rct_consort.md` — CONSORT flow, baseline by arm (no p-values), ITT primary estimate with CI,
|
|
34
|
+
secondary outcomes + harms, subgroup interaction, per-protocol beside ITT (CONSORT 2010).
|
|
33
35
|
|
|
34
36
|
## Curator guidelines (for adding more)
|
|
35
37
|
|
|
@@ -0,0 +1,38 @@
|
|
|
1
|
+
# Results structure — randomized controlled trial (CONSORT 2010)
|
|
2
|
+
|
|
3
|
+
A structure model for the Results of a parallel-group RCT. It follows its Methods CONSORT
|
|
4
|
+
sibling in Methods order. Each heading is a paragraph; each bullet is *what it must establish*.
|
|
5
|
+
Fill the `[brackets]`; do not copy this text. Report findings only — no interpretation.
|
|
6
|
+
|
|
7
|
+
## Participant flow (Figure 1 — CONSORT diagram)
|
|
8
|
+
- The CONSORT flow: assessed → randomized → allocated (received / did not receive) → followed up
|
|
9
|
+
(lost, discontinued, with reasons) → analyzed, **per arm**, with counts.
|
|
10
|
+
- The cascade reconciles (randomized = Σ analyzed + Σ excluded-with-reason); the number
|
|
11
|
+
**analyzed for the primary outcome** matches the ITT denominator. Numbers match the Abstract,
|
|
12
|
+
Methods, and Table 1.
|
|
13
|
+
|
|
14
|
+
## Recruitment and baseline (Table 1)
|
|
15
|
+
- Recruitment and follow-up dates and why the trial ended (target reached / stopped);
|
|
16
|
+
Table 1 baseline characteristics **by arm** — **no p-values for baseline balance** (randomization
|
|
17
|
+
makes them meaningless; show the descriptive comparison and, if used, the balance metric).
|
|
18
|
+
|
|
19
|
+
## Primary outcome
|
|
20
|
+
- The pre-specified **primary** outcome by arm: the **effect estimate with its 95% CI** and the
|
|
21
|
+
absolute numbers per arm (events/N or mean±SD), on the **ITT** population — not just a p-value.
|
|
22
|
+
- For non-inferiority: the CI relative to the **pre-stated margin** (the conclusion follows from
|
|
23
|
+
where the CI sits, not from a significance test).
|
|
24
|
+
|
|
25
|
+
## Secondary outcomes and harms
|
|
26
|
+
- Secondary outcomes with estimates + CIs, flagged as secondary (multiplicity in view — do not
|
|
27
|
+
elevate a significant secondary to a headline).
|
|
28
|
+
- **Harms / adverse events** reported per arm (a trial that reports only efficacy is incomplete).
|
|
29
|
+
|
|
30
|
+
## Subgroups and sensitivity
|
|
31
|
+
- Pre-specified subgroup analyses with the **interaction test**, framed as exploratory; the
|
|
32
|
+
per-protocol analysis beside ITT (concordance is reassuring; divergence is discussed, not hidden).
|
|
33
|
+
|
|
34
|
+
## Common omission
|
|
35
|
+
- **Baseline p-values** (should not appear), and the **primary reported off the per-protocol set**
|
|
36
|
+
rather than ITT, or without its absolute per-arm numbers — the Results elements RCT drafts most
|
|
37
|
+
often get wrong. Cross-reference `section_guides/results.md`, the
|
|
38
|
+
`peer-review/references/domain-probes/rct_trial.md` probes, and the CONSORT flow requirement.
|