medsci-skills 5.13.0 → 5.14.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -153,14 +153,19 @@
153
153
  },
154
154
  {
155
155
  "path": "skills/analyze-stats/SKILL.md",
156
- "size": 56422,
157
- "sha256": "ddb73598fe60601942e8b44a87771e3ba583a585304effac4ae206d53ab4f154"
156
+ "size": 56948,
157
+ "sha256": "e7f01164e2fc9db924e49ee226b13bcf78d4d7c0463099c0f666985152a91d2e"
158
158
  },
159
159
  {
160
160
  "path": "skills/analyze-stats/references/analysis_guides/agreement_reliability.md",
161
161
  "size": 6674,
162
162
  "sha256": "599ad2551f547043b2bec79e3139d9c2c3ae33c5025784359b22022f882504c2"
163
163
  },
164
+ {
165
+ "path": "skills/analyze-stats/references/analysis_guides/calibration.md",
166
+ "size": 6470,
167
+ "sha256": "ac33862c6c707997be5a2511ed6a6d546c1e829fc9fc033d1c85eca9afa3ca9a"
168
+ },
164
169
  {
165
170
  "path": "skills/analyze-stats/references/analysis_guides/diagnostic_accuracy.md",
166
171
  "size": 9303,
@@ -3053,8 +3058,8 @@
3053
3058
  },
3054
3059
  {
3055
3060
  "path": "skills/peer-review/references/domain-probes/survival_prognostic.md",
3056
- "size": 14022,
3057
- "sha256": "96195cd1b575535bc12a133125b9d85e2ca7e756b96a0649e03d4c3696e5312b"
3061
+ "size": 14297,
3062
+ "sha256": "32cd929ff2e151f33bc5b106eddef0d38d5bd15672054e690c603ee82a22a817"
3058
3063
  },
3059
3064
  {
3060
3065
  "path": "skills/peer-review/references/exemplar_reviews/README.md",
@@ -3613,8 +3618,8 @@
3613
3618
  },
3614
3619
  {
3615
3620
  "path": "skills/self-review/references/domain-probes/survival_prognostic.md",
3616
- "size": 14022,
3617
- "sha256": "96195cd1b575535bc12a133125b9d85e2ca7e756b96a0649e03d4c3696e5312b"
3621
+ "size": 14297,
3622
+ "sha256": "32cd929ff2e151f33bc5b106eddef0d38d5bd15672054e690c603ee82a22a817"
3618
3623
  },
3619
3624
  {
3620
3625
  "path": "skills/self-review/references/exemplar_findings/README.md",
@@ -3918,8 +3923,8 @@
3918
3923
  },
3919
3924
  {
3920
3925
  "path": "skills/write-paper/SKILL.md",
3921
- "size": 66771,
3922
- "sha256": "1a50c9ceb040a79feaa60e809c7196f9f04b85cab2f5f3402dcaed3f7aee552c"
3926
+ "size": 67054,
3927
+ "sha256": "62b3cbff56a139ed20a783c462fb6dde6b7a932bb057762a2d1c6382c8d044c1"
3923
3928
  },
3924
3929
  {
3925
3930
  "path": "skills/write-paper/references/exemplar_abstract.md",
@@ -3938,8 +3943,8 @@
3938
3943
  },
3939
3944
  {
3940
3945
  "path": "skills/write-paper/references/exemplar_discussion/README.md",
3941
- "size": 2727,
3942
- "sha256": "3bfcd4eb4df4ca0309446a606550cb11adc4d88daa676f2217249691d2e6b042"
3946
+ "size": 2902,
3947
+ "sha256": "1db58e8d3801a7fd2bd6a6bb9c271f22436a71e82ba034a8c888e9468945a026"
3943
3948
  },
3944
3949
  {
3945
3950
  "path": "skills/write-paper/references/exemplar_discussion/ai_validation_tripod_claim.md",
@@ -3961,6 +3966,11 @@
3961
3966
  "size": 3652,
3962
3967
  "sha256": "21ec3cbfcb626cc00f09e66aa27d8f2f835a202ed083bcebcba57b787ddff866"
3963
3968
  },
3969
+ {
3970
+ "path": "skills/write-paper/references/exemplar_discussion/rct_consort.md",
3971
+ "size": 2613,
3972
+ "sha256": "2fa6e9f621219974f07294e8637da83d923327207e38e1692b62bb45c7d20ab5"
3973
+ },
3964
3974
  {
3965
3975
  "path": "skills/write-paper/references/exemplar_introduction.md",
3966
3976
  "size": 2636,
@@ -3968,8 +3978,8 @@
3968
3978
  },
3969
3979
  {
3970
3980
  "path": "skills/write-paper/references/exemplar_methods/README.md",
3971
- "size": 2342,
3972
- "sha256": "6966e8b723d0c4ccc012d54e2b8a22ef59b39c22619e63a8597f9badb380b4e7"
3981
+ "size": 2426,
3982
+ "sha256": "b55addd809841fdb8ae8e834693c7a03602cb1d8aeb00131ed86ecb8fbc713f6"
3973
3983
  },
3974
3984
  {
3975
3985
  "path": "skills/write-paper/references/exemplar_methods/ai_validation_tripod_claim.md",
@@ -3991,10 +4001,15 @@
3991
4001
  "size": 2364,
3992
4002
  "sha256": "e2ea3f64be321d2fb8249d4d528cfd24231c0b69d2562e4290cf3dcddd29d419"
3993
4003
  },
4004
+ {
4005
+ "path": "skills/write-paper/references/exemplar_methods/rct_consort.md",
4006
+ "size": 3268,
4007
+ "sha256": "742e343b163336a6dda2fa2054e0f486df489e1d6cb2cf3eb782e05d25d9710d"
4008
+ },
3994
4009
  {
3995
4010
  "path": "skills/write-paper/references/exemplar_results/README.md",
3996
- "size": 2799,
3997
- "sha256": "14b588f1f4433be71c39f11627148fff797eb87b3b74be75e61514095d4c38ea"
4011
+ "size": 2989,
4012
+ "sha256": "b67ec682f899a1774af10913eaeb8f33956859da2ef787250c01ef3b437f1155"
3998
4013
  },
3999
4014
  {
4000
4015
  "path": "skills/write-paper/references/exemplar_results/ai_validation_tripod_claim.md",
@@ -4016,6 +4031,11 @@
4016
4031
  "size": 2452,
4017
4032
  "sha256": "4c8e1fc5c5d00a3c0dc26ea0d94dc80e3077a96503ada653ccdcee8eeb8b14da"
4018
4033
  },
4034
+ {
4035
+ "path": "skills/write-paper/references/exemplar_results/rct_consort.md",
4036
+ "size": 2356,
4037
+ "sha256": "6a0becfe8f24db7f686bd23f26c9caec749dd996914705679ea915398fc7c3dd"
4038
+ },
4019
4039
  {
4020
4040
  "path": "skills/write-paper/references/journal_profiles/AJNR.md",
4021
4041
  "size": 6381,
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schema_version": 1,
3
- "version": "5.13.0",
3
+ "version": "5.14.0",
4
4
  "owned_skills": [
5
5
  "academic-aio",
6
6
  "add-journal",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "medsci-skills",
3
- "version": "5.13.0",
3
+ "version": "5.14.0",
4
4
  "description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
5
5
  "license": "SEE LICENSE IN LICENSE",
6
6
  "homepage": "https://github.com/Aperivue/medsci-skills#readme",
@@ -545,7 +545,8 @@ When death or other events preclude the outcome of interest, standard KM overest
545
545
  - **Guide**: Load `analysis_guides/regression.md` before generating code
546
546
  - **Template**: `references/templates/regression.py` (set `regression_type = "logistic"`)
547
547
  - Run univariable analysis first, then multivariable with clinically selected variables
548
- - Required outputs: OR table (univariable + multivariable), C-statistic (95% CI), Hosmer-Lemeshow
548
+ - Required outputs: OR table (univariable + multivariable), C-statistic (95% CI), and **calibration** (intercept + slope + flexible plot — **not** HosmerLemeshow, which is deprecated; see the calibration guide)
549
+ - **Prediction-model calibration guide**: `references/analysis_guides/calibration.md` (**load before generating code** for any model that outputs a risk used for a decision — the apparent slope of exactly 1.00 is the in-sample tell, so produce the **bootstrap optimism-corrected** slope/intercept; Van Calster's calibration levels; scaled Brier; why Hosmer–Lemeshow is dropped; produce-side of probe S7)
549
550
  - Check VIF < 5, EPV >= 10 (warn if violated)
550
551
  - **Nested observation units**: when rows are clustered within subjects (multiple lesions/visits per patient), use cluster-robust standard errors (`cov_type="cluster"`, `cov_kwds={"groups": id}` in statsmodels) or a mixed-effects logistic model — a naive logit CI assumes independent rows and is too narrow
551
552
  - Box-Tidwell test for continuous predictor linearity
@@ -0,0 +1,132 @@
1
+ # Prediction-Model Calibration Guide
2
+
3
+ For a clinical prediction model (a logistic risk score or a Cox/survival model at a fixed
4
+ horizon), **discrimination (AUC / C-index) is not enough** — a model that ranks well can still
5
+ output probabilities that are systematically too high or too low. The ways calibration fails
6
+ review are (1) reporting **apparent (in-sample)** calibration — a slope of exactly 1.00 is the
7
+ fingerprint — with no internal-validation correction, (2) leaning on the **Hosmer–Lemeshow**
8
+ test (deprecated), and (3) omitting calibration entirely for a model meant to guide care. This
9
+ guide produces the corrected estimand; it is the produce-side of probe **S7**.
10
+
11
+ ---
12
+
13
+ ## When to Use
14
+
15
+ - Any model that outputs a **risk / probability** used for a decision (surveillance intensity,
16
+ treatment eligibility, triage) — logistic or survival-at-a-horizon.
17
+ - Reported **alongside** discrimination and, for a utility claim, decision-curve net benefit —
18
+ never discrimination alone.
19
+ - NOT a substitute for external validation: internal (bootstrap/CV) calibration corrects
20
+ optimism but does not establish transportability.
21
+
22
+ ---
23
+
24
+ ## Apparent calibration is optimistic — correct it (this is the S7 issue)
25
+
26
+ Fit a logistic model by maximum likelihood and its **apparent** calibration slope on the same
27
+ data is **exactly 1.00** and its calibration-in-the-large intercept **exactly 0** — by
28
+ construction, not because the model is well-calibrated. A slope printed as `1.00` (or metrics
29
+ with no `bootstrap` / `cross-valid` / `optimism` / `held-out` token nearby) is presumptively
30
+ in-sample. Produce the **bootstrap optimism-corrected** slope instead (Harrell/Steyerberg).
31
+
32
+ ```python
33
+ import numpy as np, statsmodels.api as sm
34
+ X = ... # design matrix (add_constant), y = 0/1 outcome
35
+ def cal_slope(y, lp):
36
+ m = sm.GLM(y, sm.add_constant(lp), family=sm.families.Binomial()).fit()
37
+ return m.params[1], m.params[0] # slope, calibration-in-the-large intercept
38
+
39
+ full = sm.GLM(y, X, family=sm.families.Binomial()).fit()
40
+ app_slope, app_int = cal_slope(y, X @ full.params) # apparent: slope ~1.00, intercept ~0
41
+
42
+ rng = np.random.default_rng(42); n = len(y); opt = []
43
+ for _ in range(500): # bootstrap optimism (Harrell)
44
+ idx = rng.integers(0, n, n)
45
+ bm = sm.GLM(y[idx], X[idx], family=sm.families.Binomial()).fit()
46
+ s_boot, _ = cal_slope(y[idx], X[idx] @ bm.params) # boot model on boot data (apparent)
47
+ s_orig, _ = cal_slope(y, X @ bm.params) # boot model on original data (test)
48
+ opt.append(s_boot - s_orig)
49
+ corrected_slope = app_slope - float(np.mean(opt)) # < 1.00 when the model overfits
50
+ print(f"apparent slope {app_slope:.3f} -> optimism-corrected {corrected_slope:.3f}")
51
+ ```
52
+
53
+ A corrected slope **< 1** means predictions are too extreme (overfit) and should be shrunk
54
+ (a penalized/uniform-shrinkage refit); a slope **> 1** means they are too moderate.
55
+
56
+ ---
57
+
58
+ ## The four levels of calibration (report weak calibration at least)
59
+
60
+ Van Calster's hierarchy — report at minimum **weak calibration** (intercept + slope):
61
+
62
+ - **Mean** (calibration-in-the-large): mean predicted = observed event rate (the intercept).
63
+ - **Weak**: intercept ≈ 0 **and** slope ≈ 1.
64
+ - **Moderate**: a **flexible calibration curve** (loess / spline of observed on predicted),
65
+ not decile bins — the plot most reviewers now expect.
66
+ - **Strong**: correct per-covariate (rarely achievable; not required).
67
+
68
+ ```python
69
+ import matplotlib.pyplot as plt
70
+ from sklearn.calibration import calibration_curve
71
+ phat = full.predict(X)
72
+ frac_pos, mean_pred = calibration_curve(y, phat, n_bins=10, strategy="quantile")
73
+ plt.plot([0, 1], [0, 1], "--"); plt.plot(mean_pred, frac_pos, "o-") # add a loess curve for moderate
74
+ plt.xlabel("Predicted probability"); plt.ylabel("Observed frequency")
75
+ ```
76
+
77
+ ---
78
+
79
+ ## Brier score; do NOT rely on Hosmer–Lemeshow
80
+
81
+ - **Brier score** = mean squared error of the probabilities; report the **scaled Brier**
82
+ (1 − Brier / Brier_null) so it is interpretable against the event rate.
83
+ - **Hosmer–Lemeshow is deprecated** (Van Calster 2016; Austin & Steyerberg): its p-value depends
84
+ on an arbitrary number of bins, it is underpowered in small samples and rejects trivially in
85
+ large ones, and it gives no direction or magnitude. Report the **calibration slope + intercept
86
+ + a flexible calibration plot** instead of an H–L p-value.
87
+
88
+ ```python
89
+ from sklearn.metrics import brier_score_loss
90
+ brier = brier_score_loss(y, phat); scaled = 1 - brier / (y.mean() * (1 - y.mean()))
91
+ ```
92
+
93
+ ## Survival models
94
+
95
+ For a Cox/survival model, calibrate the **predicted vs observed risk at a fixed horizon**
96
+ (e.g. 3-year): group by predicted-risk decile and compare to a Kaplan–Meier / pseudo-value
97
+ estimate at that time, or use `rms::calibrate` / `pec` in R with bootstrap optimism correction.
98
+ State the horizon; a model can be well-calibrated at 1 year and not at 5.
99
+
100
+ ---
101
+
102
+ ## Reporting
103
+
104
+ - Calibration **intercept and slope** with the **internal-validation method named**
105
+ (bootstrap/CV), not the apparent slope of 1.00; the flexible calibration plot.
106
+ - Scaled Brier; the horizon (survival); the cohort each metric was computed on (development vs
107
+ held-out vs external) stated explicitly.
108
+ - Discrimination **and** calibration together; add decision-curve net benefit for a utility claim
109
+ (see `table-standards/table-types/incremental_value.md` and the `make-figures` decision-curve
110
+ exemplar).
111
+
112
+ ---
113
+
114
+ ## Common failures (flag at review)
115
+
116
+ - **Apparent calibration slope of exactly 1.00** (and intercept 0) with no bootstrap/CV/external
117
+ token — in-sample fit presented as calibration (S7).
118
+ - **Hosmer–Lemeshow p-value** offered as the calibration evidence (deprecated).
119
+ - **Discrimination reported without calibration** for a model meant to guide care (S7 → MAJOR).
120
+ - **Decile-bin calibration only**, no flexible curve; or a survival calibration with no stated
121
+ horizon.
122
+
123
+ ---
124
+
125
+ ## Anti-Hallucination
126
+
127
+ - Never hand-type a calibration slope/intercept, Brier, or CI — compute it from predictions with
128
+ a seeded script.
129
+ - Do not report a calibration slope of 1.00 as evidence of good calibration — it is the
130
+ apparent-fit artifact; report the optimism-corrected value the bootstrap produced.
131
+ - Name the validation source of every calibration number (development / bootstrap-corrected /
132
+ external); do not present development-sample calibration as validated performance.
@@ -56,6 +56,7 @@ A 9-probe checklist for time-to-event outcomes and prognostic model development.
56
56
  - Decision-curve analysis at clinically relevant probability thresholds?
57
57
  - For a prognostic model intended to guide surveillance intensity, treatment intensification, or eligibility for adjuvant therapy, discrimination alone is insufficient. If Methods mention calibration but Results/supplement contain no calibration plot or numeric metrics → MAJOR.
58
58
  - **Apparent-vs-optimism-corrected deterministic tell**: discrimination/calibration reported with no internal-validation token nearby (`optimism`, `bootstrap`, `cross-valid`, `held-out`, `external`) is presumptively **apparent (in-sample) performance**. A **calibration slope reported as exactly 1.00** (and/or Uno's C, Brier, net-benefit all on the development sample) is the in-sample-fit signature — the model was evaluated on the data it was fit to. Flag MINOR (a bootstrap optimism correction is cheap and usually reproduces the estimate); escalate to MAJOR when the model is assessed for clinical utility / net-benefit, where optimistic metrics directly inflate the deployment claim.
59
+ - Produce the fix: `analyze-stats` `references/analysis_guides/calibration.md` has the bootstrap optimism-corrected calibration slope/intercept (the apparent slope is 1.00 by construction), the flexible calibration curve + scaled Brier, and why Hosmer–Lemeshow is dropped.
59
60
 
60
61
  **S8 — Estimand provenance**:
61
62
  - Is the survival estimand stated explicitly and held consistent across Abstract / Methods / Results — event-free survival, cause-specific cumulative incidence, all-cause mortality — and at the subject vs population level? A subdistribution hazard (Fine-Gray) answers a different question than a cause-specific hazard; quoting an sHR for an etiologic claim, or a cause-specific HR for an absolute-risk claim, is an estimand mismatch.
@@ -56,6 +56,7 @@ A 9-probe checklist for time-to-event outcomes and prognostic model development.
56
56
  - Decision-curve analysis at clinically relevant probability thresholds?
57
57
  - For a prognostic model intended to guide surveillance intensity, treatment intensification, or eligibility for adjuvant therapy, discrimination alone is insufficient. If Methods mention calibration but Results/supplement contain no calibration plot or numeric metrics → MAJOR.
58
58
  - **Apparent-vs-optimism-corrected deterministic tell**: discrimination/calibration reported with no internal-validation token nearby (`optimism`, `bootstrap`, `cross-valid`, `held-out`, `external`) is presumptively **apparent (in-sample) performance**. A **calibration slope reported as exactly 1.00** (and/or Uno's C, Brier, net-benefit all on the development sample) is the in-sample-fit signature — the model was evaluated on the data it was fit to. Flag MINOR (a bootstrap optimism correction is cheap and usually reproduces the estimate); escalate to MAJOR when the model is assessed for clinical utility / net-benefit, where optimistic metrics directly inflate the deployment claim.
59
+ - Produce the fix: `analyze-stats` `references/analysis_guides/calibration.md` has the bootstrap optimism-corrected calibration slope/intercept (the apparent slope is 1.00 by construction), the flexible calibration curve + scaled Brier, and why Hosmer–Lemeshow is dropped.
59
60
 
60
61
  **S8 — Estimand provenance**:
61
62
  - Is the survival estimand stated explicitly and held consistent across Abstract / Methods / Results — event-free survival, cause-specific cumulative incidence, all-cause mortality — and at the subject vs population level? A subdistribution hazard (Fine-Gray) answers a different question than a cause-specific hazard; quoting an sHR for an etiologic claim, or a cause-specific HR for an absolute-risk claim, is an estimand mismatch.
@@ -221,7 +221,7 @@ Design all tables and figures BEFORE writing prose. This ensures the narrative s
221
221
 
222
222
  Write the Methods section first -- it is the most objective and anchors the rest of the paper.
223
223
 
224
- **Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/methods.md` for PICO structure, backbone article usage, checklist cross-reference, and terminology conventions. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_methods/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — it lists, paragraph by paragraph, what each Methods paragraph must establish plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
224
+ **Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/methods.md` for PICO structure, backbone article usage, checklist cross-reference, and terminology conventions. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_methods/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020, RCT/CONSORT 2010) — it lists, paragraph by paragraph, what each Methods paragraph must establish plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
225
225
 
226
226
  **Writing order within Methods:**
227
227
  1. Study Design and Setting
@@ -255,7 +255,7 @@ Write the Methods section first -- it is the most objective and anchors the rest
255
255
  Write Results aligned to the approved tables and figures. **Results = "What did we find?"
256
256
  — nothing more.** Every sentence must be a factual statement backed by a number.
257
257
 
258
- **Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/results.md` for mirror-symmetry rules, flowchart requirements, missing data handling, and the anti-interpretation self-check. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_results/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — each follows its `exemplar_methods/` sibling in Methods order, listing what each Results paragraph must establish (flow → baseline/prevalence → primary estimate with CIs → calibration/agreement → subgroups → sensitivity; for meta-analysis, PRISMA flow → characteristics+provenance → RoB → pooled estimate with I²/τ²/prediction interval → subgroup interaction → publication bias) plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
258
+ **Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/results.md` for mirror-symmetry rules, flowchart requirements, missing data handling, and the anti-interpretation self-check. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_results/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020, RCT/CONSORT 2010) — each follows its `exemplar_methods/` sibling in Methods order, listing what each Results paragraph must establish (flow → baseline/prevalence → primary estimate with CIs → calibration/agreement → subgroups → sensitivity; for meta-analysis, PRISMA flow → characteristics+provenance → RoB → pooled estimate with I²/τ²/prediction interval → subgroup interaction → publication bias; for an RCT, CONSORT flow → baseline-by-arm with no p-values → ITT primary with CI → secondary+harms → per-protocol beside ITT) plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
259
259
 
260
260
  **Rules:**
261
261
  - Every number in the text must match the corresponding table cell exactly.
@@ -296,7 +296,7 @@ Write Results aligned to the approved tables and figures. **Results = "What did
296
296
 
297
297
  ### Phase 5: Discussion
298
298
 
299
- **Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/discussion.md` for the 4-paragraph structure, word limits, limitation writing guidelines, and Table/Figure citation rules. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_discussion/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — completing the exemplar trio, each lists what every Discussion paragraph must establish (key finding → interpretation/comparison → limitations → generalizability → conclusion matched to the evidence) plus the element that type most often omits (spectrum/verification bias; evidence-tier separation and optimism caveats; mandatory causal caution; for meta-analysis, GRADE certainty + heterogeneity source + non-independence/overlap caveat). For case reports, use `${CLAUDE_SKILL_DIR}/references/exemplar_case_report.md` instead: it controls literature-boundary wording, n=1 causal caution, and bedside teaching-point framing. Model the structure; the exemplars are synthetic, introduce no new results, and are not prose to copy.
299
+ **Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/discussion.md` for the 4-paragraph structure, word limits, limitation writing guidelines, and Table/Figure citation rules. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_discussion/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020, RCT/CONSORT 2010) — completing the exemplar trio, each lists what every Discussion paragraph must establish (key finding → interpretation/comparison → limitations → generalizability → conclusion matched to the evidence) plus the element that type most often omits (spectrum/verification bias; evidence-tier separation and optimism caveats; mandatory causal caution; for meta-analysis, GRADE certainty + heterogeneity source + non-independence/overlap caveat; for an RCT, blinding/attrition limitation + clinical-vs-statistical significance vs the MCID). For case reports, use `${CLAUDE_SKILL_DIR}/references/exemplar_case_report.md` instead: it controls literature-boundary wording, n=1 causal caution, and bedside teaching-point framing. Model the structure; the exemplars are synthetic, introduce no new results, and are not prose to copy.
300
300
 
301
301
  **Before drafting, collect user input (Discussion Planning Gate).**
302
302
 
@@ -30,6 +30,8 @@ closes with a conclusion matched to the evidence — introducing no new results.
30
30
  - `meta_analysis_prisma.md` — summary-of-evidence with certainty framing, heterogeneity-source
31
31
  and non-independence/overlap caveats, GRADE limitations, no guideline-grade conclusion
32
32
  (PRISMA 2020).
33
+ - `rct_consort.md` — primary-on-ITT key finding, clinical-vs-statistical significance vs the MCID,
34
+ blinding/attrition limitations, single-trial conclusion (CONSORT 2010).
33
35
 
34
36
  ## Curator guidelines (for adding more)
35
37
 
@@ -0,0 +1,42 @@
1
+ # Discussion structure — randomized controlled trial (CONSORT 2010)
2
+
3
+ A structure model for the Discussion of a parallel-group RCT, completing the trio with the
4
+ `exemplar_methods/` and `exemplar_results/` siblings. Each heading is a paragraph; each bullet is
5
+ *what it must establish*. Fill the `[brackets]`; do not copy this text. Follows
6
+ `section_guides/discussion.md`. Introduces **no new results** and cites **no tables/figures**.
7
+
8
+ ## Paragraph 1 — Key finding
9
+ - Restate the **primary** result (the effect estimate + CI on the ITT population) in 2–3
10
+ sentences, matched to Results — no new estimate, and no elevation of a secondary outcome.
11
+ - One sentence on clinical meaning, tied to the **pre-stated MCID** (a statistically significant
12
+ effect smaller than the MCID is not clinically meaningful).
13
+ - **If the primary is null or non-inferiority is not shown, frame it by the CI vs the margin**,
14
+ not as "no difference" — state what the interval excludes.
15
+
16
+ ## Paragraphs 2–3 — Interpretation and comparison
17
+ - Compare to prior trials and meta-analyses (agree / differ and why: population, comparator,
18
+ dose, endpoint); avoid over-generalizing beyond the enrolled population.
19
+ - Discuss mechanism / plausibility as hypothesis; keep secondary and subgroup findings explicitly
20
+ exploratory (do not narrate a subgroup as if it were a confirmatory result).
21
+
22
+ ## Limitations
23
+ - **Risk-of-bias honesty**: open-label / unblinded assessment, attrition and how missing primary
24
+ data were handled, any post-randomization exclusions, early stopping and its effect on the
25
+ estimate, single-centre / narrow eligibility limiting external validity.
26
+ - Whether the trial was powered for the primary only (secondary/subgroup analyses under-powered).
27
+
28
+ ## Generalizability
29
+ - The population and setting the result applies to, and where it may not transfer (eligibility
30
+ restrictions, standard-of-care comparator specific to the setting, adherence in a trial vs
31
+ routine care).
32
+
33
+ ## Conclusion
34
+ - Matched to the evidence — a single trial supports the primary contrast in the studied
35
+ population; use "in this trial, [intervention] reduced/did not reduce [outcome]", not a
36
+ guideline-grade recommendation the evidence tier does not license.
37
+
38
+ ## Common omission
39
+ - An explicit **blinding / attrition limitation and a clinical-vs-statistical-significance caveat**
40
+ tied to the MCID — the Discussion elements RCT drafts most often soften. Cross-reference
41
+ `section_guides/discussion.md`, the `peer-review/references/domain-probes/rct_trial.md` probes,
42
+ and, for an AI intervention, the CONSORT-AI reporting items.
@@ -24,6 +24,7 @@ fit). A missing element is a gap to fill before drafting Results.
24
24
  - `ai_validation_tripod_claim.md` — an AI/ML model development + validation study (TRIPOD+AI / CLAIM).
25
25
  - `observational_cohort_strobe.md` — an exposure→outcome observational cohort (STROBE).
26
26
  - `meta_analysis_prisma.md` — a systematic review with quantitative synthesis (PRISMA 2020).
27
+ - `rct_consort.md` — a parallel-group randomized controlled trial (CONSORT 2010).
27
28
 
28
29
  ## Curator guidelines (for adding more)
29
30
 
@@ -0,0 +1,55 @@
1
+ # Methods structure — randomized controlled trial (CONSORT 2010)
2
+
3
+ A structure model for a parallel-group RCT. Each heading is a paragraph; each bullet is *what it
4
+ must establish*. Fill the `[brackets]`; do not copy this text. Anchors to CONSORT 2010; pairs the
5
+ `peer-review/references/domain-probes/rct_trial.md` probes (and CONSORT-AI/SPIRIT-AI for an AI
6
+ intervention).
7
+
8
+ ## Trial design and registration
9
+ - Design (parallel-group / crossover / cluster), allocation ratio (e.g. 1:1), and framework
10
+ (superiority / non-inferiority / equivalence) — with the **margin** stated up front for the
11
+ latter two.
12
+ - **Prospective trial registration** (ClinicalTrials.gov / a WHO ICTRP registry) with the number,
13
+ and the protocol reference; ethics approval and informed consent.
14
+
15
+ ## Participants, setting, interventions
16
+ - Eligibility as a numbered inclusion / exclusion list; the settings and the recruitment and
17
+ follow-up dates.
18
+ - The experimental and comparator interventions in enough detail to replicate (dose, schedule,
19
+ who delivered them); the comparator is a real, defined control (placebo / standard-of-care),
20
+ not a strawman.
21
+
22
+ ## Outcomes
23
+ - The **single pre-specified primary outcome** with its metric and time point (do not re-designate
24
+ the primary after seeing data); secondary outcomes listed; any changes to outcomes after trial
25
+ commencement disclosed with dates.
26
+
27
+ ## Sample size
28
+ - The power calculation: assumed effect (the MCID, not the hoped-for effect), α, power, and the
29
+ resulting N with attrition inflation; for non-inferiority, the margin and its justification.
30
+
31
+ ## Randomization, allocation concealment, blinding
32
+ - **Sequence generation** (how the random sequence was produced), **allocation concealment**
33
+ (the mechanism that hid the next assignment — central/pharmacy/sealed opaque envelopes), and
34
+ **implementation** (who enrolled, who assigned) — three distinct items, all required.
35
+ - **Blinding**: who was blinded (participants, clinicians, outcome assessors, analysts) and how;
36
+ if open-label, state it and how assessor blinding was preserved.
37
+
38
+ ## Statistical methods
39
+ - The primary analysis and estimand: the contrast, the population (**intention-to-treat** as
40
+ primary — everyone as randomized; per-protocol as a secondary, and the primary population for a
41
+ non-inferiority trial stated), and how missing data were handled (not naive complete-case for a
42
+ dropout-prone endpoint).
43
+ - Pre-specified subgroups and interim analyses / stopping rules; multiplicity handling; software.
44
+
45
+ ## Reporting-guideline fit
46
+ - CONSORT 2010 (+ CONSORT-AI for an AI intervention). Critical items: registration, sequence
47
+ generation + allocation concealment + blinding, the pre-specified primary, ITT, and a
48
+ participant flow diagram that reconciles (randomized → received → analyzed).
49
+
50
+ ## Common omission
51
+ - **Allocation concealment described as blinding** (they are different — concealment protects
52
+ *assignment*, blinding protects *ascertainment*), and an unstated **ITT vs per-protocol**
53
+ primary population — the Methods elements RCT drafts most often conflate or skip, and the first
54
+ a trials reviewer checks. Cross-reference `section_guides/methods.md` and the
55
+ `peer-review/references/domain-probes/rct_trial.md` probes.
@@ -30,6 +30,8 @@ element that interprets rather than reports belongs in the Discussion.
30
30
  - `meta_analysis_prisma.md` — PRISMA flow, study/patient characteristics + provenance, pooled
31
31
  estimate with I²/τ²/prediction interval, subgroup interaction, sensitivity, publication bias
32
32
  (PRISMA 2020).
33
+ - `rct_consort.md` — CONSORT flow, baseline by arm (no p-values), ITT primary estimate with CI,
34
+ secondary outcomes + harms, subgroup interaction, per-protocol beside ITT (CONSORT 2010).
33
35
 
34
36
  ## Curator guidelines (for adding more)
35
37
 
@@ -0,0 +1,38 @@
1
+ # Results structure — randomized controlled trial (CONSORT 2010)
2
+
3
+ A structure model for the Results of a parallel-group RCT. It follows its Methods CONSORT
4
+ sibling in Methods order. Each heading is a paragraph; each bullet is *what it must establish*.
5
+ Fill the `[brackets]`; do not copy this text. Report findings only — no interpretation.
6
+
7
+ ## Participant flow (Figure 1 — CONSORT diagram)
8
+ - The CONSORT flow: assessed → randomized → allocated (received / did not receive) → followed up
9
+ (lost, discontinued, with reasons) → analyzed, **per arm**, with counts.
10
+ - The cascade reconciles (randomized = Σ analyzed + Σ excluded-with-reason); the number
11
+ **analyzed for the primary outcome** matches the ITT denominator. Numbers match the Abstract,
12
+ Methods, and Table 1.
13
+
14
+ ## Recruitment and baseline (Table 1)
15
+ - Recruitment and follow-up dates and why the trial ended (target reached / stopped);
16
+ Table 1 baseline characteristics **by arm** — **no p-values for baseline balance** (randomization
17
+ makes them meaningless; show the descriptive comparison and, if used, the balance metric).
18
+
19
+ ## Primary outcome
20
+ - The pre-specified **primary** outcome by arm: the **effect estimate with its 95% CI** and the
21
+ absolute numbers per arm (events/N or mean±SD), on the **ITT** population — not just a p-value.
22
+ - For non-inferiority: the CI relative to the **pre-stated margin** (the conclusion follows from
23
+ where the CI sits, not from a significance test).
24
+
25
+ ## Secondary outcomes and harms
26
+ - Secondary outcomes with estimates + CIs, flagged as secondary (multiplicity in view — do not
27
+ elevate a significant secondary to a headline).
28
+ - **Harms / adverse events** reported per arm (a trial that reports only efficacy is incomplete).
29
+
30
+ ## Subgroups and sensitivity
31
+ - Pre-specified subgroup analyses with the **interaction test**, framed as exploratory; the
32
+ per-protocol analysis beside ITT (concordance is reassuring; divergence is discussed, not hidden).
33
+
34
+ ## Common omission
35
+ - **Baseline p-values** (should not appear), and the **primary reported off the per-protocol set**
36
+ rather than ITT, or without its absolute per-arm numbers — the Results elements RCT drafts most
37
+ often get wrong. Cross-reference `section_guides/results.md`, the
38
+ `peer-review/references/domain-probes/rct_trial.md` probes, and the CONSORT flow requirement.