medsci-skills 5.11.0 → 5.12.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/metadata/distribution_files.json +49 -24
- package/metadata/distribution_manifest.json +1 -1
- package/package.json +1 -1
- package/skills/analyze-stats/SKILL.md +1 -0
- package/skills/analyze-stats/references/analysis_guides/agreement_reliability.md +130 -0
- package/skills/peer-review/SKILL.md +11 -0
- package/skills/peer-review/references/domain-probes/observational_confounding.md +1 -0
- package/skills/self-review/SKILL.md +9 -0
- package/skills/self-review/references/domain-probes/observational_confounding.md +1 -0
- package/skills/self-review/scripts/check_claim_artifact.py +52 -1
- package/skills/self-review/scripts/check_cv_leakage.py +139 -0
- package/skills/self-review/skill.yml +1 -0
- package/skills/self-review/tests/fixtures/claim_manuscript_single_primary.md +3 -0
- package/skills/self-review/tests/fixtures/claim_scripts_consistent/05_primary_cox.R +3 -0
- package/skills/self-review/tests/fixtures/claim_scripts_coprimary/05_primary_cox.R +3 -0
- package/skills/self-review/tests/fixtures/cv_leakage_bad.md +2 -0
- package/skills/self-review/tests/fixtures/cv_leakage_clean.md +2 -0
- package/skills/self-review/tests/test_claim_artifact.sh +15 -0
- package/skills/self-review/tests/test_cv_leakage.sh +35 -0
- package/skills/write-paper/SKILL.md +3 -3
- package/skills/write-paper/references/exemplar_discussion/README.md +3 -0
- package/skills/write-paper/references/exemplar_discussion/meta_analysis_prisma.md +52 -0
- package/skills/write-paper/references/exemplar_methods/README.md +1 -0
- package/skills/write-paper/references/exemplar_methods/meta_analysis_prisma.md +65 -0
- package/skills/write-paper/references/exemplar_results/README.md +3 -0
- package/skills/write-paper/references/exemplar_results/meta_analysis_prisma.md +51 -0
- package/skills/write-paper/references/paper_types/meta_analysis.md +5 -0
|
@@ -153,8 +153,13 @@
|
|
|
153
153
|
},
|
|
154
154
|
{
|
|
155
155
|
"path": "skills/analyze-stats/SKILL.md",
|
|
156
|
-
"size":
|
|
157
|
-
"sha256": "
|
|
156
|
+
"size": 55642,
|
|
157
|
+
"sha256": "5c4a966490cfb8da6e09ebda897599479bd527d7ee3957ccbfceb5abff06306a"
|
|
158
|
+
},
|
|
159
|
+
{
|
|
160
|
+
"path": "skills/analyze-stats/references/analysis_guides/agreement_reliability.md",
|
|
161
|
+
"size": 6674,
|
|
162
|
+
"sha256": "599ad2551f547043b2bec79e3139d9c2c3ae33c5025784359b22022f882504c2"
|
|
158
163
|
},
|
|
159
164
|
{
|
|
160
165
|
"path": "skills/analyze-stats/references/analysis_guides/health_economic_evaluation.md",
|
|
@@ -2923,8 +2928,8 @@
|
|
|
2923
2928
|
},
|
|
2924
2929
|
{
|
|
2925
2930
|
"path": "skills/peer-review/SKILL.md",
|
|
2926
|
-
"size":
|
|
2927
|
-
"sha256": "
|
|
2931
|
+
"size": 69384,
|
|
2932
|
+
"sha256": "94033b84b41304d7dbd7919c84c7e9502bdaff1b7e2ef8f5dd7ec9cd7147c532"
|
|
2928
2933
|
},
|
|
2929
2934
|
{
|
|
2930
2935
|
"path": "skills/peer-review/references/aczel_2021_reviewer2_patterns.md",
|
|
@@ -2993,8 +2998,8 @@
|
|
|
2993
2998
|
},
|
|
2994
2999
|
{
|
|
2995
3000
|
"path": "skills/peer-review/references/domain-probes/observational_confounding.md",
|
|
2996
|
-
"size":
|
|
2997
|
-
"sha256": "
|
|
3001
|
+
"size": 34795,
|
|
3002
|
+
"sha256": "ff8cb910c7fb83ec08a04ba88866bdf1e0d2f43d972e2861b825595a85b035ee"
|
|
2998
3003
|
},
|
|
2999
3004
|
{
|
|
3000
3005
|
"path": "skills/peer-review/references/domain-probes/polygenic_risk_score.md",
|
|
@@ -3488,8 +3493,8 @@
|
|
|
3488
3493
|
},
|
|
3489
3494
|
{
|
|
3490
3495
|
"path": "skills/self-review/SKILL.md",
|
|
3491
|
-
"size":
|
|
3492
|
-
"sha256": "
|
|
3496
|
+
"size": 104731,
|
|
3497
|
+
"sha256": "9c74f317d10c7b491e710debb619dcf536689092a59fe749050a0ad41f4aacd5"
|
|
3493
3498
|
},
|
|
3494
3499
|
{
|
|
3495
3500
|
"path": "skills/self-review/references/domain-probes/ai_overclaiming.md",
|
|
@@ -3553,8 +3558,8 @@
|
|
|
3553
3558
|
},
|
|
3554
3559
|
{
|
|
3555
3560
|
"path": "skills/self-review/references/domain-probes/observational_confounding.md",
|
|
3556
|
-
"size":
|
|
3557
|
-
"sha256": "
|
|
3561
|
+
"size": 34795,
|
|
3562
|
+
"sha256": "ff8cb910c7fb83ec08a04ba88866bdf1e0d2f43d972e2861b825595a85b035ee"
|
|
3558
3563
|
},
|
|
3559
3564
|
{
|
|
3560
3565
|
"path": "skills/self-review/references/domain-probes/polygenic_risk_score.md",
|
|
@@ -3668,8 +3673,8 @@
|
|
|
3668
3673
|
},
|
|
3669
3674
|
{
|
|
3670
3675
|
"path": "skills/self-review/scripts/check_claim_artifact.py",
|
|
3671
|
-
"size":
|
|
3672
|
-
"sha256": "
|
|
3676
|
+
"size": 18475,
|
|
3677
|
+
"sha256": "c99c8090205734b0eae981550ed09e6275d94add608e66df78f4533c35f20924"
|
|
3673
3678
|
},
|
|
3674
3679
|
{
|
|
3675
3680
|
"path": "skills/self-review/scripts/check_classical_style.py",
|
|
@@ -3686,6 +3691,11 @@
|
|
|
3686
3691
|
"size": 21506,
|
|
3687
3692
|
"sha256": "7d3e67074d58a28ffee52ce64b486231f103a3ddcaf6b3b6ee83ba5f89c63bc2"
|
|
3688
3693
|
},
|
|
3694
|
+
{
|
|
3695
|
+
"path": "skills/self-review/scripts/check_cv_leakage.py",
|
|
3696
|
+
"size": 6349,
|
|
3697
|
+
"sha256": "431c5f14ad2c4f59e85b2a5e86150ac708d476e8f2f1da859d81953f522b2117"
|
|
3698
|
+
},
|
|
3689
3699
|
{
|
|
3690
3700
|
"path": "skills/self-review/scripts/check_editorial_impression.py",
|
|
3691
3701
|
"size": 22032,
|
|
@@ -3743,8 +3753,8 @@
|
|
|
3743
3753
|
},
|
|
3744
3754
|
{
|
|
3745
3755
|
"path": "skills/self-review/skill.yml",
|
|
3746
|
-
"size":
|
|
3747
|
-
"sha256": "
|
|
3756
|
+
"size": 2381,
|
|
3757
|
+
"sha256": "9499a87d985d981eac9418d6cd95ed79c1d9724aba8854b7d6713d89679d9f1d"
|
|
3748
3758
|
},
|
|
3749
3759
|
{
|
|
3750
3760
|
"path": "skills/setup-medsci/SKILL.md",
|
|
@@ -3898,8 +3908,8 @@
|
|
|
3898
3908
|
},
|
|
3899
3909
|
{
|
|
3900
3910
|
"path": "skills/write-paper/SKILL.md",
|
|
3901
|
-
"size":
|
|
3902
|
-
"sha256": "
|
|
3911
|
+
"size": 66771,
|
|
3912
|
+
"sha256": "1a50c9ceb040a79feaa60e809c7196f9f04b85cab2f5f3402dcaed3f7aee552c"
|
|
3903
3913
|
},
|
|
3904
3914
|
{
|
|
3905
3915
|
"path": "skills/write-paper/references/exemplar_abstract.md",
|
|
@@ -3918,8 +3928,8 @@
|
|
|
3918
3928
|
},
|
|
3919
3929
|
{
|
|
3920
3930
|
"path": "skills/write-paper/references/exemplar_discussion/README.md",
|
|
3921
|
-
"size":
|
|
3922
|
-
"sha256": "
|
|
3931
|
+
"size": 2727,
|
|
3932
|
+
"sha256": "3bfcd4eb4df4ca0309446a606550cb11adc4d88daa676f2217249691d2e6b042"
|
|
3923
3933
|
},
|
|
3924
3934
|
{
|
|
3925
3935
|
"path": "skills/write-paper/references/exemplar_discussion/ai_validation_tripod_claim.md",
|
|
@@ -3931,6 +3941,11 @@
|
|
|
3931
3941
|
"size": 2500,
|
|
3932
3942
|
"sha256": "7d49d926eb9aa78c42ce0e6579ca299f528dec77dafc17ed9451ada0c2340b3c"
|
|
3933
3943
|
},
|
|
3944
|
+
{
|
|
3945
|
+
"path": "skills/write-paper/references/exemplar_discussion/meta_analysis_prisma.md",
|
|
3946
|
+
"size": 3357,
|
|
3947
|
+
"sha256": "3cc8542c754372b19acbc25b3c6d6211479b4bbec361c7638c5d2550db3ecc5f"
|
|
3948
|
+
},
|
|
3934
3949
|
{
|
|
3935
3950
|
"path": "skills/write-paper/references/exemplar_discussion/observational_cohort_strobe.md",
|
|
3936
3951
|
"size": 3652,
|
|
@@ -3943,8 +3958,8 @@
|
|
|
3943
3958
|
},
|
|
3944
3959
|
{
|
|
3945
3960
|
"path": "skills/write-paper/references/exemplar_methods/README.md",
|
|
3946
|
-
"size":
|
|
3947
|
-
"sha256": "
|
|
3961
|
+
"size": 2342,
|
|
3962
|
+
"sha256": "6966e8b723d0c4ccc012d54e2b8a22ef59b39c22619e63a8597f9badb380b4e7"
|
|
3948
3963
|
},
|
|
3949
3964
|
{
|
|
3950
3965
|
"path": "skills/write-paper/references/exemplar_methods/ai_validation_tripod_claim.md",
|
|
@@ -3956,6 +3971,11 @@
|
|
|
3956
3971
|
"size": 2739,
|
|
3957
3972
|
"sha256": "f403613a0f3625c5c7dab8fa16a21635007821ede2b58d265440db51b55165a6"
|
|
3958
3973
|
},
|
|
3974
|
+
{
|
|
3975
|
+
"path": "skills/write-paper/references/exemplar_methods/meta_analysis_prisma.md",
|
|
3976
|
+
"size": 4160,
|
|
3977
|
+
"sha256": "a0e696e2f581fa6ae8b1aeb86ff7b9a5aaa6bdb2184fee4352e74d2ee6559d90"
|
|
3978
|
+
},
|
|
3959
3979
|
{
|
|
3960
3980
|
"path": "skills/write-paper/references/exemplar_methods/observational_cohort_strobe.md",
|
|
3961
3981
|
"size": 2364,
|
|
@@ -3963,8 +3983,8 @@
|
|
|
3963
3983
|
},
|
|
3964
3984
|
{
|
|
3965
3985
|
"path": "skills/write-paper/references/exemplar_results/README.md",
|
|
3966
|
-
"size":
|
|
3967
|
-
"sha256": "
|
|
3986
|
+
"size": 2799,
|
|
3987
|
+
"sha256": "14b588f1f4433be71c39f11627148fff797eb87b3b74be75e61514095d4c38ea"
|
|
3968
3988
|
},
|
|
3969
3989
|
{
|
|
3970
3990
|
"path": "skills/write-paper/references/exemplar_results/ai_validation_tripod_claim.md",
|
|
@@ -3976,6 +3996,11 @@
|
|
|
3976
3996
|
"size": 2772,
|
|
3977
3997
|
"sha256": "8e45e0b30b4ab505b4eb8c37007f970bfb5aa81375fd011915b07d85c9419598"
|
|
3978
3998
|
},
|
|
3999
|
+
{
|
|
4000
|
+
"path": "skills/write-paper/references/exemplar_results/meta_analysis_prisma.md",
|
|
4001
|
+
"size": 3170,
|
|
4002
|
+
"sha256": "8aed3ee5d36c15515843d77b5f69fe76f433c9972827a9e18303b38802fac782"
|
|
4003
|
+
},
|
|
3979
4004
|
{
|
|
3980
4005
|
"path": "skills/write-paper/references/exemplar_results/observational_cohort_strobe.md",
|
|
3981
4006
|
"size": 2452,
|
|
@@ -4288,8 +4313,8 @@
|
|
|
4288
4313
|
},
|
|
4289
4314
|
{
|
|
4290
4315
|
"path": "skills/write-paper/references/paper_types/meta_analysis.md",
|
|
4291
|
-
"size":
|
|
4292
|
-
"sha256": "
|
|
4316
|
+
"size": 10536,
|
|
4317
|
+
"sha256": "2c237409005214baab63627966d8f2b31498e3c28bea34c571a6d5591a068ed8"
|
|
4293
4318
|
},
|
|
4294
4319
|
{
|
|
4295
4320
|
"path": "skills/write-paper/references/paper_types/nhis_cohort.md",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "medsci-skills",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.12.0",
|
|
4
4
|
"description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE",
|
|
6
6
|
"homepage": "https://github.com/Aperivue/medsci-skills#readme",
|
|
@@ -411,6 +411,7 @@ tbl %>% as_flex_table() %>% flextable::save_as_docx(path = "table.docx")
|
|
|
411
411
|
|
|
412
412
|
### Inter-rater Agreement
|
|
413
413
|
|
|
414
|
+
- **Methodology guide**: `references/analysis_guides/agreement_reliability.md` (**load before generating code** — the pseudoreplication trap for clustered/repeated measurements + the pseudoreplication-safe per-subject / mixed-effects code, ICC model/type selection, agreement-vs-reliability distinction; pairs with self-review probe O18)
|
|
414
415
|
- Table type guide: `references/table-standards/table-types/agreement.md` (ICC with model/type + CI, weighted κ for ordinal, Bland–Altman bias + LoA, reliability-vs-agreement distinction, common errors)
|
|
415
416
|
- Template: `references/templates/agreement_analysis.py`
|
|
416
417
|
- 2 raters + categorical: Cohen's kappa
|
|
@@ -0,0 +1,130 @@
|
|
|
1
|
+
# Inter-rater Agreement & Reliability Guide
|
|
2
|
+
|
|
3
|
+
Quantifying how well two or more raters (or a rater and a reference, or repeated
|
|
4
|
+
measurements) **agree**. The coefficient is easy to compute; the two ways these
|
|
5
|
+
analyses fail review are (1) treating **clustered** measurements as independent
|
|
6
|
+
(pseudoreplication) and (2) confusing **agreement** with **reliability**.
|
|
7
|
+
|
|
8
|
+
---
|
|
9
|
+
|
|
10
|
+
## When to Use
|
|
11
|
+
|
|
12
|
+
- **Cohen's kappa** — 2 raters, categorical (nominal) labels.
|
|
13
|
+
- **Weighted kappa** — 2 raters, **ordinal** labels (linear or quadratic weights; disagreement
|
|
14
|
+
by one category counts less than by three).
|
|
15
|
+
- **Fleiss' kappa** — ≥3 raters, categorical.
|
|
16
|
+
- **Krippendorff's alpha** — any number of raters, any measurement level, tolerates missing data.
|
|
17
|
+
- **ICC (intraclass correlation)** — **continuous** measurements; report the model + type (below).
|
|
18
|
+
- **Bland–Altman** — two continuous methods/raters: bias (mean difference) + 95% limits of agreement.
|
|
19
|
+
- NOT for: a single 2×2 vs a reference standard (that is diagnostic accuracy — see
|
|
20
|
+
`table-types/diagnostic_accuracy.md`); not for a multi-reader AI-vs-human comparison with reader +
|
|
21
|
+
case variance (that is an MRMC reader study — see `table-types/reader_study.md`).
|
|
22
|
+
|
|
23
|
+
---
|
|
24
|
+
|
|
25
|
+
## Pseudoreplication comes first (this, not the coefficient, is the issue)
|
|
26
|
+
|
|
27
|
+
If each **subject contributes more than one measurement** — several lesions, aneurysms, nodules,
|
|
28
|
+
slices, or time-points per patient — the rows are **not independent**. Computing agreement on the
|
|
29
|
+
**pooled** rows (or on all pairwise distances) uses an inflated *n*, narrows the CI, and gives an
|
|
30
|
+
**anti-conservative** p-value. This is the single most common reliability-study error a reviewer
|
|
31
|
+
catches (it is flagged by the self-review probe **O18** in `observational_confounding.md`).
|
|
32
|
+
|
|
33
|
+
Two correct paths — pick one and state it:
|
|
34
|
+
|
|
35
|
+
1. **Aggregate to the independent unit first**, then compute agreement per subject. This is the
|
|
36
|
+
simplest defensible analysis when a per-subject summary is meaningful (e.g. mean measurement,
|
|
37
|
+
majority label, or one index lesion per subject).
|
|
38
|
+
2. **Model the clustering** — a mixed-effects / variance-components ICC with a **subject random
|
|
39
|
+
effect** (or a GEE with an exchangeable working correlation), so the within-subject correlation
|
|
40
|
+
is estimated rather than ignored.
|
|
41
|
+
|
|
42
|
+
A pooled-pairwise test can *flip* on correction: e.g. Mann–Whitney p = 0.02 on 448 pooled
|
|
43
|
+
pairwise distances became p = 0.59 at the per-aneurysm level (n = 112). **Report the unit of
|
|
44
|
+
analysis explicitly**, and when subjects have multiple measurements report a per-subject
|
|
45
|
+
sensitivity analysis.
|
|
46
|
+
|
|
47
|
+
### Produce the pseudoreplication-safe version
|
|
48
|
+
|
|
49
|
+
```python
|
|
50
|
+
import pandas as pd
|
|
51
|
+
import pingouin as pg # ICC with model/type + CI
|
|
52
|
+
|
|
53
|
+
# long format: one row per (subject, measurement); rater columns rater1..raterK
|
|
54
|
+
df = pd.read_csv("ratings.csv")
|
|
55
|
+
|
|
56
|
+
# 1) DETECT clustering: more rows than independent subjects
|
|
57
|
+
n_rows, n_subjects = len(df), df["subject_id"].nunique()
|
|
58
|
+
if n_rows > n_subjects:
|
|
59
|
+
print(f"CLUSTERED: {n_rows} measurements from {n_subjects} subjects "
|
|
60
|
+
f"({n_rows / n_subjects:.1f} per subject) — do NOT pool as independent.")
|
|
61
|
+
|
|
62
|
+
# 2a) PER-SUBJECT AGGREGATION (continuous): mean per subject, then ICC on subject means
|
|
63
|
+
per_subj = df.groupby("subject_id")[["rater1", "rater2"]].mean().reset_index()
|
|
64
|
+
long = per_subj.melt(id_vars="subject_id", var_name="rater", value_name="score")
|
|
65
|
+
icc = pg.intraclass_corr(data=long, targets="subject_id", raters="rater", nan_policy="omit")
|
|
66
|
+
print(icc[["Type", "ICC", "CI95%"]]) # report Type (e.g. ICC2/ICC2k) + CI
|
|
67
|
+
|
|
68
|
+
# 2b) OR MODEL THE CLUSTERING (keep every measurement, subject random effect)
|
|
69
|
+
import statsmodels.formula.api as smf
|
|
70
|
+
df_long = df.melt(id_vars="subject_id", value_vars=["rater1", "rater2"],
|
|
71
|
+
var_name="rater", value_name="score")
|
|
72
|
+
m = smf.mixedlm("score ~ 1", data=df_long, groups=df_long["subject_id"])
|
|
73
|
+
res = m.fit()
|
|
74
|
+
var_between = float(res.cov_re.iloc[0, 0]); var_resid = float(res.scale)
|
|
75
|
+
icc_clustered = var_between / (var_between + var_resid)
|
|
76
|
+
print(f"variance-components ICC (subject random effect) = {icc_clustered:.3f}")
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
---
|
|
80
|
+
|
|
81
|
+
## ICC: state the model and the type (they are not interchangeable)
|
|
82
|
+
|
|
83
|
+
- **Model**: one-way random (raters differ per subject), two-way random (same raters, generalise to
|
|
84
|
+
a rater population), two-way mixed (same raters, these raters only).
|
|
85
|
+
- **Type**: **agreement** vs **consistency** (agreement penalises systematic rater bias; consistency
|
|
86
|
+
does not), and **single** vs **average** measurement (average-of-k is higher — only report it if
|
|
87
|
+
the clinical use averages k raters).
|
|
88
|
+
- Report as e.g. **ICC(2,1) = 0.82 (95% CI 0.74–0.88), two-way random, absolute agreement, single
|
|
89
|
+
rater**. An ICC with no model/type is not interpretable.
|
|
90
|
+
|
|
91
|
+
---
|
|
92
|
+
|
|
93
|
+
## Agreement is not reliability
|
|
94
|
+
|
|
95
|
+
- **Agreement** = do raters give the *same value* (absolute; Bland–Altman bias, absolute-agreement ICC).
|
|
96
|
+
- **Reliability** = can raters *rank/discriminate subjects* consistently (relative; consistency ICC,
|
|
97
|
+
Pearson/Spearman). A method can be highly reliable yet have poor agreement (a constant offset).
|
|
98
|
+
State which one the clinical claim needs, and use the matching coefficient.
|
|
99
|
+
|
|
100
|
+
---
|
|
101
|
+
|
|
102
|
+
## Reporting
|
|
103
|
+
|
|
104
|
+
- The coefficient **with a 95% CI** (bootstrap or analytic), the model/type (for ICC), and the
|
|
105
|
+
**unit of analysis** (per-subject vs per-lesion, and the clustering handling).
|
|
106
|
+
- The interpretation band used (e.g. Landis–Koch), but do not over-interpret a point estimate whose
|
|
107
|
+
CI spans two bands.
|
|
108
|
+
- For continuous methods: Bland–Altman **bias + 95% limits of agreement**, not just a correlation.
|
|
109
|
+
|
|
110
|
+
---
|
|
111
|
+
|
|
112
|
+
## Common failures (flag at review)
|
|
113
|
+
|
|
114
|
+
- **Pooled/pairwise agreement on clustered data** (pseudoreplication) — the headline coefficient's
|
|
115
|
+
CI is too narrow; re-run per-subject or with a subject random effect (probe O18).
|
|
116
|
+
- **ICC reported with no model/type** — uninterpretable; the same data yields different ICCs.
|
|
117
|
+
- **Reliability coefficient used to claim agreement** (or vice versa) — a high consistency ICC does
|
|
118
|
+
not establish that the two methods are interchangeable.
|
|
119
|
+
- **Correlation (r) reported as agreement** for two methods — r ignores a constant/proportional bias;
|
|
120
|
+
Bland–Altman is required.
|
|
121
|
+
- **Kappa on ordinal labels unweighted** — treats a one-category disagreement as a full disagreement.
|
|
122
|
+
|
|
123
|
+
---
|
|
124
|
+
|
|
125
|
+
## Anti-Hallucination
|
|
126
|
+
|
|
127
|
+
- Never hand-type a coefficient or CI — compute it from the ratings CSV with a seeded script.
|
|
128
|
+
- Do not quote an ICC without the model/type actually estimated by the code.
|
|
129
|
+
- If subjects have multiple measurements, the per-subject sensitivity analysis is **mandatory** —
|
|
130
|
+
do not report only the pooled number.
|
|
@@ -118,6 +118,17 @@ within the current submission (poolability of incommensurable studies, a broken
|
|
|
118
118
|
evaluation instrument). When both classes are present, the **unfixable** class governs the recommendation —
|
|
119
119
|
do not let a long list of fixable items reframe an unfixable core as "addressable in revision."
|
|
120
120
|
|
|
121
|
+
**Salvage-reframe that shrinks the contribution is NOT a fixable major revision.** When your proposed fix
|
|
122
|
+
for a construct/validity flaw is to *narrow the claim* (e.g. "reframe from a clinical classifier to a
|
|
123
|
+
re-identifiability signal", "scope down to a proof-of-concept"), check whether that narrower framing survives
|
|
124
|
+
the novelty/importance bar. If novelty/importance is ALREADY weak — a co-reviewer or your own scorecard flags
|
|
125
|
+
the work as "expected / well-known finding / unconvincing motivation / limited use case" (Originality or
|
|
126
|
+
Reader-interest ≤ mid) — then the reframe *reduces* the contribution and makes the importance problem worse,
|
|
127
|
+
not better. A contribution shrunk to survive a validity flaw is a **Reject-leaning** outcome (the contribution
|
|
128
|
+
is the product, not addressable-in-revision), not an encourage-major-revision. Deterministic trigger to
|
|
129
|
+
self-audit: if your confidential note says the claim is "narrower than / more modest than claimed" AND your
|
|
130
|
+
recommendation is Reject-family-adjacent, do not upgrade it to major revision on the strength of the reframe.
|
|
131
|
+
|
|
121
132
|
**Review/narrative/primer escalation** *(the contribution IS the product)*: for a review article there is no
|
|
122
133
|
data to re-analyze; the distinct contribution — novelty, integrative synthesis, domain-specificity — is the
|
|
123
134
|
deliverable itself. Therefore **weak novelty / no distinct contribution / not domain-specific is
|
|
@@ -113,6 +113,7 @@ A 17-probe checklist for observational studies (cohort, case-control, cross-sect
|
|
|
113
113
|
- When an agreement or reader study computes a test (Mann–Whitney, t-test, correlation) on **pooled pairwise distances / reader-pairs** rather than on **independent units** (subjects, lesions, aneurysms), the effective n is inflated (each subject contributes several pairwise rows) and the p-value is anti-conservative. This is the reader/agreement sibling of the analysis-unit / clustering issue in O8: the number of *observations* is not the number of *independent units*.
|
|
114
114
|
- Lead: when a reported test **n exceeds the number of independent subjects/lesions** (e.g. n=448 or 672 pairwise from 112 aneurysms) and no clustering / mixed-effects / per-subject aggregation is stated → flag. Ask for the analysis re-run at the **per-subject** level (or a mixed model with a subject random effect); a pooled-pairwise p can flip (e.g. p=0.02 pooled → p=0.59 per-aneurysm).
|
|
115
115
|
- Severity: MAJOR when a headline agreement/superiority claim rests on the pooled-pairwise test; MINOR when a per-subject sensitivity reproduces it. Applies to all multi-rater agreement / MRMC reader studies.
|
|
116
|
+
- Produce the fix: `analyze-stats` `references/analysis_guides/agreement_reliability.md` has the pseudoreplication-safe per-subject aggregation + subject-random-effect ICC code.
|
|
116
117
|
- **(e) Effect size and resolution honesty at scale.** With the very large N these scans run on, trivially small effects clear any threshold; report **effect magnitudes** and clinical relevance alongside significance, and respect the **resolution floor** (a permutation procedure with k permutations cannot resolve p below ≈ 1/k; FDR has a minimum detectable q at a given hit count). A "number of significant exposures" headline with no effect sizes overstates the finding.
|
|
117
118
|
- Severity: MAJOR when a headline causal/actionable claim rests on uncorrected, single-cohort-only, or univariate-in-a-correlated-block top hits; MINOR when correction + replication + full results are present and the claim is framed as screening. Cross-link O11 (complex-survey design — an NHANES/KNHANES ExWAS must combine design-based standard errors with the multiplicity correction, not one or the other), O12 (the single-exposure threshold/non-linearity analogue), and O2/O7 (confounding / over-adjustment for whichever hit survives). Report the tested-set size, the correction method, and the replication design explicitly.
|
|
118
119
|
|
|
@@ -362,6 +362,15 @@ These modules carry the same domain-specific critique probes used by `/peer-revi
|
|
|
362
362
|
| Scoping review (maps the breadth/nature of evidence, clarifies concepts, identifies gaps; PCC framing, charting, optional appraisal — not a focused effectiveness/accuracy question) | `references/domain-probes/scoping_review.md` (SC1–SC8) |
|
|
363
363
|
| Qualitative study (interviews, focus groups, ethnography, grounded theory, phenomenology, document analysis; reflexivity, trustworthiness, thematic analysis — not quantitative validity) | `references/domain-probes/qualitative_research.md` (QL1–QL8) |
|
|
364
364
|
|
|
365
|
+
For a **classifier / NLP / tabular ML** manuscript, also run the deterministic feature-selection-leakage gate — a data-driven selection (feature selection, log-odds / univariate filtering, vocabulary construction, a threshold) fit on the FULL dataset before cross-validation inflates the CV metric:
|
|
366
|
+
|
|
367
|
+
```bash
|
|
368
|
+
python3 "${CLAUDE_SKILL_DIR}/scripts/check_cv_leakage.py" \
|
|
369
|
+
--manuscript manuscript.md --out qc/cv_leakage.json
|
|
370
|
+
```
|
|
371
|
+
|
|
372
|
+
`CV_SELECTION_LEAKAGE` (Major) fires when a selection token co-occurs with cross-validation and no fold-nesting is disclosed ("within each fold" / "nested CV" suppresses it). This is distinct from patient-vs-image split leakage (`model-validation/check_split_leakage.py`).
|
|
373
|
+
|
|
365
374
|
When the manuscript matches a row, read `${CLAUDE_SKILL_DIR}/references/domain-probes/<module>.md` and apply each probe as an additional source of Anticipated Major / Minor Comments. The module severity words (MAJOR / MINOR) map to this skill's framing as follows: a conclusion-threatening or design-level finding becomes a **Fatal** Anticipated Major Comment, a reporting-level finding becomes a **Fixable** Anticipated Minor Comment, and each is tagged with the closest category letter (A–K). These probes **complement** categories A–K above; they do not replace them. (The modules are vendored byte-identical from `/peer-review`; do not edit one copy only — run `python3 scripts/check_domain_probe_sync.py --sync`.)
|
|
366
375
|
|
|
367
376
|
### Phase 2.5: Numerical Cross-Verification (Internal)
|
|
@@ -113,6 +113,7 @@ A 17-probe checklist for observational studies (cohort, case-control, cross-sect
|
|
|
113
113
|
- When an agreement or reader study computes a test (Mann–Whitney, t-test, correlation) on **pooled pairwise distances / reader-pairs** rather than on **independent units** (subjects, lesions, aneurysms), the effective n is inflated (each subject contributes several pairwise rows) and the p-value is anti-conservative. This is the reader/agreement sibling of the analysis-unit / clustering issue in O8: the number of *observations* is not the number of *independent units*.
|
|
114
114
|
- Lead: when a reported test **n exceeds the number of independent subjects/lesions** (e.g. n=448 or 672 pairwise from 112 aneurysms) and no clustering / mixed-effects / per-subject aggregation is stated → flag. Ask for the analysis re-run at the **per-subject** level (or a mixed model with a subject random effect); a pooled-pairwise p can flip (e.g. p=0.02 pooled → p=0.59 per-aneurysm).
|
|
115
115
|
- Severity: MAJOR when a headline agreement/superiority claim rests on the pooled-pairwise test; MINOR when a per-subject sensitivity reproduces it. Applies to all multi-rater agreement / MRMC reader studies.
|
|
116
|
+
- Produce the fix: `analyze-stats` `references/analysis_guides/agreement_reliability.md` has the pseudoreplication-safe per-subject aggregation + subject-random-effect ICC code.
|
|
116
117
|
- **(e) Effect size and resolution honesty at scale.** With the very large N these scans run on, trivially small effects clear any threshold; report **effect magnitudes** and clinical relevance alongside significance, and respect the **resolution floor** (a permutation procedure with k permutations cannot resolve p below ≈ 1/k; FDR has a minimum detectable q at a given hit count). A "number of significant exposures" headline with no effect sizes overstates the finding.
|
|
117
118
|
- Severity: MAJOR when a headline causal/actionable claim rests on uncorrected, single-cohort-only, or univariate-in-a-correlated-block top hits; MINOR when correction + replication + full results are present and the claim is framed as screening. Cross-link O11 (complex-survey design — an NHANES/KNHANES ExWAS must combine design-based standard errors with the multiplicity correction, not one or the other), O12 (the single-exposure threshold/non-linearity analogue), and O2/O7 (confounding / over-adjustment for whichever hit survives). Report the tested-set size, the correction method, and the replication design explicitly.
|
|
118
119
|
|
|
@@ -83,6 +83,16 @@ EVALUE_RE = re.compile(
|
|
|
83
83
|
NONPRIMARY_KW = ("secondary", "exploratory", "subgroup", "sensitivity", "supporting",
|
|
84
84
|
"cause-specific", "cancer-specific", "post-hoc", "post hoc", "non-primary")
|
|
85
85
|
|
|
86
|
+
# The manuscript asserts exactly ONE primary model/analysis (so a script annotating a
|
|
87
|
+
# model as "co-primary" is a third-SSOT drift).
|
|
88
|
+
SINGLE_PRIMARY = re.compile(
|
|
89
|
+
r"\bsingle\s+primary\b|\ba\s+single\s+primary\b|\bone\s+primary\s+(?:model|analysis|endpoint|outcome)\b"
|
|
90
|
+
r"|\bthe\s+primary\s+(?:model|analysis|endpoint|outcome)\b[^.]{0,70}?"
|
|
91
|
+
r"(?:consistent\s+with\s+the\s+(?:registered|pre-?specified)|registered\s+analysis\s+plan)",
|
|
92
|
+
re.I)
|
|
93
|
+
# A model annotated "co-primary" in analysis code (a comment, string, or variable).
|
|
94
|
+
CO_PRIMARY_CODE = re.compile(r"\bco[-\s]?primary\b", re.I)
|
|
95
|
+
|
|
86
96
|
STOP = set("the a an of for in on to and or with by is was were are be been being this that "
|
|
87
97
|
"between association associated estimated using model analysis primary outcome "
|
|
88
98
|
"endpoint study patients group as at from".split())
|
|
@@ -271,6 +281,45 @@ def check_evalue(manuscript: str) -> list[dict]:
|
|
|
271
281
|
# manual confirmation against the registration before either is acted on, and a P0
|
|
272
282
|
# that needs hand-confirmation is not a P0. Only explicit re-designation and a
|
|
273
283
|
# non-recomputing E-value are Major.
|
|
284
|
+
def check_code_labels(manuscript: str, scripts_dir: str | None) -> list[dict]:
|
|
285
|
+
"""Reconcile the manuscript's declared primary against analysis-script labels.
|
|
286
|
+
|
|
287
|
+
Fires only the specific conflict: the manuscript asserts a SINGLE primary while an
|
|
288
|
+
analysis script annotates a model as 'co-primary' — the code label is a third SSOT
|
|
289
|
+
that drifts across revisions. Advisory (code comments can lag)."""
|
|
290
|
+
claims: list[dict] = []
|
|
291
|
+
if not scripts_dir:
|
|
292
|
+
return claims
|
|
293
|
+
d = Path(scripts_dir)
|
|
294
|
+
if not d.exists() or not SINGLE_PRIMARY.search(manuscript):
|
|
295
|
+
return claims
|
|
296
|
+
for p in sorted(d.rglob("*")):
|
|
297
|
+
if p.suffix.lower() not in (".r", ".py"):
|
|
298
|
+
continue
|
|
299
|
+
try:
|
|
300
|
+
txt = p.read_text(encoding="utf-8", errors="replace")
|
|
301
|
+
except OSError:
|
|
302
|
+
continue
|
|
303
|
+
m = CO_PRIMARY_CODE.search(txt)
|
|
304
|
+
if not m:
|
|
305
|
+
continue
|
|
306
|
+
ln = txt[:m.start()].count("\n") + 1
|
|
307
|
+
snippet = txt.splitlines()[ln - 1].strip()[:80] if ln - 1 < len(txt.splitlines()) else ""
|
|
308
|
+
claims.append({
|
|
309
|
+
"claim_id": "EST-code-label",
|
|
310
|
+
"type": "estimand",
|
|
311
|
+
"prose_value": "manuscript asserts a single primary model/analysis",
|
|
312
|
+
"artifact_source": f"{p.name}:{ln} labels a model 'co-primary'",
|
|
313
|
+
"verdict": "PRIMARY_LABEL_CODE_DRIFT",
|
|
314
|
+
"detail": (f"the manuscript declares a SINGLE primary while an analysis script "
|
|
315
|
+
f"annotates a model as co-primary ({p.name}:{ln}: '{snippet}'); reconcile "
|
|
316
|
+
f"the code's primary/co-primary label with the declared estimand — code "
|
|
317
|
+
f"labels are a third SSOT that can drift across revisions. ADVISORY."),
|
|
318
|
+
})
|
|
319
|
+
break # one is enough to prompt a reconcile
|
|
320
|
+
return claims
|
|
321
|
+
|
|
322
|
+
|
|
274
323
|
MAJOR = {"PRIMARY_REASSIGNED", "EVALUE_ARITHMETIC"}
|
|
275
324
|
|
|
276
325
|
|
|
@@ -278,6 +327,7 @@ def main() -> int:
|
|
|
278
327
|
ap = argparse.ArgumentParser(description="Claim-vs-artifact cross-check (estimand + E-value).")
|
|
279
328
|
ap.add_argument("--manuscript", required=True, help="manuscript markdown/text")
|
|
280
329
|
ap.add_argument("--prereg", help="pre-registration / protocol / project.yaml text")
|
|
330
|
+
ap.add_argument("--scripts", help="analysis-scripts directory (reconcile code primary/co-primary labels)")
|
|
281
331
|
ap.add_argument("--out", help="write JSON artifact to this path")
|
|
282
332
|
ap.add_argument("--strict", action="store_true", help="exit 1 if any Major verdict")
|
|
283
333
|
args = ap.parse_args()
|
|
@@ -302,7 +352,8 @@ def main() -> int:
|
|
|
302
352
|
else:
|
|
303
353
|
sys.stderr.write(f"WARN: prereg not found: {args.prereg} (estimand provenance limited)\n")
|
|
304
354
|
|
|
305
|
-
claims = check_estimand(manuscript, prereg, prereg_raw) + check_evalue(manuscript)
|
|
355
|
+
claims = (check_estimand(manuscript, prereg, prereg_raw) + check_evalue(manuscript)
|
|
356
|
+
+ check_code_labels(manuscript, args.scripts))
|
|
306
357
|
n_major = sum(1 for c in claims if c["verdict"] in MAJOR)
|
|
307
358
|
n_flag = sum(1 for c in claims if c["verdict"] not in MAJOR and c["verdict"] != "OK")
|
|
308
359
|
|
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Feature-selection-outside-CV leakage gate (self-review Phase 2.5 / data-prep).
|
|
3
|
+
|
|
4
|
+
For a classifier / NLP / tabular manuscript, if feature selection, vocabulary
|
|
5
|
+
construction, log-odds / univariate filtering, or a threshold is chosen on the WHOLE
|
|
6
|
+
dataset and only THEN cross-validation is run, the CV performance is optimistically
|
|
7
|
+
inflated: the selection has already seen the held-out folds. The fix is to nest the
|
|
8
|
+
selection inside each training fold (nested CV). This is a class a statistical
|
|
9
|
+
reviewer catches deterministically, and it is distinct from patient-vs-image split
|
|
10
|
+
leakage (`model-validation/check_split_leakage.py`).
|
|
11
|
+
|
|
12
|
+
Verdict:
|
|
13
|
+
CV_SELECTION_LEAKAGE (Major) a feature-selection / vocabulary / threshold step
|
|
14
|
+
co-occurs with a cross-validation description AND no
|
|
15
|
+
fold-nesting disclosure ("within each fold", "nested
|
|
16
|
+
CV", "inside the training fold") is present. The
|
|
17
|
+
headline CV metric is likely optimistic.
|
|
18
|
+
|
|
19
|
+
Conservative by construction: fires only when BOTH a selection token AND a CV token
|
|
20
|
+
appear AND no nesting-disclosure token is anywhere in the document. A single
|
|
21
|
+
"within each training fold" / "nested cross-validation" sentence suppresses it.
|
|
22
|
+
|
|
23
|
+
Exit codes: 0 clean/report-only, 1 with --strict when any Major, 2 usage. Stdlib-only.
|
|
24
|
+
|
|
25
|
+
Usage:
|
|
26
|
+
python3 check_cv_leakage.py --manuscript manuscript.md \
|
|
27
|
+
[--out qc/cv_leakage.json] [--strict] [--quiet]
|
|
28
|
+
"""
|
|
29
|
+
from __future__ import annotations
|
|
30
|
+
|
|
31
|
+
import argparse
|
|
32
|
+
import json
|
|
33
|
+
import re
|
|
34
|
+
import sys
|
|
35
|
+
from pathlib import Path
|
|
36
|
+
|
|
37
|
+
# A data-driven selection / construction step that must be nested inside CV.
|
|
38
|
+
SELECTION = re.compile(
|
|
39
|
+
r"\bfeature\s+selection\b|\bselected\s+(?:the\s+)?(?:top\s+)?\d+\s+(?:features|variables|predictors)"
|
|
40
|
+
r"|\blog[-\s]?odds\b|\bvocabulary\s+(?:construction|was\s+built|building)\b|\bbuilt\s+(?:a\s+)?vocabulary"
|
|
41
|
+
r"|\bfeature\s+ranking\b|\bunivariate\s+(?:filter|screening|selection)\b|\btop[-\s]?k\s+features"
|
|
42
|
+
r"|\bmutual[-\s]information\s+(?:selection|ranking)\b|\bchi[-\s]?square(?:d)?\s+selection"
|
|
43
|
+
r"|\b(?:selected|chose|retained|kept)\s+(?:the\s+)?(?:most\s+)?(?:informative|discriminative|predictive)\s+features"
|
|
44
|
+
r"|\bthreshold(?:ed|ing)?\b[^.\n]{0,40}\b(?:on|over|across|using)\s+the\s+(?:entire|full|whole|complete)\s+(?:data|dataset|cohort|corpus)"
|
|
45
|
+
r"|\b(?:LASSO|elastic[-\s]net|recursive\s+feature\s+elimination|RFE|Boruta)\b",
|
|
46
|
+
re.IGNORECASE)
|
|
47
|
+
|
|
48
|
+
# A cross-validation evaluation.
|
|
49
|
+
CV = re.compile(
|
|
50
|
+
r"\bcross[-\s]?validat(?:ion|ed)\b|\b\d+[-\s]?fold\b|\bk[-\s]?fold\b"
|
|
51
|
+
r"|\bleave[-\s]one[-\s]out\b|\bLOOCV\b|\bstratified\s+(?:\d+[-\s]?)?fold",
|
|
52
|
+
re.IGNORECASE)
|
|
53
|
+
|
|
54
|
+
# Disclosure that the selection is correctly nested inside the CV training folds.
|
|
55
|
+
NESTING = re.compile(
|
|
56
|
+
r"\bnested\s+(?:cross[-\s]?validat|CV)\b"
|
|
57
|
+
r"|\bwithin\s+each\s+(?:training\s+)?fold\b|\binside\s+(?:the\s+)?training\s+(?:fold|partition|split)"
|
|
58
|
+
r"|\bper[-\s]?fold\b|\bfold[-\s]specific\b|\bfor\s+each\s+(?:training\s+)?fold\b"
|
|
59
|
+
r"|\bon\s+the\s+training\s+(?:fold|partition|split|set)\s+only\b"
|
|
60
|
+
r"|\brefit(?:ted)?\s+within\b|\brepeated\s+(?:in|within|inside)\s+each\s+fold"
|
|
61
|
+
r"|\bselection\s+was\s+(?:performed|done|repeated)\s+(?:in|within|inside)\s+each\b",
|
|
62
|
+
re.IGNORECASE)
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
def check(text: str) -> list[dict]:
|
|
66
|
+
sel = SELECTION.search(text)
|
|
67
|
+
cv = CV.search(text)
|
|
68
|
+
if not (sel and cv):
|
|
69
|
+
return []
|
|
70
|
+
if NESTING.search(text):
|
|
71
|
+
return [] # nesting is disclosed
|
|
72
|
+
return [{
|
|
73
|
+
"verdict": "CV_SELECTION_LEAKAGE",
|
|
74
|
+
"severity": "Major",
|
|
75
|
+
"detail": (f"a data-driven selection step ('{sel.group(0).strip()}') co-occurs with "
|
|
76
|
+
f"cross-validation ('{cv.group(0).strip()}') but no fold-nesting is disclosed "
|
|
77
|
+
f"(no 'within each fold' / 'nested CV'); if the selection was fit on the full "
|
|
78
|
+
f"dataset the CV metric is optimistically inflated — nest it in each training fold"),
|
|
79
|
+
"where": text[max(0, sel.start() - 30):sel.end() + 50].replace("\n", " ").strip()[:170],
|
|
80
|
+
}]
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def analyze(manuscript: str) -> dict:
|
|
84
|
+
p = Path(manuscript)
|
|
85
|
+
if not p.is_file():
|
|
86
|
+
sys.stderr.write(f"ERROR: manuscript not found: {manuscript}\n")
|
|
87
|
+
sys.exit(2)
|
|
88
|
+
claims = check(p.read_text(encoding="utf-8"))
|
|
89
|
+
n_major = sum(1 for c in claims if c["severity"] == "Major")
|
|
90
|
+
return {
|
|
91
|
+
"manuscript": str(p),
|
|
92
|
+
"claims": claims,
|
|
93
|
+
"summary": {"n_claims": len(claims), "n_major": n_major, "n_flag": len(claims) - n_major,
|
|
94
|
+
"verdict": "MAJOR_CANDIDATE" if n_major else "OK"},
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
|
|
98
|
+
def render(result: dict) -> str:
|
|
99
|
+
lines = ["| Check | Severity | Detail |", "|---|---|---|"]
|
|
100
|
+
for c in result["claims"]:
|
|
101
|
+
lines.append(f"| {c['verdict']} | {c['severity']} | {c['detail']} |")
|
|
102
|
+
if len(lines) == 2:
|
|
103
|
+
lines.append("| (none) | — | no feature-selection-outside-CV leakage detected |")
|
|
104
|
+
return "\n".join(lines)
|
|
105
|
+
|
|
106
|
+
|
|
107
|
+
def main() -> int:
|
|
108
|
+
ap = argparse.ArgumentParser(
|
|
109
|
+
description="Feature-selection-outside-CV leakage gate (Phase 2.5 / data-prep).")
|
|
110
|
+
ap.add_argument("--manuscript", required=True, help="manuscript markdown/text")
|
|
111
|
+
ap.add_argument("--out", help="write JSON artifact to this path")
|
|
112
|
+
ap.add_argument("--strict", action="store_true", help="exit 1 if any Major claim exists")
|
|
113
|
+
ap.add_argument("--quiet", action="store_true", help="suppress stdout table")
|
|
114
|
+
args = ap.parse_args()
|
|
115
|
+
|
|
116
|
+
result = analyze(args.manuscript)
|
|
117
|
+
|
|
118
|
+
if not args.quiet:
|
|
119
|
+
print("=" * 44)
|
|
120
|
+
print(" CV selection-leakage (§2.5 / data-prep)")
|
|
121
|
+
print("=" * 44)
|
|
122
|
+
print(render(result))
|
|
123
|
+
print()
|
|
124
|
+
if result["summary"]["n_major"]:
|
|
125
|
+
print("MAJOR candidate: a selection step co-occurs with CV and no fold-nesting is disclosed.")
|
|
126
|
+
else:
|
|
127
|
+
print("OK: no feature-selection-outside-CV leakage detected.")
|
|
128
|
+
|
|
129
|
+
if args.out:
|
|
130
|
+
Path(args.out).parent.mkdir(parents=True, exist_ok=True)
|
|
131
|
+
Path(args.out).write_text(json.dumps(result, indent=2), encoding="utf-8")
|
|
132
|
+
if not args.quiet:
|
|
133
|
+
print(f"\nwrote {args.out}")
|
|
134
|
+
|
|
135
|
+
return 1 if (args.strict and result["summary"]["n_major"]) else 0
|
|
136
|
+
|
|
137
|
+
|
|
138
|
+
if __name__ == "__main__":
|
|
139
|
+
sys.exit(main())
|
|
@@ -86,5 +86,20 @@ raise SystemExit(0 if not any(c['verdict']=='ESTIMAND_DRIFT' for c in d['claims'
|
|
|
86
86
|
python3 "$SCRIPT" --manuscript "$SMAN" --prereg "$SPRE" --strict >/dev/null 2>&1
|
|
87
87
|
check "exit 0 on structured consistent prereg (no Major)" test "$?" -eq 0
|
|
88
88
|
|
|
89
|
+
# Code-label reconciliation (--scripts): a manuscript asserting a SINGLE primary vs an
|
|
90
|
+
# analysis script annotating a model 'co-primary' -> PRIMARY_LABEL_CODE_DRIFT (advisory,
|
|
91
|
+
# not Major). A consistent scripts dir and the no-flag backward-compatible default.
|
|
92
|
+
SP="$HERE/fixtures/claim_manuscript_single_primary.md"
|
|
93
|
+
python3 "$SCRIPT" --manuscript "$SP" --scripts "$HERE/fixtures/claim_scripts_coprimary" --out "$OUT" >/dev/null 2>&1
|
|
94
|
+
check "PRIMARY_LABEL_CODE_DRIFT on code co-primary vs single-primary manuscript" has_verdict PRIMARY_LABEL_CODE_DRIFT
|
|
95
|
+
python3 "$SCRIPT" --manuscript "$SP" --scripts "$HERE/fixtures/claim_scripts_coprimary" --strict >/dev/null 2>&1
|
|
96
|
+
check "code-label drift is advisory (exit 0 under --strict)" test "$?" -eq 0
|
|
97
|
+
python3 "$SCRIPT" --manuscript "$SP" --scripts "$HERE/fixtures/claim_scripts_consistent" --out "$OUT" >/dev/null 2>&1
|
|
98
|
+
check "no drift when scripts carry no co-primary label" python3 -c "
|
|
99
|
+
import json
|
|
100
|
+
d=json.load(open('$OUT'))
|
|
101
|
+
raise SystemExit(0 if not any(c['verdict']=='PRIMARY_LABEL_CODE_DRIFT' for c in d['claims']) else 1)
|
|
102
|
+
"
|
|
103
|
+
|
|
89
104
|
echo "fail=$fail"; [[ "$fail" -eq 0 ]] && echo "ALL PASS" || echo "FAILURES: $fail"
|
|
90
105
|
exit "$fail"
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Regression test for the feature-selection-outside-CV leakage gate.
|
|
3
|
+
# (bad) log-odds feature selection on the full dataset + 5-fold CV, no nesting ->
|
|
4
|
+
# CV_SELECTION_LEAKAGE; (clean) selection repeated within each training fold (nested
|
|
5
|
+
# CV) -> no flag.
|
|
6
|
+
set -u
|
|
7
|
+
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
8
|
+
SCRIPT="$HERE/../scripts/check_cv_leakage.py"
|
|
9
|
+
BAD="$HERE/fixtures/cv_leakage_bad.md"
|
|
10
|
+
CLEAN="$HERE/fixtures/cv_leakage_clean.md"
|
|
11
|
+
OUT="$(mktemp -t cvl_XXXX).json"
|
|
12
|
+
trap 'rm -f "$OUT"' EXIT
|
|
13
|
+
fail=0
|
|
14
|
+
check() { local label="$1"; shift
|
|
15
|
+
if "$@" >/dev/null 2>&1; then printf ' PASS %s\n' "$label"
|
|
16
|
+
else printf ' FAIL %s\n' "$label"; fail=$((fail+1)); fi; }
|
|
17
|
+
[[ -f "$SCRIPT" ]] || { echo "ENV-ERR: script missing" >&2; exit 2; }
|
|
18
|
+
|
|
19
|
+
python3 "$SCRIPT" --manuscript "$BAD" --out "$OUT" --strict --quiet >/dev/null 2>&1
|
|
20
|
+
check "exit 1 under --strict (Major present)" test "$?" -eq 1
|
|
21
|
+
check "CV_SELECTION_LEAKAGE on selection+CV without nesting" python3 -c "
|
|
22
|
+
import json
|
|
23
|
+
d=json.load(open('$OUT'))
|
|
24
|
+
assert any(c['verdict']=='CV_SELECTION_LEAKAGE' for c in d['claims']), 'not flagged'
|
|
25
|
+
"
|
|
26
|
+
python3 "$SCRIPT" --manuscript "$CLEAN" --strict --quiet >/dev/null 2>&1
|
|
27
|
+
check "exit 0 when selection is nested within each fold" test "$?" -eq 0
|
|
28
|
+
python3 "$SCRIPT" --manuscript "$CLEAN" --out "$OUT" --quiet >/dev/null 2>&1
|
|
29
|
+
check "no leakage flag on nested CV" python3 -c "
|
|
30
|
+
import json
|
|
31
|
+
d=json.load(open('$OUT'))
|
|
32
|
+
assert not d['claims'], d['claims']
|
|
33
|
+
"
|
|
34
|
+
echo "fail=$fail"; [[ "$fail" -eq 0 ]] && echo "ALL PASS" || echo "FAILURES: $fail"
|
|
35
|
+
exit "$fail"
|
|
@@ -221,7 +221,7 @@ Design all tables and figures BEFORE writing prose. This ensures the narrative s
|
|
|
221
221
|
|
|
222
222
|
Write the Methods section first -- it is the most objective and anchors the rest of the paper.
|
|
223
223
|
|
|
224
|
-
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/methods.md` for PICO structure, backbone article usage, checklist cross-reference, and terminology conventions. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_methods/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE) — it lists, paragraph by paragraph, what each Methods paragraph must establish plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
|
|
224
|
+
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/methods.md` for PICO structure, backbone article usage, checklist cross-reference, and terminology conventions. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_methods/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — it lists, paragraph by paragraph, what each Methods paragraph must establish plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
|
|
225
225
|
|
|
226
226
|
**Writing order within Methods:**
|
|
227
227
|
1. Study Design and Setting
|
|
@@ -255,7 +255,7 @@ Write the Methods section first -- it is the most objective and anchors the rest
|
|
|
255
255
|
Write Results aligned to the approved tables and figures. **Results = "What did we find?"
|
|
256
256
|
— nothing more.** Every sentence must be a factual statement backed by a number.
|
|
257
257
|
|
|
258
|
-
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/results.md` for mirror-symmetry rules, flowchart requirements, missing data handling, and the anti-interpretation self-check. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_results/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE) — each follows its `exemplar_methods/` sibling in Methods order, listing what each Results paragraph must establish (flow → baseline/prevalence → primary estimate with CIs → calibration/agreement → subgroups → sensitivity) plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
|
|
258
|
+
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/results.md` for mirror-symmetry rules, flowchart requirements, missing data handling, and the anti-interpretation self-check. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_results/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — each follows its `exemplar_methods/` sibling in Methods order, listing what each Results paragraph must establish (flow → baseline/prevalence → primary estimate with CIs → calibration/agreement → subgroups → sensitivity; for meta-analysis, PRISMA flow → characteristics+provenance → RoB → pooled estimate with I²/τ²/prediction interval → subgroup interaction → publication bias) plus the element that type most often omits. Model the structure; the exemplars are synthetic, with placeholder specifics, not prose to copy.
|
|
259
259
|
|
|
260
260
|
**Rules:**
|
|
261
261
|
- Every number in the text must match the corresponding table cell exactly.
|
|
@@ -296,7 +296,7 @@ Write Results aligned to the approved tables and figures. **Results = "What did
|
|
|
296
296
|
|
|
297
297
|
### Phase 5: Discussion
|
|
298
298
|
|
|
299
|
-
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/discussion.md` for the 4-paragraph structure, word limits, limitation writing guidelines, and Table/Figure citation rules. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_discussion/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE) — completing the exemplar trio, each lists what every Discussion paragraph must establish (key finding → interpretation/comparison → limitations → generalizability → conclusion matched to the evidence) plus the element that type most often omits (spectrum/verification bias; evidence-tier separation and optimism caveats; mandatory causal caution). For case reports, use `${CLAUDE_SKILL_DIR}/references/exemplar_case_report.md` instead: it controls literature-boundary wording, n=1 causal caution, and bedside teaching-point framing. Model the structure; the exemplars are synthetic, introduce no new results, and are not prose to copy.
|
|
299
|
+
**Before writing:** Load `${CLAUDE_SKILL_DIR}/references/section_guides/discussion.md` for the 4-paragraph structure, word limits, limitation writing guidelines, and Table/Figure citation rules. For the matching study type, also skim the structure model in `${CLAUDE_SKILL_DIR}/references/exemplar_discussion/` (diagnostic-accuracy/STARD, AI-validation/TRIPOD+AI·CLAIM, observational-cohort/STROBE, meta-analysis/PRISMA 2020) — completing the exemplar trio, each lists what every Discussion paragraph must establish (key finding → interpretation/comparison → limitations → generalizability → conclusion matched to the evidence) plus the element that type most often omits (spectrum/verification bias; evidence-tier separation and optimism caveats; mandatory causal caution; for meta-analysis, GRADE certainty + heterogeneity source + non-independence/overlap caveat). For case reports, use `${CLAUDE_SKILL_DIR}/references/exemplar_case_report.md` instead: it controls literature-boundary wording, n=1 causal caution, and bedside teaching-point framing. Model the structure; the exemplars are synthetic, introduce no new results, and are not prose to copy.
|
|
300
300
|
|
|
301
301
|
**Before drafting, collect user input (Discussion Planning Gate).**
|
|
302
302
|
|
|
@@ -27,6 +27,9 @@ closes with a conclusion matched to the evidence — introducing no new results.
|
|
|
27
27
|
CLAIM).
|
|
28
28
|
- `observational_cohort_strobe.md` — mandatory causal caution, reverse-causation and
|
|
29
29
|
unmeasured-confounding limitations, no care-directive conclusion (STROBE).
|
|
30
|
+
- `meta_analysis_prisma.md` — summary-of-evidence with certainty framing, heterogeneity-source
|
|
31
|
+
and non-independence/overlap caveats, GRADE limitations, no guideline-grade conclusion
|
|
32
|
+
(PRISMA 2020).
|
|
30
33
|
|
|
31
34
|
## Curator guidelines (for adding more)
|
|
32
35
|
|
|
@@ -0,0 +1,52 @@
|
|
|
1
|
+
# Discussion structure — systematic review / meta-analysis (PRISMA 2020)
|
|
2
|
+
|
|
3
|
+
A structure model for the Discussion of a systematic review with quantitative synthesis,
|
|
4
|
+
completing the trio with the `exemplar_methods/` and `exemplar_results/` siblings. Each heading
|
|
5
|
+
is a paragraph; each bullet is *what it must establish*. Fill the `[brackets]`; do not copy this
|
|
6
|
+
text. Follows `section_guides/discussion.md`. Introduces **no new results** and cites **no
|
|
7
|
+
tables/figures**.
|
|
8
|
+
|
|
9
|
+
## Paragraph 1 — Summary of evidence
|
|
10
|
+
- Restate the primary pooled estimate (effect + 95% CI, k studies, N patients) in 2–3
|
|
11
|
+
sentences, matched to the Results — introduce no pooled number not already reported.
|
|
12
|
+
- One sentence on what it means clinically, framed to the **certainty of evidence**, not as
|
|
13
|
+
proof.
|
|
14
|
+
- **If the pooled estimate is null or the CI is wide, frame it by precision, not absence** —
|
|
15
|
+
state what the CI and the prediction interval *exclude*; an imprecise pool with few studies is
|
|
16
|
+
"not yet established," not "no effect."
|
|
17
|
+
|
|
18
|
+
## Paragraphs 2–3 — Interpretation, comparison, and heterogeneity
|
|
19
|
+
- Compare to the 2–3 most recent related syntheses (agree / differ and why: eligibility,
|
|
20
|
+
populations, measurement, analysis).
|
|
21
|
+
- **Explain heterogeneity**, distinguishing **clinical** (population, technique, threshold) from
|
|
22
|
+
**statistical** (I²/τ²), and tie it to the subgroup / meta-regression findings — a high I²
|
|
23
|
+
narrated only as a number, without a source, is a reviewer target.
|
|
24
|
+
- Note whether small-study effects / publication bias plausibly move the estimate.
|
|
25
|
+
|
|
26
|
+
## Limitations
|
|
27
|
+
- **Study-level risk of bias** carried into the pooled estimate (name the driving domain); the
|
|
28
|
+
**certainty of evidence** downgrades (GRADE: risk of bias, inconsistency, indirectness,
|
|
29
|
+
imprecision, publication bias).
|
|
30
|
+
- **Review-level limits**: overlapping populations / shared datasets or benchmarks
|
|
31
|
+
(non-independence) and how a leave-one-dataset-out check addressed it; few studies for
|
|
32
|
+
subgroups; language/publication restrictions; a low-powered or absent publication-bias test.
|
|
33
|
+
- If reviewer count, search currency, or a synthesis-method choice deviated from protocol, state
|
|
34
|
+
it plainly here — consistent with what Methods declared.
|
|
35
|
+
|
|
36
|
+
## Generalizability
|
|
37
|
+
- The populations and settings the pooled estimate represents and where it may not transfer;
|
|
38
|
+
how the prediction interval bounds — not the CI — describe the plausible effect in a new study.
|
|
39
|
+
|
|
40
|
+
## Conclusion
|
|
41
|
+
- Matched to the evidence and its certainty — "the pooled evidence suggests", "moderate-certainty
|
|
42
|
+
evidence for", or a call for adequately powered / prospective studies; never a care directive or
|
|
43
|
+
a guideline-grade recommendation the evidence tier does not license (guideline recommendations
|
|
44
|
+
are the panel's, not a single review's — and disclose any guideline-co-authorship intellectual
|
|
45
|
+
COI).
|
|
46
|
+
|
|
47
|
+
## Common omission
|
|
48
|
+
- An explicit **certainty-of-evidence (GRADE) treatment plus a heterogeneity source and a
|
|
49
|
+
non-independence/overlap caveat** — the Discussion elements MA drafts most often soften, and the
|
|
50
|
+
ones a synthesis reviewer checks first. Cross-reference `section_guides/discussion.md`, the
|
|
51
|
+
`peer-review/references/domain-probes/sr_ma.md` probes (P2 non-independence, P4 k=1), and the
|
|
52
|
+
PRISMA critical items in `peer-review/references/reviewer_calibration/compliance_floor.md`.
|
|
@@ -23,6 +23,7 @@ fit). A missing element is a gap to fill before drafting Results.
|
|
|
23
23
|
- `diagnostic_accuracy_stard.md` — an index-test vs reference-standard accuracy study (STARD).
|
|
24
24
|
- `ai_validation_tripod_claim.md` — an AI/ML model development + validation study (TRIPOD+AI / CLAIM).
|
|
25
25
|
- `observational_cohort_strobe.md` — an exposure→outcome observational cohort (STROBE).
|
|
26
|
+
- `meta_analysis_prisma.md` — a systematic review with quantitative synthesis (PRISMA 2020).
|
|
26
27
|
|
|
27
28
|
## Curator guidelines (for adding more)
|
|
28
29
|
|
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
# Methods structure — systematic review / meta-analysis (PRISMA 2020)
|
|
2
|
+
|
|
3
|
+
A structure model for a systematic review with quantitative synthesis. Each heading is a
|
|
4
|
+
paragraph; each bullet is *what it must establish*. Fill the `[brackets]`; do not copy this
|
|
5
|
+
text. Pairs with `paper_types/meta_analysis.md` (which carries the fuller prose templates) —
|
|
6
|
+
this file is the paragraph-order skeleton and the reviewer-critical checklist.
|
|
7
|
+
|
|
8
|
+
## Protocol and registration
|
|
9
|
+
- Prospective registration (PROSPERO `CRD[number]`) or an OSF/protocol reference, **lodged
|
|
10
|
+
before data extraction began** — state the registration date relative to the search.
|
|
11
|
+
- Adherence to PRISMA 2020 (name the base guideline *and* the extension used, e.g. PRISMA
|
|
12
|
+
base + **PRISMA-DTA** for diagnostic accuracy, PRISMA-NMA for network — never just the
|
|
13
|
+
extension acronym).
|
|
14
|
+
- Any protocol amendments as dated entries (what changed, when, why), not a silent deviation.
|
|
15
|
+
|
|
16
|
+
## Eligibility criteria
|
|
17
|
+
- PICO(S) as a numbered inclusion list and a **separate** explicit exclusion list (not "did
|
|
18
|
+
not meet inclusion") — Population, Intervention/Index test, Comparator, Outcome, Study
|
|
19
|
+
design, and any language/date limits.
|
|
20
|
+
- The effect measure the review is built around (OR / RR / HR / MD / SMD / sensitivity &
|
|
21
|
+
specificity / AUC) named here, so eligibility and synthesis agree.
|
|
22
|
+
|
|
23
|
+
## Information sources and search strategy
|
|
24
|
+
- Every database with its coverage dates and the last-search date; grey literature and
|
|
25
|
+
register searches (ClinicalTrials.gov, conference abstracts) if used.
|
|
26
|
+
- The **full search string for ≥1 database, verbatim** (in-text or as a supplement) — a
|
|
27
|
+
PRISMA-critical, first-line reviewer item; a paraphrased strategy is not reproducible.
|
|
28
|
+
|
|
29
|
+
## Study selection and data extraction
|
|
30
|
+
- Dual independent screening (title/abstract then full text) with the agreement metric
|
|
31
|
+
(Cohen's κ or % agreement) at each stage and how disagreements were resolved; the operational
|
|
32
|
+
reviewer count here must match what Limitations later concedes (no "dual" here / "single
|
|
33
|
+
reviewer" there).
|
|
34
|
+
- Standardized dual extraction; the variables extracted; how the reference-standard / outcome
|
|
35
|
+
cells (e.g. 2×2 TP/FP/FN/TN) were obtained and reconciled against the source (cell-level
|
|
36
|
+
audit for hand-entered accuracy data).
|
|
37
|
+
|
|
38
|
+
## Risk of bias / quality assessment
|
|
39
|
+
- The tool matched to the design (**QUADAS-2** for DTA, **RoB 2** for RCTs, ROBINS-I / NOS /
|
|
40
|
+
MINORS for non-randomized), applied in duplicate; how domain judgements were reached.
|
|
41
|
+
- Whether risk of bias fed a sensitivity analysis (low-RoB-only pool), not just a figure.
|
|
42
|
+
|
|
43
|
+
## Statistical synthesis
|
|
44
|
+
- The model and estimand: **random-effects a priori** (state the estimator, e.g.
|
|
45
|
+
DerSimonian–Laird or REML; bivariate / HSROC for DTA) with the pooled effect and **95% CIs**.
|
|
46
|
+
- Heterogeneity: I² **with τ² and a 95% prediction interval** (I² is a proportion, not a
|
|
47
|
+
magnitude — the PI shows the range of true effects); Cochran's Q threshold.
|
|
48
|
+
- **Pre-specified** subgroup / meta-regression covariates (post-hoc ones labeled exploratory);
|
|
49
|
+
sensitivity analyses (leave-one-out, prospective-only, low-RoB-only); overlapping-population /
|
|
50
|
+
shared-benchmark non-independence check.
|
|
51
|
+
- Publication bias (Egger / Deeks funnel asymmetry) **only when ≥10 studies**; certainty of
|
|
52
|
+
evidence (**GRADE**) if claimed. Software + version + seed.
|
|
53
|
+
|
|
54
|
+
## Reporting-guideline fit
|
|
55
|
+
- PRISMA 2020 (+ the relevant extension). Critical items: verbatim search strategy for ≥1
|
|
56
|
+
database, a flow diagram that reconciles, per-study risk of bias, and the protocol/registration
|
|
57
|
+
reference (see `peer-review/references/reviewer_calibration/compliance_floor.md`).
|
|
58
|
+
|
|
59
|
+
## Common omission
|
|
60
|
+
- A **pre-specified subgroup set + an overlapping-population / shared-dataset non-independence
|
|
61
|
+
check**, and naming **base + extension** guideline correctly — the Methods elements MA drafts
|
|
62
|
+
most often skip, and among the first a synthesis reviewer checks. Cross-reference
|
|
63
|
+
`section_guides/methods.md`, `paper_types/meta_analysis.md`, and the
|
|
64
|
+
`peer-review/references/domain-probes/sr_ma.md` probes (P1 comparator existence, P2
|
|
65
|
+
non-independence, P4 k=1 subgroup).
|
|
@@ -27,6 +27,9 @@ element that interprets rather than reports belongs in the Discussion.
|
|
|
27
27
|
calibration, operating point, clinical utility (TRIPOD+AI / CLAIM).
|
|
28
28
|
- `observational_cohort_strobe.md` — assembly, Table 1 by exposure, crude-then-adjusted
|
|
29
29
|
estimate with CIs, subgroups, sensitivity analyses (STROBE).
|
|
30
|
+
- `meta_analysis_prisma.md` — PRISMA flow, study/patient characteristics + provenance, pooled
|
|
31
|
+
estimate with I²/τ²/prediction interval, subgroup interaction, sensitivity, publication bias
|
|
32
|
+
(PRISMA 2020).
|
|
30
33
|
|
|
31
34
|
## Curator guidelines (for adding more)
|
|
32
35
|
|
|
@@ -0,0 +1,51 @@
|
|
|
1
|
+
# Results structure — systematic review / meta-analysis (PRISMA 2020)
|
|
2
|
+
|
|
3
|
+
A structure model for the Results of a systematic review with quantitative synthesis. It
|
|
4
|
+
follows its Methods PRISMA sibling in Methods order. Each heading is a paragraph; each bullet
|
|
5
|
+
is *what it must establish*. Fill the `[brackets]`; do not copy this text. Report findings
|
|
6
|
+
only — no interpretation.
|
|
7
|
+
|
|
8
|
+
## Study selection (Figure 1 — PRISMA flow)
|
|
9
|
+
- The identification→screening→eligibility→inclusion cascade as the PRISMA 2020 flow diagram:
|
|
10
|
+
records identified, duplicates removed, screened, full texts assessed, and excluded **with
|
|
11
|
+
reasons and counts**.
|
|
12
|
+
- The cascade reconciles (identified − duplicates − excluded = included); state N studies in the
|
|
13
|
+
review vs N in the meta-analysis (they can differ). These numbers match the Abstract, Methods,
|
|
14
|
+
and Table 1 exactly (one locked count, re-derived — not re-typed per document).
|
|
15
|
+
|
|
16
|
+
## Study and patient characteristics (Table 1)
|
|
17
|
+
- Publication-year span, design mix (prospective / retrospective; RCT / observational), total
|
|
18
|
+
and per-study N (median, IQR/range), countries, and the effect-relevant descriptors.
|
|
19
|
+
- A **data-provenance column** when studies may share an institution / public database / imaging
|
|
20
|
+
benchmark — so overlapping-population non-independence is visible, not hidden.
|
|
21
|
+
|
|
22
|
+
## Risk of bias across studies (Figure 2)
|
|
23
|
+
- The RoB summary (QUADAS-2 / RoB 2 domain plot): how many studies were low risk overall and
|
|
24
|
+
the domain that drove concern, with counts (N/N).
|
|
25
|
+
|
|
26
|
+
## Quantitative synthesis — primary outcome (Figure 3, forest plot)
|
|
27
|
+
- The pooled effect with **95% CI**, the study count **k** and pooled **N**, and heterogeneity
|
|
28
|
+
reported as **I² with τ² and the 95% prediction interval** (state what the PI's bounds imply
|
|
29
|
+
for a new setting).
|
|
30
|
+
- The forest plot cited; per-study weights visible. For DTA, the bivariate/HSROC summary point
|
|
31
|
+
(pooled sensitivity & specificity with CIs) — check no sensitivity/specificity swap versus a
|
|
32
|
+
source.
|
|
33
|
+
|
|
34
|
+
## Subgroups, meta-regression, and sensitivity
|
|
35
|
+
- Each **pre-specified** subgroup with its estimate + CI **and the test for subgroup interaction**
|
|
36
|
+
(report the interaction p, not two bare stratum estimates); flag k=1 strata as descriptive, not
|
|
37
|
+
pooled.
|
|
38
|
+
- Leave-one-out, prospective-only, and low-RoB-only sensitivity results, reported as results — not
|
|
39
|
+
deferred to a supplement mention only.
|
|
40
|
+
|
|
41
|
+
## Publication bias / small-study effects
|
|
42
|
+
- Egger / Deeks asymmetry **only if ≥10 studies** (state the count gate); if trim-and-fill was
|
|
43
|
+
run, the adjusted estimate. Do not over-read a non-significant test — its power is low; say so.
|
|
44
|
+
|
|
45
|
+
## Common omission
|
|
46
|
+
- The **prediction interval alongside I²**, a **subgroup interaction test** (rather than two
|
|
47
|
+
separate stratum estimates), and the **overlapping-population/provenance** disclosure — the
|
|
48
|
+
Results elements MA drafts most often skip. Cross-reference `section_guides/results.md`, the
|
|
49
|
+
`peer-review/references/domain-probes/sr_ma.md` probes (P1, P2 non-independence, P3 subset-N
|
|
50
|
+
transparency, P4 k=1), and the PRISMA critical items in
|
|
51
|
+
`peer-review/references/reviewer_calibration/compliance_floor.md`.
|
|
@@ -8,6 +8,11 @@
|
|
|
8
8
|
- **Typical word count:** 4500–6000 words
|
|
9
9
|
- **Structure:** Abstract → Introduction → Methods → Results → Discussion → Conclusions
|
|
10
10
|
|
|
11
|
+
**Worked structure models** (paragraph order + what each paragraph must establish, synthetic):
|
|
12
|
+
`exemplar_methods/meta_analysis_prisma.md`, `exemplar_results/meta_analysis_prisma.md`,
|
|
13
|
+
`exemplar_discussion/meta_analysis_prisma.md`. This file carries the fuller prose templates;
|
|
14
|
+
the exemplars are the compact skeletons for confirming every load-bearing paragraph is present.
|
|
15
|
+
|
|
11
16
|
---
|
|
12
17
|
|
|
13
18
|
## Required at Phase 0 (Init)
|