medsci-skills 5.5.0 → 5.6.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/metadata/distribution_files.json +11 -6
- package/metadata/distribution_manifest.json +1 -1
- package/package.json +1 -1
- package/skills/self-review/SKILL.md +121 -0
- package/skills/self-review/references/panel_review_template.md +45 -4
- package/skills/self-review/scripts/check_editorial_impression.py +461 -0
- package/skills/self-review/skill.yml +2 -0
- package/skills/self-review/tests/fixtures/editorial_clean.md +29 -0
- package/skills/self-review/tests/fixtures/editorial_defensive.md +25 -0
- package/skills/self-review/tests/test_editorial_impression.sh +61 -0
|
@@ -3393,8 +3393,8 @@
|
|
|
3393
3393
|
},
|
|
3394
3394
|
{
|
|
3395
3395
|
"path": "skills/self-review/SKILL.md",
|
|
3396
|
-
"size":
|
|
3397
|
-
"sha256": "
|
|
3396
|
+
"size": 104683,
|
|
3397
|
+
"sha256": "8cc62fbae95f1dd18c5c62f65931904d981daa8a565bad2f8418e3671ca56e7c"
|
|
3398
3398
|
},
|
|
3399
3399
|
{
|
|
3400
3400
|
"path": "skills/self-review/references/domain-probes/ai_overclaiming.md",
|
|
@@ -3538,8 +3538,8 @@
|
|
|
3538
3538
|
},
|
|
3539
3539
|
{
|
|
3540
3540
|
"path": "skills/self-review/references/panel_review_template.md",
|
|
3541
|
-
"size":
|
|
3542
|
-
"sha256": "
|
|
3541
|
+
"size": 14863,
|
|
3542
|
+
"sha256": "a44567b5ea0c47971bcc75b0e6c46f3846d687240b0607ded07183d6fbaad86f"
|
|
3543
3543
|
},
|
|
3544
3544
|
{
|
|
3545
3545
|
"path": "skills/self-review/scripts/check_artifact_coverage.py",
|
|
@@ -3576,6 +3576,11 @@
|
|
|
3576
3576
|
"size": 21506,
|
|
3577
3577
|
"sha256": "7d3e67074d58a28ffee52ce64b486231f103a3ddcaf6b3b6ee83ba5f89c63bc2"
|
|
3578
3578
|
},
|
|
3579
|
+
{
|
|
3580
|
+
"path": "skills/self-review/scripts/check_editorial_impression.py",
|
|
3581
|
+
"size": 22032,
|
|
3582
|
+
"sha256": "42e0e9315e1c97ca0f9943213c65ea91d7064773d25287b2577f0386dfb7fc41"
|
|
3583
|
+
},
|
|
3579
3584
|
{
|
|
3580
3585
|
"path": "skills/self-review/scripts/check_null_calibration.py",
|
|
3581
3586
|
"size": 7594,
|
|
@@ -3613,8 +3618,8 @@
|
|
|
3613
3618
|
},
|
|
3614
3619
|
{
|
|
3615
3620
|
"path": "skills/self-review/skill.yml",
|
|
3616
|
-
"size":
|
|
3617
|
-
"sha256": "
|
|
3621
|
+
"size": 2223,
|
|
3622
|
+
"sha256": "4009f3148776fab2096da3dad8a15e503c5073b4cc66b42c57498948e2040270"
|
|
3618
3623
|
},
|
|
3619
3624
|
{
|
|
3620
3625
|
"path": "skills/setup-medsci/SKILL.md",
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "medsci-skills",
|
|
3
|
-
"version": "5.
|
|
3
|
+
"version": "5.6.0",
|
|
4
4
|
"description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
|
|
5
5
|
"license": "SEE LICENSE IN LICENSE",
|
|
6
6
|
"homepage": "https://github.com/Aperivue/medsci-skills#readme",
|
|
@@ -33,6 +33,29 @@ When flagging issues, classify severity:
|
|
|
33
33
|
|
|
34
34
|
Most issues are Fixable. Reserve Fatal for true design-level problems.
|
|
35
35
|
|
|
36
|
+
## Two Objectives: the Floor and the Ceiling
|
|
37
|
+
|
|
38
|
+
A submission-ready manuscript optimizes **two** things at once, and most of this skill (and
|
|
39
|
+
the gate stack behind it) only optimizes the first:
|
|
40
|
+
|
|
41
|
+
- **Floor — minimize rejection-for-cause.** Fabricated citations, numbers that do not
|
|
42
|
+
reconcile, overclaims, missing checklist items, leakage. Categories A–K and the
|
|
43
|
+
deterministic gates (Phases 2.5–2.5f) do this, and they are right to. Many of them raise the
|
|
44
|
+
floor by **adding** material: a hedge, a caveat, a disclosure, an audit trail, a checklist row.
|
|
45
|
+
- **Ceiling — maximize editorial-championing.** Will a handling editor read a *confident
|
|
46
|
+
narrative* (problem → design → result → meaning) and want to send it out, or a *defensive
|
|
47
|
+
audit* and bounce it? Nothing in the floor stack pushes here, and several floor gates push the
|
|
48
|
+
other way. Iterated, a manuscript over-hardens: every individual gate finding is correct, yet
|
|
49
|
+
the **accumulated** product reads as a rebuttal letter — over-hedged, audit-trail-heavy,
|
|
50
|
+
Abstract buried under caveats, the strongest sensitivity result hidden in Limitations, too long.
|
|
51
|
+
|
|
52
|
+
These objectives can conflict, so the order matters: **the floor gates run first and secure
|
|
53
|
+
accuracy; then the ceiling pass (category L / Phase 2.5g) reads the accurate manuscript as a
|
|
54
|
+
whole and recommends SUBTRACTION — REMOVE, MOVE, or TIGHTEN — so the same content is read
|
|
55
|
+
confidently.** The ceiling pass is advisory and never blocks; it cannot relax a floor gate.
|
|
56
|
+
Without it, repeated self-review monotonically over-defends. Surface the ceiling findings as
|
|
57
|
+
their own first-class output (Phase 3), not folded silently into the "add this" comments.
|
|
58
|
+
|
|
36
59
|
## Workflow
|
|
37
60
|
|
|
38
61
|
### Phase 1: Intake
|
|
@@ -250,6 +273,38 @@ fabrication. Resolution path:
|
|
|
250
273
|
1. Honest Methods/PROSPERO update (single-reviewer execution disclosed), OR
|
|
251
274
|
2. Limitations confession rewritten if dual review was actually completed.
|
|
252
275
|
|
|
276
|
+
#### L. Editorial impression & defensiveness (advisory; the counterweight)
|
|
277
|
+
|
|
278
|
+
This is the **ceiling** category (see "Two Objectives" above) and the inverse of the floor
|
|
279
|
+
gates: where A–K and the numerical gates ask "what is missing or wrong?" (and answer by
|
|
280
|
+
**adding**), L asks "does the accurate manuscript read confidently, or has it over-defended?"
|
|
281
|
+
(and answers by **subtracting**). Every L finding is **advisory (Minor / impression) and
|
|
282
|
+
non-blocking** — it never converts to a Major and never blocks submission. The fixes are
|
|
283
|
+
REMOVE / MOVE / TIGHTEN, not "add a caveat."
|
|
284
|
+
|
|
285
|
+
| Check | What to look for | Action |
|
|
286
|
+
|-------|-----------------|--------|
|
|
287
|
+
| Hedge density | Defensive-caveat tokens stacking up per 1,000 narrative words — the prose hedges faster than it asserts. Keep the load-bearing caveats; cut the reflexive ones. | TIGHTEN |
|
|
288
|
+
| Repeated caveat | The same caveat motif ("no deployable claim", "not generalizable", "hypothesis-generating") repeated across body + Abstract. Say it once, firmly. | TIGHTEN |
|
|
289
|
+
| Audit minutiae in body | Provenance tokens (SHA / git commit / unit-test / post-lock timeline / manifest / seed=N / audit trail) in the Introduction / Results / Discussion narrative. Reproducibility detail belongs in a Methods statement or a supplement. | MOVE |
|
|
290
|
+
| Limitations volume | A Limitations passage that enumerates a long list of discrete items reads as a rebuttal letter; consolidate related items. | TIGHTEN |
|
|
291
|
+
| Abstract caveat load | The Abstract carries several caveat clauses, burying the headline result before a reader reaches it. Lead with the result; keep one or two essential qualifiers. | TIGHTEN |
|
|
292
|
+
| Buried defense | A strong numeric robustness / sensitivity result sitting only in Limitations or the supplement, with no robustness mention in Results. Promote it into Results — it is *evidence for* the finding, not a caveat against it. (The inverse of the scope-coherence gate, which pushes a *weak* analysis out of Results.) | MOVE |
|
|
293
|
+
|
|
294
|
+
Run the deterministic gate (Phase 2.5g) rather than eyeballing it — these are all counts and placements:
|
|
295
|
+
|
|
296
|
+
```bash
|
|
297
|
+
python3 "${CLAUDE_SKILL_DIR}/scripts/check_editorial_impression.py" \
|
|
298
|
+
--manuscript manuscript.md --out qc/editorial_impression.json
|
|
299
|
+
```
|
|
300
|
+
|
|
301
|
+
`HEDGE_DENSITY`, `HEDGE_REPEAT`, `AUDIT_IN_BODY`, `LIMITATIONS_VOLUME`, `ABSTRACT_CAVEAT_LOAD`,
|
|
302
|
+
and `BURIED_DEFENSE` are Anticipated **Minor** Comments (category: L. Editorial impression),
|
|
303
|
+
each carrying a REMOVE / MOVE / TIGHTEN `action`. The gate never blocks (it has no Major and
|
|
304
|
+
exits 0 even under `--strict`); thresholds are tunable (`--hedge-per-1k`, `--repeat-threshold`,
|
|
305
|
+
`--limitations-max`, `--abstract-caveat-max`). It is conservative — each probe fires only on an
|
|
306
|
+
explicit, locatable signal.
|
|
307
|
+
|
|
253
308
|
### Research-Type Adaptation
|
|
254
309
|
|
|
255
310
|
Not all categories apply equally to every study type. Use this routing table:
|
|
@@ -267,6 +322,7 @@ Not all categories apply equally to every study type. Use this routing table:
|
|
|
267
322
|
| I. Protocol Heterogeneity | Full | Full | N/A | Per-study | N/A | Full |
|
|
268
323
|
| J. Method Transparency | Full | Partial | Partial | N/A | N/A | Partial |
|
|
269
324
|
| K. Reviewer-team consistency | N/A | N/A | N/A | Full | N/A | N/A |
|
|
325
|
+
| L. Editorial impression | Full | Full | Full | Full | Full | Full |
|
|
270
326
|
|
|
271
327
|
*Meta-analysis: Replace C with heterogeneity assessment (I-squared, prediction intervals),
|
|
272
328
|
publication bias (funnel plot, Egger), and sensitivity/subgroup analyses.
|
|
@@ -1016,6 +1072,48 @@ reconciliation in `qc/claim_artifact.json` and confirm against the actual regist
|
|
|
1016
1072
|
before raising `ESTIMAND_DRIFT`. For time-to-event manuscripts, also apply probe **S8
|
|
1017
1073
|
(estimand provenance)** of `references/domain-probes/survival_prognostic.md`.
|
|
1018
1074
|
|
|
1075
|
+
### Phase 2.5g: Editorial-Impression / Defensiveness Scan (the ceiling pass)
|
|
1076
|
+
|
|
1077
|
+
Run this **after** the floor gates (Phases 2.5–2.5f), because it reads the *accurate* manuscript
|
|
1078
|
+
and recommends what to take back out. It is the operational form of category L and the
|
|
1079
|
+
counterweight to the additive bias of the rest of the stack: every other phase can only make the
|
|
1080
|
+
manuscript longer and more defended; this one is the only phase that can make it shorter and more
|
|
1081
|
+
confident. It is advisory and **non-blocking** — it never produces a Major and never gates
|
|
1082
|
+
submission.
|
|
1083
|
+
|
|
1084
|
+
```bash
|
|
1085
|
+
python3 "${CLAUDE_SKILL_DIR}/scripts/check_editorial_impression.py" \
|
|
1086
|
+
--manuscript manuscript.md --out qc/editorial_impression.json
|
|
1087
|
+
```
|
|
1088
|
+
|
|
1089
|
+
The gate reads the manuscript as a whole, segments it by IMRAD heading, and emits up to six
|
|
1090
|
+
verdicts, each tagged with a SUBTRACTION `action`:
|
|
1091
|
+
|
|
1092
|
+
| Verdict | Reads as | Action |
|
|
1093
|
+
|---|---|---|
|
|
1094
|
+
| `HEDGE_DENSITY` | defensive-caveat tokens per 1,000 narrative words over threshold | TIGHTEN |
|
|
1095
|
+
| `HEDGE_REPEAT` | one caveat motif repeated across body + Abstract | TIGHTEN |
|
|
1096
|
+
| `AUDIT_IN_BODY` | SHA / commit / unit-test / post-lock / manifest / seed in the narrative | MOVE (→ Methods/supplement) |
|
|
1097
|
+
| `LIMITATIONS_VOLUME` | a long enumerated Limitations list | TIGHTEN (consolidate) |
|
|
1098
|
+
| `ABSTRACT_CAVEAT_LOAD` | several caveat clauses in the Abstract | TIGHTEN |
|
|
1099
|
+
| `BURIED_DEFENSE` | strong numeric robustness result only in Limitations/supplement | MOVE (→ Results) |
|
|
1100
|
+
|
|
1101
|
+
**Fold the findings into the report as the SUBTRACTION axis, not the additive one.** Each
|
|
1102
|
+
becomes a Minor `issues[]` entry under `category: "L" / category_name: "Editorial impression"`,
|
|
1103
|
+
additively carrying `issue_type: "editorial_impression"`, `subtype: <verdict>`, and
|
|
1104
|
+
`action: "REMOVE" | "MOVE" | "TIGHTEN"`. They are summarized in their own Phase 3 block
|
|
1105
|
+
("Editorial-Impression Risks — REMOVE / MOVE / TIGHTEN"), kept visually separate from the
|
|
1106
|
+
"Anticipated Major / Minor Comments (ADD / FIX)" so the author sees both forces. Mark them
|
|
1107
|
+
`fixable_by_ai: false` by default — TIGHTEN-ing a hedge or MOVE-ing a robustness result is a
|
|
1108
|
+
voice-and-judgment edit the author should own — except a clearly-redundant repeated caveat
|
|
1109
|
+
(`HEDGE_REPEAT`), which `--fix` may collapse to a single statement.
|
|
1110
|
+
|
|
1111
|
+
**Net-impact note.** When an *earlier* phase recommends adding a caveat or disclosure, weigh it
|
|
1112
|
+
against L: an integrity-critical disclosure is a **must (state it once, crisply)**, but a
|
|
1113
|
+
defensive over-disclosure is a **cut / move**. The two are not symmetric — keep the disclosure,
|
|
1114
|
+
but place it once and point to the supplement rather than repeating it at every claim site
|
|
1115
|
+
(placement discipline: main text narrates, auditability lives in the supplement).
|
|
1116
|
+
|
|
1019
1117
|
### Phase 2.6: Multi-Agent Panel Review (--panel, opt-in)
|
|
1020
1118
|
|
|
1021
1119
|
Run this phase **only when `--panel` is passed**. The default single-pass review (Phases 2–2.5d) stays the fast path; the panel is the high-cost, high-precision option for a pre-submission final pass on a top-tier target. Run it after the numerical audits (Phases 2.5–2.5d) so the reviewers see source-verified numbers, and before the Phase 3 report, which it feeds.
|
|
@@ -1038,6 +1136,12 @@ The panel simulates independent peer reviewers who do not see each other's comme
|
|
|
1038
1136
|
|
|
1039
1137
|
If the type is ambiguous, ask the user before composing the set.
|
|
1040
1138
|
|
|
1139
|
+
Append the **handling-editor desk-impression** persona (the ceiling lens) to every reviewer set:
|
|
1140
|
+
it loads no domain probe, reads only for narrative confidence vs over-defensiveness, and returns
|
|
1141
|
+
Minor REMOVE / MOVE / TIGHTEN findings (category L) that the editor routes to the separate
|
|
1142
|
+
Editorial-Impression Risks block. Its focus checklist is in `references/panel_review_template.md`.
|
|
1143
|
+
It does not count toward the Step 3.5 lens-diversity axes.
|
|
1144
|
+
|
|
1041
1145
|
**Step 2 — Run the reviewers (portable execution).** When the host provides a parallel subagent / Task capability (Claude Code, or any harness exposing an Agent tool), spawn the reviewer set as independent parallel subagents, each blinded to the others, then run the editor as a final synthesis agent. **Fallback (no subagent capability — e.g. a minimal Codex/Cursor harness):** a single agent role-plays each reviewer sequentially and in isolation — it completes and writes out reviewer R1's full structured review before reading the manuscript "fresh" as R2, so a later reviewer never sees an earlier reviewer's comments. The panel is defined by these instructions; it does **not** depend on the `Workflow` tool or any Claude-Code-only orchestration.
|
|
1042
1146
|
|
|
1043
1147
|
A reusable reviewer schema, a generic harsh-but-fair reviewer prompt skeleton with per-domain focus checklists, and the editor synthesis prompt skeleton live in `${CLAUDE_SKILL_DIR}/references/panel_review_template.md`.
|
|
@@ -1103,6 +1207,15 @@ M2. ...
|
|
|
1103
1207
|
m1. **{Issue}** [{Category}]: {1 sentence with location + fix}
|
|
1104
1208
|
m2. ...
|
|
1105
1209
|
|
|
1210
|
+
## Editorial-Impression Risks (REMOVE / MOVE / TIGHTEN)
|
|
1211
|
+
|
|
1212
|
+
*The subtraction axis — what to take out, move, or tighten so the accurate manuscript reads
|
|
1213
|
+
confidently. Advisory and non-blocking; from Phase 2.5g / category L. Omit this block only if the
|
|
1214
|
+
scan returned nothing.*
|
|
1215
|
+
|
|
1216
|
+
L1. **{Issue}** [{REMOVE | MOVE | TIGHTEN}]: {1 sentence — what reads as over-defensive and where, with the subtraction to make}
|
|
1217
|
+
L2. ...
|
|
1218
|
+
|
|
1106
1219
|
## Strengths (emphasize in cover letter)
|
|
1107
1220
|
|
|
1108
1221
|
- {Specific strength 1}
|
|
@@ -1110,9 +1223,15 @@ m2. ...
|
|
|
1110
1223
|
- ...
|
|
1111
1224
|
```
|
|
1112
1225
|
|
|
1226
|
+
The report carries **two** axes, kept visually separate: the **ADD / FIX** axis (Anticipated
|
|
1227
|
+
Major / Minor Comments — what is missing or wrong) and the **SUBTRACTION** axis
|
|
1228
|
+
(Editorial-Impression Risks — what to remove, move, or tighten). Do not fold the L items into the
|
|
1229
|
+
Minor Comments; an author who sees only "add this" will monotonically over-defend.
|
|
1230
|
+
|
|
1113
1231
|
**Conciseness targets**:
|
|
1114
1232
|
- Anticipated Major Comments: 3-7 items, each 3-5 lines
|
|
1115
1233
|
- Anticipated Minor Comments: 3-6 items, each 1-2 sentences
|
|
1234
|
+
- Editorial-Impression Risks: 0-6 items, each 1 sentence (only what the Phase 2.5g gate flagged)
|
|
1116
1235
|
- Strengths: 3-5 items, each 1 sentence
|
|
1117
1236
|
- Total report: 400-800 words (excluding optional R0 section)
|
|
1118
1237
|
|
|
@@ -1181,6 +1300,7 @@ When `--json` is passed, or when invoked by `/write-paper` Phase 7, append a mac
|
|
|
1181
1300
|
- `requires_reanalysis` *(optional, default `false`)*: `true` when closing the finding needs a **committed analysis re-run against the real data**, not a prose edit — power/MDE re-simulation under the full model, first-visit/one-record-per-subject dedup, an extended- or reduced-adjustment sensitivity model, optimism correction of calibration. Always implies `fixable_by_ai: false`. Additive and backwards-compatible; parsers that do not expect it must ignore it. Route these to `/analyze-stats` (see Phase 4).
|
|
1182
1301
|
- `suggested_fix`: Specific, actionable instruction. If `fixable_by_ai` is true, this must be concrete enough for the fixer to execute without ambiguity.
|
|
1183
1302
|
- `consensus` *(optional, panel mode only)*: array of reviewer ids that raised the issue, e.g. `["R1","R3"]`. Additive and backwards-compatible — present only when Phase 2.6 ran; parsers that do not expect it must ignore it.
|
|
1303
|
+
- `action` *(optional, editorial-impression findings only)*: `"REMOVE" | "MOVE" | "TIGHTEN"` — the SUBTRACTION direction for a category-L finding (Phase 2.5g). Present alongside `issue_type: "editorial_impression"` and `subtype: <verdict>` (e.g. `HEDGE_REPEAT`). Additive and backwards-compatible; these are always `severity: "minor"`, never block, and are `fixable_by_ai: false` by default (except a redundant `HEDGE_REPEAT`, which `--fix` may collapse). Parsers that do not expect it must ignore it.
|
|
1184
1304
|
|
|
1185
1305
|
### Phase 4: Fix Support
|
|
1186
1306
|
|
|
@@ -1272,5 +1392,6 @@ Here is how to address it with your existing data."
|
|
|
1272
1392
|
| Phase 2.5c reference hallucination scan (delegate `/verify-refs`) | ENFORCED | `FABRICATED` in `records[]` OR nonempty `duplicate_findings[]` | P0 Major Comment, blocks submission |
|
|
1273
1393
|
| Phase 2.5a-2 design/power statistic provenance | ENFORCED | a reported MDE / power / sample-size value is not reproduced by committed code, or is reproducible only by a method the committed script does not implement | Major Comment (P0 if a headline claim); recompute and either correct the value or update the committed code to reproduce it |
|
|
1274
1394
|
| `--fix` auto-fix loop (max 2 iterations) | ENFORCED in `/write-paper` Phase 7.4 chain | score still below threshold after 2 iterations | Route to write-paper Phase 7.4a Audit Recovery |
|
|
1395
|
+
| Phase 2.5g editorial-impression scan (`check_editorial_impression.py`) | ADVISORY (non-blocking) | HEDGE_DENSITY / HEDGE_REPEAT / AUDIT_IN_BODY / LIMITATIONS_VOLUME / ABSTRACT_CAVEAT_LOAD / BURIED_DEFENSE | Minor REMOVE/MOVE/TIGHTEN recommendation in the Editorial-Impression Risks block; never blocks submission |
|
|
1275
1396
|
| R0 numbering output | OPT-IN | `--r0-numbering` flag or downstream `/revise` consumer | Emits structured Anticipated Major/Minor Comments — consumable by `/revise` |
|
|
1276
1397
|
| `--json` machine-readable output | OPT-IN | `--json` flag | Emits parseable JSON block consumed by `/orchestrate` post-skill validation |
|
|
@@ -126,6 +126,36 @@ design-level finding, **Fixable** for a reporting-level finding.
|
|
|
126
126
|
- RV6 single-anchor overload: does a load-bearing clinical claim rest on essentially one (often abstract-only/unreplicated) study while the Abstract calls it "landmark" and the body concedes the base "is thin"? Flag the Abstract↔body register mismatch.
|
|
127
127
|
- RV8 self-citation architecture: do the weakest/most-deferred axes coincide with the authors' own forthcoming/companion work without a body-level disclosure, and does each load-bearing axis carry ≥1 independent (other-group) source?
|
|
128
128
|
|
|
129
|
+
**Handling editor — desk-impression / champion-or-bounce (cross-type, the ceiling lens)**
|
|
130
|
+
|
|
131
|
+
This persona is the counterweight to the rigor reviewers above. It is **not** a domain expert and
|
|
132
|
+
loads **no** domain-probe module; it reads the manuscript exactly as a handling editor skimming
|
|
133
|
+
for the desk decision, and its only question is whether the manuscript reads as a **confident
|
|
134
|
+
narrative** an editor would champion or a **defensive audit** an editor would bounce. It raises no
|
|
135
|
+
Major and no Fatal — every finding is a Minor with a SUBTRACTION action (REMOVE / MOVE / TIGHTEN),
|
|
136
|
+
mirroring category L / Phase 2.5g. Append it to any reviewer set on a `--panel` run; it does
|
|
137
|
+
**not** count toward the lens-diversity axes (its findings fall to the `other` family and the gate
|
|
138
|
+
never penalizes an extra lens). Add the optional `action` field (`"REMOVE" | "MOVE" | "TIGHTEN"`)
|
|
139
|
+
to each of its `minor[]` items.
|
|
140
|
+
|
|
141
|
+
- Does the manuscript open by stating the problem, the design, and the headline result, or does it
|
|
142
|
+
bury the result behind qualifiers? Is the Abstract carrying several caveat clauses before a
|
|
143
|
+
reader reaches the finding? (TIGHTEN)
|
|
144
|
+
- Is the strongest robustness / sensitivity result up front in Results, or hidden in Limitations or
|
|
145
|
+
the supplement where it reads as a caveat rather than as evidence? (MOVE → Results)
|
|
146
|
+
- Does the narrative (Introduction / Results / Discussion) carry audit minutiae — hashes, commit
|
|
147
|
+
ids, unit-test mentions, post-lock timelines, manifests — that belong in a Methods statement or a
|
|
148
|
+
supplement? (MOVE)
|
|
149
|
+
- Is the same caveat repeated at multiple claim sites? Does the Limitations section read as a
|
|
150
|
+
consolidated honest disclosure or as a long rebuttal-letter enumeration? Say each caveat once,
|
|
151
|
+
firmly. (TIGHTEN / REMOVE)
|
|
152
|
+
- Overall: is the manuscript longer and more defended than its evidence requires? Name the single
|
|
153
|
+
change that would most raise an editor's confidence on a skim — and it should be a subtraction.
|
|
154
|
+
|
|
155
|
+
Run `scripts/check_editorial_impression.py` first and use its verdicts as the deterministic spine
|
|
156
|
+
for this persona's findings, then add anything the gate cannot see (tone, narrative order, a
|
|
157
|
+
result that is technically present but framed apologetically).
|
|
158
|
+
|
|
129
159
|
---
|
|
130
160
|
|
|
131
161
|
## Editor synthesis prompt skeleton
|
|
@@ -147,6 +177,14 @@ design-level finding, **Fixable** for a reporting-level finding.
|
|
|
147
177
|
> undisclosed near-identical prior publication, or (for a review/primer) weak novelty /
|
|
148
178
|
> no distinct contribution since the contribution IS the product — let it dominate the
|
|
149
179
|
> tier over fixable reporting defects rather than letting the fixable framing soften it.
|
|
180
|
+
> Then run a THIRD, opposite-direction lens — **defensiveness / narrative** — symmetric to
|
|
181
|
+
> the contribution lens but guarding the other failure: is the manuscript *over-defended*?
|
|
182
|
+
> Does it read as a confident narrative or as a rebuttal letter (over-hedged, audit-trail in
|
|
183
|
+
> the body, Abstract buried under caveats, the strongest sensitivity result hidden in
|
|
184
|
+
> Limitations, too long)? Treat a defensive over-disclosure as a **cut / move**, not a virtue,
|
|
185
|
+
> while keeping any integrity-critical disclosure (stated once, crisply). The contribution lens
|
|
186
|
+
> guards against a too-lenient verdict; this lens guards against blessing an over-hardened
|
|
187
|
+
> manuscript an editor would bounce on impression.
|
|
150
188
|
> 2. De-duplicate and consolidate the major comments by theme. For each consolidated
|
|
151
189
|
> point, flag CONSENSUS (raised by ≥2 reviewers) or single-reviewer, and attribute
|
|
152
190
|
> (R1/R2/R3).
|
|
@@ -159,10 +197,13 @@ design-level finding, **Fixable** for a reporting-level finding.
|
|
|
159
197
|
> rather than implicit in the attribution.
|
|
160
198
|
>
|
|
161
199
|
> Map every finding onto the self-review framing (Fatal / Fixable, category letters
|
|
162
|
-
> A–
|
|
163
|
-
> JSON, adding the optional `consensus` field where ≥2 reviewers agreed.
|
|
164
|
-
>
|
|
165
|
-
>
|
|
200
|
+
> A–L) and emit it through the Phase 3 report, Phase 3b R0 numbering, and Phase 3c
|
|
201
|
+
> JSON, adding the optional `consensus` field where ≥2 reviewers agreed. Route the
|
|
202
|
+
> handling-editor desk-impression findings to the separate Editorial-Impression Risks
|
|
203
|
+
> block (category L, each with a REMOVE / MOVE / TIGHTEN `action`); do not fold them into
|
|
204
|
+
> the Anticipated Major / Minor (ADD / FIX) comments, so the author sees both forces.
|
|
205
|
+
> Follow the manuscript-style rules: no "§" symbols, minimal em-dashes, full prose, cite
|
|
206
|
+
> specific locations.
|
|
166
207
|
|
|
167
208
|
---
|
|
168
209
|
|
|
@@ -0,0 +1,461 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""Editorial-impression / defensiveness gate (self-review §L) — the counterweight pass.
|
|
3
|
+
|
|
4
|
+
The rest of the MedSci-Audit stack minimizes *rejection-for-cause* (the floor):
|
|
5
|
+
fabricated citations, drifting numbers, overclaims, missing checklist items. Several
|
|
6
|
+
of those gates raise the floor by *adding* material — a hedge, a caveat, a disclosure,
|
|
7
|
+
a checklist row — and nothing in the stack pushes back. Iterated, a manuscript
|
|
8
|
+
monotonically over-hardens: a confident narrative turns into a defensive audit that an
|
|
9
|
+
editor reads as a risk signal even when every individual gate finding was correct.
|
|
10
|
+
|
|
11
|
+
This gate is the missing opposite force. It does not relax any integrity gate; it scans
|
|
12
|
+
the *manuscript as a whole* for editorial-impression risks and recommends SUBTRACTION —
|
|
13
|
+
REMOVE, MOVE, or TIGHTEN — so the accurate content the gates secured is also read
|
|
14
|
+
confidently. Every finding is advisory (Minor / impression) and NON-BLOCKING: this gate
|
|
15
|
+
never returns a submission blocker. It raises the ceiling; it does not gate the floor.
|
|
16
|
+
|
|
17
|
+
HEDGE_DENSITY defensive-caveat tokens per 1,000 body-narrative words exceed a
|
|
18
|
+
threshold — the prose hedges faster than it asserts. TIGHTEN.
|
|
19
|
+
HEDGE_REPEAT one caveat motif ("no deployable claim", "not generalizable",
|
|
20
|
+
"none evaluated here") repeats >=N times across body + abstract.
|
|
21
|
+
Say it once, firmly. TIGHTEN.
|
|
22
|
+
AUDIT_IN_BODY provenance/audit minutiae (SHA / git commit / unit-test /
|
|
23
|
+
post-lock timeline / manifest / seed=N / audit trail) appear in
|
|
24
|
+
the Introduction / Results / Discussion narrative rather than a
|
|
25
|
+
Methods reproducibility statement or a supplement. MOVE.
|
|
26
|
+
LIMITATIONS_VOLUME the Limitations passage enumerates more than N discrete items;
|
|
27
|
+
a wall of limitations reads as a rebuttal letter. TIGHTEN.
|
|
28
|
+
ABSTRACT_CAVEAT_LOAD the Abstract carries >=N caveat clauses; the headline result is
|
|
29
|
+
buried under qualifiers before a reader reaches it. TIGHTEN.
|
|
30
|
+
BURIED_DEFENSE a strong numeric robustness / sensitivity result sits only in the
|
|
31
|
+
Limitations / supplement, with no robustness mention in Results.
|
|
32
|
+
This is the inverse of the scope-coherence gate: scope-coherence
|
|
33
|
+
pushes a *weak* analysis out of Results; BURIED_DEFENSE pulls a
|
|
34
|
+
*strong* confound rebuttal back into Results. MOVE (promote).
|
|
35
|
+
|
|
36
|
+
Conservative by construction: each probe fires only on an explicit, locatable signal, to
|
|
37
|
+
keep false positives low on a widely-used skill. The gate needs IMRAD-style headings to
|
|
38
|
+
locate sections; with none it degrades to a whole-document density read.
|
|
39
|
+
|
|
40
|
+
INPUTS
|
|
41
|
+
--manuscript manuscript markdown/text (required).
|
|
42
|
+
thresholds --hedge-per-1k (10.0), --repeat-threshold (3), --limitations-max (6),
|
|
43
|
+
--abstract-caveat-max (2). A probe fires when its count exceeds the max.
|
|
44
|
+
|
|
45
|
+
OUTPUT
|
|
46
|
+
A reconciliation table (stdout) and, with --out, a JSON artifact:
|
|
47
|
+
{manuscript, claims[{verdict, severity, action, detail, where}], summary}
|
|
48
|
+
Every claim is severity "Minor" with an action of REMOVE / MOVE / TIGHTEN. Exit code is
|
|
49
|
+
always 0 for the findings themselves (advisory); --strict is accepted for CLI parity
|
|
50
|
+
with the other gates but never blocks, since this gate emits no Major.
|
|
51
|
+
|
|
52
|
+
Stdlib-only (json / re / argparse / pathlib). Exit codes: 0 clean or advisory findings,
|
|
53
|
+
2 input/usage error.
|
|
54
|
+
"""
|
|
55
|
+
|
|
56
|
+
from __future__ import annotations
|
|
57
|
+
|
|
58
|
+
import argparse
|
|
59
|
+
import json
|
|
60
|
+
import re
|
|
61
|
+
import sys
|
|
62
|
+
from pathlib import Path
|
|
63
|
+
|
|
64
|
+
# --------------------------------------------------------------------------- #
|
|
65
|
+
# Section segmentation
|
|
66
|
+
# --------------------------------------------------------------------------- #
|
|
67
|
+
|
|
68
|
+
HEADING_RE = re.compile(r"^(#{1,6})\s*\*{0,2}(.+?)\*{0,2}\s*$", re.MULTILINE)
|
|
69
|
+
|
|
70
|
+
# Narrative regions where defensive prose and out-of-place audit minutiae read worst.
|
|
71
|
+
# Methods is deliberately excluded (a reproducibility statement belongs there), as is
|
|
72
|
+
# any supplement / availability / declarations region.
|
|
73
|
+
BODY_NARRATIVE = {"introduction", "results", "discussion", "conclusion", "limitations"}
|
|
74
|
+
|
|
75
|
+
|
|
76
|
+
def classify_heading(h: str) -> str:
|
|
77
|
+
t = h.lower().strip()
|
|
78
|
+
if "abstract" in t or t == "summary":
|
|
79
|
+
return "abstract"
|
|
80
|
+
if any(k in t for k in (
|
|
81
|
+
"data availability", "code availability", "availability", "supplement",
|
|
82
|
+
"appendix", "acknowledg", "funding", "declaration", "competing interest",
|
|
83
|
+
"conflict of interest", "reproducibility", "references", "author contribution",
|
|
84
|
+
)):
|
|
85
|
+
return "supplement"
|
|
86
|
+
if "limitation" in t:
|
|
87
|
+
return "limitations"
|
|
88
|
+
if "introduction" in t or "background" in t:
|
|
89
|
+
return "introduction"
|
|
90
|
+
if any(k in t for k in (
|
|
91
|
+
"method", "material", "statistical analys", "study design",
|
|
92
|
+
"patients and", "data collection", "study population",
|
|
93
|
+
)):
|
|
94
|
+
return "methods"
|
|
95
|
+
if "result" in t or "finding" in t:
|
|
96
|
+
return "results"
|
|
97
|
+
if "discussion" in t:
|
|
98
|
+
return "discussion"
|
|
99
|
+
if "conclusion" in t:
|
|
100
|
+
return "conclusion"
|
|
101
|
+
return "other"
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
def segment(text: str) -> list[tuple[str, str]]:
|
|
105
|
+
"""Return an ordered list of (region, body_text) pairs. Text before the first
|
|
106
|
+
heading is a 'preamble' region. Region names follow classify_heading()."""
|
|
107
|
+
matches = list(HEADING_RE.finditer(text))
|
|
108
|
+
regions: list[tuple[str, str]] = []
|
|
109
|
+
if not matches:
|
|
110
|
+
return [("preamble", text)]
|
|
111
|
+
if matches[0].start() > 0:
|
|
112
|
+
pre = text[: matches[0].start()].strip()
|
|
113
|
+
if pre:
|
|
114
|
+
regions.append(("preamble", pre))
|
|
115
|
+
for i, m in enumerate(matches):
|
|
116
|
+
start = m.end()
|
|
117
|
+
end = matches[i + 1].start() if i + 1 < len(matches) else len(text)
|
|
118
|
+
body = text[start:end].strip()
|
|
119
|
+
regions.append((classify_heading(m.group(2)), body))
|
|
120
|
+
return regions
|
|
121
|
+
|
|
122
|
+
|
|
123
|
+
def region_text(regions: list[tuple[str, str]], names: set[str]) -> str:
|
|
124
|
+
return "\n".join(b for r, b in regions if r in names)
|
|
125
|
+
|
|
126
|
+
|
|
127
|
+
def abstract_text(regions: list[tuple[str, str]], full: str) -> str:
|
|
128
|
+
"""The Abstract region; fall back to the text before the first Introduction/Methods
|
|
129
|
+
heading (a structured abstract without its own heading), capped to ~350 words."""
|
|
130
|
+
abs_regions = [b for r, b in regions if r == "abstract"]
|
|
131
|
+
if abs_regions:
|
|
132
|
+
return "\n".join(abs_regions)
|
|
133
|
+
# Fallback: preamble + everything up to the first intro/methods region.
|
|
134
|
+
out: list[str] = []
|
|
135
|
+
for r, b in regions:
|
|
136
|
+
if r in ("introduction", "methods", "results", "discussion"):
|
|
137
|
+
break
|
|
138
|
+
if r in ("preamble", "other"):
|
|
139
|
+
out.append(b)
|
|
140
|
+
joined = "\n".join(out)
|
|
141
|
+
words = joined.split()
|
|
142
|
+
return " ".join(words[:350]) if len(words) > 350 else joined
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def word_count(text: str) -> int:
|
|
146
|
+
return sum(1 for w in text.split() if any(c.isalpha() for c in w))
|
|
147
|
+
|
|
148
|
+
|
|
149
|
+
def sentences(text: str) -> list[str]:
|
|
150
|
+
# Lightweight sentence split on ., !, ? followed by whitespace.
|
|
151
|
+
parts = re.split(r"(?<=[.!?])\s+", text.strip())
|
|
152
|
+
return [s.strip() for s in parts if s.strip()]
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
# --------------------------------------------------------------------------- #
|
|
156
|
+
# Lexicons
|
|
157
|
+
# --------------------------------------------------------------------------- #
|
|
158
|
+
|
|
159
|
+
# Defensive caveats — explicitly hedging phrases, not ordinary modal verbs. Stacking
|
|
160
|
+
# these is the defensiveness tell HEDGE_DENSITY measures.
|
|
161
|
+
CAVEAT = re.compile(
|
|
162
|
+
r"\bcaveats?\b|should be interpreted with caution|with caution\b|"
|
|
163
|
+
r"must be interpreted|interpreted? with care|"
|
|
164
|
+
r"cannot be (?:inferred|established|excluded|determined|generaliz\w+|ruled out|drawn|assumed)|"
|
|
165
|
+
r"no (?:causal|deployable|clinical|definitive) (?:claim|inference|conclusion|relationship)|"
|
|
166
|
+
r"not (?:be )?(?:generaliz\w+|definitive|conclusive|deployable|warranted)|"
|
|
167
|
+
r"\bpreliminary\b|\bexploratory\b|hypothesis[-\s]generating|"
|
|
168
|
+
r"warrants? (?:caution|further (?:study|validation|research|investigation))|"
|
|
169
|
+
r"remains? (?:unclear|uncertain|to be (?:established|determined|confirmed))|"
|
|
170
|
+
r"\blimited (?:by|generaliz\w+|sample|to|in scope)|"
|
|
171
|
+
r"single[-\s](?:cent(?:er|re)|institution|site)|retrospective (?:design|nature)|"
|
|
172
|
+
r"should not be (?:used|interpreted|construed)|"
|
|
173
|
+
r"do(?:es)? not (?:establish|imply|permit|support|prove)|"
|
|
174
|
+
r"not (?:yet )?(?:ready|validated|intended) for (?:clinical|deployment|practice)|"
|
|
175
|
+
r"\bunderpowered\b|\bmodest\b|no (?:firm|strong) (?:conclusion|inference)",
|
|
176
|
+
re.IGNORECASE)
|
|
177
|
+
|
|
178
|
+
# Repeated caveat motifs (HEDGE_REPEAT): family key -> regex. A family repeating across
|
|
179
|
+
# body + abstract above the threshold should be stated once, firmly.
|
|
180
|
+
MOTIFS: dict[str, re.Pattern] = {
|
|
181
|
+
"no_deployable_claim": re.compile(
|
|
182
|
+
r"no (?:deployable|deployment|clinical|practice|diagnostic) (?:claim|use|recommendation)|"
|
|
183
|
+
r"not (?:ready|intended|validated) for (?:clinical|deployment|practice)",
|
|
184
|
+
re.IGNORECASE),
|
|
185
|
+
"not_generalizable": re.compile(r"not (?:be )?generaliz\w+|limited generaliz\w+", re.IGNORECASE),
|
|
186
|
+
"none_evaluated_here": re.compile(
|
|
187
|
+
r"(?:none|not|no \w+) (?:were |was |are |is )?evaluated (?:here|in this (?:study|work|analysis))|"
|
|
188
|
+
r"not (?:assessed|examined|tested) (?:here|in this (?:study|work))", re.IGNORECASE),
|
|
189
|
+
"no_causal": re.compile(r"no causal (?:claim|inference|relationship|conclusion|interpretation)", re.IGNORECASE),
|
|
190
|
+
"hypothesis_generating": re.compile(r"hypothesis[-\s]generating", re.IGNORECASE),
|
|
191
|
+
"interpret_with_caution": re.compile(
|
|
192
|
+
r"interpret\w* with caution|should be interpreted with caution|with caution", re.IGNORECASE),
|
|
193
|
+
"single_center": re.compile(r"single[-\s](?:cent(?:er|re)|institution|site)", re.IGNORECASE),
|
|
194
|
+
"retrospective_design": re.compile(r"retrospective (?:design|nature|study|cohort)", re.IGNORECASE),
|
|
195
|
+
"preliminary": re.compile(r"\bpreliminary\b", re.IGNORECASE),
|
|
196
|
+
}
|
|
197
|
+
|
|
198
|
+
# Motifs that are factual study-design descriptors rather than defensive caveats. A
|
|
199
|
+
# single-centre / retrospective study must state its design in Methods, and naming it
|
|
200
|
+
# again in the Abstract and Limitations is normal, not over-hedging. Count these only
|
|
201
|
+
# in the non-Methods narrative so an honestly-written single-centre retrospective study
|
|
202
|
+
# (the most common observational design) is not flagged for stating a true fact.
|
|
203
|
+
FACTUAL_DESCRIPTOR_MOTIFS = {"single_center", "retrospective_design"}
|
|
204
|
+
|
|
205
|
+
# Provenance / audit minutiae that belong in Methods or a supplement, not the narrative.
|
|
206
|
+
AUDIT = re.compile(
|
|
207
|
+
r"\bsha-?256\b|\bmd5\b|\bchecksum\b|(?:git\s+)?(?:commit|hash|sha)\s*[:=]?\s*[0-9a-f]{7,40}\b|"
|
|
208
|
+
r"\bcommit\s+[0-9a-f]{7,40}\b|\bunit[-\s]?test(?:s|ing|ed)?\b|\bpost[-\s]?lock\b|"
|
|
209
|
+
r"seed\s*=\s*\d+|\brandom seed\s+\d+\b|\baudit trail\b|"
|
|
210
|
+
r"reproducibility (?:manifest|hash|record)|data lock(?:ed)? on|\bcontent[-\s]hash\b",
|
|
211
|
+
re.IGNORECASE)
|
|
212
|
+
|
|
213
|
+
# Robustness / sensitivity vocabulary (for BURIED_DEFENSE).
|
|
214
|
+
ROBUST = re.compile(
|
|
215
|
+
r"sensitivity analys\w+|robustness|leave[-\s]one[-\s]out|leave[-\s]pair[-\s]out|"
|
|
216
|
+
r"\bE[-\s]?value\b|remained (?:significant|robust|consistent|unchanged|stable)|"
|
|
217
|
+
r"did not (?:materially |substantially |meaningfully )?(?:change|alter|differ)|"
|
|
218
|
+
r"results were (?:similar|consistent|robust|unchanged)|consistent across|"
|
|
219
|
+
r"after (?:excluding|adjusting for|accounting for)|tipping[-\s]point", re.IGNORECASE)
|
|
220
|
+
|
|
221
|
+
# A strong numeric token next to a robustness statement makes it Results-worthy.
|
|
222
|
+
NUMERIC = re.compile(
|
|
223
|
+
r"\b\d+\.\d+\b|\b\d{1,3}%|95%\s*ci|(?:OR|HR|RR|AUC|aHR|aOR)\s*[=:]?\s*\d|"
|
|
224
|
+
r"p\s*[<=>]\s*0?\.\d+", re.IGNORECASE)
|
|
225
|
+
|
|
226
|
+
ORDINALS = ["first", "second", "third", "fourth", "fifth", "sixth", "seventh",
|
|
227
|
+
"eighth", "ninth", "tenth"]
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
# --------------------------------------------------------------------------- #
|
|
231
|
+
# Probes
|
|
232
|
+
# --------------------------------------------------------------------------- #
|
|
233
|
+
|
|
234
|
+
def probe_hedge_density(regions, body, threshold) -> list[dict]:
|
|
235
|
+
words = word_count(body)
|
|
236
|
+
if words < 80: # too little narrative to judge density reliably
|
|
237
|
+
return []
|
|
238
|
+
n = len(CAVEAT.findall(body))
|
|
239
|
+
density = n / words * 1000
|
|
240
|
+
if density > threshold:
|
|
241
|
+
return [{
|
|
242
|
+
"verdict": "HEDGE_DENSITY", "severity": "Minor", "action": "TIGHTEN",
|
|
243
|
+
"detail": (f"defensive-caveat density is {density:.1f} per 1,000 body words "
|
|
244
|
+
f"({n} caveat tokens / {words} words; threshold {threshold:.0f}); the "
|
|
245
|
+
f"prose hedges faster than it asserts — keep the load-bearing caveats, "
|
|
246
|
+
f"cut the reflexive ones"),
|
|
247
|
+
"where": f"body narrative ({words} words)",
|
|
248
|
+
}]
|
|
249
|
+
return []
|
|
250
|
+
|
|
251
|
+
|
|
252
|
+
def probe_hedge_repeat(regions, full, threshold) -> list[dict]:
|
|
253
|
+
non_methods = "\n".join(b for r, b in regions if r != "methods")
|
|
254
|
+
claims = []
|
|
255
|
+
for key, rx in MOTIFS.items():
|
|
256
|
+
# Factual design descriptors are not over-hedging when stated in Methods;
|
|
257
|
+
# count them only in the non-Methods narrative.
|
|
258
|
+
hay = non_methods if key in FACTUAL_DESCRIPTOR_MOTIFS else full
|
|
259
|
+
n = len(rx.findall(hay))
|
|
260
|
+
if n >= threshold:
|
|
261
|
+
m = rx.search(hay)
|
|
262
|
+
phrase = m.group(0).strip() if m else key
|
|
263
|
+
claims.append({
|
|
264
|
+
"verdict": "HEDGE_REPEAT", "severity": "Minor", "action": "TIGHTEN",
|
|
265
|
+
"detail": (f"the caveat '{phrase}' (motif: {key}) appears {n} times across "
|
|
266
|
+
f"the narrative; state it once, firmly, and remove the repeats"),
|
|
267
|
+
"where": phrase[:120],
|
|
268
|
+
})
|
|
269
|
+
return claims
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def probe_audit_in_body(regions) -> list[dict]:
|
|
273
|
+
body = region_text(regions, BODY_NARRATIVE)
|
|
274
|
+
claims = []
|
|
275
|
+
seen = set()
|
|
276
|
+
for m in AUDIT.finditer(body):
|
|
277
|
+
tok = m.group(0).strip().lower()
|
|
278
|
+
norm = re.sub(r"[0-9a-f]{7,40}", "<hash>", tok)
|
|
279
|
+
if norm in seen:
|
|
280
|
+
continue
|
|
281
|
+
seen.add(norm)
|
|
282
|
+
claims.append({
|
|
283
|
+
"verdict": "AUDIT_IN_BODY", "severity": "Minor", "action": "MOVE",
|
|
284
|
+
"detail": (f"provenance/audit token '{m.group(0).strip()}' appears in the "
|
|
285
|
+
f"Introduction/Results/Discussion narrative; move reproducibility "
|
|
286
|
+
f"detail to a Methods statement or a supplement"),
|
|
287
|
+
"where": body[max(0, m.start() - 40):m.end() + 40].strip()[:160],
|
|
288
|
+
})
|
|
289
|
+
return claims
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def probe_limitations_volume(regions, full, max_items) -> list[dict]:
|
|
293
|
+
lim = region_text(regions, {"limitations"})
|
|
294
|
+
if not lim:
|
|
295
|
+
# Inline limitations paragraph inside Discussion.
|
|
296
|
+
disc = region_text(regions, {"discussion", "conclusion"})
|
|
297
|
+
m = re.search(r"(?:our |this )?stud(?:y|ies) (?:has|have)[^.]{0,40}limitations?|"
|
|
298
|
+
r"several (?:important )?limitations|limitations? (?:of this|warrant)",
|
|
299
|
+
disc, re.IGNORECASE)
|
|
300
|
+
if not m:
|
|
301
|
+
return []
|
|
302
|
+
lim = disc[m.start():]
|
|
303
|
+
# Count discrete items: max of ordinal markers, (N) enumerators, bullet lines.
|
|
304
|
+
low = lim.lower()
|
|
305
|
+
n_ord = sum(1 for o in ORDINALS if re.search(rf"(?:^|[\s(,;])\b{o}\b\s*,", low))
|
|
306
|
+
n_enum = len(set(re.findall(r"\((\d{1,2})\)", lim)))
|
|
307
|
+
n_bullet = len(re.findall(r"^\s*[-*]\s+\S", lim, re.MULTILINE))
|
|
308
|
+
n = max(n_ord, n_enum, n_bullet)
|
|
309
|
+
if n > max_items:
|
|
310
|
+
return [{
|
|
311
|
+
"verdict": "LIMITATIONS_VOLUME", "severity": "Minor", "action": "TIGHTEN",
|
|
312
|
+
"detail": (f"the Limitations passage enumerates {n} discrete items "
|
|
313
|
+
f"(threshold {max_items}); consolidate related items so the section "
|
|
314
|
+
f"reads as honest disclosure, not a rebuttal letter"),
|
|
315
|
+
"where": f"Limitations ({n} items)",
|
|
316
|
+
}]
|
|
317
|
+
return []
|
|
318
|
+
|
|
319
|
+
|
|
320
|
+
def probe_abstract_caveat_load(regions, full, max_caveats) -> list[dict]:
|
|
321
|
+
abs_t = abstract_text(regions, full)
|
|
322
|
+
if word_count(abs_t) < 40:
|
|
323
|
+
return []
|
|
324
|
+
caveat_sents = [s for s in sentences(abs_t) if CAVEAT.search(s)]
|
|
325
|
+
n = len(caveat_sents)
|
|
326
|
+
if n > max_caveats:
|
|
327
|
+
return [{
|
|
328
|
+
"verdict": "ABSTRACT_CAVEAT_LOAD", "severity": "Minor", "action": "TIGHTEN",
|
|
329
|
+
"detail": (f"the Abstract carries {n} caveat-bearing clauses (threshold "
|
|
330
|
+
f"{max_caveats}); lead with the result and keep at most one or two "
|
|
331
|
+
f"essential qualifiers so the headline is not buried"),
|
|
332
|
+
"where": (caveat_sents[0][:140] if caveat_sents else "Abstract"),
|
|
333
|
+
}]
|
|
334
|
+
return []
|
|
335
|
+
|
|
336
|
+
|
|
337
|
+
def probe_buried_defense(regions) -> list[dict]:
|
|
338
|
+
results = region_text(regions, {"results"})
|
|
339
|
+
buried_src = region_text(regions, {"limitations", "supplement"})
|
|
340
|
+
if not buried_src:
|
|
341
|
+
return []
|
|
342
|
+
# If Results already discusses robustness, nothing is buried.
|
|
343
|
+
if ROBUST.search(results):
|
|
344
|
+
return []
|
|
345
|
+
claims = []
|
|
346
|
+
for s in sentences(buried_src):
|
|
347
|
+
if ROBUST.search(s) and NUMERIC.search(s):
|
|
348
|
+
claims.append({
|
|
349
|
+
"verdict": "BURIED_DEFENSE", "severity": "Minor", "action": "MOVE",
|
|
350
|
+
"detail": ("a numeric robustness/sensitivity result sits in the "
|
|
351
|
+
"Limitations/supplement with no robustness mention in Results; "
|
|
352
|
+
"promote it into Results — it is evidence for the finding, not a "
|
|
353
|
+
"caveat against it"),
|
|
354
|
+
"where": s[:160],
|
|
355
|
+
})
|
|
356
|
+
break # one promotion recommendation is enough
|
|
357
|
+
return claims
|
|
358
|
+
|
|
359
|
+
|
|
360
|
+
# --------------------------------------------------------------------------- #
|
|
361
|
+
# Driver
|
|
362
|
+
# --------------------------------------------------------------------------- #
|
|
363
|
+
|
|
364
|
+
def check(text: str, *, hedge_per_1k: float, repeat_threshold: int,
|
|
365
|
+
limitations_max: int, abstract_caveat_max: int) -> list[dict]:
|
|
366
|
+
regions = segment(text)
|
|
367
|
+
body = region_text(regions, BODY_NARRATIVE)
|
|
368
|
+
if not body: # no IMRAD headings — degrade to a whole-document read (minus supplement)
|
|
369
|
+
body = region_text(regions, {"preamble", "other"}) or text
|
|
370
|
+
claims: list[dict] = []
|
|
371
|
+
claims += probe_hedge_density(regions, body, hedge_per_1k)
|
|
372
|
+
claims += probe_hedge_repeat(regions, text, repeat_threshold)
|
|
373
|
+
claims += probe_audit_in_body(regions)
|
|
374
|
+
claims += probe_limitations_volume(regions, text, limitations_max)
|
|
375
|
+
claims += probe_abstract_caveat_load(regions, text, abstract_caveat_max)
|
|
376
|
+
claims += probe_buried_defense(regions)
|
|
377
|
+
return claims
|
|
378
|
+
|
|
379
|
+
|
|
380
|
+
def analyze(manuscript: str, **kw) -> dict:
|
|
381
|
+
p = Path(manuscript)
|
|
382
|
+
if not p.is_file():
|
|
383
|
+
sys.stderr.write(f"ERROR: manuscript not found: {manuscript}\n")
|
|
384
|
+
sys.exit(2)
|
|
385
|
+
claims = check(p.read_text(encoding="utf-8"), **kw)
|
|
386
|
+
by_action = {"REMOVE": 0, "MOVE": 0, "TIGHTEN": 0}
|
|
387
|
+
for c in claims:
|
|
388
|
+
by_action[c["action"]] = by_action.get(c["action"], 0) + 1
|
|
389
|
+
return {
|
|
390
|
+
"manuscript": str(p),
|
|
391
|
+
"claims": claims,
|
|
392
|
+
"summary": {
|
|
393
|
+
"n_claims": len(claims),
|
|
394
|
+
"by_action": by_action,
|
|
395
|
+
"verdict": "IMPRESSION_FLAGS" if claims else "OK",
|
|
396
|
+
},
|
|
397
|
+
}
|
|
398
|
+
|
|
399
|
+
|
|
400
|
+
def render(result: dict) -> str:
|
|
401
|
+
lines = ["| Check | Action | Detail |", "|---|---|---|"]
|
|
402
|
+
for c in result["claims"]:
|
|
403
|
+
lines.append(f"| {c['verdict']} | {c['action']} | {c['detail']} |")
|
|
404
|
+
if len(lines) == 2:
|
|
405
|
+
lines.append("| (none) | — | narrative reads confidently; no subtraction needed |")
|
|
406
|
+
return "\n".join(lines)
|
|
407
|
+
|
|
408
|
+
|
|
409
|
+
def main() -> int:
|
|
410
|
+
ap = argparse.ArgumentParser(
|
|
411
|
+
description="Editorial-impression / defensiveness gate (§L) — advisory, non-blocking.")
|
|
412
|
+
ap.add_argument("--manuscript", required=True, help="manuscript markdown/text")
|
|
413
|
+
ap.add_argument("--out", help="write JSON artifact to this path")
|
|
414
|
+
ap.add_argument("--strict", action="store_true",
|
|
415
|
+
help="accepted for CLI parity; this gate emits no Major, so it never blocks")
|
|
416
|
+
ap.add_argument("--quiet", action="store_true", help="suppress stdout table")
|
|
417
|
+
ap.add_argument("--hedge-per-1k", type=float, default=10.0,
|
|
418
|
+
help="HEDGE_DENSITY: caveat tokens per 1,000 body words before firing (default 10)")
|
|
419
|
+
ap.add_argument("--repeat-threshold", type=int, default=3,
|
|
420
|
+
help="HEDGE_REPEAT: motif occurrences across body+abstract before firing (default 3)")
|
|
421
|
+
ap.add_argument("--limitations-max", type=int, default=6,
|
|
422
|
+
help="LIMITATIONS_VOLUME: discrete Limitations items allowed (default 6)")
|
|
423
|
+
ap.add_argument("--abstract-caveat-max", type=int, default=2,
|
|
424
|
+
help="ABSTRACT_CAVEAT_LOAD: caveat clauses allowed in the Abstract (default 2)")
|
|
425
|
+
args = ap.parse_args()
|
|
426
|
+
|
|
427
|
+
result = analyze(
|
|
428
|
+
args.manuscript,
|
|
429
|
+
hedge_per_1k=args.hedge_per_1k,
|
|
430
|
+
repeat_threshold=args.repeat_threshold,
|
|
431
|
+
limitations_max=args.limitations_max,
|
|
432
|
+
abstract_caveat_max=args.abstract_caveat_max,
|
|
433
|
+
)
|
|
434
|
+
|
|
435
|
+
if not args.quiet:
|
|
436
|
+
print("=" * 41)
|
|
437
|
+
print(" Editorial Impression / Defensiveness (§L)")
|
|
438
|
+
print("=" * 41)
|
|
439
|
+
print(render(result))
|
|
440
|
+
print()
|
|
441
|
+
s = result["summary"]
|
|
442
|
+
if s["n_claims"]:
|
|
443
|
+
ba = s["by_action"]
|
|
444
|
+
print(f"IMPRESSION flags: {s['n_claims']} advisory finding(s) "
|
|
445
|
+
f"(REMOVE {ba['REMOVE']} / MOVE {ba['MOVE']} / TIGHTEN {ba['TIGHTEN']}). "
|
|
446
|
+
f"Non-blocking — these raise the ceiling, they do not gate submission.")
|
|
447
|
+
else:
|
|
448
|
+
print("OK: narrative reads confidently; no subtraction recommended.")
|
|
449
|
+
|
|
450
|
+
if args.out:
|
|
451
|
+
Path(args.out).parent.mkdir(parents=True, exist_ok=True)
|
|
452
|
+
Path(args.out).write_text(json.dumps(result, indent=2), encoding="utf-8")
|
|
453
|
+
if not args.quiet:
|
|
454
|
+
print(f"\nwrote {args.out}")
|
|
455
|
+
|
|
456
|
+
# Advisory: never blocks. --strict is accepted for parity but this gate has no Major.
|
|
457
|
+
return 0
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
if __name__ == "__main__":
|
|
461
|
+
sys.exit(main())
|
|
@@ -22,6 +22,7 @@ outputs:
|
|
|
22
22
|
- qc/reference_adequacy.json
|
|
23
23
|
deterministic_scripts:
|
|
24
24
|
- scripts/check_reference_adequacy.py
|
|
25
|
+
- scripts/check_editorial_impression.py
|
|
25
26
|
side_effects:
|
|
26
27
|
- may_edit_manuscript_when_fix_flag_set
|
|
27
28
|
downstream_consumers:
|
|
@@ -44,5 +45,6 @@ validation_commands:
|
|
|
44
45
|
- "python3 scripts/check_domain_probe_sync.py --strict"
|
|
45
46
|
- "bash tests/test_panel_mode.sh"
|
|
46
47
|
- "bash tests/test_reference_adequacy.sh"
|
|
48
|
+
- "bash tests/test_editorial_impression.sh"
|
|
47
49
|
- "feed R0-numbered output into /revise"
|
|
48
50
|
evidence_surface: demo
|
|
@@ -0,0 +1,29 @@
|
|
|
1
|
+
# A Deep-Learning Marker for Synthetic Outcome X: Development and Internal Validation
|
|
2
|
+
|
|
3
|
+
## Abstract
|
|
4
|
+
|
|
5
|
+
**Background:** Marker X may aid early triage of outcome X. **Methods:** We developed and internally validated a model on a patient-level split. **Results:** The model discriminated outcome X with an area under the curve of 0.84 (95% CI 0.79–0.89) and was well calibrated. **Conclusion:** Marker X identifies patients at higher risk of outcome X and supports triage. These findings are preliminary and require external validation.
|
|
6
|
+
|
|
7
|
+
## Introduction
|
|
8
|
+
|
|
9
|
+
Outcome X is common and its early identification changes triage. Existing tools rely on manual scoring, which is slow and operator-dependent. We asked whether a deep-learning marker derived from routine inputs could identify patients at higher risk and inform the triage decision. This study develops such a marker and tests it on a held-out internal cohort, addressing a gap left by prior manual approaches.
|
|
10
|
+
|
|
11
|
+
## Methods
|
|
12
|
+
|
|
13
|
+
We assembled a cohort and split it at the patient level into development and held-out sets before any preprocessing. A regularised model was trained with five-fold cross-validation. All analysis code, the dataset schema, and a content-hash manifest are archived (see Data Availability).
|
|
14
|
+
|
|
15
|
+
## Results
|
|
16
|
+
|
|
17
|
+
The model discriminated outcome X with an area under the curve of 0.84 (95% CI 0.79–0.89). Calibration was good, with a slope of 0.97 and a Brier score of 0.12. At the triage threshold, sensitivity was 0.82 and specificity was 0.79. In a sensitivity analysis excluding the 41 borderline cases, the adjusted odds ratio was unchanged at 2.18 (95% CI 1.40–3.39), and results were consistent across the two recruitment years. Decision-curve analysis showed net benefit over the manual score across the clinically relevant threshold range.
|
|
18
|
+
|
|
19
|
+
## Discussion
|
|
20
|
+
|
|
21
|
+
A routine-input marker identified patients at higher risk of outcome X and added net benefit over the existing manual score, which is the decision it is meant to inform. The effect size corresponds to a clinically meaningful shift in pretest probability. The marker is reproducible and its calibration supports use at the stated threshold.
|
|
22
|
+
|
|
23
|
+
## Limitations
|
|
24
|
+
|
|
25
|
+
This single-centre study has three main limitations: it was developed on retrospective data, the marker was measured once, and external validation in an independent cohort is the necessary next step before deployment.
|
|
26
|
+
|
|
27
|
+
## Data Availability
|
|
28
|
+
|
|
29
|
+
The analysis code, the dataset schema, the patient-level split assignment, and a reproducibility manifest with the dataset content hash are archived in the project repository, so every reported number can be regenerated from a single committed pipeline.
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# A Deep-Learning Marker for Synthetic Outcome X: A Preliminary Single-Centre Study
|
|
2
|
+
|
|
3
|
+
## Abstract
|
|
4
|
+
|
|
5
|
+
**Background:** Marker X has been proposed as a screening adjunct. **Methods:** We trained a model on a retrospective cohort. **Results:** The model reached an area under the curve of 0.84. **Conclusion:** These findings are preliminary and exploratory and should be interpreted with caution. The results are hypothesis-generating and not generalizable beyond this single-centre cohort, and no deployable claim is made. Because the analysis is underpowered, no causal inference can be inferred.
|
|
6
|
+
|
|
7
|
+
## Introduction
|
|
8
|
+
|
|
9
|
+
Marker X is of interest, although the evidence remains uncertain and the prior literature is limited by small samples. We hypothesise that a model may help, but the work is exploratory and any conclusion should be interpreted with caution. We make no deployable claim and the results are not generalizable; the study is hypothesis-generating only.
|
|
10
|
+
|
|
11
|
+
## Methods
|
|
12
|
+
|
|
13
|
+
We used a retrospective cohort and fit a regularised model. Splits were at the patient level.
|
|
14
|
+
|
|
15
|
+
## Results
|
|
16
|
+
|
|
17
|
+
The model reached an area under the curve of 0.84 (95% CI 0.79–0.89). The reproducibility manifest and the sha256 checksum of the locked dataset are reported. Every metric in this paragraph was produced by a committed unit-test against the post-lock data with seed=42, and the git commit a1b2c3d4e5f6 pins the exact run. The audit trail for each figure is recorded.
|
|
18
|
+
|
|
19
|
+
## Discussion
|
|
20
|
+
|
|
21
|
+
Our marker may help, but the finding is preliminary and modest and cannot be established as causal. The result is not generalizable and remains uncertain; it is hypothesis-generating only and warrants further study. We reiterate that no deployable claim is made and that any inference should be interpreted with caution. The exact manifest hash and the post-lock timeline are given above for full auditability.
|
|
22
|
+
|
|
23
|
+
## Limitations
|
|
24
|
+
|
|
25
|
+
This study has several limitations. First, the design is retrospective and single-centre, so selection bias cannot be excluded. Second, the sample is limited and the analysis is underpowered. Third, the marker was measured once. Fourth, no external validation was performed. Fifth, residual confounding cannot be ruled out. Sixth, the outcome label is registry-derived. Seventh, generalizability is uncertain. In a sensitivity analysis excluding the 41 borderline cases, the adjusted odds ratio remained 2.18 (95% CI 1.40–3.39), which we note here for completeness.
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
#!/usr/bin/env bash
|
|
2
|
+
# Regression test for the editorial-impression / defensiveness gate (self-review §L).
|
|
3
|
+
# Synthetic, PII-free fixtures: (a) an over-defensive manuscript that trips all six
|
|
4
|
+
# probes (HEDGE_DENSITY, HEDGE_REPEAT, AUDIT_IN_BODY, LIMITATIONS_VOLUME,
|
|
5
|
+
# ABSTRACT_CAVEAT_LOAD, BURIED_DEFENSE); (b) a confident clean manuscript that trips
|
|
6
|
+
# none (false-positive guard). The gate is advisory and non-blocking — it must exit 0
|
|
7
|
+
# even under --strict, since it emits no Major.
|
|
8
|
+
# Stdlib-only (python3).
|
|
9
|
+
set -u
|
|
10
|
+
|
|
11
|
+
HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
|
|
12
|
+
SCRIPT="$HERE/../scripts/check_editorial_impression.py"
|
|
13
|
+
DEF="$HERE/fixtures/editorial_defensive.md"
|
|
14
|
+
CLEAN="$HERE/fixtures/editorial_clean.md"
|
|
15
|
+
OUT="$(mktemp -t editorial_XXXX).json"
|
|
16
|
+
trap 'rm -f "$OUT"' EXIT
|
|
17
|
+
|
|
18
|
+
fail=0
|
|
19
|
+
check() { local label="$1"; shift
|
|
20
|
+
if "$@" >/dev/null 2>&1; then printf ' PASS %s\n' "$label"
|
|
21
|
+
else printf ' FAIL %s\n' "$label"; fail=$((fail+1)); fi
|
|
22
|
+
}
|
|
23
|
+
has_verdict() { python3 -c "
|
|
24
|
+
import json
|
|
25
|
+
d=json.load(open('$OUT'))
|
|
26
|
+
assert any(c['verdict']=='$1' for c in d['claims']), '$1 not found'
|
|
27
|
+
"; }
|
|
28
|
+
|
|
29
|
+
[[ -f "$SCRIPT" ]] || { echo "ENV-ERR: script missing" >&2; exit 2; }
|
|
30
|
+
|
|
31
|
+
# (1) defensive fixture -> all six verdicts fire
|
|
32
|
+
python3 "$SCRIPT" --manuscript "$DEF" --out "$OUT" --quiet >/dev/null 2>&1
|
|
33
|
+
for v in HEDGE_DENSITY HEDGE_REPEAT AUDIT_IN_BODY LIMITATIONS_VOLUME ABSTRACT_CAVEAT_LOAD BURIED_DEFENSE; do
|
|
34
|
+
check "$v detected (defensive)" has_verdict "$v"
|
|
35
|
+
done
|
|
36
|
+
|
|
37
|
+
# (2) every claim carries a SUBTRACTION action in {REMOVE, MOVE, TIGHTEN}
|
|
38
|
+
check "every claim has a REMOVE/MOVE/TIGHTEN action" python3 -c "
|
|
39
|
+
import json
|
|
40
|
+
d=json.load(open('$OUT'))
|
|
41
|
+
assert d['claims'], 'expected findings'
|
|
42
|
+
assert all(c.get('action') in ('REMOVE','MOVE','TIGHTEN') for c in d['claims']), 'bad action'
|
|
43
|
+
assert all(c.get('severity')=='Minor' for c in d['claims']), 'all findings must be Minor (advisory)'
|
|
44
|
+
"
|
|
45
|
+
|
|
46
|
+
# (3) advisory / non-blocking: exit 0 even under --strict on a fully-flagged manuscript
|
|
47
|
+
python3 "$SCRIPT" --manuscript "$DEF" --strict --quiet >/dev/null 2>&1
|
|
48
|
+
check "exit 0 under --strict (non-blocking)" test "$?" -eq 0
|
|
49
|
+
|
|
50
|
+
# (4) clean fixture -> zero claims (false-positive guard)
|
|
51
|
+
python3 "$SCRIPT" --manuscript "$CLEAN" --out "$OUT" --quiet >/dev/null 2>&1
|
|
52
|
+
check "clean manuscript yields no flags" python3 -c "
|
|
53
|
+
import json
|
|
54
|
+
d=json.load(open('$OUT'))
|
|
55
|
+
assert d['summary']['n_claims']==0, d['claims']
|
|
56
|
+
"
|
|
57
|
+
python3 "$SCRIPT" --manuscript "$CLEAN" --strict --quiet >/dev/null 2>&1
|
|
58
|
+
check "exit 0 on clean manuscript" test "$?" -eq 0
|
|
59
|
+
|
|
60
|
+
echo "fail=$fail"; [[ "$fail" -eq 0 ]] && echo "ALL PASS" || echo "FAILURES: $fail"
|
|
61
|
+
exit "$fail"
|