medsci-skills 5.5.0 → 5.6.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -3393,8 +3393,8 @@
3393
3393
  },
3394
3394
  {
3395
3395
  "path": "skills/self-review/SKILL.md",
3396
- "size": 95312,
3397
- "sha256": "ff45218dcd5cc5233787b95c8f6ce9168dc2ff27db1ce98d4b121884e13ae526"
3396
+ "size": 104683,
3397
+ "sha256": "8cc62fbae95f1dd18c5c62f65931904d981daa8a565bad2f8418e3671ca56e7c"
3398
3398
  },
3399
3399
  {
3400
3400
  "path": "skills/self-review/references/domain-probes/ai_overclaiming.md",
@@ -3538,8 +3538,8 @@
3538
3538
  },
3539
3539
  {
3540
3540
  "path": "skills/self-review/references/panel_review_template.md",
3541
- "size": 11648,
3542
- "sha256": "07a714771f9a0a07804c0f56db3191627ce3555aad605e9b2e1bdaf31e242afa"
3541
+ "size": 14863,
3542
+ "sha256": "a44567b5ea0c47971bcc75b0e6c46f3846d687240b0607ded07183d6fbaad86f"
3543
3543
  },
3544
3544
  {
3545
3545
  "path": "skills/self-review/scripts/check_artifact_coverage.py",
@@ -3576,6 +3576,11 @@
3576
3576
  "size": 21506,
3577
3577
  "sha256": "7d3e67074d58a28ffee52ce64b486231f103a3ddcaf6b3b6ee83ba5f89c63bc2"
3578
3578
  },
3579
+ {
3580
+ "path": "skills/self-review/scripts/check_editorial_impression.py",
3581
+ "size": 22032,
3582
+ "sha256": "42e0e9315e1c97ca0f9943213c65ea91d7064773d25287b2577f0386dfb7fc41"
3583
+ },
3579
3584
  {
3580
3585
  "path": "skills/self-review/scripts/check_null_calibration.py",
3581
3586
  "size": 7594,
@@ -3613,8 +3618,8 @@
3613
3618
  },
3614
3619
  {
3615
3620
  "path": "skills/self-review/skill.yml",
3616
- "size": 2135,
3617
- "sha256": "803c868da4e5bdfc0e3447c67dc54809f78c0f64cee448033c919693e9e6537c"
3621
+ "size": 2223,
3622
+ "sha256": "4009f3148776fab2096da3dad8a15e503c5073b4cc66b42c57498948e2040270"
3618
3623
  },
3619
3624
  {
3620
3625
  "path": "skills/setup-medsci/SKILL.md",
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schema_version": 1,
3
- "version": "5.5.0",
3
+ "version": "5.6.0",
4
4
  "owned_skills": [
5
5
  "academic-aio",
6
6
  "add-journal",
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "medsci-skills",
3
- "version": "5.5.0",
3
+ "version": "5.6.0",
4
4
  "description": "MedSci Skills — a medical/scientific research skill suite for AI coding agents (Claude Code, Codex, Cursor, Copilot). The npm package is a terminal-friendly installer shortcut; the canonical distribution remains the GitHub repository and the Claude Code plugin marketplace.",
5
5
  "license": "SEE LICENSE IN LICENSE",
6
6
  "homepage": "https://github.com/Aperivue/medsci-skills#readme",
@@ -33,6 +33,29 @@ When flagging issues, classify severity:
33
33
 
34
34
  Most issues are Fixable. Reserve Fatal for true design-level problems.
35
35
 
36
+ ## Two Objectives: the Floor and the Ceiling
37
+
38
+ A submission-ready manuscript optimizes **two** things at once, and most of this skill (and
39
+ the gate stack behind it) only optimizes the first:
40
+
41
+ - **Floor — minimize rejection-for-cause.** Fabricated citations, numbers that do not
42
+ reconcile, overclaims, missing checklist items, leakage. Categories A–K and the
43
+ deterministic gates (Phases 2.5–2.5f) do this, and they are right to. Many of them raise the
44
+ floor by **adding** material: a hedge, a caveat, a disclosure, an audit trail, a checklist row.
45
+ - **Ceiling — maximize editorial-championing.** Will a handling editor read a *confident
46
+ narrative* (problem → design → result → meaning) and want to send it out, or a *defensive
47
+ audit* and bounce it? Nothing in the floor stack pushes here, and several floor gates push the
48
+ other way. Iterated, a manuscript over-hardens: every individual gate finding is correct, yet
49
+ the **accumulated** product reads as a rebuttal letter — over-hedged, audit-trail-heavy,
50
+ Abstract buried under caveats, the strongest sensitivity result hidden in Limitations, too long.
51
+
52
+ These objectives can conflict, so the order matters: **the floor gates run first and secure
53
+ accuracy; then the ceiling pass (category L / Phase 2.5g) reads the accurate manuscript as a
54
+ whole and recommends SUBTRACTION — REMOVE, MOVE, or TIGHTEN — so the same content is read
55
+ confidently.** The ceiling pass is advisory and never blocks; it cannot relax a floor gate.
56
+ Without it, repeated self-review monotonically over-defends. Surface the ceiling findings as
57
+ their own first-class output (Phase 3), not folded silently into the "add this" comments.
58
+
36
59
  ## Workflow
37
60
 
38
61
  ### Phase 1: Intake
@@ -250,6 +273,38 @@ fabrication. Resolution path:
250
273
  1. Honest Methods/PROSPERO update (single-reviewer execution disclosed), OR
251
274
  2. Limitations confession rewritten if dual review was actually completed.
252
275
 
276
+ #### L. Editorial impression & defensiveness (advisory; the counterweight)
277
+
278
+ This is the **ceiling** category (see "Two Objectives" above) and the inverse of the floor
279
+ gates: where A–K and the numerical gates ask "what is missing or wrong?" (and answer by
280
+ **adding**), L asks "does the accurate manuscript read confidently, or has it over-defended?"
281
+ (and answers by **subtracting**). Every L finding is **advisory (Minor / impression) and
282
+ non-blocking** — it never converts to a Major and never blocks submission. The fixes are
283
+ REMOVE / MOVE / TIGHTEN, not "add a caveat."
284
+
285
+ | Check | What to look for | Action |
286
+ |-------|-----------------|--------|
287
+ | Hedge density | Defensive-caveat tokens stacking up per 1,000 narrative words — the prose hedges faster than it asserts. Keep the load-bearing caveats; cut the reflexive ones. | TIGHTEN |
288
+ | Repeated caveat | The same caveat motif ("no deployable claim", "not generalizable", "hypothesis-generating") repeated across body + Abstract. Say it once, firmly. | TIGHTEN |
289
+ | Audit minutiae in body | Provenance tokens (SHA / git commit / unit-test / post-lock timeline / manifest / seed=N / audit trail) in the Introduction / Results / Discussion narrative. Reproducibility detail belongs in a Methods statement or a supplement. | MOVE |
290
+ | Limitations volume | A Limitations passage that enumerates a long list of discrete items reads as a rebuttal letter; consolidate related items. | TIGHTEN |
291
+ | Abstract caveat load | The Abstract carries several caveat clauses, burying the headline result before a reader reaches it. Lead with the result; keep one or two essential qualifiers. | TIGHTEN |
292
+ | Buried defense | A strong numeric robustness / sensitivity result sitting only in Limitations or the supplement, with no robustness mention in Results. Promote it into Results — it is *evidence for* the finding, not a caveat against it. (The inverse of the scope-coherence gate, which pushes a *weak* analysis out of Results.) | MOVE |
293
+
294
+ Run the deterministic gate (Phase 2.5g) rather than eyeballing it — these are all counts and placements:
295
+
296
+ ```bash
297
+ python3 "${CLAUDE_SKILL_DIR}/scripts/check_editorial_impression.py" \
298
+ --manuscript manuscript.md --out qc/editorial_impression.json
299
+ ```
300
+
301
+ `HEDGE_DENSITY`, `HEDGE_REPEAT`, `AUDIT_IN_BODY`, `LIMITATIONS_VOLUME`, `ABSTRACT_CAVEAT_LOAD`,
302
+ and `BURIED_DEFENSE` are Anticipated **Minor** Comments (category: L. Editorial impression),
303
+ each carrying a REMOVE / MOVE / TIGHTEN `action`. The gate never blocks (it has no Major and
304
+ exits 0 even under `--strict`); thresholds are tunable (`--hedge-per-1k`, `--repeat-threshold`,
305
+ `--limitations-max`, `--abstract-caveat-max`). It is conservative — each probe fires only on an
306
+ explicit, locatable signal.
307
+
253
308
  ### Research-Type Adaptation
254
309
 
255
310
  Not all categories apply equally to every study type. Use this routing table:
@@ -267,6 +322,7 @@ Not all categories apply equally to every study type. Use this routing table:
267
322
  | I. Protocol Heterogeneity | Full | Full | N/A | Per-study | N/A | Full |
268
323
  | J. Method Transparency | Full | Partial | Partial | N/A | N/A | Partial |
269
324
  | K. Reviewer-team consistency | N/A | N/A | N/A | Full | N/A | N/A |
325
+ | L. Editorial impression | Full | Full | Full | Full | Full | Full |
270
326
 
271
327
  *Meta-analysis: Replace C with heterogeneity assessment (I-squared, prediction intervals),
272
328
  publication bias (funnel plot, Egger), and sensitivity/subgroup analyses.
@@ -1016,6 +1072,48 @@ reconciliation in `qc/claim_artifact.json` and confirm against the actual regist
1016
1072
  before raising `ESTIMAND_DRIFT`. For time-to-event manuscripts, also apply probe **S8
1017
1073
  (estimand provenance)** of `references/domain-probes/survival_prognostic.md`.
1018
1074
 
1075
+ ### Phase 2.5g: Editorial-Impression / Defensiveness Scan (the ceiling pass)
1076
+
1077
+ Run this **after** the floor gates (Phases 2.5–2.5f), because it reads the *accurate* manuscript
1078
+ and recommends what to take back out. It is the operational form of category L and the
1079
+ counterweight to the additive bias of the rest of the stack: every other phase can only make the
1080
+ manuscript longer and more defended; this one is the only phase that can make it shorter and more
1081
+ confident. It is advisory and **non-blocking** — it never produces a Major and never gates
1082
+ submission.
1083
+
1084
+ ```bash
1085
+ python3 "${CLAUDE_SKILL_DIR}/scripts/check_editorial_impression.py" \
1086
+ --manuscript manuscript.md --out qc/editorial_impression.json
1087
+ ```
1088
+
1089
+ The gate reads the manuscript as a whole, segments it by IMRAD heading, and emits up to six
1090
+ verdicts, each tagged with a SUBTRACTION `action`:
1091
+
1092
+ | Verdict | Reads as | Action |
1093
+ |---|---|---|
1094
+ | `HEDGE_DENSITY` | defensive-caveat tokens per 1,000 narrative words over threshold | TIGHTEN |
1095
+ | `HEDGE_REPEAT` | one caveat motif repeated across body + Abstract | TIGHTEN |
1096
+ | `AUDIT_IN_BODY` | SHA / commit / unit-test / post-lock / manifest / seed in the narrative | MOVE (→ Methods/supplement) |
1097
+ | `LIMITATIONS_VOLUME` | a long enumerated Limitations list | TIGHTEN (consolidate) |
1098
+ | `ABSTRACT_CAVEAT_LOAD` | several caveat clauses in the Abstract | TIGHTEN |
1099
+ | `BURIED_DEFENSE` | strong numeric robustness result only in Limitations/supplement | MOVE (→ Results) |
1100
+
1101
+ **Fold the findings into the report as the SUBTRACTION axis, not the additive one.** Each
1102
+ becomes a Minor `issues[]` entry under `category: "L" / category_name: "Editorial impression"`,
1103
+ additively carrying `issue_type: "editorial_impression"`, `subtype: <verdict>`, and
1104
+ `action: "REMOVE" | "MOVE" | "TIGHTEN"`. They are summarized in their own Phase 3 block
1105
+ ("Editorial-Impression Risks — REMOVE / MOVE / TIGHTEN"), kept visually separate from the
1106
+ "Anticipated Major / Minor Comments (ADD / FIX)" so the author sees both forces. Mark them
1107
+ `fixable_by_ai: false` by default — TIGHTEN-ing a hedge or MOVE-ing a robustness result is a
1108
+ voice-and-judgment edit the author should own — except a clearly-redundant repeated caveat
1109
+ (`HEDGE_REPEAT`), which `--fix` may collapse to a single statement.
1110
+
1111
+ **Net-impact note.** When an *earlier* phase recommends adding a caveat or disclosure, weigh it
1112
+ against L: an integrity-critical disclosure is a **must (state it once, crisply)**, but a
1113
+ defensive over-disclosure is a **cut / move**. The two are not symmetric — keep the disclosure,
1114
+ but place it once and point to the supplement rather than repeating it at every claim site
1115
+ (placement discipline: main text narrates, auditability lives in the supplement).
1116
+
1019
1117
  ### Phase 2.6: Multi-Agent Panel Review (--panel, opt-in)
1020
1118
 
1021
1119
  Run this phase **only when `--panel` is passed**. The default single-pass review (Phases 2–2.5d) stays the fast path; the panel is the high-cost, high-precision option for a pre-submission final pass on a top-tier target. Run it after the numerical audits (Phases 2.5–2.5d) so the reviewers see source-verified numbers, and before the Phase 3 report, which it feeds.
@@ -1038,6 +1136,12 @@ The panel simulates independent peer reviewers who do not see each other's comme
1038
1136
 
1039
1137
  If the type is ambiguous, ask the user before composing the set.
1040
1138
 
1139
+ Append the **handling-editor desk-impression** persona (the ceiling lens) to every reviewer set:
1140
+ it loads no domain probe, reads only for narrative confidence vs over-defensiveness, and returns
1141
+ Minor REMOVE / MOVE / TIGHTEN findings (category L) that the editor routes to the separate
1142
+ Editorial-Impression Risks block. Its focus checklist is in `references/panel_review_template.md`.
1143
+ It does not count toward the Step 3.5 lens-diversity axes.
1144
+
1041
1145
  **Step 2 — Run the reviewers (portable execution).** When the host provides a parallel subagent / Task capability (Claude Code, or any harness exposing an Agent tool), spawn the reviewer set as independent parallel subagents, each blinded to the others, then run the editor as a final synthesis agent. **Fallback (no subagent capability — e.g. a minimal Codex/Cursor harness):** a single agent role-plays each reviewer sequentially and in isolation — it completes and writes out reviewer R1's full structured review before reading the manuscript "fresh" as R2, so a later reviewer never sees an earlier reviewer's comments. The panel is defined by these instructions; it does **not** depend on the `Workflow` tool or any Claude-Code-only orchestration.
1042
1146
 
1043
1147
  A reusable reviewer schema, a generic harsh-but-fair reviewer prompt skeleton with per-domain focus checklists, and the editor synthesis prompt skeleton live in `${CLAUDE_SKILL_DIR}/references/panel_review_template.md`.
@@ -1103,6 +1207,15 @@ M2. ...
1103
1207
  m1. **{Issue}** [{Category}]: {1 sentence with location + fix}
1104
1208
  m2. ...
1105
1209
 
1210
+ ## Editorial-Impression Risks (REMOVE / MOVE / TIGHTEN)
1211
+
1212
+ *The subtraction axis — what to take out, move, or tighten so the accurate manuscript reads
1213
+ confidently. Advisory and non-blocking; from Phase 2.5g / category L. Omit this block only if the
1214
+ scan returned nothing.*
1215
+
1216
+ L1. **{Issue}** [{REMOVE | MOVE | TIGHTEN}]: {1 sentence — what reads as over-defensive and where, with the subtraction to make}
1217
+ L2. ...
1218
+
1106
1219
  ## Strengths (emphasize in cover letter)
1107
1220
 
1108
1221
  - {Specific strength 1}
@@ -1110,9 +1223,15 @@ m2. ...
1110
1223
  - ...
1111
1224
  ```
1112
1225
 
1226
+ The report carries **two** axes, kept visually separate: the **ADD / FIX** axis (Anticipated
1227
+ Major / Minor Comments — what is missing or wrong) and the **SUBTRACTION** axis
1228
+ (Editorial-Impression Risks — what to remove, move, or tighten). Do not fold the L items into the
1229
+ Minor Comments; an author who sees only "add this" will monotonically over-defend.
1230
+
1113
1231
  **Conciseness targets**:
1114
1232
  - Anticipated Major Comments: 3-7 items, each 3-5 lines
1115
1233
  - Anticipated Minor Comments: 3-6 items, each 1-2 sentences
1234
+ - Editorial-Impression Risks: 0-6 items, each 1 sentence (only what the Phase 2.5g gate flagged)
1116
1235
  - Strengths: 3-5 items, each 1 sentence
1117
1236
  - Total report: 400-800 words (excluding optional R0 section)
1118
1237
 
@@ -1181,6 +1300,7 @@ When `--json` is passed, or when invoked by `/write-paper` Phase 7, append a mac
1181
1300
  - `requires_reanalysis` *(optional, default `false`)*: `true` when closing the finding needs a **committed analysis re-run against the real data**, not a prose edit — power/MDE re-simulation under the full model, first-visit/one-record-per-subject dedup, an extended- or reduced-adjustment sensitivity model, optimism correction of calibration. Always implies `fixable_by_ai: false`. Additive and backwards-compatible; parsers that do not expect it must ignore it. Route these to `/analyze-stats` (see Phase 4).
1182
1301
  - `suggested_fix`: Specific, actionable instruction. If `fixable_by_ai` is true, this must be concrete enough for the fixer to execute without ambiguity.
1183
1302
  - `consensus` *(optional, panel mode only)*: array of reviewer ids that raised the issue, e.g. `["R1","R3"]`. Additive and backwards-compatible — present only when Phase 2.6 ran; parsers that do not expect it must ignore it.
1303
+ - `action` *(optional, editorial-impression findings only)*: `"REMOVE" | "MOVE" | "TIGHTEN"` — the SUBTRACTION direction for a category-L finding (Phase 2.5g). Present alongside `issue_type: "editorial_impression"` and `subtype: <verdict>` (e.g. `HEDGE_REPEAT`). Additive and backwards-compatible; these are always `severity: "minor"`, never block, and are `fixable_by_ai: false` by default (except a redundant `HEDGE_REPEAT`, which `--fix` may collapse). Parsers that do not expect it must ignore it.
1184
1304
 
1185
1305
  ### Phase 4: Fix Support
1186
1306
 
@@ -1272,5 +1392,6 @@ Here is how to address it with your existing data."
1272
1392
  | Phase 2.5c reference hallucination scan (delegate `/verify-refs`) | ENFORCED | `FABRICATED` in `records[]` OR nonempty `duplicate_findings[]` | P0 Major Comment, blocks submission |
1273
1393
  | Phase 2.5a-2 design/power statistic provenance | ENFORCED | a reported MDE / power / sample-size value is not reproduced by committed code, or is reproducible only by a method the committed script does not implement | Major Comment (P0 if a headline claim); recompute and either correct the value or update the committed code to reproduce it |
1274
1394
  | `--fix` auto-fix loop (max 2 iterations) | ENFORCED in `/write-paper` Phase 7.4 chain | score still below threshold after 2 iterations | Route to write-paper Phase 7.4a Audit Recovery |
1395
+ | Phase 2.5g editorial-impression scan (`check_editorial_impression.py`) | ADVISORY (non-blocking) | HEDGE_DENSITY / HEDGE_REPEAT / AUDIT_IN_BODY / LIMITATIONS_VOLUME / ABSTRACT_CAVEAT_LOAD / BURIED_DEFENSE | Minor REMOVE/MOVE/TIGHTEN recommendation in the Editorial-Impression Risks block; never blocks submission |
1275
1396
  | R0 numbering output | OPT-IN | `--r0-numbering` flag or downstream `/revise` consumer | Emits structured Anticipated Major/Minor Comments — consumable by `/revise` |
1276
1397
  | `--json` machine-readable output | OPT-IN | `--json` flag | Emits parseable JSON block consumed by `/orchestrate` post-skill validation |
@@ -126,6 +126,36 @@ design-level finding, **Fixable** for a reporting-level finding.
126
126
  - RV6 single-anchor overload: does a load-bearing clinical claim rest on essentially one (often abstract-only/unreplicated) study while the Abstract calls it "landmark" and the body concedes the base "is thin"? Flag the Abstract↔body register mismatch.
127
127
  - RV8 self-citation architecture: do the weakest/most-deferred axes coincide with the authors' own forthcoming/companion work without a body-level disclosure, and does each load-bearing axis carry ≥1 independent (other-group) source?
128
128
 
129
+ **Handling editor — desk-impression / champion-or-bounce (cross-type, the ceiling lens)**
130
+
131
+ This persona is the counterweight to the rigor reviewers above. It is **not** a domain expert and
132
+ loads **no** domain-probe module; it reads the manuscript exactly as a handling editor skimming
133
+ for the desk decision, and its only question is whether the manuscript reads as a **confident
134
+ narrative** an editor would champion or a **defensive audit** an editor would bounce. It raises no
135
+ Major and no Fatal — every finding is a Minor with a SUBTRACTION action (REMOVE / MOVE / TIGHTEN),
136
+ mirroring category L / Phase 2.5g. Append it to any reviewer set on a `--panel` run; it does
137
+ **not** count toward the lens-diversity axes (its findings fall to the `other` family and the gate
138
+ never penalizes an extra lens). Add the optional `action` field (`"REMOVE" | "MOVE" | "TIGHTEN"`)
139
+ to each of its `minor[]` items.
140
+
141
+ - Does the manuscript open by stating the problem, the design, and the headline result, or does it
142
+ bury the result behind qualifiers? Is the Abstract carrying several caveat clauses before a
143
+ reader reaches the finding? (TIGHTEN)
144
+ - Is the strongest robustness / sensitivity result up front in Results, or hidden in Limitations or
145
+ the supplement where it reads as a caveat rather than as evidence? (MOVE → Results)
146
+ - Does the narrative (Introduction / Results / Discussion) carry audit minutiae — hashes, commit
147
+ ids, unit-test mentions, post-lock timelines, manifests — that belong in a Methods statement or a
148
+ supplement? (MOVE)
149
+ - Is the same caveat repeated at multiple claim sites? Does the Limitations section read as a
150
+ consolidated honest disclosure or as a long rebuttal-letter enumeration? Say each caveat once,
151
+ firmly. (TIGHTEN / REMOVE)
152
+ - Overall: is the manuscript longer and more defended than its evidence requires? Name the single
153
+ change that would most raise an editor's confidence on a skim — and it should be a subtraction.
154
+
155
+ Run `scripts/check_editorial_impression.py` first and use its verdicts as the deterministic spine
156
+ for this persona's findings, then add anything the gate cannot see (tone, narrative order, a
157
+ result that is technically present but framed apologetically).
158
+
129
159
  ---
130
160
 
131
161
  ## Editor synthesis prompt skeleton
@@ -147,6 +177,14 @@ design-level finding, **Fixable** for a reporting-level finding.
147
177
  > undisclosed near-identical prior publication, or (for a review/primer) weak novelty /
148
178
  > no distinct contribution since the contribution IS the product — let it dominate the
149
179
  > tier over fixable reporting defects rather than letting the fixable framing soften it.
180
+ > Then run a THIRD, opposite-direction lens — **defensiveness / narrative** — symmetric to
181
+ > the contribution lens but guarding the other failure: is the manuscript *over-defended*?
182
+ > Does it read as a confident narrative or as a rebuttal letter (over-hedged, audit-trail in
183
+ > the body, Abstract buried under caveats, the strongest sensitivity result hidden in
184
+ > Limitations, too long)? Treat a defensive over-disclosure as a **cut / move**, not a virtue,
185
+ > while keeping any integrity-critical disclosure (stated once, crisply). The contribution lens
186
+ > guards against a too-lenient verdict; this lens guards against blessing an over-hardened
187
+ > manuscript an editor would bounce on impression.
150
188
  > 2. De-duplicate and consolidate the major comments by theme. For each consolidated
151
189
  > point, flag CONSENSUS (raised by ≥2 reviewers) or single-reviewer, and attribute
152
190
  > (R1/R2/R3).
@@ -159,10 +197,13 @@ design-level finding, **Fixable** for a reporting-level finding.
159
197
  > rather than implicit in the attribution.
160
198
  >
161
199
  > Map every finding onto the self-review framing (Fatal / Fixable, category letters
162
- > A–K) and emit it through the Phase 3 report, Phase 3b R0 numbering, and Phase 3c
163
- > JSON, adding the optional `consensus` field where ≥2 reviewers agreed. Follow the
164
- > manuscript-style rules: no "§" symbols, minimal em-dashes, full prose, cite specific
165
- > locations.
200
+ > A–L) and emit it through the Phase 3 report, Phase 3b R0 numbering, and Phase 3c
201
+ > JSON, adding the optional `consensus` field where ≥2 reviewers agreed. Route the
202
+ > handling-editor desk-impression findings to the separate Editorial-Impression Risks
203
+ > block (category L, each with a REMOVE / MOVE / TIGHTEN `action`); do not fold them into
204
+ > the Anticipated Major / Minor (ADD / FIX) comments, so the author sees both forces.
205
+ > Follow the manuscript-style rules: no "§" symbols, minimal em-dashes, full prose, cite
206
+ > specific locations.
166
207
 
167
208
  ---
168
209
 
@@ -0,0 +1,461 @@
1
+ #!/usr/bin/env python3
2
+ """Editorial-impression / defensiveness gate (self-review §L) — the counterweight pass.
3
+
4
+ The rest of the MedSci-Audit stack minimizes *rejection-for-cause* (the floor):
5
+ fabricated citations, drifting numbers, overclaims, missing checklist items. Several
6
+ of those gates raise the floor by *adding* material — a hedge, a caveat, a disclosure,
7
+ a checklist row — and nothing in the stack pushes back. Iterated, a manuscript
8
+ monotonically over-hardens: a confident narrative turns into a defensive audit that an
9
+ editor reads as a risk signal even when every individual gate finding was correct.
10
+
11
+ This gate is the missing opposite force. It does not relax any integrity gate; it scans
12
+ the *manuscript as a whole* for editorial-impression risks and recommends SUBTRACTION —
13
+ REMOVE, MOVE, or TIGHTEN — so the accurate content the gates secured is also read
14
+ confidently. Every finding is advisory (Minor / impression) and NON-BLOCKING: this gate
15
+ never returns a submission blocker. It raises the ceiling; it does not gate the floor.
16
+
17
+ HEDGE_DENSITY defensive-caveat tokens per 1,000 body-narrative words exceed a
18
+ threshold — the prose hedges faster than it asserts. TIGHTEN.
19
+ HEDGE_REPEAT one caveat motif ("no deployable claim", "not generalizable",
20
+ "none evaluated here") repeats >=N times across body + abstract.
21
+ Say it once, firmly. TIGHTEN.
22
+ AUDIT_IN_BODY provenance/audit minutiae (SHA / git commit / unit-test /
23
+ post-lock timeline / manifest / seed=N / audit trail) appear in
24
+ the Introduction / Results / Discussion narrative rather than a
25
+ Methods reproducibility statement or a supplement. MOVE.
26
+ LIMITATIONS_VOLUME the Limitations passage enumerates more than N discrete items;
27
+ a wall of limitations reads as a rebuttal letter. TIGHTEN.
28
+ ABSTRACT_CAVEAT_LOAD the Abstract carries >=N caveat clauses; the headline result is
29
+ buried under qualifiers before a reader reaches it. TIGHTEN.
30
+ BURIED_DEFENSE a strong numeric robustness / sensitivity result sits only in the
31
+ Limitations / supplement, with no robustness mention in Results.
32
+ This is the inverse of the scope-coherence gate: scope-coherence
33
+ pushes a *weak* analysis out of Results; BURIED_DEFENSE pulls a
34
+ *strong* confound rebuttal back into Results. MOVE (promote).
35
+
36
+ Conservative by construction: each probe fires only on an explicit, locatable signal, to
37
+ keep false positives low on a widely-used skill. The gate needs IMRAD-style headings to
38
+ locate sections; with none it degrades to a whole-document density read.
39
+
40
+ INPUTS
41
+ --manuscript manuscript markdown/text (required).
42
+ thresholds --hedge-per-1k (10.0), --repeat-threshold (3), --limitations-max (6),
43
+ --abstract-caveat-max (2). A probe fires when its count exceeds the max.
44
+
45
+ OUTPUT
46
+ A reconciliation table (stdout) and, with --out, a JSON artifact:
47
+ {manuscript, claims[{verdict, severity, action, detail, where}], summary}
48
+ Every claim is severity "Minor" with an action of REMOVE / MOVE / TIGHTEN. Exit code is
49
+ always 0 for the findings themselves (advisory); --strict is accepted for CLI parity
50
+ with the other gates but never blocks, since this gate emits no Major.
51
+
52
+ Stdlib-only (json / re / argparse / pathlib). Exit codes: 0 clean or advisory findings,
53
+ 2 input/usage error.
54
+ """
55
+
56
+ from __future__ import annotations
57
+
58
+ import argparse
59
+ import json
60
+ import re
61
+ import sys
62
+ from pathlib import Path
63
+
64
+ # --------------------------------------------------------------------------- #
65
+ # Section segmentation
66
+ # --------------------------------------------------------------------------- #
67
+
68
+ HEADING_RE = re.compile(r"^(#{1,6})\s*\*{0,2}(.+?)\*{0,2}\s*$", re.MULTILINE)
69
+
70
+ # Narrative regions where defensive prose and out-of-place audit minutiae read worst.
71
+ # Methods is deliberately excluded (a reproducibility statement belongs there), as is
72
+ # any supplement / availability / declarations region.
73
+ BODY_NARRATIVE = {"introduction", "results", "discussion", "conclusion", "limitations"}
74
+
75
+
76
+ def classify_heading(h: str) -> str:
77
+ t = h.lower().strip()
78
+ if "abstract" in t or t == "summary":
79
+ return "abstract"
80
+ if any(k in t for k in (
81
+ "data availability", "code availability", "availability", "supplement",
82
+ "appendix", "acknowledg", "funding", "declaration", "competing interest",
83
+ "conflict of interest", "reproducibility", "references", "author contribution",
84
+ )):
85
+ return "supplement"
86
+ if "limitation" in t:
87
+ return "limitations"
88
+ if "introduction" in t or "background" in t:
89
+ return "introduction"
90
+ if any(k in t for k in (
91
+ "method", "material", "statistical analys", "study design",
92
+ "patients and", "data collection", "study population",
93
+ )):
94
+ return "methods"
95
+ if "result" in t or "finding" in t:
96
+ return "results"
97
+ if "discussion" in t:
98
+ return "discussion"
99
+ if "conclusion" in t:
100
+ return "conclusion"
101
+ return "other"
102
+
103
+
104
+ def segment(text: str) -> list[tuple[str, str]]:
105
+ """Return an ordered list of (region, body_text) pairs. Text before the first
106
+ heading is a 'preamble' region. Region names follow classify_heading()."""
107
+ matches = list(HEADING_RE.finditer(text))
108
+ regions: list[tuple[str, str]] = []
109
+ if not matches:
110
+ return [("preamble", text)]
111
+ if matches[0].start() > 0:
112
+ pre = text[: matches[0].start()].strip()
113
+ if pre:
114
+ regions.append(("preamble", pre))
115
+ for i, m in enumerate(matches):
116
+ start = m.end()
117
+ end = matches[i + 1].start() if i + 1 < len(matches) else len(text)
118
+ body = text[start:end].strip()
119
+ regions.append((classify_heading(m.group(2)), body))
120
+ return regions
121
+
122
+
123
+ def region_text(regions: list[tuple[str, str]], names: set[str]) -> str:
124
+ return "\n".join(b for r, b in regions if r in names)
125
+
126
+
127
+ def abstract_text(regions: list[tuple[str, str]], full: str) -> str:
128
+ """The Abstract region; fall back to the text before the first Introduction/Methods
129
+ heading (a structured abstract without its own heading), capped to ~350 words."""
130
+ abs_regions = [b for r, b in regions if r == "abstract"]
131
+ if abs_regions:
132
+ return "\n".join(abs_regions)
133
+ # Fallback: preamble + everything up to the first intro/methods region.
134
+ out: list[str] = []
135
+ for r, b in regions:
136
+ if r in ("introduction", "methods", "results", "discussion"):
137
+ break
138
+ if r in ("preamble", "other"):
139
+ out.append(b)
140
+ joined = "\n".join(out)
141
+ words = joined.split()
142
+ return " ".join(words[:350]) if len(words) > 350 else joined
143
+
144
+
145
+ def word_count(text: str) -> int:
146
+ return sum(1 for w in text.split() if any(c.isalpha() for c in w))
147
+
148
+
149
+ def sentences(text: str) -> list[str]:
150
+ # Lightweight sentence split on ., !, ? followed by whitespace.
151
+ parts = re.split(r"(?<=[.!?])\s+", text.strip())
152
+ return [s.strip() for s in parts if s.strip()]
153
+
154
+
155
+ # --------------------------------------------------------------------------- #
156
+ # Lexicons
157
+ # --------------------------------------------------------------------------- #
158
+
159
+ # Defensive caveats — explicitly hedging phrases, not ordinary modal verbs. Stacking
160
+ # these is the defensiveness tell HEDGE_DENSITY measures.
161
+ CAVEAT = re.compile(
162
+ r"\bcaveats?\b|should be interpreted with caution|with caution\b|"
163
+ r"must be interpreted|interpreted? with care|"
164
+ r"cannot be (?:inferred|established|excluded|determined|generaliz\w+|ruled out|drawn|assumed)|"
165
+ r"no (?:causal|deployable|clinical|definitive) (?:claim|inference|conclusion|relationship)|"
166
+ r"not (?:be )?(?:generaliz\w+|definitive|conclusive|deployable|warranted)|"
167
+ r"\bpreliminary\b|\bexploratory\b|hypothesis[-\s]generating|"
168
+ r"warrants? (?:caution|further (?:study|validation|research|investigation))|"
169
+ r"remains? (?:unclear|uncertain|to be (?:established|determined|confirmed))|"
170
+ r"\blimited (?:by|generaliz\w+|sample|to|in scope)|"
171
+ r"single[-\s](?:cent(?:er|re)|institution|site)|retrospective (?:design|nature)|"
172
+ r"should not be (?:used|interpreted|construed)|"
173
+ r"do(?:es)? not (?:establish|imply|permit|support|prove)|"
174
+ r"not (?:yet )?(?:ready|validated|intended) for (?:clinical|deployment|practice)|"
175
+ r"\bunderpowered\b|\bmodest\b|no (?:firm|strong) (?:conclusion|inference)",
176
+ re.IGNORECASE)
177
+
178
+ # Repeated caveat motifs (HEDGE_REPEAT): family key -> regex. A family repeating across
179
+ # body + abstract above the threshold should be stated once, firmly.
180
+ MOTIFS: dict[str, re.Pattern] = {
181
+ "no_deployable_claim": re.compile(
182
+ r"no (?:deployable|deployment|clinical|practice|diagnostic) (?:claim|use|recommendation)|"
183
+ r"not (?:ready|intended|validated) for (?:clinical|deployment|practice)",
184
+ re.IGNORECASE),
185
+ "not_generalizable": re.compile(r"not (?:be )?generaliz\w+|limited generaliz\w+", re.IGNORECASE),
186
+ "none_evaluated_here": re.compile(
187
+ r"(?:none|not|no \w+) (?:were |was |are |is )?evaluated (?:here|in this (?:study|work|analysis))|"
188
+ r"not (?:assessed|examined|tested) (?:here|in this (?:study|work))", re.IGNORECASE),
189
+ "no_causal": re.compile(r"no causal (?:claim|inference|relationship|conclusion|interpretation)", re.IGNORECASE),
190
+ "hypothesis_generating": re.compile(r"hypothesis[-\s]generating", re.IGNORECASE),
191
+ "interpret_with_caution": re.compile(
192
+ r"interpret\w* with caution|should be interpreted with caution|with caution", re.IGNORECASE),
193
+ "single_center": re.compile(r"single[-\s](?:cent(?:er|re)|institution|site)", re.IGNORECASE),
194
+ "retrospective_design": re.compile(r"retrospective (?:design|nature|study|cohort)", re.IGNORECASE),
195
+ "preliminary": re.compile(r"\bpreliminary\b", re.IGNORECASE),
196
+ }
197
+
198
+ # Motifs that are factual study-design descriptors rather than defensive caveats. A
199
+ # single-centre / retrospective study must state its design in Methods, and naming it
200
+ # again in the Abstract and Limitations is normal, not over-hedging. Count these only
201
+ # in the non-Methods narrative so an honestly-written single-centre retrospective study
202
+ # (the most common observational design) is not flagged for stating a true fact.
203
+ FACTUAL_DESCRIPTOR_MOTIFS = {"single_center", "retrospective_design"}
204
+
205
+ # Provenance / audit minutiae that belong in Methods or a supplement, not the narrative.
206
+ AUDIT = re.compile(
207
+ r"\bsha-?256\b|\bmd5\b|\bchecksum\b|(?:git\s+)?(?:commit|hash|sha)\s*[:=]?\s*[0-9a-f]{7,40}\b|"
208
+ r"\bcommit\s+[0-9a-f]{7,40}\b|\bunit[-\s]?test(?:s|ing|ed)?\b|\bpost[-\s]?lock\b|"
209
+ r"seed\s*=\s*\d+|\brandom seed\s+\d+\b|\baudit trail\b|"
210
+ r"reproducibility (?:manifest|hash|record)|data lock(?:ed)? on|\bcontent[-\s]hash\b",
211
+ re.IGNORECASE)
212
+
213
+ # Robustness / sensitivity vocabulary (for BURIED_DEFENSE).
214
+ ROBUST = re.compile(
215
+ r"sensitivity analys\w+|robustness|leave[-\s]one[-\s]out|leave[-\s]pair[-\s]out|"
216
+ r"\bE[-\s]?value\b|remained (?:significant|robust|consistent|unchanged|stable)|"
217
+ r"did not (?:materially |substantially |meaningfully )?(?:change|alter|differ)|"
218
+ r"results were (?:similar|consistent|robust|unchanged)|consistent across|"
219
+ r"after (?:excluding|adjusting for|accounting for)|tipping[-\s]point", re.IGNORECASE)
220
+
221
+ # A strong numeric token next to a robustness statement makes it Results-worthy.
222
+ NUMERIC = re.compile(
223
+ r"\b\d+\.\d+\b|\b\d{1,3}%|95%\s*ci|(?:OR|HR|RR|AUC|aHR|aOR)\s*[=:]?\s*\d|"
224
+ r"p\s*[<=>]\s*0?\.\d+", re.IGNORECASE)
225
+
226
+ ORDINALS = ["first", "second", "third", "fourth", "fifth", "sixth", "seventh",
227
+ "eighth", "ninth", "tenth"]
228
+
229
+
230
+ # --------------------------------------------------------------------------- #
231
+ # Probes
232
+ # --------------------------------------------------------------------------- #
233
+
234
+ def probe_hedge_density(regions, body, threshold) -> list[dict]:
235
+ words = word_count(body)
236
+ if words < 80: # too little narrative to judge density reliably
237
+ return []
238
+ n = len(CAVEAT.findall(body))
239
+ density = n / words * 1000
240
+ if density > threshold:
241
+ return [{
242
+ "verdict": "HEDGE_DENSITY", "severity": "Minor", "action": "TIGHTEN",
243
+ "detail": (f"defensive-caveat density is {density:.1f} per 1,000 body words "
244
+ f"({n} caveat tokens / {words} words; threshold {threshold:.0f}); the "
245
+ f"prose hedges faster than it asserts — keep the load-bearing caveats, "
246
+ f"cut the reflexive ones"),
247
+ "where": f"body narrative ({words} words)",
248
+ }]
249
+ return []
250
+
251
+
252
+ def probe_hedge_repeat(regions, full, threshold) -> list[dict]:
253
+ non_methods = "\n".join(b for r, b in regions if r != "methods")
254
+ claims = []
255
+ for key, rx in MOTIFS.items():
256
+ # Factual design descriptors are not over-hedging when stated in Methods;
257
+ # count them only in the non-Methods narrative.
258
+ hay = non_methods if key in FACTUAL_DESCRIPTOR_MOTIFS else full
259
+ n = len(rx.findall(hay))
260
+ if n >= threshold:
261
+ m = rx.search(hay)
262
+ phrase = m.group(0).strip() if m else key
263
+ claims.append({
264
+ "verdict": "HEDGE_REPEAT", "severity": "Minor", "action": "TIGHTEN",
265
+ "detail": (f"the caveat '{phrase}' (motif: {key}) appears {n} times across "
266
+ f"the narrative; state it once, firmly, and remove the repeats"),
267
+ "where": phrase[:120],
268
+ })
269
+ return claims
270
+
271
+
272
+ def probe_audit_in_body(regions) -> list[dict]:
273
+ body = region_text(regions, BODY_NARRATIVE)
274
+ claims = []
275
+ seen = set()
276
+ for m in AUDIT.finditer(body):
277
+ tok = m.group(0).strip().lower()
278
+ norm = re.sub(r"[0-9a-f]{7,40}", "<hash>", tok)
279
+ if norm in seen:
280
+ continue
281
+ seen.add(norm)
282
+ claims.append({
283
+ "verdict": "AUDIT_IN_BODY", "severity": "Minor", "action": "MOVE",
284
+ "detail": (f"provenance/audit token '{m.group(0).strip()}' appears in the "
285
+ f"Introduction/Results/Discussion narrative; move reproducibility "
286
+ f"detail to a Methods statement or a supplement"),
287
+ "where": body[max(0, m.start() - 40):m.end() + 40].strip()[:160],
288
+ })
289
+ return claims
290
+
291
+
292
+ def probe_limitations_volume(regions, full, max_items) -> list[dict]:
293
+ lim = region_text(regions, {"limitations"})
294
+ if not lim:
295
+ # Inline limitations paragraph inside Discussion.
296
+ disc = region_text(regions, {"discussion", "conclusion"})
297
+ m = re.search(r"(?:our |this )?stud(?:y|ies) (?:has|have)[^.]{0,40}limitations?|"
298
+ r"several (?:important )?limitations|limitations? (?:of this|warrant)",
299
+ disc, re.IGNORECASE)
300
+ if not m:
301
+ return []
302
+ lim = disc[m.start():]
303
+ # Count discrete items: max of ordinal markers, (N) enumerators, bullet lines.
304
+ low = lim.lower()
305
+ n_ord = sum(1 for o in ORDINALS if re.search(rf"(?:^|[\s(,;])\b{o}\b\s*,", low))
306
+ n_enum = len(set(re.findall(r"\((\d{1,2})\)", lim)))
307
+ n_bullet = len(re.findall(r"^\s*[-*]\s+\S", lim, re.MULTILINE))
308
+ n = max(n_ord, n_enum, n_bullet)
309
+ if n > max_items:
310
+ return [{
311
+ "verdict": "LIMITATIONS_VOLUME", "severity": "Minor", "action": "TIGHTEN",
312
+ "detail": (f"the Limitations passage enumerates {n} discrete items "
313
+ f"(threshold {max_items}); consolidate related items so the section "
314
+ f"reads as honest disclosure, not a rebuttal letter"),
315
+ "where": f"Limitations ({n} items)",
316
+ }]
317
+ return []
318
+
319
+
320
+ def probe_abstract_caveat_load(regions, full, max_caveats) -> list[dict]:
321
+ abs_t = abstract_text(regions, full)
322
+ if word_count(abs_t) < 40:
323
+ return []
324
+ caveat_sents = [s for s in sentences(abs_t) if CAVEAT.search(s)]
325
+ n = len(caveat_sents)
326
+ if n > max_caveats:
327
+ return [{
328
+ "verdict": "ABSTRACT_CAVEAT_LOAD", "severity": "Minor", "action": "TIGHTEN",
329
+ "detail": (f"the Abstract carries {n} caveat-bearing clauses (threshold "
330
+ f"{max_caveats}); lead with the result and keep at most one or two "
331
+ f"essential qualifiers so the headline is not buried"),
332
+ "where": (caveat_sents[0][:140] if caveat_sents else "Abstract"),
333
+ }]
334
+ return []
335
+
336
+
337
+ def probe_buried_defense(regions) -> list[dict]:
338
+ results = region_text(regions, {"results"})
339
+ buried_src = region_text(regions, {"limitations", "supplement"})
340
+ if not buried_src:
341
+ return []
342
+ # If Results already discusses robustness, nothing is buried.
343
+ if ROBUST.search(results):
344
+ return []
345
+ claims = []
346
+ for s in sentences(buried_src):
347
+ if ROBUST.search(s) and NUMERIC.search(s):
348
+ claims.append({
349
+ "verdict": "BURIED_DEFENSE", "severity": "Minor", "action": "MOVE",
350
+ "detail": ("a numeric robustness/sensitivity result sits in the "
351
+ "Limitations/supplement with no robustness mention in Results; "
352
+ "promote it into Results — it is evidence for the finding, not a "
353
+ "caveat against it"),
354
+ "where": s[:160],
355
+ })
356
+ break # one promotion recommendation is enough
357
+ return claims
358
+
359
+
360
+ # --------------------------------------------------------------------------- #
361
+ # Driver
362
+ # --------------------------------------------------------------------------- #
363
+
364
+ def check(text: str, *, hedge_per_1k: float, repeat_threshold: int,
365
+ limitations_max: int, abstract_caveat_max: int) -> list[dict]:
366
+ regions = segment(text)
367
+ body = region_text(regions, BODY_NARRATIVE)
368
+ if not body: # no IMRAD headings — degrade to a whole-document read (minus supplement)
369
+ body = region_text(regions, {"preamble", "other"}) or text
370
+ claims: list[dict] = []
371
+ claims += probe_hedge_density(regions, body, hedge_per_1k)
372
+ claims += probe_hedge_repeat(regions, text, repeat_threshold)
373
+ claims += probe_audit_in_body(regions)
374
+ claims += probe_limitations_volume(regions, text, limitations_max)
375
+ claims += probe_abstract_caveat_load(regions, text, abstract_caveat_max)
376
+ claims += probe_buried_defense(regions)
377
+ return claims
378
+
379
+
380
+ def analyze(manuscript: str, **kw) -> dict:
381
+ p = Path(manuscript)
382
+ if not p.is_file():
383
+ sys.stderr.write(f"ERROR: manuscript not found: {manuscript}\n")
384
+ sys.exit(2)
385
+ claims = check(p.read_text(encoding="utf-8"), **kw)
386
+ by_action = {"REMOVE": 0, "MOVE": 0, "TIGHTEN": 0}
387
+ for c in claims:
388
+ by_action[c["action"]] = by_action.get(c["action"], 0) + 1
389
+ return {
390
+ "manuscript": str(p),
391
+ "claims": claims,
392
+ "summary": {
393
+ "n_claims": len(claims),
394
+ "by_action": by_action,
395
+ "verdict": "IMPRESSION_FLAGS" if claims else "OK",
396
+ },
397
+ }
398
+
399
+
400
+ def render(result: dict) -> str:
401
+ lines = ["| Check | Action | Detail |", "|---|---|---|"]
402
+ for c in result["claims"]:
403
+ lines.append(f"| {c['verdict']} | {c['action']} | {c['detail']} |")
404
+ if len(lines) == 2:
405
+ lines.append("| (none) | — | narrative reads confidently; no subtraction needed |")
406
+ return "\n".join(lines)
407
+
408
+
409
+ def main() -> int:
410
+ ap = argparse.ArgumentParser(
411
+ description="Editorial-impression / defensiveness gate (§L) — advisory, non-blocking.")
412
+ ap.add_argument("--manuscript", required=True, help="manuscript markdown/text")
413
+ ap.add_argument("--out", help="write JSON artifact to this path")
414
+ ap.add_argument("--strict", action="store_true",
415
+ help="accepted for CLI parity; this gate emits no Major, so it never blocks")
416
+ ap.add_argument("--quiet", action="store_true", help="suppress stdout table")
417
+ ap.add_argument("--hedge-per-1k", type=float, default=10.0,
418
+ help="HEDGE_DENSITY: caveat tokens per 1,000 body words before firing (default 10)")
419
+ ap.add_argument("--repeat-threshold", type=int, default=3,
420
+ help="HEDGE_REPEAT: motif occurrences across body+abstract before firing (default 3)")
421
+ ap.add_argument("--limitations-max", type=int, default=6,
422
+ help="LIMITATIONS_VOLUME: discrete Limitations items allowed (default 6)")
423
+ ap.add_argument("--abstract-caveat-max", type=int, default=2,
424
+ help="ABSTRACT_CAVEAT_LOAD: caveat clauses allowed in the Abstract (default 2)")
425
+ args = ap.parse_args()
426
+
427
+ result = analyze(
428
+ args.manuscript,
429
+ hedge_per_1k=args.hedge_per_1k,
430
+ repeat_threshold=args.repeat_threshold,
431
+ limitations_max=args.limitations_max,
432
+ abstract_caveat_max=args.abstract_caveat_max,
433
+ )
434
+
435
+ if not args.quiet:
436
+ print("=" * 41)
437
+ print(" Editorial Impression / Defensiveness (§L)")
438
+ print("=" * 41)
439
+ print(render(result))
440
+ print()
441
+ s = result["summary"]
442
+ if s["n_claims"]:
443
+ ba = s["by_action"]
444
+ print(f"IMPRESSION flags: {s['n_claims']} advisory finding(s) "
445
+ f"(REMOVE {ba['REMOVE']} / MOVE {ba['MOVE']} / TIGHTEN {ba['TIGHTEN']}). "
446
+ f"Non-blocking — these raise the ceiling, they do not gate submission.")
447
+ else:
448
+ print("OK: narrative reads confidently; no subtraction recommended.")
449
+
450
+ if args.out:
451
+ Path(args.out).parent.mkdir(parents=True, exist_ok=True)
452
+ Path(args.out).write_text(json.dumps(result, indent=2), encoding="utf-8")
453
+ if not args.quiet:
454
+ print(f"\nwrote {args.out}")
455
+
456
+ # Advisory: never blocks. --strict is accepted for parity but this gate has no Major.
457
+ return 0
458
+
459
+
460
+ if __name__ == "__main__":
461
+ sys.exit(main())
@@ -22,6 +22,7 @@ outputs:
22
22
  - qc/reference_adequacy.json
23
23
  deterministic_scripts:
24
24
  - scripts/check_reference_adequacy.py
25
+ - scripts/check_editorial_impression.py
25
26
  side_effects:
26
27
  - may_edit_manuscript_when_fix_flag_set
27
28
  downstream_consumers:
@@ -44,5 +45,6 @@ validation_commands:
44
45
  - "python3 scripts/check_domain_probe_sync.py --strict"
45
46
  - "bash tests/test_panel_mode.sh"
46
47
  - "bash tests/test_reference_adequacy.sh"
48
+ - "bash tests/test_editorial_impression.sh"
47
49
  - "feed R0-numbered output into /revise"
48
50
  evidence_surface: demo
@@ -0,0 +1,29 @@
1
+ # A Deep-Learning Marker for Synthetic Outcome X: Development and Internal Validation
2
+
3
+ ## Abstract
4
+
5
+ **Background:** Marker X may aid early triage of outcome X. **Methods:** We developed and internally validated a model on a patient-level split. **Results:** The model discriminated outcome X with an area under the curve of 0.84 (95% CI 0.79–0.89) and was well calibrated. **Conclusion:** Marker X identifies patients at higher risk of outcome X and supports triage. These findings are preliminary and require external validation.
6
+
7
+ ## Introduction
8
+
9
+ Outcome X is common and its early identification changes triage. Existing tools rely on manual scoring, which is slow and operator-dependent. We asked whether a deep-learning marker derived from routine inputs could identify patients at higher risk and inform the triage decision. This study develops such a marker and tests it on a held-out internal cohort, addressing a gap left by prior manual approaches.
10
+
11
+ ## Methods
12
+
13
+ We assembled a cohort and split it at the patient level into development and held-out sets before any preprocessing. A regularised model was trained with five-fold cross-validation. All analysis code, the dataset schema, and a content-hash manifest are archived (see Data Availability).
14
+
15
+ ## Results
16
+
17
+ The model discriminated outcome X with an area under the curve of 0.84 (95% CI 0.79–0.89). Calibration was good, with a slope of 0.97 and a Brier score of 0.12. At the triage threshold, sensitivity was 0.82 and specificity was 0.79. In a sensitivity analysis excluding the 41 borderline cases, the adjusted odds ratio was unchanged at 2.18 (95% CI 1.40–3.39), and results were consistent across the two recruitment years. Decision-curve analysis showed net benefit over the manual score across the clinically relevant threshold range.
18
+
19
+ ## Discussion
20
+
21
+ A routine-input marker identified patients at higher risk of outcome X and added net benefit over the existing manual score, which is the decision it is meant to inform. The effect size corresponds to a clinically meaningful shift in pretest probability. The marker is reproducible and its calibration supports use at the stated threshold.
22
+
23
+ ## Limitations
24
+
25
+ This single-centre study has three main limitations: it was developed on retrospective data, the marker was measured once, and external validation in an independent cohort is the necessary next step before deployment.
26
+
27
+ ## Data Availability
28
+
29
+ The analysis code, the dataset schema, the patient-level split assignment, and a reproducibility manifest with the dataset content hash are archived in the project repository, so every reported number can be regenerated from a single committed pipeline.
@@ -0,0 +1,25 @@
1
+ # A Deep-Learning Marker for Synthetic Outcome X: A Preliminary Single-Centre Study
2
+
3
+ ## Abstract
4
+
5
+ **Background:** Marker X has been proposed as a screening adjunct. **Methods:** We trained a model on a retrospective cohort. **Results:** The model reached an area under the curve of 0.84. **Conclusion:** These findings are preliminary and exploratory and should be interpreted with caution. The results are hypothesis-generating and not generalizable beyond this single-centre cohort, and no deployable claim is made. Because the analysis is underpowered, no causal inference can be inferred.
6
+
7
+ ## Introduction
8
+
9
+ Marker X is of interest, although the evidence remains uncertain and the prior literature is limited by small samples. We hypothesise that a model may help, but the work is exploratory and any conclusion should be interpreted with caution. We make no deployable claim and the results are not generalizable; the study is hypothesis-generating only.
10
+
11
+ ## Methods
12
+
13
+ We used a retrospective cohort and fit a regularised model. Splits were at the patient level.
14
+
15
+ ## Results
16
+
17
+ The model reached an area under the curve of 0.84 (95% CI 0.79–0.89). The reproducibility manifest and the sha256 checksum of the locked dataset are reported. Every metric in this paragraph was produced by a committed unit-test against the post-lock data with seed=42, and the git commit a1b2c3d4e5f6 pins the exact run. The audit trail for each figure is recorded.
18
+
19
+ ## Discussion
20
+
21
+ Our marker may help, but the finding is preliminary and modest and cannot be established as causal. The result is not generalizable and remains uncertain; it is hypothesis-generating only and warrants further study. We reiterate that no deployable claim is made and that any inference should be interpreted with caution. The exact manifest hash and the post-lock timeline are given above for full auditability.
22
+
23
+ ## Limitations
24
+
25
+ This study has several limitations. First, the design is retrospective and single-centre, so selection bias cannot be excluded. Second, the sample is limited and the analysis is underpowered. Third, the marker was measured once. Fourth, no external validation was performed. Fifth, residual confounding cannot be ruled out. Sixth, the outcome label is registry-derived. Seventh, generalizability is uncertain. In a sensitivity analysis excluding the 41 borderline cases, the adjusted odds ratio remained 2.18 (95% CI 1.40–3.39), which we note here for completeness.
@@ -0,0 +1,61 @@
1
+ #!/usr/bin/env bash
2
+ # Regression test for the editorial-impression / defensiveness gate (self-review §L).
3
+ # Synthetic, PII-free fixtures: (a) an over-defensive manuscript that trips all six
4
+ # probes (HEDGE_DENSITY, HEDGE_REPEAT, AUDIT_IN_BODY, LIMITATIONS_VOLUME,
5
+ # ABSTRACT_CAVEAT_LOAD, BURIED_DEFENSE); (b) a confident clean manuscript that trips
6
+ # none (false-positive guard). The gate is advisory and non-blocking — it must exit 0
7
+ # even under --strict, since it emits no Major.
8
+ # Stdlib-only (python3).
9
+ set -u
10
+
11
+ HERE="$(cd "$(dirname "${BASH_SOURCE[0]}")" && pwd)"
12
+ SCRIPT="$HERE/../scripts/check_editorial_impression.py"
13
+ DEF="$HERE/fixtures/editorial_defensive.md"
14
+ CLEAN="$HERE/fixtures/editorial_clean.md"
15
+ OUT="$(mktemp -t editorial_XXXX).json"
16
+ trap 'rm -f "$OUT"' EXIT
17
+
18
+ fail=0
19
+ check() { local label="$1"; shift
20
+ if "$@" >/dev/null 2>&1; then printf ' PASS %s\n' "$label"
21
+ else printf ' FAIL %s\n' "$label"; fail=$((fail+1)); fi
22
+ }
23
+ has_verdict() { python3 -c "
24
+ import json
25
+ d=json.load(open('$OUT'))
26
+ assert any(c['verdict']=='$1' for c in d['claims']), '$1 not found'
27
+ "; }
28
+
29
+ [[ -f "$SCRIPT" ]] || { echo "ENV-ERR: script missing" >&2; exit 2; }
30
+
31
+ # (1) defensive fixture -> all six verdicts fire
32
+ python3 "$SCRIPT" --manuscript "$DEF" --out "$OUT" --quiet >/dev/null 2>&1
33
+ for v in HEDGE_DENSITY HEDGE_REPEAT AUDIT_IN_BODY LIMITATIONS_VOLUME ABSTRACT_CAVEAT_LOAD BURIED_DEFENSE; do
34
+ check "$v detected (defensive)" has_verdict "$v"
35
+ done
36
+
37
+ # (2) every claim carries a SUBTRACTION action in {REMOVE, MOVE, TIGHTEN}
38
+ check "every claim has a REMOVE/MOVE/TIGHTEN action" python3 -c "
39
+ import json
40
+ d=json.load(open('$OUT'))
41
+ assert d['claims'], 'expected findings'
42
+ assert all(c.get('action') in ('REMOVE','MOVE','TIGHTEN') for c in d['claims']), 'bad action'
43
+ assert all(c.get('severity')=='Minor' for c in d['claims']), 'all findings must be Minor (advisory)'
44
+ "
45
+
46
+ # (3) advisory / non-blocking: exit 0 even under --strict on a fully-flagged manuscript
47
+ python3 "$SCRIPT" --manuscript "$DEF" --strict --quiet >/dev/null 2>&1
48
+ check "exit 0 under --strict (non-blocking)" test "$?" -eq 0
49
+
50
+ # (4) clean fixture -> zero claims (false-positive guard)
51
+ python3 "$SCRIPT" --manuscript "$CLEAN" --out "$OUT" --quiet >/dev/null 2>&1
52
+ check "clean manuscript yields no flags" python3 -c "
53
+ import json
54
+ d=json.load(open('$OUT'))
55
+ assert d['summary']['n_claims']==0, d['claims']
56
+ "
57
+ python3 "$SCRIPT" --manuscript "$CLEAN" --strict --quiet >/dev/null 2>&1
58
+ check "exit 0 on clean manuscript" test "$?" -eq 0
59
+
60
+ echo "fail=$fail"; [[ "$fail" -eq 0 ]] && echo "ALL PASS" || echo "FAILURES: $fail"
61
+ exit "$fail"