replication-radar 0.3.2__tar.gz → 0.3.4__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (36) hide show
  1. replication_radar-0.3.4/DEMO.md +149 -0
  2. {replication_radar-0.3.2 → replication_radar-0.3.4}/PKG-INFO +1 -1
  3. replication_radar-0.3.4/STORY.md +130 -0
  4. {replication_radar-0.3.2 → replication_radar-0.3.4}/pyproject.toml +1 -1
  5. {replication_radar-0.3.2 → replication_radar-0.3.4}/site/app.js +147 -47
  6. {replication_radar-0.3.2 → replication_radar-0.3.4}/site/index.html +7 -4
  7. replication_radar-0.3.4/site/methodology.html +91 -0
  8. replication_radar-0.3.4/site/methodology.json +59 -0
  9. {replication_radar-0.3.2 → replication_radar-0.3.4}/site/style.css +50 -18
  10. replication_radar-0.3.4/site/verdicts.json +394 -0
  11. replication_radar-0.3.4/src/replication_radar/data/verdicts.json +584 -0
  12. {replication_radar-0.3.2 → replication_radar-0.3.4}/src/replication_radar/network.py +37 -1
  13. {replication_radar-0.3.2 → replication_radar-0.3.4}/src/replication_radar/openaire.py +22 -0
  14. {replication_radar-0.3.2 → replication_radar-0.3.4}/src/replication_radar/radar.py +10 -1
  15. {replication_radar-0.3.2 → replication_radar-0.3.4}/src/replication_radar/server.py +13 -3
  16. replication_radar-0.3.4/src/replication_radar/verdicts.py +128 -0
  17. replication_radar-0.3.2/STORY.md +0 -111
  18. replication_radar-0.3.2/site/verdicts.json +0 -346
  19. replication_radar-0.3.2/src/replication_radar/data/verdicts.json +0 -346
  20. replication_radar-0.3.2/src/replication_radar/verdicts.py +0 -56
  21. {replication_radar-0.3.2 → replication_radar-0.3.4}/.github/workflows/publish-pypi.yml +0 -0
  22. {replication_radar-0.3.2 → replication_radar-0.3.4}/.gitignore +0 -0
  23. {replication_radar-0.3.2 → replication_radar-0.3.4}/CLAUDE.md +0 -0
  24. {replication_radar-0.3.2 → replication_radar-0.3.4}/LICENSE +0 -0
  25. {replication_radar-0.3.2 → replication_radar-0.3.4}/README.md +0 -0
  26. {replication_radar-0.3.2 → replication_radar-0.3.4}/demo_sdm.py +0 -0
  27. {replication_radar-0.3.2 → replication_radar-0.3.4}/docs/app-ui.md +0 -0
  28. {replication_radar-0.3.2 → replication_radar-0.3.4}/docs/link-types.md +0 -0
  29. {replication_radar-0.3.2 → replication_radar-0.3.4}/docs/next-layers-plan.md +0 -0
  30. {replication_radar-0.3.2 → replication_radar-0.3.4}/docs/openaire-mcp.md +0 -0
  31. {replication_radar-0.3.2 → replication_radar-0.3.4}/docs/readiness-scoring-plan.md +0 -0
  32. {replication_radar-0.3.2 → replication_radar-0.3.4}/netlify.toml +0 -0
  33. {replication_radar-0.3.2 → replication_radar-0.3.4}/scripts/build_verdicts.py +0 -0
  34. {replication_radar-0.3.2 → replication_radar-0.3.4}/site/README.md +0 -0
  35. {replication_radar-0.3.2 → replication_radar-0.3.4}/site/curated.json +0 -0
  36. {replication_radar-0.3.2 → replication_radar-0.3.4}/src/replication_radar/__init__.py +0 -0
@@ -0,0 +1,149 @@
1
+ # Replication Radar — MCP demo runbook
2
+
3
+ A clean ~60-second screen recording showing the **verified-knowledge MCP** in an AI agent:
4
+ the agent gives a *cited, verified* answer about a research claim instead of a confident,
5
+ unchecked one. This is the "cite instead of hallucinate" moment — the AI-hackathon hook.
6
+
7
+ The MCP is **read-only**: it answers *"has this claim been independently checked, and did it
8
+ hold?"* It does **not** start replications (that's the FORRT template). Don't demo a "start a
9
+ replication" flow with it.
10
+
11
+ ---
12
+
13
+ ## 1 · One-time setup (~5 minutes)
14
+
15
+ **Install the MCP in an isolated environment** (so the path is stable for the client):
16
+
17
+ ```bash
18
+ python3 -m venv ~/.venvs/radar
19
+ ~/.venvs/radar/bin/pip install replication-radar # pulls in the `mcp` runtime too
20
+ ```
21
+
22
+ **Smoke-test it works** (should print `True` then a number):
23
+
24
+ ```bash
25
+ ~/.venvs/radar/bin/python - <<'PY'
26
+ from replication_radar.radar import replication_status, verified_claims
27
+ print("replicated:", replication_status("10.1126/science.aax8591")["replicated"])
28
+ print("verified claims in corpus:", verified_claims()["count"])
29
+ PY
30
+ ```
31
+
32
+ **Register it with Claude Desktop.** Edit
33
+ `~/Library/Application Support/Claude/claude_desktop_config.json` (create it if missing) — use the
34
+ **absolute** python path (Claude Desktop does not use your shell PATH):
35
+
36
+ ```json
37
+ {
38
+ "mcpServers": {
39
+ "replication-radar": {
40
+ "command": "/Users/annef/.venvs/radar/bin/python",
41
+ "args": ["-m", "replication_radar.server"]
42
+ }
43
+ }
44
+ }
45
+ ```
46
+
47
+ Quit and reopen Claude Desktop. Click the tools/🔨 icon — you should see **replication-radar**
48
+ with 4 tools: `radar`, `replication_status`, `find_independent_software`, `verified_claims`.
49
+
50
+ **Optional — add the OpenAIRE / Alien Gateway MCP** alongside it (you/Jean have the connection).
51
+ It makes the "two MCPs together" point explicit. If wiring it up is fiddly, **skip it** — the
52
+ demo lands with just `replication-radar`.
53
+
54
+ ---
55
+
56
+ ## 2 · Pre-flight (right before recording)
57
+
58
+ 1. Ask: *"Which tools do you have from replication-radar?"* → it should list the 4. (Warms it up.)
59
+ 2. Do **one off-record dry run** of Beat 1 below — it warms the network/HTTP caches so the real
60
+ take is fast and identical.
61
+
62
+ ---
63
+
64
+ ## 3 · The recording — three beats
65
+
66
+ ### Beat 1 — the money shot: verify + cite (≈35 s)
67
+
68
+ **Type this prompt:**
69
+
70
+ > I want to cite the finding from Soroye et al. 2020 (Science, DOI 10.1126/science.aax8591) —
71
+ > that projected per-species bumble-bee extirpation rankings are robust. Before I do: has that
72
+ > claim actually been independently replicated, and did it hold? Give me something citable.
73
+
74
+ **What to expect:** the agent calls **`replication_status("10.1126/science.aax8591")`** and gets
75
+ back `replicated: true` with **5 independent verdicts** — 4 `confirms` (Validated) and 1
76
+ `qualifies` (PartiallySupported) — each with a **signed Outcome nanopublication URL** and the
77
+ replication's deposit DOI. The answer should say roughly: *"independently replicated 5×, 4
78
+ confirmed, 1 qualifies it; here are the signed verdicts to cite,"* with links.
79
+
80
+ **The point to land (caption or voiceover):** OpenAIRE/citation count would call this paper
81
+ "settled"; the MCP shows it's been checked 5 times and hands you **signed, citable verdicts** —
82
+ including the one that *qualifies* it. That's the difference between paraphrasing and citing.
83
+
84
+ ### Beat 2 — abstract → atomic claim → verdict: the two-project synergy (≈30 s)
85
+
86
+ This is the strongest AI beat: the agent turns a paper's **text into a structured claim** and
87
+ then **verifies** it, in one turn — your two hackathon projects composing (Jean's Hackaweek
88
+ claim-*extraction* pipeline + the Radar's claim-*verification* layer). Needs the MCP **≥ 0.3.3**
89
+ (it now returns the paper's abstract).
90
+
91
+ **Type this prompt:**
92
+
93
+ > Read the abstract of Soroye et al. 2020 (DOI 10.1126/science.aax8591), extract its single
94
+ > central finding as one atomic AIDA sentence, and then tell me whether that exact claim has
95
+ > been independently replicated.
96
+
97
+ **What to expect:** the agent calls **`replication_status("10.1126/science.aax8591")`**, which now
98
+ returns the **abstract** alongside the verdicts. The agent reads the abstract, **writes the atomic
99
+ claim itself** (something like *"An increasing frequency of unusually hot days raises local
100
+ extinction and lowers site occupancy of bumble bees, independent of land-use change"*), and then
101
+ reports it has been **independently replicated 5× (4 confirm, 1 qualifies)** with the signed
102
+ nanopubs.
103
+
104
+ **The point to land:** *from a paper's text → a structured, atomic claim → a signed, citable
105
+ verdict, in one turn.* Generation meets verification — the full "graph of verified knowledge" arc,
106
+ live.
107
+
108
+ ### Beat 3 — the discovery side (≈20 s, optional)
109
+
110
+ **Type this prompt:**
111
+
112
+ > What high-impact work on marine heatwaves and species distributions is worth replicating —
113
+ > and what's already been checked?
114
+
115
+ **What to expect:** the agent calls **`radar("marine heatwave species")`** and returns
116
+ impact-ranked papers, each flagged **OPEN** (a replication opportunity) or **VERIFIED** (already
117
+ checked, with the verdict).
118
+
119
+ **The point to land:** the same layer also tells an agent *where the replication gaps are*.
120
+
121
+ ---
122
+
123
+ ## 4 · Recording tips
124
+
125
+ - Resize the Claude window to a clean 1280×800-ish; hide other panels.
126
+ - ~45–75 seconds total; no audio needed — burn in 2-3 short captions for the "point to land" lines.
127
+ - Keep the tool-call expansion **visible** for a beat (it's proof the answer came from the MCP,
128
+ not the model's memory).
129
+ - Export as MP4 or GIF for the submission.
130
+
131
+ ---
132
+
133
+ ## 5 · If the OpenAIRE MCP is set up too
134
+
135
+ Add a one-line framing before Beat 1: *"Two MCPs are connected — OpenAIRE for the structural
136
+ graph, replication-radar for the verification layer."* You don't need to force OpenAIRE to fire;
137
+ its presence in the tools list is enough to make the pairing point. The verification answer from
138
+ `replication-radar` is the star.
139
+
140
+ ---
141
+
142
+ *Tools reference — what the agent can call:*
143
+
144
+ | tool | answers |
145
+ |---|---|
146
+ | `replication_status(doi)` | Has this DOI been replicated, did it hold? Verdicts + signed nanopub links. |
147
+ | `verified_claims()` | The whole verified-knowledge corpus (every claim with a verdict). |
148
+ | `radar(topic)` | Impact-ranked replication targets in a field — OPEN vs VERIFIED. |
149
+ | `find_independent_software(doi, topic)` | Reusable engines *not* authored by the original team. |
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: replication-radar
3
- Version: 0.3.2
3
+ Version: 0.3.4
4
4
  Summary: MCP server that turns the OpenAIRE Graph into a ranked replication queue — impact-ranked targets and the Science Live verification overlay (retraction/supersession-aware).
5
5
  Project-URL: Homepage, https://github.com/ScienceLiveHub/replication-radar
6
6
  Project-URL: Repository, https://github.com/ScienceLiveHub/replication-radar
@@ -0,0 +1,130 @@
1
+ # Replication Radar — adding the signals the OpenAIRE Graph can't hold
2
+
3
+ *OpenAIRE AI Hackathon · Theme B (Build) · a Science Live contribution*
4
+ **Live app: https://openaire-hackathon.netlify.app · how it works: /methodology.html · `pip install replication-radar`**
5
+
6
+ ## The question
7
+
8
+ The OpenAIRE Graph is a network of **structural links** between research entities — papers,
9
+ authors, institutions, funding. It can tell you how *visible* a paper is (citation influence,
10
+ popularity, the BIP! classes C1–C5), but not what it *means*: it links documents to one another
11
+ without representing the **claims** inside them, their level of evidence, their **epistemic
12
+ status** (confirmed, contested, retracted, superseded), or the **semantic relations between
13
+ results** — replication, contradiction, refinement, not just "cites".
14
+
15
+ That gap matters more than ever in the age of LLMs. A model fed the Graph swallows everything
16
+ equally: a result replicated fifty times reads the same as a single study on twelve mice or an
17
+ unreviewed preprint. The difference between *recognising text patterns* and *understanding* is
18
+ exactly this missing layer — verified, status-aware, traceable knowledge that a system can **cite
19
+ instead of paraphrase**. OpenAIRE is the infrastructure best placed to start closing that gap at
20
+ European scale.
21
+
22
+ So we asked a concrete build question toward it: **can we add the two most actionable missing
23
+ signals — is a claim *reliable* (independently checked, and did it hold) and is its *software*
24
+ reusable — live, on top of the Graph, without changing it?** A heavily-cited paper looks
25
+ identical, in the Graph today, to one nobody ever reproduced; a widely-used research tool has the
26
+ same "0 citations, class C5" as an abandoned script. Both are signals the Graph structurally
27
+ cannot hold.
28
+
29
+ ## The journey
30
+
31
+ We started simply: rank papers by impact to find what's worth replicating. That worked, but it
32
+ just re-served the Graph's one signal. The turn came when we tried to answer "has this been
33
+ replicated?" — and realised the Graph *structurally cannot* hold that answer. A verification isn't
34
+ a paper, gets no citations, and has no node in the Graph.
35
+
36
+ But it does exist elsewhere. Science Live publishes replication outcomes as cryptographically
37
+ signed **nanopublications** (the FORRT chain: Quote → Claim → Study → Outcome → CiTO). So the
38
+ Radar pulls the verdict layer **live from the nanopub network** and overlays it on the Graph by
39
+ DOI. Several corrections shaped the design along the way:
40
+
41
+ - **Verification is author-agnostic.** We don't care *who* ran the replication, so the index is
42
+ built **by template, not by person** — querying every FORRT Outcome and CiTO on the network and
43
+ joining them on the nanopub trusty hash. Today that surfaces **31 independent, signed
44
+ replication outcomes across 21 papers**; as more people publish replications, they flow in
45
+ automatically.
46
+ - **Enumeration has to be the right shape.** We verified empirically that walking the nanopub
47
+ graph outward from a paper *bleeds* into adjacent chains (it once pulled a lizard study into a
48
+ bumble-bee paper's replications) and *misses* disconnected ones, so we enumerate by the
49
+ **CiTO→DOI verdict-citation** instead — the set that is actually correct.
50
+ - **Validity is part of the verdict.** A retracted or superseded outcome must not count. The
51
+ overlay filters any outcome retracted/invalidated/superseded **by its own signer**, via the
52
+ nanopub admin graph — only the original author can retract their own work.
53
+ - **Reproduce ≠ replicate, and agreement matters.** We surface the FORRT distinction (materials
54
+ available = reproducible; tested by a different route = replicated) and an **agreement pattern**
55
+ — robustly-validated, validated, contested, refuted — computed from the verdict spread, so
56
+ "five replications that all agree" reads differently from "five that disagree".
57
+ - **What, not just whether.** Each verdict carries the **claim it actually tested** — the atomic
58
+ AIDA statement, traversed Outcome → Study → Claim — so a card says not "Validated" but
59
+ *"Validated: ‘per-species extirpation rankings are sensitive to the grid resolution’"*.
60
+
61
+ Then the software side. The Graph makes research software *findable* but not *assessable*. We
62
+ first tried to *recommend* tooling and it failed badly (keyword-matching surfaced off-topic
63
+ repos), so we pivoted from recommendation to **assessment**, and after checking that standard
64
+ FAIR services (F-UJI, OSTrails) had no usable API, computed the **fair-software.eu** five
65
+ recommendations ourselves, live, from the GitHub and Software Heritage APIs.
66
+
67
+ Two disciplines run through all of it. **Everything is grounded** — every signal comes from a
68
+ named, verifiable source, and is documented, signal by signal, in a **machine- and human-readable
69
+ methodology page** (`methodology.json` + `/methodology.html`) that states where each label and
70
+ score comes from and how it is computed. And everything runs **client-side** against public,
71
+ CORS-enabled APIs — no backend, no keys — so the whole thing is a static site anyone can open,
72
+ and it ships accessible (Lighthouse accessibility 100, colour-blind-safe, keyboard-operable).
73
+
74
+ ## The insight
75
+
76
+ - **Nanopublications are the substrate the Graph is missing — and the AI hook.** The verdict
77
+ layer isn't scraped text; it's built from claim-level, cryptographically-signed assertions that
78
+ already carry what the Graph lacks: the claim, its epistemic relation (`cito:confirms` /
79
+ `disputes` / `qualifies`), and its provenance. So the Graph gains, for a paper, not "this
80
+ document exists" but "*this specific claim was independently checked → validated → here is the
81
+ signed verdict*". That is exactly what an LLM needs to **cite rather than hallucinate**. We
82
+ package it as an **MCP server** that an agent runs **next to the OpenAIRE/Alien MCP**: one gives
83
+ the structural graph, the other answers "has this been checked, and did it hold". Together they
84
+ are the first bricks of a graph of **verified knowledge**.
85
+ - **Reliability and reusability are *different categories* of signal**, not better metrics. You
86
+ can't repair the citation axis into a truth axis or a reuse axis — you have to *add* them, and
87
+ you can add them *live*, on top of the Graph, without waiting for it to change.
88
+ - **Verification is author-agnostic and network-wide.** Keying it on a template rather than a
89
+ person turns a personal portfolio into a community trust layer.
90
+ - **Grounded-and-transparent is a discipline, not a nicety.** Every signal is sourced and
91
+ documented; the one feature we built on a guess (keyword tooling) we deleted — and the project
92
+ is stronger for it.
93
+
94
+ ## What others can reuse
95
+
96
+ - **The live web app** — pure static, queries OpenAIRE + the nanopub network + GitHub/Software
97
+ Heritage from the browser. Fork it, point it elsewhere.
98
+ - **An MCP server** (`pip install replication-radar`) exposing the same engine to any agent, to
99
+ run alongside the OpenAIRE MCP — the verified-knowledge layer for agentic workflows.
100
+ - **A reproducible, author-agnostic, retraction-aware verdict-index method** — FORRT
101
+ Outcome/CiTO templates joined on the trusty hash, with the admin-graph validity guard. Any
102
+ replication network can be read this way.
103
+ - **A machine-readable provenance & methodology spec** (`methodology.json`, CC-BY) — every
104
+ signal's source and formula, reusable as a transparency pattern for any composite-score tool.
105
+ - **A grounded software-FAIR assessment** — the fair-software.eu recommendations + usage,
106
+ computed from GitHub + Software Heritage (no third-party scorer needed).
107
+ - **A feasibility map of the open-science API landscape** — what's reachable and CORS-friendly
108
+ (OpenAIRE Graph API, the nanopub SPARQL + admin graph, GitHub/SWH/Zenodo) and what isn't
109
+ (F-UJI/OSTrails assessment APIs; per-paper relations from the public Graph API) — so the next
110
+ builder doesn't re-discover it.
111
+
112
+ *A complementary facet by Jean Iaquinta uses the **OpenAIRE MCP's** citation-graph tools to trace
113
+ the relationships around a verified paper — and shows the citation graph contains everything
114
+ **except** the verification edge, which is exactly the gap the Radar fills.*
115
+
116
+ ## Honest limits
117
+
118
+ Discovery recall is keyword-bound (OpenAIRE free-text terms are AND-ed); the verdict overlay is
119
+ network-wide but only covers claims with a DOI a search can reach; FAIR-software runs only where a
120
+ real repository resolves, and GitHub's unauthenticated rate limit caps how many it scores per hour
121
+ (results are cached so repeated use stays stable); OpenAIRE's own subject classification is
122
+ sometimes quirky and is shown faithfully, not corrected. None of this is hidden in the output. And
123
+ we add only two of the missing layers — reliability and reusability; the fuller graph of *verified
124
+ knowledge* (claim-level extraction at scale, temporal obsolescence, distinguishing hypothesis from
125
+ result from interpretation) is the direction this points at, not something we finished.
126
+
127
+ ---
128
+
129
+ *Materials are dual-licensed: **source code under MIT**, and this write-up together with the
130
+ verdict index and methodology spec under **[CC-BY 4.0](https://creativecommons.org/licenses/by/4.0/)**.*
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "replication-radar"
3
- version = "0.3.2"
3
+ version = "0.3.4"
4
4
  description = "MCP server that turns the OpenAIRE Graph into a ranked replication queue — impact-ranked targets and the Science Live verification overlay (retraction/supersession-aware)."
5
5
  readme = "README.md"
6
6
  requires-python = ">=3.10"