paper-repro 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (64) hide show
  1. paper_repro-0.1.0/.github/workflows/ci.yml +31 -0
  2. paper_repro-0.1.0/.github/workflows/release.yml +34 -0
  3. paper_repro-0.1.0/.gitignore +13 -0
  4. paper_repro-0.1.0/AGENTS.md +33 -0
  5. paper_repro-0.1.0/CHANGELOG.md +25 -0
  6. paper_repro-0.1.0/CONTRIBUTING.md +15 -0
  7. paper_repro-0.1.0/LICENSE +21 -0
  8. paper_repro-0.1.0/PKG-INFO +194 -0
  9. paper_repro-0.1.0/README.md +172 -0
  10. paper_repro-0.1.0/SECURITY.md +9 -0
  11. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/evidence.jsonl +13 -0
  12. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/inspect.json +613 -0
  13. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/lock.txt +62 -0
  14. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/report.json +264 -0
  15. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/report.md +78 -0
  16. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r1/stderr.txt +0 -0
  17. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r1/stdout.txt +6 -0
  18. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r2/stderr.txt +4 -0
  19. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r2/stdout.txt +70 -0
  20. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r3/stderr.txt +4 -0
  21. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r3/stdout.txt +2076 -0
  22. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r4/stderr.txt +4 -0
  23. paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r4/stdout.txt +2076 -0
  24. paper_repro-0.1.0/examples/pygat-cora/env-inputs/e3-unpinned-requirements.txt +3 -0
  25. paper_repro-0.1.0/examples/pygat-cora/evidence.jsonl +14 -0
  26. paper_repro-0.1.0/examples/pygat-cora/inspect.json +175 -0
  27. paper_repro-0.1.0/examples/pygat-cora/lock.txt +12 -0
  28. paper_repro-0.1.0/examples/pygat-cora/report.json +290 -0
  29. paper_repro-0.1.0/examples/pygat-cora/report.md +86 -0
  30. paper_repro-0.1.0/examples/pygat-cora/runs/r1/stderr.txt +0 -0
  31. paper_repro-0.1.0/examples/pygat-cora/runs/r1/stdout.txt +7 -0
  32. paper_repro-0.1.0/examples/pygat-cora/runs/r2/stderr.txt +0 -0
  33. paper_repro-0.1.0/examples/pygat-cora/runs/r2/stdout.txt +840 -0
  34. paper_repro-0.1.0/examples/pygat-cora/runs/r3/stderr.txt +0 -0
  35. paper_repro-0.1.0/examples/pygat-cora/runs/r3/stdout.txt +816 -0
  36. paper_repro-0.1.0/examples/pygat-cora/runs/r4/stderr.txt +0 -0
  37. paper_repro-0.1.0/examples/pygat-cora/runs/r4/stdout.txt +834 -0
  38. paper_repro-0.1.0/pyproject.toml +54 -0
  39. paper_repro-0.1.0/skills/paper-repro/SKILL.md +76 -0
  40. paper_repro-0.1.0/src/paper_repro/__init__.py +3 -0
  41. paper_repro-0.1.0/src/paper_repro/__main__.py +3 -0
  42. paper_repro-0.1.0/src/paper_repro/agent_setup.py +277 -0
  43. paper_repro-0.1.0/src/paper_repro/claims.py +237 -0
  44. paper_repro-0.1.0/src/paper_repro/cli.py +541 -0
  45. paper_repro-0.1.0/src/paper_repro/compare.py +427 -0
  46. paper_repro-0.1.0/src/paper_repro/envs.py +323 -0
  47. paper_repro-0.1.0/src/paper_repro/inspect_repo.py +504 -0
  48. paper_repro-0.1.0/src/paper_repro/mcp_server.py +277 -0
  49. paper_repro-0.1.0/src/paper_repro/metrics.py +345 -0
  50. paper_repro-0.1.0/src/paper_repro/report.py +557 -0
  51. paper_repro-0.1.0/src/paper_repro/runner.py +329 -0
  52. paper_repro-0.1.0/src/paper_repro/study.py +213 -0
  53. paper_repro-0.1.0/src/paper_repro/util.py +297 -0
  54. paper_repro-0.1.0/tests/conftest.py +58 -0
  55. paper_repro-0.1.0/tests/fixtures.py +75 -0
  56. paper_repro-0.1.0/tests/test_claims.py +61 -0
  57. paper_repro-0.1.0/tests/test_compare.py +89 -0
  58. paper_repro-0.1.0/tests/test_flow.py +163 -0
  59. paper_repro-0.1.0/tests/test_inspect.py +78 -0
  60. paper_repro-0.1.0/tests/test_mcp.py +90 -0
  61. paper_repro-0.1.0/tests/test_metrics.py +75 -0
  62. paper_repro-0.1.0/tests/test_runner.py +87 -0
  63. paper_repro-0.1.0/tests/test_setup.py +81 -0
  64. paper_repro-0.1.0/uv.lock +906 -0
@@ -0,0 +1,31 @@
1
+ name: CI
2
+
3
+ on:
4
+ push:
5
+ pull_request:
6
+
7
+ jobs:
8
+ test:
9
+ runs-on: ${{ matrix.os }}
10
+ strategy:
11
+ fail-fast: false
12
+ matrix:
13
+ os: [ubuntu-latest, macos-latest]
14
+ python: ["3.11", "3.12", "3.13"]
15
+ steps:
16
+ - uses: actions/checkout@v4
17
+ - uses: astral-sh/setup-uv@v6
18
+ with:
19
+ python-version: ${{ matrix.python }}
20
+ - name: Install
21
+ run: uv sync --python ${{ matrix.python }}
22
+ - name: Lint
23
+ run: |
24
+ uv run ruff check src tests
25
+ uv run ruff format --check src tests
26
+ - name: Test
27
+ run: uv run pytest -q
28
+ - name: CLI smoke
29
+ run: |
30
+ uv run paper-repro --version
31
+ uv run paper-repro --help
@@ -0,0 +1,34 @@
1
+ name: release
2
+
3
+ # Publishes to PyPI when a version tag (v*) is pushed, using PyPI trusted publishing (no API token).
4
+ on:
5
+ push:
6
+ tags: ["v*"]
7
+
8
+ permissions:
9
+ contents: read
10
+
11
+ jobs:
12
+ build:
13
+ runs-on: ubuntu-latest
14
+ steps:
15
+ - uses: actions/checkout@v4
16
+ - uses: astral-sh/setup-uv@v6
17
+ - run: uv build
18
+ - uses: actions/upload-artifact@v4
19
+ with:
20
+ name: dist
21
+ path: dist/
22
+
23
+ publish:
24
+ needs: build
25
+ runs-on: ubuntu-latest
26
+ environment: pypi
27
+ permissions:
28
+ id-token: write
29
+ steps:
30
+ - uses: actions/download-artifact@v4
31
+ with:
32
+ name: dist
33
+ path: dist/
34
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,13 @@
1
+ __pycache__/
2
+ *.py[cod]
3
+ .venv/
4
+ build/
5
+ dist/
6
+ *.egg-info/
7
+ .pytest_cache/
8
+ .ruff_cache/
9
+ .mypy_cache/
10
+ paper-repro-runs/
11
+ .DS_Store
12
+ CLAUDE.md
13
+ .claude/
@@ -0,0 +1,33 @@
1
+ # Notes for contributors and coding agents
2
+
3
+ paper-repro is a deterministic recorder. The host agent decides what to run; this package
4
+ runs it, captures it, and writes reports. It never calls an LLM.
5
+
6
+ ## Layout
7
+
8
+ ```
9
+ src/paper_repro/
10
+ study.py Study directory, append-only hash-chained evidence.jsonl, state replay
11
+ inspect_repo.py clone/copy, dependency and entry-point detection, README scan
12
+ claims.py claimed numbers from Markdown (tables, sentences, ranges, ±)
13
+ metrics.py metric values from logs, JSON, CSV; selectors (m1:acc:last@each)
14
+ envs.py uv venv + installs, conda translation, unpin, lock.txt
15
+ runner.py run one shell command with limits, capture outputs and written files
16
+ compare.py verdict logic (tolerance, rounding, ranges, scale, scope)
17
+ report.py report.json and report.md, built only from the log
18
+ agent_setup.py `setup` for Claude Code, Codex, Cursor
19
+ mcp_server.py MCP tools over stdio, thin wrappers
20
+ cli.py argparse CLI, --json everywhere
21
+ skills/paper-repro/SKILL.md the workflow the agent follows (shipped in the wheel)
22
+ examples/ real evidence bundles from real runs
23
+ ```
24
+
25
+ ## Invariants
26
+
27
+ - `evidence.jsonl` is append-only. Every derived view (status, reports) replays it.
28
+ Never mutate an entry; add a new one.
29
+ - A shortened or smoke run can never yield `counts_as_reproduction: true`.
30
+ - Values typed in by hand are flagged `unsourced` and can never count as a reproduction.
31
+ - `setup` must show its plan, require `--yes` or an interactive yes, back up what it edits,
32
+ and be idempotent. Tests run it against a temp HOME only.
33
+ - No em dashes or en dashes in code, docs or messages.
@@ -0,0 +1,25 @@
1
+ # Changelog
2
+
3
+ ## 0.1.0 (unreleased)
4
+
5
+ First version.
6
+
7
+ - `inspect`: shallow clone (or copy of a local path), detection of languages, dependency
8
+ files, Python version hints, README commands and `pip install` lines, entry points,
9
+ data and weight downloads, GPU hints, and claimed numbers (tables, sentences, ranges,
10
+ `±` values).
11
+ - `env`: uv virtualenv per study, requirements and conda files (conda translated to pip),
12
+ `--from-readme`, `--unpin` and `--extra` recorded as deviations, `lock.txt` from
13
+ `uv pip freeze`, exact error on failure.
14
+ - `run`: shell command in the repo with the study env active, timeout that kills the
15
+ process tree, CPU-time limit, peak memory, stdout/stderr to files with SHA-256, files
16
+ written, and copies of small result files taken when the run ends.
17
+ - `metrics`: `name: value`, `name=value%`, `name value`, `n/d` fractions, JSON, JSON lines,
18
+ CSV/TSV, each with its source line; selectors such as `m1:val_loss:last@each`.
19
+ - `compare`: verdicts reproduced, close, not reproduced, could not run; percent/fraction
20
+ rescaling, rounding-aware tolerance, ranges, repeated runs, shortened runs never count.
21
+ - `report`: `report.md` and `report.json` with commit, environment, commands, claim,
22
+ measured values with sources, reasoning, deviations and what was not checked.
23
+ - Hash-chained `evidence.jsonl` and `verify`.
24
+ - MCP server (`paper-repro mcp`), `setup` for Claude Code, Codex and Cursor, and the
25
+ `paper-repro` agent skill.
@@ -0,0 +1,15 @@
1
+ # Contributing
2
+
3
+ ```bash
4
+ uv sync
5
+ uv run pytest
6
+ uv run ruff check src tests && uv run ruff format --check src tests
7
+ ```
8
+
9
+ Tests must stay offline and fast. They build small fixture repos in temp directories and
10
+ run real processes; they never touch your home directory (HOME is redirected) and keep uv
11
+ offline. If you add a metric format or a claim pattern, add a test with the exact line it
12
+ should match and one it should not.
13
+
14
+ Two rules for changes: the tool must not call any model API, and nothing it writes to a
15
+ report may come from anywhere but the evidence log. See `AGENTS.md` for the layout.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Abel Yagubyan
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,194 @@
1
+ Metadata-Version: 2.5
2
+ Name: paper-repro
3
+ Version: 0.1.0
4
+ Summary: Point your agent at a paper's code; it gets it running and tells you whether the headline number reproduces, with the evidence.
5
+ Project-URL: Homepage, https://github.com/Abelo9996/paper-repro
6
+ Author: Abel Yagubyan
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Keywords: agents,machine-learning,mcp,reproducibility,research
10
+ Classifier: Development Status :: 3 - Alpha
11
+ Classifier: Intended Audience :: Science/Research
12
+ Classifier: License :: OSI Approved :: MIT License
13
+ Classifier: Programming Language :: Python :: 3
14
+ Classifier: Programming Language :: Python :: 3.11
15
+ Classifier: Programming Language :: Python :: 3.12
16
+ Classifier: Programming Language :: Python :: 3.13
17
+ Classifier: Topic :: Scientific/Engineering
18
+ Requires-Python: >=3.11
19
+ Requires-Dist: mcp>=1.2
20
+ Requires-Dist: pyyaml>=6
21
+ Description-Content-Type: text/markdown
22
+
23
+ # paper-repro
24
+
25
+ Point your agent at a paper's code; it gets it running and tells you whether the headline number reproduces, with the evidence.
26
+
27
+ ```bash
28
+ uvx paper-repro setup --yes # register the MCP server and skill with Claude Code, Codex, Cursor
29
+ # then ask your agent: "Does github.com/karpathy/nanoGPT reproduce its CPU val loss of 1.88?"
30
+ uvx paper-repro inspect https://github.com/karpathy/nanoGPT # or drive it yourself from the CLI
31
+ ```
32
+
33
+ > Not on PyPI yet. Until the first release, run it straight from GitHub by replacing `uvx paper-repro` with
34
+ > `uvx --from git+https://github.com/Abelo9996/paper-repro paper-repro`. `setup` registers `uvx paper-repro mcp`, so it works once the
35
+ > package is on PyPI.
36
+
37
+ Papers with Code went offline on 24 July 2025, taking its 79,817 paper-to-code links with it
38
+ ([shutdown record](https://www.codesota.com/papers-with-code/shutdown)). There's no common place
39
+ that tracks whether a paper's released code actually reproduces the numbers in the paper. Research benchmarks for this exist (CORE-Bench, PaperBench), but they score agents,
40
+ not papers. paper-repro is the user-facing half: it gives a coding agent reliable operations to
41
+ clone a repo, build its environment, run it, pull numbers out of the logs and compare them with
42
+ the claim, and it writes everything down so a person can check the verdict without trusting
43
+ the agent.
44
+
45
+ ## Example output
46
+
47
+ A real run on this machine (Apple M4, macOS, CPU only) against
48
+ [karpathy/nanoGPT](https://github.com/karpathy/nanoGPT) at commit `3adf61e`. Its README says the
49
+ small CPU configuration of the Shakespeare character model "gets us a loss of only 1.88". The full
50
+ evidence bundle is in [examples/nanogpt-shakespeare-char-cpu](examples/nanogpt-shakespeare-char-cpu/report.md).
51
+
52
+ ```text
53
+ $ paper-repro inspect https://github.com/karpathy/nanoGPT
54
+ claimed numbers (12):
55
+ c1: best validation loss = 1.4697 (README.md:51)
56
+ c2: loss = 1.88 (README.md:88)
57
+ c3: loss = 2.85 (README.md:121)
58
+ ...
59
+ $ paper-repro env --python 3.12 --from-readme
60
+ $ paper-repro run --scope setup -- python data/shakespeare_char/prepare.py
61
+ $ paper-repro run -- "python train.py config/train_shakespeare_char.py --device=cpu --compile=False --eval_iters=20 ..."
62
+ r3: exit 0 in 236.5s, peak memory 194.6 MB, 1 files written
63
+ $ paper-repro metrics --run r3 --run r4 --name val_loss
64
+ $ paper-repro compare --claim c2 --measured 'm1:val_loss:last@each' --why '...'
65
+ k1: REPRODUCED
66
+ Reproduced: loss claimed 1.88, measured 1.8857 (mean of 2 runs).
67
+ - Claim: loss 1.88 (README.md line 88).
68
+ - Measured over 2 runs: mean 1.8857, std 0, min 1.8857, max 1.8857.
69
+ - Distance from the claim: 0.0057 (0.3% of the claim).
70
+ - The measured value is worse than claimed (higher, and lower is better for this metric).
71
+ - Tolerance: ±0.0188 (1% of the claimed value, default). Close band: ±0.0564.
72
+ ```
73
+
74
+ The report also lists what was not checked: the A100 claim (1.4697), the GPT-2 numbers, the
75
+ checksums of downloaded data, and that the seed is hard-coded, so the two runs measure
76
+ determinism (identical losses at every eval step) rather than seed variance.
77
+
78
+ ## How it works
79
+
80
+ paper-repro is a recorder, not a brain. It never calls a model. Your agent (Claude Code, Codex,
81
+ Cursor) decides what to run; paper-repro does it the same way every time and keeps the receipts.
82
+
83
+ | Step | What it does | Tools it drives |
84
+ |---|---|---|
85
+ | `inspect` | Shallow-clones the repo (or copies a local path), records the commit SHA, finds dependency files, Python version hints, README commands and `pip install` lines, entry points and their flags, data and weight downloads, GPU hints, and every number the README claims (table cells, sentences, ranges, `±` values), each with file and line. | `git` |
86
+ | `env` | Creates an isolated virtualenv per study and installs the repo's own dependency file with pins as written. Conda `environment.yml` files are translated to pip. `--unpin`, `--extra`, `--from-readme` and a different Python are allowed but recorded as deviations. Writes `lock.txt` from `uv pip freeze`. On failure, records the exact error. | `uv venv`, `uv pip` |
87
+ | `run` | Runs one shell command from the repo root inside that env. Captures stdout and stderr to files with SHA-256, exit code, wall time, CPU time, peak memory, every file created or modified, and a copy of small result files as they were when the run ended. Timeout kills the whole process tree. Each run is labeled `full`, `shortened` (needs a note saying what was cut), `smoke` or `setup`. | your shell, POSIX rlimits |
88
+ | `metrics` | Extracts values like `accuracy: 0.913`, `acc=91.3%`, `F1 81.2`, `val loss 1.8857`, `Accuracy: 9897/10000`, plus JSON, JSON lines and CSV results, each with its source file, line and text. Select one with `m1:val_loss:last`, or one per run with `m1:val_loss:last@each`. | |
89
+ | `compare` | Puts measured values next to the claim and gives one of four verdicts: **reproduced**, **close**, **not reproduced**, **could not run**. Handles percent versus fraction, ranges, `±`, repeated runs (mean, std, min, max) and the rounding of the stated value. Default tolerance is 1% relative, close band 3x that; pass your own with a reason. A shortened run or a hand-typed value never counts as a reproduction. | |
90
+ | `report` | Writes `report.md` (the shareable artifact) and `report.json` from the evidence log only: repo URL and commit, machine, Python and key package versions, every command with exit code and duration, the claim, measured values with source lines, the reasoning, deviations, and what was not checked. | |
91
+
92
+ Every step appends to `evidence.jsonl`, a hash-chained log. `paper-repro verify` rechecks the
93
+ chain and the hashes of captured outputs, so an edited log or a swapped stdout file shows up.
94
+
95
+ A study directory looks like this:
96
+
97
+ ```
98
+ paper-repro-runs/karpathy-nanoGPT/
99
+ repo/ the checkout
100
+ env/ the virtualenv
101
+ runs/r3/ stdout.txt, stderr.txt, files/ (results as the run left them)
102
+ lock.txt resolved packages
103
+ evidence.jsonl the log everything else is derived from
104
+ report.md the verdict and evidence, for people
105
+ report.json the same, for programs
106
+ ```
107
+
108
+ `paper-repro report --bundle DIR` copies the shareable parts (no checkout, no env) to `DIR`.
109
+
110
+ ### CLI
111
+
112
+ The second example, [examples/pygat-cora](examples/pygat-cora/report.md), was produced with
113
+ these commands against [Diego999/pyGAT](https://github.com/Diego999/pyGAT), whose README says
114
+ "The final accuracy is between 84.2 and 85.3 (obtained on 5 different runs)":
115
+
116
+ ```bash
117
+ paper-repro inspect Diego999/pyGAT
118
+ paper-repro env # fails: README asks for Python 3.5, which uv does not support
119
+ paper-repro env --python 3.12 # fails: torch==0.4.1.post2 has no wheel for this machine
120
+ paper-repro env --python 3.12 --unpin
121
+ paper-repro run --scope smoke --note "2 epochs ..." -- python train.py --epochs 2
122
+ paper-repro run --seed 72 --timeout 10800 --note "..." -- "rm -f *.pkl && python train.py --seed 72"
123
+ paper-repro run --seed 1 ... && paper-repro run --seed 2 ...
124
+ paper-repro metrics --run r2 --run r3 --run r4 --name accuracy
125
+ paper-repro compare --claim c1 --measured 'm1:accuracy:last@each' --tol 0.1 --close-tol 1.0 \
126
+ --why "Cora's test split has 1000 nodes, so accuracy moves in steps of 0.1 points; ..."
127
+ paper-repro note --kind not_checked "The README's range comes from 5 runs; 3 seeds were run here ..."
128
+ paper-repro report --bundle examples/pygat-cora
129
+ ```
130
+
131
+ Result: test accuracy 84.2, 84.7 and 84.2 over three seeds, mean 84.37, inside the claimed range,
132
+ verdict reproduced. Both failed environment attempts, the Python and unpinning deviations, and
133
+ the fact that only 3 of the authors' 5 runs were repeated are all in the report.
134
+
135
+ Every subcommand takes `--json`. `paper-repro status` shows what a study has recorded so far.
136
+
137
+ ## Setup for agents
138
+
139
+ ```bash
140
+ uvx paper-repro setup # shows what it would change, asks before applying
141
+ uvx paper-repro setup --yes # apply without asking
142
+ ```
143
+
144
+ It detects each agent and, for the ones present:
145
+
146
+ - Claude Code: `claude mcp add --scope user paper-repro -- uvx paper-repro mcp`, and copies the
147
+ skill to `~/.claude/skills/paper-repro/SKILL.md`. Use `--project DIR` to write a project
148
+ `.mcp.json` instead.
149
+ - Codex: adds `[mcp_servers.paper-repro]` to `~/.codex/config.toml` and copies the skill to
150
+ `~/.codex/skills/paper-repro/`.
151
+ - Cursor: adds `paper-repro` to `mcpServers` in `~/.cursor/mcp.json`.
152
+
153
+ It backs up any file it edits (`*.bak-paper-repro-<time>`) and does nothing on a second run.
154
+ Manual registration is the same command everywhere: `uvx paper-repro mcp` over stdio.
155
+
156
+ The skill ([skills/paper-repro/SKILL.md](skills/paper-repro/SKILL.md)) is the workflow the agent
157
+ follows: inspect, pick one claim and say why, check feasibility, set up the env with the
158
+ least invasive fix, run the smallest faithful configuration first, extract, compare with a
159
+ tolerance it can defend, report. It also sets the rules: never fabricate or estimate a number,
160
+ never change the method to make a run succeed, and report "could not run" with the blocking
161
+ error instead of pushing through.
162
+
163
+ ## What it can't do
164
+
165
+ - It does not judge whether the code implements the method in the paper. It checks the code's
166
+ output against a stated number, nothing more.
167
+ - No GPU orchestration. On a machine without the GPU the paper used, GPU-only claims end as
168
+ "could not run" or as shortened CPU runs that never count as reproductions.
169
+ - No container builds yet. A Dockerfile is detected and reported but not built; environments
170
+ are Python virtualenvs. R, Julia and other languages are detected but `env` only builds
171
+ Python environments (you can still `run` commands that use a system R or Julia).
172
+ - Conda environments are translated to pip, which can resolve different builds than conda
173
+ would. The translation is recorded as a deviation.
174
+ - Claim detection reads the README (and other Markdown in the repo), not the paper PDF. Numbers
175
+ that only appear in the paper must be added with `paper-repro claim --source "paper Table 2"`.
176
+ - Metric extraction is pattern-based. Unusual log formats may need `--file` on a results file or
177
+ a careful choice among the extracted values; the report always shows the exact source line.
178
+ - Memory limits are only enforced on Linux; on macOS peak memory is recorded but not capped.
179
+ - The default tolerance (1% relative) is a convention, not a statistical test. Choose one per
180
+ claim and say why.
181
+
182
+ ## Privacy and safety
183
+
184
+ Everything runs on your machine. paper-repro sends nothing anywhere and has no telemetry. The
185
+ only network traffic is what you ask for: `git clone`, `uv` downloading packages, and whatever
186
+ the repository's own scripts download.
187
+
188
+ That last part matters: running a paper's code means running a stranger's code with your
189
+ user's permissions. paper-repro isolates the Python environment and can cap time and CPU, but
190
+ it is not a sandbox. For code you do not trust, run it inside a VM or a disposable container.
191
+
192
+ ## License
193
+
194
+ MIT, see [LICENSE](LICENSE).
@@ -0,0 +1,172 @@
1
+ # paper-repro
2
+
3
+ Point your agent at a paper's code; it gets it running and tells you whether the headline number reproduces, with the evidence.
4
+
5
+ ```bash
6
+ uvx paper-repro setup --yes # register the MCP server and skill with Claude Code, Codex, Cursor
7
+ # then ask your agent: "Does github.com/karpathy/nanoGPT reproduce its CPU val loss of 1.88?"
8
+ uvx paper-repro inspect https://github.com/karpathy/nanoGPT # or drive it yourself from the CLI
9
+ ```
10
+
11
+ > Not on PyPI yet. Until the first release, run it straight from GitHub by replacing `uvx paper-repro` with
12
+ > `uvx --from git+https://github.com/Abelo9996/paper-repro paper-repro`. `setup` registers `uvx paper-repro mcp`, so it works once the
13
+ > package is on PyPI.
14
+
15
+ Papers with Code went offline on 24 July 2025, taking its 79,817 paper-to-code links with it
16
+ ([shutdown record](https://www.codesota.com/papers-with-code/shutdown)). There's no common place
17
+ that tracks whether a paper's released code actually reproduces the numbers in the paper. Research benchmarks for this exist (CORE-Bench, PaperBench), but they score agents,
18
+ not papers. paper-repro is the user-facing half: it gives a coding agent reliable operations to
19
+ clone a repo, build its environment, run it, pull numbers out of the logs and compare them with
20
+ the claim, and it writes everything down so a person can check the verdict without trusting
21
+ the agent.
22
+
23
+ ## Example output
24
+
25
+ A real run on this machine (Apple M4, macOS, CPU only) against
26
+ [karpathy/nanoGPT](https://github.com/karpathy/nanoGPT) at commit `3adf61e`. Its README says the
27
+ small CPU configuration of the Shakespeare character model "gets us a loss of only 1.88". The full
28
+ evidence bundle is in [examples/nanogpt-shakespeare-char-cpu](examples/nanogpt-shakespeare-char-cpu/report.md).
29
+
30
+ ```text
31
+ $ paper-repro inspect https://github.com/karpathy/nanoGPT
32
+ claimed numbers (12):
33
+ c1: best validation loss = 1.4697 (README.md:51)
34
+ c2: loss = 1.88 (README.md:88)
35
+ c3: loss = 2.85 (README.md:121)
36
+ ...
37
+ $ paper-repro env --python 3.12 --from-readme
38
+ $ paper-repro run --scope setup -- python data/shakespeare_char/prepare.py
39
+ $ paper-repro run -- "python train.py config/train_shakespeare_char.py --device=cpu --compile=False --eval_iters=20 ..."
40
+ r3: exit 0 in 236.5s, peak memory 194.6 MB, 1 files written
41
+ $ paper-repro metrics --run r3 --run r4 --name val_loss
42
+ $ paper-repro compare --claim c2 --measured 'm1:val_loss:last@each' --why '...'
43
+ k1: REPRODUCED
44
+ Reproduced: loss claimed 1.88, measured 1.8857 (mean of 2 runs).
45
+ - Claim: loss 1.88 (README.md line 88).
46
+ - Measured over 2 runs: mean 1.8857, std 0, min 1.8857, max 1.8857.
47
+ - Distance from the claim: 0.0057 (0.3% of the claim).
48
+ - The measured value is worse than claimed (higher, and lower is better for this metric).
49
+ - Tolerance: ±0.0188 (1% of the claimed value, default). Close band: ±0.0564.
50
+ ```
51
+
52
+ The report also lists what was not checked: the A100 claim (1.4697), the GPT-2 numbers, the
53
+ checksums of downloaded data, and that the seed is hard-coded, so the two runs measure
54
+ determinism (identical losses at every eval step) rather than seed variance.
55
+
56
+ ## How it works
57
+
58
+ paper-repro is a recorder, not a brain. It never calls a model. Your agent (Claude Code, Codex,
59
+ Cursor) decides what to run; paper-repro does it the same way every time and keeps the receipts.
60
+
61
+ | Step | What it does | Tools it drives |
62
+ |---|---|---|
63
+ | `inspect` | Shallow-clones the repo (or copies a local path), records the commit SHA, finds dependency files, Python version hints, README commands and `pip install` lines, entry points and their flags, data and weight downloads, GPU hints, and every number the README claims (table cells, sentences, ranges, `±` values), each with file and line. | `git` |
64
+ | `env` | Creates an isolated virtualenv per study and installs the repo's own dependency file with pins as written. Conda `environment.yml` files are translated to pip. `--unpin`, `--extra`, `--from-readme` and a different Python are allowed but recorded as deviations. Writes `lock.txt` from `uv pip freeze`. On failure, records the exact error. | `uv venv`, `uv pip` |
65
+ | `run` | Runs one shell command from the repo root inside that env. Captures stdout and stderr to files with SHA-256, exit code, wall time, CPU time, peak memory, every file created or modified, and a copy of small result files as they were when the run ended. Timeout kills the whole process tree. Each run is labeled `full`, `shortened` (needs a note saying what was cut), `smoke` or `setup`. | your shell, POSIX rlimits |
66
+ | `metrics` | Extracts values like `accuracy: 0.913`, `acc=91.3%`, `F1 81.2`, `val loss 1.8857`, `Accuracy: 9897/10000`, plus JSON, JSON lines and CSV results, each with its source file, line and text. Select one with `m1:val_loss:last`, or one per run with `m1:val_loss:last@each`. | |
67
+ | `compare` | Puts measured values next to the claim and gives one of four verdicts: **reproduced**, **close**, **not reproduced**, **could not run**. Handles percent versus fraction, ranges, `±`, repeated runs (mean, std, min, max) and the rounding of the stated value. Default tolerance is 1% relative, close band 3x that; pass your own with a reason. A shortened run or a hand-typed value never counts as a reproduction. | |
68
+ | `report` | Writes `report.md` (the shareable artifact) and `report.json` from the evidence log only: repo URL and commit, machine, Python and key package versions, every command with exit code and duration, the claim, measured values with source lines, the reasoning, deviations, and what was not checked. | |
69
+
70
+ Every step appends to `evidence.jsonl`, a hash-chained log. `paper-repro verify` rechecks the
71
+ chain and the hashes of captured outputs, so an edited log or a swapped stdout file shows up.
72
+
73
+ A study directory looks like this:
74
+
75
+ ```
76
+ paper-repro-runs/karpathy-nanoGPT/
77
+ repo/ the checkout
78
+ env/ the virtualenv
79
+ runs/r3/ stdout.txt, stderr.txt, files/ (results as the run left them)
80
+ lock.txt resolved packages
81
+ evidence.jsonl the log everything else is derived from
82
+ report.md the verdict and evidence, for people
83
+ report.json the same, for programs
84
+ ```
85
+
86
+ `paper-repro report --bundle DIR` copies the shareable parts (no checkout, no env) to `DIR`.
87
+
88
+ ### CLI
89
+
90
+ The second example, [examples/pygat-cora](examples/pygat-cora/report.md), was produced with
91
+ these commands against [Diego999/pyGAT](https://github.com/Diego999/pyGAT), whose README says
92
+ "The final accuracy is between 84.2 and 85.3 (obtained on 5 different runs)":
93
+
94
+ ```bash
95
+ paper-repro inspect Diego999/pyGAT
96
+ paper-repro env # fails: README asks for Python 3.5, which uv does not support
97
+ paper-repro env --python 3.12 # fails: torch==0.4.1.post2 has no wheel for this machine
98
+ paper-repro env --python 3.12 --unpin
99
+ paper-repro run --scope smoke --note "2 epochs ..." -- python train.py --epochs 2
100
+ paper-repro run --seed 72 --timeout 10800 --note "..." -- "rm -f *.pkl && python train.py --seed 72"
101
+ paper-repro run --seed 1 ... && paper-repro run --seed 2 ...
102
+ paper-repro metrics --run r2 --run r3 --run r4 --name accuracy
103
+ paper-repro compare --claim c1 --measured 'm1:accuracy:last@each' --tol 0.1 --close-tol 1.0 \
104
+ --why "Cora's test split has 1000 nodes, so accuracy moves in steps of 0.1 points; ..."
105
+ paper-repro note --kind not_checked "The README's range comes from 5 runs; 3 seeds were run here ..."
106
+ paper-repro report --bundle examples/pygat-cora
107
+ ```
108
+
109
+ Result: test accuracy 84.2, 84.7 and 84.2 over three seeds, mean 84.37, inside the claimed range,
110
+ verdict reproduced. Both failed environment attempts, the Python and unpinning deviations, and
111
+ the fact that only 3 of the authors' 5 runs were repeated are all in the report.
112
+
113
+ Every subcommand takes `--json`. `paper-repro status` shows what a study has recorded so far.
114
+
115
+ ## Setup for agents
116
+
117
+ ```bash
118
+ uvx paper-repro setup # shows what it would change, asks before applying
119
+ uvx paper-repro setup --yes # apply without asking
120
+ ```
121
+
122
+ It detects each agent and, for the ones present:
123
+
124
+ - Claude Code: `claude mcp add --scope user paper-repro -- uvx paper-repro mcp`, and copies the
125
+ skill to `~/.claude/skills/paper-repro/SKILL.md`. Use `--project DIR` to write a project
126
+ `.mcp.json` instead.
127
+ - Codex: adds `[mcp_servers.paper-repro]` to `~/.codex/config.toml` and copies the skill to
128
+ `~/.codex/skills/paper-repro/`.
129
+ - Cursor: adds `paper-repro` to `mcpServers` in `~/.cursor/mcp.json`.
130
+
131
+ It backs up any file it edits (`*.bak-paper-repro-<time>`) and does nothing on a second run.
132
+ Manual registration is the same command everywhere: `uvx paper-repro mcp` over stdio.
133
+
134
+ The skill ([skills/paper-repro/SKILL.md](skills/paper-repro/SKILL.md)) is the workflow the agent
135
+ follows: inspect, pick one claim and say why, check feasibility, set up the env with the
136
+ least invasive fix, run the smallest faithful configuration first, extract, compare with a
137
+ tolerance it can defend, report. It also sets the rules: never fabricate or estimate a number,
138
+ never change the method to make a run succeed, and report "could not run" with the blocking
139
+ error instead of pushing through.
140
+
141
+ ## What it can't do
142
+
143
+ - It does not judge whether the code implements the method in the paper. It checks the code's
144
+ output against a stated number, nothing more.
145
+ - No GPU orchestration. On a machine without the GPU the paper used, GPU-only claims end as
146
+ "could not run" or as shortened CPU runs that never count as reproductions.
147
+ - No container builds yet. A Dockerfile is detected and reported but not built; environments
148
+ are Python virtualenvs. R, Julia and other languages are detected but `env` only builds
149
+ Python environments (you can still `run` commands that use a system R or Julia).
150
+ - Conda environments are translated to pip, which can resolve different builds than conda
151
+ would. The translation is recorded as a deviation.
152
+ - Claim detection reads the README (and other Markdown in the repo), not the paper PDF. Numbers
153
+ that only appear in the paper must be added with `paper-repro claim --source "paper Table 2"`.
154
+ - Metric extraction is pattern-based. Unusual log formats may need `--file` on a results file or
155
+ a careful choice among the extracted values; the report always shows the exact source line.
156
+ - Memory limits are only enforced on Linux; on macOS peak memory is recorded but not capped.
157
+ - The default tolerance (1% relative) is a convention, not a statistical test. Choose one per
158
+ claim and say why.
159
+
160
+ ## Privacy and safety
161
+
162
+ Everything runs on your machine. paper-repro sends nothing anywhere and has no telemetry. The
163
+ only network traffic is what you ask for: `git clone`, `uv` downloading packages, and whatever
164
+ the repository's own scripts download.
165
+
166
+ That last part matters: running a paper's code means running a stranger's code with your
167
+ user's permissions. paper-repro isolates the Python environment and can cap time and CPU, but
168
+ it is not a sandbox. For code you do not trust, run it inside a VM or a disposable container.
169
+
170
+ ## License
171
+
172
+ MIT, see [LICENSE](LICENSE).
@@ -0,0 +1,9 @@
1
+ # Security
2
+
3
+ paper-repro clones and runs other people's code on your machine. That is the point of the
4
+ tool, and it is also the main risk: a repository's training script can do anything your user
5
+ account can do. The tool does not sandbox it beyond a separate virtualenv, a timeout and
6
+ optional CPU-time limits. Run untrusted repositories in a VM or container you can throw away.
7
+
8
+ To report a vulnerability in paper-repro itself, open a private security advisory on the
9
+ GitHub repository or email the maintainer. Please do not file a public issue first.