paper-repro 0.1.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- paper_repro-0.1.0/.github/workflows/ci.yml +31 -0
- paper_repro-0.1.0/.github/workflows/release.yml +34 -0
- paper_repro-0.1.0/.gitignore +13 -0
- paper_repro-0.1.0/AGENTS.md +33 -0
- paper_repro-0.1.0/CHANGELOG.md +25 -0
- paper_repro-0.1.0/CONTRIBUTING.md +15 -0
- paper_repro-0.1.0/LICENSE +21 -0
- paper_repro-0.1.0/PKG-INFO +194 -0
- paper_repro-0.1.0/README.md +172 -0
- paper_repro-0.1.0/SECURITY.md +9 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/evidence.jsonl +13 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/inspect.json +613 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/lock.txt +62 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/report.json +264 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/report.md +78 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r1/stderr.txt +0 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r1/stdout.txt +6 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r2/stderr.txt +4 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r2/stdout.txt +70 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r3/stderr.txt +4 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r3/stdout.txt +2076 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r4/stderr.txt +4 -0
- paper_repro-0.1.0/examples/nanogpt-shakespeare-char-cpu/runs/r4/stdout.txt +2076 -0
- paper_repro-0.1.0/examples/pygat-cora/env-inputs/e3-unpinned-requirements.txt +3 -0
- paper_repro-0.1.0/examples/pygat-cora/evidence.jsonl +14 -0
- paper_repro-0.1.0/examples/pygat-cora/inspect.json +175 -0
- paper_repro-0.1.0/examples/pygat-cora/lock.txt +12 -0
- paper_repro-0.1.0/examples/pygat-cora/report.json +290 -0
- paper_repro-0.1.0/examples/pygat-cora/report.md +86 -0
- paper_repro-0.1.0/examples/pygat-cora/runs/r1/stderr.txt +0 -0
- paper_repro-0.1.0/examples/pygat-cora/runs/r1/stdout.txt +7 -0
- paper_repro-0.1.0/examples/pygat-cora/runs/r2/stderr.txt +0 -0
- paper_repro-0.1.0/examples/pygat-cora/runs/r2/stdout.txt +840 -0
- paper_repro-0.1.0/examples/pygat-cora/runs/r3/stderr.txt +0 -0
- paper_repro-0.1.0/examples/pygat-cora/runs/r3/stdout.txt +816 -0
- paper_repro-0.1.0/examples/pygat-cora/runs/r4/stderr.txt +0 -0
- paper_repro-0.1.0/examples/pygat-cora/runs/r4/stdout.txt +834 -0
- paper_repro-0.1.0/pyproject.toml +54 -0
- paper_repro-0.1.0/skills/paper-repro/SKILL.md +76 -0
- paper_repro-0.1.0/src/paper_repro/__init__.py +3 -0
- paper_repro-0.1.0/src/paper_repro/__main__.py +3 -0
- paper_repro-0.1.0/src/paper_repro/agent_setup.py +277 -0
- paper_repro-0.1.0/src/paper_repro/claims.py +237 -0
- paper_repro-0.1.0/src/paper_repro/cli.py +541 -0
- paper_repro-0.1.0/src/paper_repro/compare.py +427 -0
- paper_repro-0.1.0/src/paper_repro/envs.py +323 -0
- paper_repro-0.1.0/src/paper_repro/inspect_repo.py +504 -0
- paper_repro-0.1.0/src/paper_repro/mcp_server.py +277 -0
- paper_repro-0.1.0/src/paper_repro/metrics.py +345 -0
- paper_repro-0.1.0/src/paper_repro/report.py +557 -0
- paper_repro-0.1.0/src/paper_repro/runner.py +329 -0
- paper_repro-0.1.0/src/paper_repro/study.py +213 -0
- paper_repro-0.1.0/src/paper_repro/util.py +297 -0
- paper_repro-0.1.0/tests/conftest.py +58 -0
- paper_repro-0.1.0/tests/fixtures.py +75 -0
- paper_repro-0.1.0/tests/test_claims.py +61 -0
- paper_repro-0.1.0/tests/test_compare.py +89 -0
- paper_repro-0.1.0/tests/test_flow.py +163 -0
- paper_repro-0.1.0/tests/test_inspect.py +78 -0
- paper_repro-0.1.0/tests/test_mcp.py +90 -0
- paper_repro-0.1.0/tests/test_metrics.py +75 -0
- paper_repro-0.1.0/tests/test_runner.py +87 -0
- paper_repro-0.1.0/tests/test_setup.py +81 -0
- paper_repro-0.1.0/uv.lock +906 -0
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
name: CI
|
|
2
|
+
|
|
3
|
+
on:
|
|
4
|
+
push:
|
|
5
|
+
pull_request:
|
|
6
|
+
|
|
7
|
+
jobs:
|
|
8
|
+
test:
|
|
9
|
+
runs-on: ${{ matrix.os }}
|
|
10
|
+
strategy:
|
|
11
|
+
fail-fast: false
|
|
12
|
+
matrix:
|
|
13
|
+
os: [ubuntu-latest, macos-latest]
|
|
14
|
+
python: ["3.11", "3.12", "3.13"]
|
|
15
|
+
steps:
|
|
16
|
+
- uses: actions/checkout@v4
|
|
17
|
+
- uses: astral-sh/setup-uv@v6
|
|
18
|
+
with:
|
|
19
|
+
python-version: ${{ matrix.python }}
|
|
20
|
+
- name: Install
|
|
21
|
+
run: uv sync --python ${{ matrix.python }}
|
|
22
|
+
- name: Lint
|
|
23
|
+
run: |
|
|
24
|
+
uv run ruff check src tests
|
|
25
|
+
uv run ruff format --check src tests
|
|
26
|
+
- name: Test
|
|
27
|
+
run: uv run pytest -q
|
|
28
|
+
- name: CLI smoke
|
|
29
|
+
run: |
|
|
30
|
+
uv run paper-repro --version
|
|
31
|
+
uv run paper-repro --help
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
name: release
|
|
2
|
+
|
|
3
|
+
# Publishes to PyPI when a version tag (v*) is pushed, using PyPI trusted publishing (no API token).
|
|
4
|
+
on:
|
|
5
|
+
push:
|
|
6
|
+
tags: ["v*"]
|
|
7
|
+
|
|
8
|
+
permissions:
|
|
9
|
+
contents: read
|
|
10
|
+
|
|
11
|
+
jobs:
|
|
12
|
+
build:
|
|
13
|
+
runs-on: ubuntu-latest
|
|
14
|
+
steps:
|
|
15
|
+
- uses: actions/checkout@v4
|
|
16
|
+
- uses: astral-sh/setup-uv@v6
|
|
17
|
+
- run: uv build
|
|
18
|
+
- uses: actions/upload-artifact@v4
|
|
19
|
+
with:
|
|
20
|
+
name: dist
|
|
21
|
+
path: dist/
|
|
22
|
+
|
|
23
|
+
publish:
|
|
24
|
+
needs: build
|
|
25
|
+
runs-on: ubuntu-latest
|
|
26
|
+
environment: pypi
|
|
27
|
+
permissions:
|
|
28
|
+
id-token: write
|
|
29
|
+
steps:
|
|
30
|
+
- uses: actions/download-artifact@v4
|
|
31
|
+
with:
|
|
32
|
+
name: dist
|
|
33
|
+
path: dist/
|
|
34
|
+
- uses: pypa/gh-action-pypi-publish@release/v1
|
|
@@ -0,0 +1,33 @@
|
|
|
1
|
+
# Notes for contributors and coding agents
|
|
2
|
+
|
|
3
|
+
paper-repro is a deterministic recorder. The host agent decides what to run; this package
|
|
4
|
+
runs it, captures it, and writes reports. It never calls an LLM.
|
|
5
|
+
|
|
6
|
+
## Layout
|
|
7
|
+
|
|
8
|
+
```
|
|
9
|
+
src/paper_repro/
|
|
10
|
+
study.py Study directory, append-only hash-chained evidence.jsonl, state replay
|
|
11
|
+
inspect_repo.py clone/copy, dependency and entry-point detection, README scan
|
|
12
|
+
claims.py claimed numbers from Markdown (tables, sentences, ranges, ±)
|
|
13
|
+
metrics.py metric values from logs, JSON, CSV; selectors (m1:acc:last@each)
|
|
14
|
+
envs.py uv venv + installs, conda translation, unpin, lock.txt
|
|
15
|
+
runner.py run one shell command with limits, capture outputs and written files
|
|
16
|
+
compare.py verdict logic (tolerance, rounding, ranges, scale, scope)
|
|
17
|
+
report.py report.json and report.md, built only from the log
|
|
18
|
+
agent_setup.py `setup` for Claude Code, Codex, Cursor
|
|
19
|
+
mcp_server.py MCP tools over stdio, thin wrappers
|
|
20
|
+
cli.py argparse CLI, --json everywhere
|
|
21
|
+
skills/paper-repro/SKILL.md the workflow the agent follows (shipped in the wheel)
|
|
22
|
+
examples/ real evidence bundles from real runs
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
## Invariants
|
|
26
|
+
|
|
27
|
+
- `evidence.jsonl` is append-only. Every derived view (status, reports) replays it.
|
|
28
|
+
Never mutate an entry; add a new one.
|
|
29
|
+
- A shortened or smoke run can never yield `counts_as_reproduction: true`.
|
|
30
|
+
- Values typed in by hand are flagged `unsourced` and can never count as a reproduction.
|
|
31
|
+
- `setup` must show its plan, require `--yes` or an interactive yes, back up what it edits,
|
|
32
|
+
and be idempotent. Tests run it against a temp HOME only.
|
|
33
|
+
- No em dashes or en dashes in code, docs or messages.
|
|
@@ -0,0 +1,25 @@
|
|
|
1
|
+
# Changelog
|
|
2
|
+
|
|
3
|
+
## 0.1.0 (unreleased)
|
|
4
|
+
|
|
5
|
+
First version.
|
|
6
|
+
|
|
7
|
+
- `inspect`: shallow clone (or copy of a local path), detection of languages, dependency
|
|
8
|
+
files, Python version hints, README commands and `pip install` lines, entry points,
|
|
9
|
+
data and weight downloads, GPU hints, and claimed numbers (tables, sentences, ranges,
|
|
10
|
+
`±` values).
|
|
11
|
+
- `env`: uv virtualenv per study, requirements and conda files (conda translated to pip),
|
|
12
|
+
`--from-readme`, `--unpin` and `--extra` recorded as deviations, `lock.txt` from
|
|
13
|
+
`uv pip freeze`, exact error on failure.
|
|
14
|
+
- `run`: shell command in the repo with the study env active, timeout that kills the
|
|
15
|
+
process tree, CPU-time limit, peak memory, stdout/stderr to files with SHA-256, files
|
|
16
|
+
written, and copies of small result files taken when the run ends.
|
|
17
|
+
- `metrics`: `name: value`, `name=value%`, `name value`, `n/d` fractions, JSON, JSON lines,
|
|
18
|
+
CSV/TSV, each with its source line; selectors such as `m1:val_loss:last@each`.
|
|
19
|
+
- `compare`: verdicts reproduced, close, not reproduced, could not run; percent/fraction
|
|
20
|
+
rescaling, rounding-aware tolerance, ranges, repeated runs, shortened runs never count.
|
|
21
|
+
- `report`: `report.md` and `report.json` with commit, environment, commands, claim,
|
|
22
|
+
measured values with sources, reasoning, deviations and what was not checked.
|
|
23
|
+
- Hash-chained `evidence.jsonl` and `verify`.
|
|
24
|
+
- MCP server (`paper-repro mcp`), `setup` for Claude Code, Codex and Cursor, and the
|
|
25
|
+
`paper-repro` agent skill.
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
# Contributing
|
|
2
|
+
|
|
3
|
+
```bash
|
|
4
|
+
uv sync
|
|
5
|
+
uv run pytest
|
|
6
|
+
uv run ruff check src tests && uv run ruff format --check src tests
|
|
7
|
+
```
|
|
8
|
+
|
|
9
|
+
Tests must stay offline and fast. They build small fixture repos in temp directories and
|
|
10
|
+
run real processes; they never touch your home directory (HOME is redirected) and keep uv
|
|
11
|
+
offline. If you add a metric format or a claim pattern, add a test with the exact line it
|
|
12
|
+
should match and one it should not.
|
|
13
|
+
|
|
14
|
+
Two rules for changes: the tool must not call any model API, and nothing it writes to a
|
|
15
|
+
report may come from anywhere but the evidence log. See `AGENTS.md` for the layout.
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Abel Yagubyan
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,194 @@
|
|
|
1
|
+
Metadata-Version: 2.5
|
|
2
|
+
Name: paper-repro
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Point your agent at a paper's code; it gets it running and tells you whether the headline number reproduces, with the evidence.
|
|
5
|
+
Project-URL: Homepage, https://github.com/Abelo9996/paper-repro
|
|
6
|
+
Author: Abel Yagubyan
|
|
7
|
+
License-Expression: MIT
|
|
8
|
+
License-File: LICENSE
|
|
9
|
+
Keywords: agents,machine-learning,mcp,reproducibility,research
|
|
10
|
+
Classifier: Development Status :: 3 - Alpha
|
|
11
|
+
Classifier: Intended Audience :: Science/Research
|
|
12
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
13
|
+
Classifier: Programming Language :: Python :: 3
|
|
14
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
15
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
17
|
+
Classifier: Topic :: Scientific/Engineering
|
|
18
|
+
Requires-Python: >=3.11
|
|
19
|
+
Requires-Dist: mcp>=1.2
|
|
20
|
+
Requires-Dist: pyyaml>=6
|
|
21
|
+
Description-Content-Type: text/markdown
|
|
22
|
+
|
|
23
|
+
# paper-repro
|
|
24
|
+
|
|
25
|
+
Point your agent at a paper's code; it gets it running and tells you whether the headline number reproduces, with the evidence.
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
uvx paper-repro setup --yes # register the MCP server and skill with Claude Code, Codex, Cursor
|
|
29
|
+
# then ask your agent: "Does github.com/karpathy/nanoGPT reproduce its CPU val loss of 1.88?"
|
|
30
|
+
uvx paper-repro inspect https://github.com/karpathy/nanoGPT # or drive it yourself from the CLI
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
> Not on PyPI yet. Until the first release, run it straight from GitHub by replacing `uvx paper-repro` with
|
|
34
|
+
> `uvx --from git+https://github.com/Abelo9996/paper-repro paper-repro`. `setup` registers `uvx paper-repro mcp`, so it works once the
|
|
35
|
+
> package is on PyPI.
|
|
36
|
+
|
|
37
|
+
Papers with Code went offline on 24 July 2025, taking its 79,817 paper-to-code links with it
|
|
38
|
+
([shutdown record](https://www.codesota.com/papers-with-code/shutdown)). There's no common place
|
|
39
|
+
that tracks whether a paper's released code actually reproduces the numbers in the paper. Research benchmarks for this exist (CORE-Bench, PaperBench), but they score agents,
|
|
40
|
+
not papers. paper-repro is the user-facing half: it gives a coding agent reliable operations to
|
|
41
|
+
clone a repo, build its environment, run it, pull numbers out of the logs and compare them with
|
|
42
|
+
the claim, and it writes everything down so a person can check the verdict without trusting
|
|
43
|
+
the agent.
|
|
44
|
+
|
|
45
|
+
## Example output
|
|
46
|
+
|
|
47
|
+
A real run on this machine (Apple M4, macOS, CPU only) against
|
|
48
|
+
[karpathy/nanoGPT](https://github.com/karpathy/nanoGPT) at commit `3adf61e`. Its README says the
|
|
49
|
+
small CPU configuration of the Shakespeare character model "gets us a loss of only 1.88". The full
|
|
50
|
+
evidence bundle is in [examples/nanogpt-shakespeare-char-cpu](examples/nanogpt-shakespeare-char-cpu/report.md).
|
|
51
|
+
|
|
52
|
+
```text
|
|
53
|
+
$ paper-repro inspect https://github.com/karpathy/nanoGPT
|
|
54
|
+
claimed numbers (12):
|
|
55
|
+
c1: best validation loss = 1.4697 (README.md:51)
|
|
56
|
+
c2: loss = 1.88 (README.md:88)
|
|
57
|
+
c3: loss = 2.85 (README.md:121)
|
|
58
|
+
...
|
|
59
|
+
$ paper-repro env --python 3.12 --from-readme
|
|
60
|
+
$ paper-repro run --scope setup -- python data/shakespeare_char/prepare.py
|
|
61
|
+
$ paper-repro run -- "python train.py config/train_shakespeare_char.py --device=cpu --compile=False --eval_iters=20 ..."
|
|
62
|
+
r3: exit 0 in 236.5s, peak memory 194.6 MB, 1 files written
|
|
63
|
+
$ paper-repro metrics --run r3 --run r4 --name val_loss
|
|
64
|
+
$ paper-repro compare --claim c2 --measured 'm1:val_loss:last@each' --why '...'
|
|
65
|
+
k1: REPRODUCED
|
|
66
|
+
Reproduced: loss claimed 1.88, measured 1.8857 (mean of 2 runs).
|
|
67
|
+
- Claim: loss 1.88 (README.md line 88).
|
|
68
|
+
- Measured over 2 runs: mean 1.8857, std 0, min 1.8857, max 1.8857.
|
|
69
|
+
- Distance from the claim: 0.0057 (0.3% of the claim).
|
|
70
|
+
- The measured value is worse than claimed (higher, and lower is better for this metric).
|
|
71
|
+
- Tolerance: ±0.0188 (1% of the claimed value, default). Close band: ±0.0564.
|
|
72
|
+
```
|
|
73
|
+
|
|
74
|
+
The report also lists what was not checked: the A100 claim (1.4697), the GPT-2 numbers, the
|
|
75
|
+
checksums of downloaded data, and that the seed is hard-coded, so the two runs measure
|
|
76
|
+
determinism (identical losses at every eval step) rather than seed variance.
|
|
77
|
+
|
|
78
|
+
## How it works
|
|
79
|
+
|
|
80
|
+
paper-repro is a recorder, not a brain. It never calls a model. Your agent (Claude Code, Codex,
|
|
81
|
+
Cursor) decides what to run; paper-repro does it the same way every time and keeps the receipts.
|
|
82
|
+
|
|
83
|
+
| Step | What it does | Tools it drives |
|
|
84
|
+
|---|---|---|
|
|
85
|
+
| `inspect` | Shallow-clones the repo (or copies a local path), records the commit SHA, finds dependency files, Python version hints, README commands and `pip install` lines, entry points and their flags, data and weight downloads, GPU hints, and every number the README claims (table cells, sentences, ranges, `±` values), each with file and line. | `git` |
|
|
86
|
+
| `env` | Creates an isolated virtualenv per study and installs the repo's own dependency file with pins as written. Conda `environment.yml` files are translated to pip. `--unpin`, `--extra`, `--from-readme` and a different Python are allowed but recorded as deviations. Writes `lock.txt` from `uv pip freeze`. On failure, records the exact error. | `uv venv`, `uv pip` |
|
|
87
|
+
| `run` | Runs one shell command from the repo root inside that env. Captures stdout and stderr to files with SHA-256, exit code, wall time, CPU time, peak memory, every file created or modified, and a copy of small result files as they were when the run ended. Timeout kills the whole process tree. Each run is labeled `full`, `shortened` (needs a note saying what was cut), `smoke` or `setup`. | your shell, POSIX rlimits |
|
|
88
|
+
| `metrics` | Extracts values like `accuracy: 0.913`, `acc=91.3%`, `F1 81.2`, `val loss 1.8857`, `Accuracy: 9897/10000`, plus JSON, JSON lines and CSV results, each with its source file, line and text. Select one with `m1:val_loss:last`, or one per run with `m1:val_loss:last@each`. | |
|
|
89
|
+
| `compare` | Puts measured values next to the claim and gives one of four verdicts: **reproduced**, **close**, **not reproduced**, **could not run**. Handles percent versus fraction, ranges, `±`, repeated runs (mean, std, min, max) and the rounding of the stated value. Default tolerance is 1% relative, close band 3x that; pass your own with a reason. A shortened run or a hand-typed value never counts as a reproduction. | |
|
|
90
|
+
| `report` | Writes `report.md` (the shareable artifact) and `report.json` from the evidence log only: repo URL and commit, machine, Python and key package versions, every command with exit code and duration, the claim, measured values with source lines, the reasoning, deviations, and what was not checked. | |
|
|
91
|
+
|
|
92
|
+
Every step appends to `evidence.jsonl`, a hash-chained log. `paper-repro verify` rechecks the
|
|
93
|
+
chain and the hashes of captured outputs, so an edited log or a swapped stdout file shows up.
|
|
94
|
+
|
|
95
|
+
A study directory looks like this:
|
|
96
|
+
|
|
97
|
+
```
|
|
98
|
+
paper-repro-runs/karpathy-nanoGPT/
|
|
99
|
+
repo/ the checkout
|
|
100
|
+
env/ the virtualenv
|
|
101
|
+
runs/r3/ stdout.txt, stderr.txt, files/ (results as the run left them)
|
|
102
|
+
lock.txt resolved packages
|
|
103
|
+
evidence.jsonl the log everything else is derived from
|
|
104
|
+
report.md the verdict and evidence, for people
|
|
105
|
+
report.json the same, for programs
|
|
106
|
+
```
|
|
107
|
+
|
|
108
|
+
`paper-repro report --bundle DIR` copies the shareable parts (no checkout, no env) to `DIR`.
|
|
109
|
+
|
|
110
|
+
### CLI
|
|
111
|
+
|
|
112
|
+
The second example, [examples/pygat-cora](examples/pygat-cora/report.md), was produced with
|
|
113
|
+
these commands against [Diego999/pyGAT](https://github.com/Diego999/pyGAT), whose README says
|
|
114
|
+
"The final accuracy is between 84.2 and 85.3 (obtained on 5 different runs)":
|
|
115
|
+
|
|
116
|
+
```bash
|
|
117
|
+
paper-repro inspect Diego999/pyGAT
|
|
118
|
+
paper-repro env # fails: README asks for Python 3.5, which uv does not support
|
|
119
|
+
paper-repro env --python 3.12 # fails: torch==0.4.1.post2 has no wheel for this machine
|
|
120
|
+
paper-repro env --python 3.12 --unpin
|
|
121
|
+
paper-repro run --scope smoke --note "2 epochs ..." -- python train.py --epochs 2
|
|
122
|
+
paper-repro run --seed 72 --timeout 10800 --note "..." -- "rm -f *.pkl && python train.py --seed 72"
|
|
123
|
+
paper-repro run --seed 1 ... && paper-repro run --seed 2 ...
|
|
124
|
+
paper-repro metrics --run r2 --run r3 --run r4 --name accuracy
|
|
125
|
+
paper-repro compare --claim c1 --measured 'm1:accuracy:last@each' --tol 0.1 --close-tol 1.0 \
|
|
126
|
+
--why "Cora's test split has 1000 nodes, so accuracy moves in steps of 0.1 points; ..."
|
|
127
|
+
paper-repro note --kind not_checked "The README's range comes from 5 runs; 3 seeds were run here ..."
|
|
128
|
+
paper-repro report --bundle examples/pygat-cora
|
|
129
|
+
```
|
|
130
|
+
|
|
131
|
+
Result: test accuracy 84.2, 84.7 and 84.2 over three seeds, mean 84.37, inside the claimed range,
|
|
132
|
+
verdict reproduced. Both failed environment attempts, the Python and unpinning deviations, and
|
|
133
|
+
the fact that only 3 of the authors' 5 runs were repeated are all in the report.
|
|
134
|
+
|
|
135
|
+
Every subcommand takes `--json`. `paper-repro status` shows what a study has recorded so far.
|
|
136
|
+
|
|
137
|
+
## Setup for agents
|
|
138
|
+
|
|
139
|
+
```bash
|
|
140
|
+
uvx paper-repro setup # shows what it would change, asks before applying
|
|
141
|
+
uvx paper-repro setup --yes # apply without asking
|
|
142
|
+
```
|
|
143
|
+
|
|
144
|
+
It detects each agent and, for the ones present:
|
|
145
|
+
|
|
146
|
+
- Claude Code: `claude mcp add --scope user paper-repro -- uvx paper-repro mcp`, and copies the
|
|
147
|
+
skill to `~/.claude/skills/paper-repro/SKILL.md`. Use `--project DIR` to write a project
|
|
148
|
+
`.mcp.json` instead.
|
|
149
|
+
- Codex: adds `[mcp_servers.paper-repro]` to `~/.codex/config.toml` and copies the skill to
|
|
150
|
+
`~/.codex/skills/paper-repro/`.
|
|
151
|
+
- Cursor: adds `paper-repro` to `mcpServers` in `~/.cursor/mcp.json`.
|
|
152
|
+
|
|
153
|
+
It backs up any file it edits (`*.bak-paper-repro-<time>`) and does nothing on a second run.
|
|
154
|
+
Manual registration is the same command everywhere: `uvx paper-repro mcp` over stdio.
|
|
155
|
+
|
|
156
|
+
The skill ([skills/paper-repro/SKILL.md](skills/paper-repro/SKILL.md)) is the workflow the agent
|
|
157
|
+
follows: inspect, pick one claim and say why, check feasibility, set up the env with the
|
|
158
|
+
least invasive fix, run the smallest faithful configuration first, extract, compare with a
|
|
159
|
+
tolerance it can defend, report. It also sets the rules: never fabricate or estimate a number,
|
|
160
|
+
never change the method to make a run succeed, and report "could not run" with the blocking
|
|
161
|
+
error instead of pushing through.
|
|
162
|
+
|
|
163
|
+
## What it can't do
|
|
164
|
+
|
|
165
|
+
- It does not judge whether the code implements the method in the paper. It checks the code's
|
|
166
|
+
output against a stated number, nothing more.
|
|
167
|
+
- No GPU orchestration. On a machine without the GPU the paper used, GPU-only claims end as
|
|
168
|
+
"could not run" or as shortened CPU runs that never count as reproductions.
|
|
169
|
+
- No container builds yet. A Dockerfile is detected and reported but not built; environments
|
|
170
|
+
are Python virtualenvs. R, Julia and other languages are detected but `env` only builds
|
|
171
|
+
Python environments (you can still `run` commands that use a system R or Julia).
|
|
172
|
+
- Conda environments are translated to pip, which can resolve different builds than conda
|
|
173
|
+
would. The translation is recorded as a deviation.
|
|
174
|
+
- Claim detection reads the README (and other Markdown in the repo), not the paper PDF. Numbers
|
|
175
|
+
that only appear in the paper must be added with `paper-repro claim --source "paper Table 2"`.
|
|
176
|
+
- Metric extraction is pattern-based. Unusual log formats may need `--file` on a results file or
|
|
177
|
+
a careful choice among the extracted values; the report always shows the exact source line.
|
|
178
|
+
- Memory limits are only enforced on Linux; on macOS peak memory is recorded but not capped.
|
|
179
|
+
- The default tolerance (1% relative) is a convention, not a statistical test. Choose one per
|
|
180
|
+
claim and say why.
|
|
181
|
+
|
|
182
|
+
## Privacy and safety
|
|
183
|
+
|
|
184
|
+
Everything runs on your machine. paper-repro sends nothing anywhere and has no telemetry. The
|
|
185
|
+
only network traffic is what you ask for: `git clone`, `uv` downloading packages, and whatever
|
|
186
|
+
the repository's own scripts download.
|
|
187
|
+
|
|
188
|
+
That last part matters: running a paper's code means running a stranger's code with your
|
|
189
|
+
user's permissions. paper-repro isolates the Python environment and can cap time and CPU, but
|
|
190
|
+
it is not a sandbox. For code you do not trust, run it inside a VM or a disposable container.
|
|
191
|
+
|
|
192
|
+
## License
|
|
193
|
+
|
|
194
|
+
MIT, see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,172 @@
|
|
|
1
|
+
# paper-repro
|
|
2
|
+
|
|
3
|
+
Point your agent at a paper's code; it gets it running and tells you whether the headline number reproduces, with the evidence.
|
|
4
|
+
|
|
5
|
+
```bash
|
|
6
|
+
uvx paper-repro setup --yes # register the MCP server and skill with Claude Code, Codex, Cursor
|
|
7
|
+
# then ask your agent: "Does github.com/karpathy/nanoGPT reproduce its CPU val loss of 1.88?"
|
|
8
|
+
uvx paper-repro inspect https://github.com/karpathy/nanoGPT # or drive it yourself from the CLI
|
|
9
|
+
```
|
|
10
|
+
|
|
11
|
+
> Not on PyPI yet. Until the first release, run it straight from GitHub by replacing `uvx paper-repro` with
|
|
12
|
+
> `uvx --from git+https://github.com/Abelo9996/paper-repro paper-repro`. `setup` registers `uvx paper-repro mcp`, so it works once the
|
|
13
|
+
> package is on PyPI.
|
|
14
|
+
|
|
15
|
+
Papers with Code went offline on 24 July 2025, taking its 79,817 paper-to-code links with it
|
|
16
|
+
([shutdown record](https://www.codesota.com/papers-with-code/shutdown)). There's no common place
|
|
17
|
+
that tracks whether a paper's released code actually reproduces the numbers in the paper. Research benchmarks for this exist (CORE-Bench, PaperBench), but they score agents,
|
|
18
|
+
not papers. paper-repro is the user-facing half: it gives a coding agent reliable operations to
|
|
19
|
+
clone a repo, build its environment, run it, pull numbers out of the logs and compare them with
|
|
20
|
+
the claim, and it writes everything down so a person can check the verdict without trusting
|
|
21
|
+
the agent.
|
|
22
|
+
|
|
23
|
+
## Example output
|
|
24
|
+
|
|
25
|
+
A real run on this machine (Apple M4, macOS, CPU only) against
|
|
26
|
+
[karpathy/nanoGPT](https://github.com/karpathy/nanoGPT) at commit `3adf61e`. Its README says the
|
|
27
|
+
small CPU configuration of the Shakespeare character model "gets us a loss of only 1.88". The full
|
|
28
|
+
evidence bundle is in [examples/nanogpt-shakespeare-char-cpu](examples/nanogpt-shakespeare-char-cpu/report.md).
|
|
29
|
+
|
|
30
|
+
```text
|
|
31
|
+
$ paper-repro inspect https://github.com/karpathy/nanoGPT
|
|
32
|
+
claimed numbers (12):
|
|
33
|
+
c1: best validation loss = 1.4697 (README.md:51)
|
|
34
|
+
c2: loss = 1.88 (README.md:88)
|
|
35
|
+
c3: loss = 2.85 (README.md:121)
|
|
36
|
+
...
|
|
37
|
+
$ paper-repro env --python 3.12 --from-readme
|
|
38
|
+
$ paper-repro run --scope setup -- python data/shakespeare_char/prepare.py
|
|
39
|
+
$ paper-repro run -- "python train.py config/train_shakespeare_char.py --device=cpu --compile=False --eval_iters=20 ..."
|
|
40
|
+
r3: exit 0 in 236.5s, peak memory 194.6 MB, 1 files written
|
|
41
|
+
$ paper-repro metrics --run r3 --run r4 --name val_loss
|
|
42
|
+
$ paper-repro compare --claim c2 --measured 'm1:val_loss:last@each' --why '...'
|
|
43
|
+
k1: REPRODUCED
|
|
44
|
+
Reproduced: loss claimed 1.88, measured 1.8857 (mean of 2 runs).
|
|
45
|
+
- Claim: loss 1.88 (README.md line 88).
|
|
46
|
+
- Measured over 2 runs: mean 1.8857, std 0, min 1.8857, max 1.8857.
|
|
47
|
+
- Distance from the claim: 0.0057 (0.3% of the claim).
|
|
48
|
+
- The measured value is worse than claimed (higher, and lower is better for this metric).
|
|
49
|
+
- Tolerance: ±0.0188 (1% of the claimed value, default). Close band: ±0.0564.
|
|
50
|
+
```
|
|
51
|
+
|
|
52
|
+
The report also lists what was not checked: the A100 claim (1.4697), the GPT-2 numbers, the
|
|
53
|
+
checksums of downloaded data, and that the seed is hard-coded, so the two runs measure
|
|
54
|
+
determinism (identical losses at every eval step) rather than seed variance.
|
|
55
|
+
|
|
56
|
+
## How it works
|
|
57
|
+
|
|
58
|
+
paper-repro is a recorder, not a brain. It never calls a model. Your agent (Claude Code, Codex,
|
|
59
|
+
Cursor) decides what to run; paper-repro does it the same way every time and keeps the receipts.
|
|
60
|
+
|
|
61
|
+
| Step | What it does | Tools it drives |
|
|
62
|
+
|---|---|---|
|
|
63
|
+
| `inspect` | Shallow-clones the repo (or copies a local path), records the commit SHA, finds dependency files, Python version hints, README commands and `pip install` lines, entry points and their flags, data and weight downloads, GPU hints, and every number the README claims (table cells, sentences, ranges, `±` values), each with file and line. | `git` |
|
|
64
|
+
| `env` | Creates an isolated virtualenv per study and installs the repo's own dependency file with pins as written. Conda `environment.yml` files are translated to pip. `--unpin`, `--extra`, `--from-readme` and a different Python are allowed but recorded as deviations. Writes `lock.txt` from `uv pip freeze`. On failure, records the exact error. | `uv venv`, `uv pip` |
|
|
65
|
+
| `run` | Runs one shell command from the repo root inside that env. Captures stdout and stderr to files with SHA-256, exit code, wall time, CPU time, peak memory, every file created or modified, and a copy of small result files as they were when the run ended. Timeout kills the whole process tree. Each run is labeled `full`, `shortened` (needs a note saying what was cut), `smoke` or `setup`. | your shell, POSIX rlimits |
|
|
66
|
+
| `metrics` | Extracts values like `accuracy: 0.913`, `acc=91.3%`, `F1 81.2`, `val loss 1.8857`, `Accuracy: 9897/10000`, plus JSON, JSON lines and CSV results, each with its source file, line and text. Select one with `m1:val_loss:last`, or one per run with `m1:val_loss:last@each`. | |
|
|
67
|
+
| `compare` | Puts measured values next to the claim and gives one of four verdicts: **reproduced**, **close**, **not reproduced**, **could not run**. Handles percent versus fraction, ranges, `±`, repeated runs (mean, std, min, max) and the rounding of the stated value. Default tolerance is 1% relative, close band 3x that; pass your own with a reason. A shortened run or a hand-typed value never counts as a reproduction. | |
|
|
68
|
+
| `report` | Writes `report.md` (the shareable artifact) and `report.json` from the evidence log only: repo URL and commit, machine, Python and key package versions, every command with exit code and duration, the claim, measured values with source lines, the reasoning, deviations, and what was not checked. | |
|
|
69
|
+
|
|
70
|
+
Every step appends to `evidence.jsonl`, a hash-chained log. `paper-repro verify` rechecks the
|
|
71
|
+
chain and the hashes of captured outputs, so an edited log or a swapped stdout file shows up.
|
|
72
|
+
|
|
73
|
+
A study directory looks like this:
|
|
74
|
+
|
|
75
|
+
```
|
|
76
|
+
paper-repro-runs/karpathy-nanoGPT/
|
|
77
|
+
repo/ the checkout
|
|
78
|
+
env/ the virtualenv
|
|
79
|
+
runs/r3/ stdout.txt, stderr.txt, files/ (results as the run left them)
|
|
80
|
+
lock.txt resolved packages
|
|
81
|
+
evidence.jsonl the log everything else is derived from
|
|
82
|
+
report.md the verdict and evidence, for people
|
|
83
|
+
report.json the same, for programs
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
`paper-repro report --bundle DIR` copies the shareable parts (no checkout, no env) to `DIR`.
|
|
87
|
+
|
|
88
|
+
### CLI
|
|
89
|
+
|
|
90
|
+
The second example, [examples/pygat-cora](examples/pygat-cora/report.md), was produced with
|
|
91
|
+
these commands against [Diego999/pyGAT](https://github.com/Diego999/pyGAT), whose README says
|
|
92
|
+
"The final accuracy is between 84.2 and 85.3 (obtained on 5 different runs)":
|
|
93
|
+
|
|
94
|
+
```bash
|
|
95
|
+
paper-repro inspect Diego999/pyGAT
|
|
96
|
+
paper-repro env # fails: README asks for Python 3.5, which uv does not support
|
|
97
|
+
paper-repro env --python 3.12 # fails: torch==0.4.1.post2 has no wheel for this machine
|
|
98
|
+
paper-repro env --python 3.12 --unpin
|
|
99
|
+
paper-repro run --scope smoke --note "2 epochs ..." -- python train.py --epochs 2
|
|
100
|
+
paper-repro run --seed 72 --timeout 10800 --note "..." -- "rm -f *.pkl && python train.py --seed 72"
|
|
101
|
+
paper-repro run --seed 1 ... && paper-repro run --seed 2 ...
|
|
102
|
+
paper-repro metrics --run r2 --run r3 --run r4 --name accuracy
|
|
103
|
+
paper-repro compare --claim c1 --measured 'm1:accuracy:last@each' --tol 0.1 --close-tol 1.0 \
|
|
104
|
+
--why "Cora's test split has 1000 nodes, so accuracy moves in steps of 0.1 points; ..."
|
|
105
|
+
paper-repro note --kind not_checked "The README's range comes from 5 runs; 3 seeds were run here ..."
|
|
106
|
+
paper-repro report --bundle examples/pygat-cora
|
|
107
|
+
```
|
|
108
|
+
|
|
109
|
+
Result: test accuracy 84.2, 84.7 and 84.2 over three seeds, mean 84.37, inside the claimed range,
|
|
110
|
+
verdict reproduced. Both failed environment attempts, the Python and unpinning deviations, and
|
|
111
|
+
the fact that only 3 of the authors' 5 runs were repeated are all in the report.
|
|
112
|
+
|
|
113
|
+
Every subcommand takes `--json`. `paper-repro status` shows what a study has recorded so far.
|
|
114
|
+
|
|
115
|
+
## Setup for agents
|
|
116
|
+
|
|
117
|
+
```bash
|
|
118
|
+
uvx paper-repro setup # shows what it would change, asks before applying
|
|
119
|
+
uvx paper-repro setup --yes # apply without asking
|
|
120
|
+
```
|
|
121
|
+
|
|
122
|
+
It detects each agent and, for the ones present:
|
|
123
|
+
|
|
124
|
+
- Claude Code: `claude mcp add --scope user paper-repro -- uvx paper-repro mcp`, and copies the
|
|
125
|
+
skill to `~/.claude/skills/paper-repro/SKILL.md`. Use `--project DIR` to write a project
|
|
126
|
+
`.mcp.json` instead.
|
|
127
|
+
- Codex: adds `[mcp_servers.paper-repro]` to `~/.codex/config.toml` and copies the skill to
|
|
128
|
+
`~/.codex/skills/paper-repro/`.
|
|
129
|
+
- Cursor: adds `paper-repro` to `mcpServers` in `~/.cursor/mcp.json`.
|
|
130
|
+
|
|
131
|
+
It backs up any file it edits (`*.bak-paper-repro-<time>`) and does nothing on a second run.
|
|
132
|
+
Manual registration is the same command everywhere: `uvx paper-repro mcp` over stdio.
|
|
133
|
+
|
|
134
|
+
The skill ([skills/paper-repro/SKILL.md](skills/paper-repro/SKILL.md)) is the workflow the agent
|
|
135
|
+
follows: inspect, pick one claim and say why, check feasibility, set up the env with the
|
|
136
|
+
least invasive fix, run the smallest faithful configuration first, extract, compare with a
|
|
137
|
+
tolerance it can defend, report. It also sets the rules: never fabricate or estimate a number,
|
|
138
|
+
never change the method to make a run succeed, and report "could not run" with the blocking
|
|
139
|
+
error instead of pushing through.
|
|
140
|
+
|
|
141
|
+
## What it can't do
|
|
142
|
+
|
|
143
|
+
- It does not judge whether the code implements the method in the paper. It checks the code's
|
|
144
|
+
output against a stated number, nothing more.
|
|
145
|
+
- No GPU orchestration. On a machine without the GPU the paper used, GPU-only claims end as
|
|
146
|
+
"could not run" or as shortened CPU runs that never count as reproductions.
|
|
147
|
+
- No container builds yet. A Dockerfile is detected and reported but not built; environments
|
|
148
|
+
are Python virtualenvs. R, Julia and other languages are detected but `env` only builds
|
|
149
|
+
Python environments (you can still `run` commands that use a system R or Julia).
|
|
150
|
+
- Conda environments are translated to pip, which can resolve different builds than conda
|
|
151
|
+
would. The translation is recorded as a deviation.
|
|
152
|
+
- Claim detection reads the README (and other Markdown in the repo), not the paper PDF. Numbers
|
|
153
|
+
that only appear in the paper must be added with `paper-repro claim --source "paper Table 2"`.
|
|
154
|
+
- Metric extraction is pattern-based. Unusual log formats may need `--file` on a results file or
|
|
155
|
+
a careful choice among the extracted values; the report always shows the exact source line.
|
|
156
|
+
- Memory limits are only enforced on Linux; on macOS peak memory is recorded but not capped.
|
|
157
|
+
- The default tolerance (1% relative) is a convention, not a statistical test. Choose one per
|
|
158
|
+
claim and say why.
|
|
159
|
+
|
|
160
|
+
## Privacy and safety
|
|
161
|
+
|
|
162
|
+
Everything runs on your machine. paper-repro sends nothing anywhere and has no telemetry. The
|
|
163
|
+
only network traffic is what you ask for: `git clone`, `uv` downloading packages, and whatever
|
|
164
|
+
the repository's own scripts download.
|
|
165
|
+
|
|
166
|
+
That last part matters: running a paper's code means running a stranger's code with your
|
|
167
|
+
user's permissions. paper-repro isolates the Python environment and can cap time and CPU, but
|
|
168
|
+
it is not a sandbox. For code you do not trust, run it inside a VM or a disposable container.
|
|
169
|
+
|
|
170
|
+
## License
|
|
171
|
+
|
|
172
|
+
MIT, see [LICENSE](LICENSE).
|
|
@@ -0,0 +1,9 @@
|
|
|
1
|
+
# Security
|
|
2
|
+
|
|
3
|
+
paper-repro clones and runs other people's code on your machine. That is the point of the
|
|
4
|
+
tool, and it is also the main risk: a repository's training script can do anything your user
|
|
5
|
+
account can do. The tool does not sandbox it beyond a separate virtualenv, a timeout and
|
|
6
|
+
optional CPU-time limits. Run untrusted repositories in a VM or container you can throw away.
|
|
7
|
+
|
|
8
|
+
To report a vulnerability in paper-repro itself, open a private security advisory on the
|
|
9
|
+
GitHub repository or email the maintainer. Please do not file a public issue first.
|