repo-bug-hunter 0.2.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (34) hide show
  1. repo_bug_hunter-0.2.0/.gitignore +19 -0
  2. repo_bug_hunter-0.2.0/LICENSE +21 -0
  3. repo_bug_hunter-0.2.0/PKG-INFO +150 -0
  4. repo_bug_hunter-0.2.0/README.md +127 -0
  5. repo_bug_hunter-0.2.0/pyproject.toml +71 -0
  6. repo_bug_hunter-0.2.0/repo_bug_hunter/__init__.py +0 -0
  7. repo_bug_hunter-0.2.0/repo_bug_hunter/agent.py +176 -0
  8. repo_bug_hunter-0.2.0/repo_bug_hunter/analyze.py +282 -0
  9. repo_bug_hunter-0.2.0/repo_bug_hunter/cli.py +35 -0
  10. repo_bug_hunter-0.2.0/repo_bug_hunter/doctor.py +83 -0
  11. repo_bug_hunter-0.2.0/repo_bug_hunter/env.py +157 -0
  12. repo_bug_hunter-0.2.0/repo_bug_hunter/evaluate.py +97 -0
  13. repo_bug_hunter-0.2.0/repo_bug_hunter/merge.py +67 -0
  14. repo_bug_hunter-0.2.0/repo_bug_hunter/openai_compat.py +226 -0
  15. repo_bug_hunter-0.2.0/repo_bug_hunter/pr.py +350 -0
  16. repo_bug_hunter-0.2.0/repo_bug_hunter/providers.py +165 -0
  17. repo_bug_hunter-0.2.0/repo_bug_hunter/run.py +169 -0
  18. repo_bug_hunter-0.2.0/repo_bug_hunter/site/index.html +362 -0
  19. repo_bug_hunter-0.2.0/repo_bug_hunter/smoke.py +105 -0
  20. repo_bug_hunter-0.2.0/repo_bug_hunter/tasks.py +24 -0
  21. repo_bug_hunter-0.2.0/repo_bug_hunter/tools.py +276 -0
  22. repo_bug_hunter-0.2.0/repo_bug_hunter/viewer.py +51 -0
  23. repo_bug_hunter-0.2.0/tests/__init__.py +0 -0
  24. repo_bug_hunter-0.2.0/tests/conftest.py +58 -0
  25. repo_bug_hunter-0.2.0/tests/test_agent.py +153 -0
  26. repo_bug_hunter-0.2.0/tests/test_analyze.py +130 -0
  27. repo_bug_hunter-0.2.0/tests/test_commands.py +87 -0
  28. repo_bug_hunter-0.2.0/tests/test_env.py +75 -0
  29. repo_bug_hunter-0.2.0/tests/test_merge.py +44 -0
  30. repo_bug_hunter-0.2.0/tests/test_openai_compat.py +203 -0
  31. repo_bug_hunter-0.2.0/tests/test_pipeline.py +230 -0
  32. repo_bug_hunter-0.2.0/tests/test_pr.py +187 -0
  33. repo_bug_hunter-0.2.0/tests/test_providers.py +97 -0
  34. repo_bug_hunter-0.2.0/tests/test_tools.py +180 -0
@@ -0,0 +1,19 @@
1
+ .venv/
2
+ __pycache__/
3
+ *.egg-info/
4
+ .pytest_cache/
5
+ dist/
6
+ .dist/
7
+ # SWE-bench harness logs: large, and re-read by `repo-bug-hunter evaluate --collect-only`
8
+ logs/
9
+ # Your runs stay local (the GitHub Actions workflow keeps its own on the repo-bug-hunter-results branch).
10
+ # The pilot runs are kept as worked examples (LEARNING_GUIDE.md, IMPLEMENTATION.md).
11
+ runs/*
12
+ !runs/local_pilot/
13
+ !runs/local_pilot_tf/
14
+ !runs/openrouter_pilot/
15
+ !runs/free_pilot/
16
+ /site/
17
+ # Secrets never belong in the repository
18
+ .env
19
+ *.key
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Manish Maurya
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,150 @@
1
+ Metadata-Version: 2.5
2
+ Name: repo-bug-hunter
3
+ Version: 0.2.0
4
+ Summary: An AI coding agent that fixes real GitHub bugs, and a lab that measures it: graded by hidden tests, every run replayable.
5
+ Project-URL: Homepage, https://github.com/Manishmaurya89/repo-bug-hunter
6
+ Project-URL: Documentation, https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/IMPLEMENTATION.md
7
+ Project-URL: Issues, https://github.com/Manishmaurya89/repo-bug-hunter/issues
8
+ Author: Manish Maurya
9
+ License-Expression: MIT
10
+ License-File: LICENSE
11
+ Keywords: ai-agents,benchmark,coding-agent,evaluation,llm,swe-bench
12
+ Classifier: Development Status :: 4 - Beta
13
+ Classifier: Environment :: Console
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: Intended Audience :: Science/Research
16
+ Classifier: Operating System :: MacOS
17
+ Classifier: Operating System :: POSIX :: Linux
18
+ Classifier: Programming Language :: Python :: 3
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Classifier: Topic :: Software Development :: Quality Assurance
24
+ Classifier: Topic :: Software Development :: Testing
25
+ Requires-Python: >=3.10
26
+ Requires-Dist: anthropic>=1.0
27
+ Requires-Dist: datasets>=2.18
28
+ Requires-Dist: openai>=3.22.0
29
+ Requires-Dist: swebench>=4.0
30
+ Description-Content-Type: text/markdown
31
+
32
+ # repo-bug-hunter
33
+
34
+ [![PyPI](https://img.shields.io/pypi/v/repo-bug-hunter)](https://pypi.org/project/repo-bug-hunter/)
35
+ [![tests](https://github.com/Manishmaurya89/repo-bug-hunter/actions/workflows/tests.yml/badge.svg)](https://github.com/Manishmaurya89/repo-bug-hunter/actions/workflows/tests.yml)
36
+ [![license: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/LICENSE)
37
+ [![Open in GitHub Codespaces](https://github.com/codespaces/badge.svg)](https://codespaces.new/Manishmaurya89/repo-bug-hunter)
38
+
39
+ An AI coding agent that fixes real GitHub bugs, and a lab that measures how well it does it.
40
+
41
+ Give it a GitHub issue and it works the way a developer would: it explores the repository, reproduces the bug, edits the code and runs the tests until it can submit a fix. The fix is then graded by the project's own hidden tests from [SWE-bench Verified](https://www.swebench.com/), and every step the agent took can be replayed in the browser.
42
+
43
+ ## What it does
44
+
45
+ - Fixes bugs from real GitHub issues inside a Docker sandbox, using six simple tools.
46
+ - Grades each fix with the tests the project's maintainers wrote for the real fix.
47
+ - Replays every run step by step on a static web page.
48
+ - Compares two versions of the agent on the same tasks, with paired statistics.
49
+ - Builds new tasks from merged GitHub pull requests, to test on bugs newer than the model.
50
+ - Works with Claude, any OpenRouter model, local models in Ollama, or any OpenAI-compatible server.
51
+
52
+ The agent is deliberately small: one file of under 200 lines, a plain loop around a chat model, with no agent framework, so every prompt and tool call is easy to follow. The sandbox is the same Docker image the grader uses and has no network, so the agent can't install packages or look up the real fix.
53
+
54
+ ## Quick start
55
+
56
+ You need Python 3.10+ and Docker, with about 30 GB of free disk space, since each task's Docker image is a few GB. On Apple Silicon, turn on Rosetta in Docker Desktop's settings; the images are x86-64.
57
+
58
+ ```bash
59
+ pip install repo-bug-hunter # or: uv tool install repo-bug-hunter
60
+ export OPENROUTER_API_KEY=... # from openrouter.ai/settings/keys; see Models for Claude
61
+ repo-bug-hunter doctor # checks Docker, disk, the model and the key
62
+ repo-bug-hunter smoke # fixes a toy bug end to end, in a few minutes
63
+ ```
64
+
65
+ Then run it on real bugs. Results go to `runs/` in the current folder:
66
+
67
+ ```bash
68
+ repo-bug-hunter run --name pilot --n 5 --difficulty "<15 min fix"
69
+ repo-bug-hunter evaluate runs/pilot # grade with the official SWE-bench harness
70
+ repo-bug-hunter analyze runs/pilot # resolve rate, cost, steps, why tasks failed
71
+ repo-bug-hunter viewer runs/pilot # build the replay site in ./site
72
+ python3 -m http.server -d site 8000 # open http://localhost:8000
73
+ ```
74
+
75
+ ### Without installing anything
76
+
77
+ - **GitHub Actions:** fork this repository and enable workflows in the fork's **Actions** tab. Add `OPENROUTER_API_KEY` under **Settings → Secrets and variables → Actions**, and set **Settings → Pages → Source** to **GitHub Actions**. Then run the **experiment** workflow. Each task runs on its own GitHub machine, the official grader scores it, and the replay site is published to your GitHub Pages.
78
+ - **Codespaces:** click the badge at the top for a ready-made environment in the browser, then run the commands above with `uv run` in front, such as `uv run repo-bug-hunter doctor`.
79
+
80
+ ## The test-first experiment
81
+
82
+ The question this project was built to answer: does an agent fix more bugs if it must first reproduce the bug with a failing test?
83
+
84
+ In `test_first` mode, the harness enforces it, not the prompt:
85
+
86
+ 1. Source files are locked until the agent registers a test command that fails on the current code.
87
+ 2. When the agent submits, the harness runs that test again. If it still fails, the submission is sent back once.
88
+ 3. So the agent can't get stuck: after three tests that don't fail, the source unlocks anyway, and a second submit is always accepted.
89
+
90
+ Run both modes on the same tasks and compare:
91
+
92
+ ```bash
93
+ repo-bug-hunter run --name baseline --variant baseline --n 50
94
+ repo-bug-hunter run --name test_first --variant test_first --n 50
95
+ repo-bug-hunter evaluate runs/baseline && repo-bug-hunter evaluate runs/test_first
96
+ repo-bug-hunter analyze runs/baseline runs/test_first --out results.md
97
+ ```
98
+
99
+ The report pairs the two runs task by task, using McNemar's exact test and a bootstrap confidence interval for the difference, so noise from a small sample isn't mistaken for an improvement.
100
+
101
+ ## How it works
102
+
103
+ - [agent.py](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/repo_bug_hunter/agent.py) is the loop. It sends the conversation to the model, runs the tools the model asks for, and repeats until the agent submits or hits a step or cost limit. With Claude it uses prompt caching and adaptive thinking.
104
+ - [tools.py](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/repo_bug_hunter/tools.py) has the tools (`read_file`, `search`, `edit_file`, `write_file`, `bash`, `submit`), the test-first gate, and the code that turns the final repository into a patch.
105
+ - [env.py](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/repo_bug_hunter/env.py) runs everything in the task's own SWE-bench Docker image, without network access, and caps output, file reads and processes so a runaway command can't hurt the machine.
106
+ - [evaluate.py](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/repo_bug_hunter/evaluate.py) grades with the official harness. A bug counts as fixed only if the hidden tests for the real fix pass and nothing that passed before breaks.
107
+ - [analyze.py](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/repo_bug_hunter/analyze.py) reports the resolve rate, cost and steps, and gives every failure one reason: gave up, wrong file, broke other tests, and so on.
108
+ - [viewer.py](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/repo_bug_hunter/viewer.py) builds the replay site, and [pr.py](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/repo_bug_hunter/pr.py) builds tasks from merged GitHub pull requests:
109
+
110
+ ```bash
111
+ repo-bug-hunter pr more-itertools/more-itertools#1305 --name fresh
112
+ ```
113
+
114
+ ## Models
115
+
116
+ | Provider | How to choose it | API key |
117
+ |---|---|---|
118
+ | Anthropic | `--model claude-sonnet-5-5` | `ANTHROPIC_API_KEY` |
119
+ | OpenRouter | `--model vendor/model` | `OPENROUTER_API_KEY` |
120
+ | Ollama (local) | `--provider ollama --model NAME` | none |
121
+ | Other OpenAI-compatible servers | `--base-url URL --model NAME` | `LLM_API_KEY` |
122
+
123
+ Without `--model`, a free OpenRouter model is used, so you can try everything without paying. `repo-bug-hunter doctor` lists the free models that currently support tool calling. Free models allow about 50 requests a day, roughly two tasks, or 1,000 a day after buying $10 of OpenRouter credits once. When a limit runs out, the run stops cleanly, and running the same command later continues where it stopped. Free providers may log prompts, so don't point them at private code.
124
+
125
+ Local models need tool calling and a context window of at least 32k tokens. Ollama's default is 4k, so create a variant first:
126
+
127
+ ```bash
128
+ printf 'FROM gemma4\nPARAMETER num_ctx 32768\n' > Modelfile && ollama create gemma4-32k -f Modelfile
129
+ ```
130
+
131
+ ## Limitations
132
+
133
+ - SWE-bench Verified is public, so strong models may have seen some of the real fixes; OpenAI stopped reporting it in February 2026 for this reason. Paired comparisons are less affected, and `repo-bug-hunter pr` tests on fresh bugs.
134
+ - `repo-bug-hunter pr` supports Python projects tested with pytest, and the pull request must name the issue it fixes.
135
+ - Failure reasons are heuristics. For example, "wrong file" compares the patch with the files the real fix changed, so a correct fix in another file would be miscounted.
136
+
137
+ ## Development
138
+
139
+ ```bash
140
+ git clone https://github.com/Manishmaurya89/repo-bug-hunter && cd repo-bug-hunter
141
+ uv sync
142
+ uv run pytest # the Docker tests run only when Docker is up
143
+ ```
144
+
145
+ - [IMPLEMENTATION.md](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/IMPLEMENTATION.md): how each part is built and why, including the bugs found along the way.
146
+ - [LEARNING_GUIDE.md](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/LEARNING_GUIDE.md): every idea explained from zero, with exercises.
147
+
148
+ ## License
149
+
150
+ [MIT](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/LICENSE)
@@ -0,0 +1,127 @@
1
+ # repo-bug-hunter
2
+
3
+ [![PyPI](https://img.shields.io/pypi/v/repo-bug-hunter)](https://pypi.org/project/repo-bug-hunter/)
4
+ [![tests](https://github.com/Manishmaurya89/repo-bug-hunter/actions/workflows/tests.yml/badge.svg)](https://github.com/Manishmaurya89/repo-bug-hunter/actions/workflows/tests.yml)
5
+ [![license: MIT](https://img.shields.io/badge/license-MIT-blue.svg)](LICENSE)
6
+ [![Open in GitHub Codespaces](https://github.com/codespaces/badge.svg)](https://codespaces.new/Manishmaurya89/repo-bug-hunter)
7
+
8
+ An AI coding agent that fixes real GitHub bugs, and a lab that measures how well it does it.
9
+
10
+ Give it a GitHub issue and it works the way a developer would: it explores the repository, reproduces the bug, edits the code and runs the tests until it can submit a fix. The fix is then graded by the project's own hidden tests from [SWE-bench Verified](https://www.swebench.com/), and every step the agent took can be replayed in the browser.
11
+
12
+ ```mermaid
13
+ flowchart LR
14
+ issue["GitHub issue"] --> agent["Agent loop in a Docker sandbox<br/>read · search · edit · run tests"]
15
+ agent --> patch["Patch"]
16
+ patch --> grader["Hidden tests<br/>official SWE-bench grader"]
17
+ grader --> report["Resolved or not<br/>cost · steps · replay"]
18
+ ```
19
+
20
+ ## What it does
21
+
22
+ - Fixes bugs from real GitHub issues inside a Docker sandbox, using six simple tools.
23
+ - Grades each fix with the tests the project's maintainers wrote for the real fix.
24
+ - Replays every run step by step on a static web page.
25
+ - Compares two versions of the agent on the same tasks, with paired statistics.
26
+ - Builds new tasks from merged GitHub pull requests, to test on bugs newer than the model.
27
+ - Works with Claude, any OpenRouter model, local models in Ollama, or any OpenAI-compatible server.
28
+
29
+ The agent is deliberately small: one file of under 200 lines, a plain loop around a chat model, with no agent framework, so every prompt and tool call is easy to follow. The sandbox is the same Docker image the grader uses and has no network, so the agent can't install packages or look up the real fix.
30
+
31
+ ## Quick start
32
+
33
+ You need Python 3.10+ and Docker, with about 30 GB of free disk space, since each task's Docker image is a few GB. On Apple Silicon, turn on Rosetta in Docker Desktop's settings; the images are x86-64.
34
+
35
+ ```bash
36
+ pip install repo-bug-hunter # or: uv tool install repo-bug-hunter
37
+ export OPENROUTER_API_KEY=... # from openrouter.ai/settings/keys; see Models for Claude
38
+ repo-bug-hunter doctor # checks Docker, disk, the model and the key
39
+ repo-bug-hunter smoke # fixes a toy bug end to end, in a few minutes
40
+ ```
41
+
42
+ Then run it on real bugs. Results go to `runs/` in the current folder:
43
+
44
+ ```bash
45
+ repo-bug-hunter run --name pilot --n 5 --difficulty "<15 min fix"
46
+ repo-bug-hunter evaluate runs/pilot # grade with the official SWE-bench harness
47
+ repo-bug-hunter analyze runs/pilot # resolve rate, cost, steps, why tasks failed
48
+ repo-bug-hunter viewer runs/pilot # build the replay site in ./site
49
+ python3 -m http.server -d site 8000 # open http://localhost:8000
50
+ ```
51
+
52
+ ### Without installing anything
53
+
54
+ - **GitHub Actions:** fork this repository and enable workflows in the fork's **Actions** tab. Add `OPENROUTER_API_KEY` under **Settings → Secrets and variables → Actions**, and set **Settings → Pages → Source** to **GitHub Actions**. Then run the **experiment** workflow. Each task runs on its own GitHub machine, the official grader scores it, and the replay site is published to your GitHub Pages.
55
+ - **Codespaces:** click the badge at the top for a ready-made environment in the browser, then run the commands above with `uv run` in front, such as `uv run repo-bug-hunter doctor`.
56
+
57
+ ## The test-first experiment
58
+
59
+ The question this project was built to answer: does an agent fix more bugs if it must first reproduce the bug with a failing test?
60
+
61
+ In `test_first` mode, the harness enforces it, not the prompt:
62
+
63
+ 1. Source files are locked until the agent registers a test command that fails on the current code.
64
+ 2. When the agent submits, the harness runs that test again. If it still fails, the submission is sent back once.
65
+ 3. So the agent can't get stuck: after three tests that don't fail, the source unlocks anyway, and a second submit is always accepted.
66
+
67
+ Run both modes on the same tasks and compare:
68
+
69
+ ```bash
70
+ repo-bug-hunter run --name baseline --variant baseline --n 50
71
+ repo-bug-hunter run --name test_first --variant test_first --n 50
72
+ repo-bug-hunter evaluate runs/baseline && repo-bug-hunter evaluate runs/test_first
73
+ repo-bug-hunter analyze runs/baseline runs/test_first --out results.md
74
+ ```
75
+
76
+ The report pairs the two runs task by task, using McNemar's exact test and a bootstrap confidence interval for the difference, so noise from a small sample isn't mistaken for an improvement.
77
+
78
+ ## How it works
79
+
80
+ - [agent.py](repo_bug_hunter/agent.py) is the loop. It sends the conversation to the model, runs the tools the model asks for, and repeats until the agent submits or hits a step or cost limit. With Claude it uses prompt caching and adaptive thinking.
81
+ - [tools.py](repo_bug_hunter/tools.py) has the tools (`read_file`, `search`, `edit_file`, `write_file`, `bash`, `submit`), the test-first gate, and the code that turns the final repository into a patch.
82
+ - [env.py](repo_bug_hunter/env.py) runs everything in the task's own SWE-bench Docker image, without network access, and caps output, file reads and processes so a runaway command can't hurt the machine.
83
+ - [evaluate.py](repo_bug_hunter/evaluate.py) grades with the official harness. A bug counts as fixed only if the hidden tests for the real fix pass and nothing that passed before breaks.
84
+ - [analyze.py](repo_bug_hunter/analyze.py) reports the resolve rate, cost and steps, and gives every failure one reason: gave up, wrong file, broke other tests, and so on.
85
+ - [viewer.py](repo_bug_hunter/viewer.py) builds the replay site, and [pr.py](repo_bug_hunter/pr.py) builds tasks from merged GitHub pull requests:
86
+
87
+ ```bash
88
+ repo-bug-hunter pr more-itertools/more-itertools#1305 --name fresh
89
+ ```
90
+
91
+ ## Models
92
+
93
+ | Provider | How to choose it | API key |
94
+ |---|---|---|
95
+ | Anthropic | `--model claude-sonnet-5-5` | `ANTHROPIC_API_KEY` |
96
+ | OpenRouter | `--model vendor/model` | `OPENROUTER_API_KEY` |
97
+ | Ollama (local) | `--provider ollama --model NAME` | none |
98
+ | Other OpenAI-compatible servers | `--base-url URL --model NAME` | `LLM_API_KEY` |
99
+
100
+ Without `--model`, a free OpenRouter model is used, so you can try everything without paying. `repo-bug-hunter doctor` lists the free models that currently support tool calling. Free models allow about 50 requests a day, roughly two tasks, or 1,000 a day after buying $10 of OpenRouter credits once. When a limit runs out, the run stops cleanly, and running the same command later continues where it stopped. Free providers may log prompts, so don't point them at private code.
101
+
102
+ Local models need tool calling and a context window of at least 32k tokens. Ollama's default is 4k, so create a variant first:
103
+
104
+ ```bash
105
+ printf 'FROM gemma4\nPARAMETER num_ctx 32768\n' > Modelfile && ollama create gemma4-32k -f Modelfile
106
+ ```
107
+
108
+ ## Limitations
109
+
110
+ - SWE-bench Verified is public, so strong models may have seen some of the real fixes; OpenAI stopped reporting it in February 2026 for this reason. Paired comparisons are less affected, and `repo-bug-hunter pr` tests on fresh bugs.
111
+ - `repo-bug-hunter pr` supports Python projects tested with pytest, and the pull request must name the issue it fixes.
112
+ - Failure reasons are heuristics. For example, "wrong file" compares the patch with the files the real fix changed, so a correct fix in another file would be miscounted.
113
+
114
+ ## Development
115
+
116
+ ```bash
117
+ git clone https://github.com/Manishmaurya89/repo-bug-hunter && cd repo-bug-hunter
118
+ uv sync
119
+ uv run pytest # the Docker tests run only when Docker is up
120
+ ```
121
+
122
+ - [IMPLEMENTATION.md](IMPLEMENTATION.md): how each part is built and why, including the bugs found along the way.
123
+ - [LEARNING_GUIDE.md](LEARNING_GUIDE.md): every idea explained from zero, with exercises.
124
+
125
+ ## License
126
+
127
+ [MIT](LICENSE)
@@ -0,0 +1,71 @@
1
+ [project]
2
+ name = "repo-bug-hunter"
3
+ version = "0.2.0"
4
+ description = "An AI coding agent that fixes real GitHub bugs, and a lab that measures it: graded by hidden tests, every run replayable."
5
+ dynamic = ["readme"]
6
+ license = "MIT"
7
+ license-files = ["LICENSE"]
8
+ authors = [{ name = "Manish Maurya" }]
9
+ requires-python = ">=3.10"
10
+ keywords = ["ai-agents", "coding-agent", "llm", "swe-bench", "evaluation", "benchmark"]
11
+ classifiers = [
12
+ "Development Status :: 4 - Beta",
13
+ "Environment :: Console",
14
+ "Intended Audience :: Developers",
15
+ "Intended Audience :: Science/Research",
16
+ "Operating System :: MacOS",
17
+ "Operating System :: POSIX :: Linux",
18
+ "Programming Language :: Python :: 3",
19
+ "Programming Language :: Python :: 3.10",
20
+ "Programming Language :: Python :: 3.11",
21
+ "Programming Language :: Python :: 3.12",
22
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
23
+ "Topic :: Software Development :: Quality Assurance",
24
+ "Topic :: Software Development :: Testing",
25
+ ]
26
+ dependencies = [
27
+ "anthropic>=1.0",
28
+ "datasets>=2.18",
29
+ "openai>=3.22.0",
30
+ "swebench>=4.0",
31
+ ]
32
+
33
+ [project.urls]
34
+ Homepage = "https://github.com/Manishmaurya89/repo-bug-hunter"
35
+ Documentation = "https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/IMPLEMENTATION.md"
36
+ Issues = "https://github.com/Manishmaurya89/repo-bug-hunter/issues"
37
+
38
+ [project.scripts]
39
+ repo-bug-hunter = "repo_bug_hunter.cli:main"
40
+
41
+ [dependency-groups]
42
+ dev = ["pytest>=8"]
43
+
44
+ [build-system]
45
+ requires = ["hatchling", "hatch-fancy-pypi-readme"]
46
+ build-backend = "hatchling.build"
47
+
48
+ [tool.hatch.build.targets.wheel]
49
+ packages = ["repo_bug_hunter"]
50
+
51
+ [tool.hatch.build.targets.sdist]
52
+ include = ["/repo_bug_hunter", "/tests", "/README.md"]
53
+
54
+ # PyPI shows the README without the Mermaid diagram, which only GitHub draws, and with
55
+ # repository links made absolute: relative ones would point into pypi.org.
56
+ [tool.hatch.metadata.hooks.fancy-pypi-readme]
57
+ content-type = "text/markdown"
58
+
59
+ [[tool.hatch.metadata.hooks.fancy-pypi-readme.fragments]]
60
+ path = "README.md"
61
+
62
+ [[tool.hatch.metadata.hooks.fancy-pypi-readme.substitutions]]
63
+ pattern = '(?s)```mermaid\n.*?```\n\n'
64
+ replacement = ''
65
+
66
+ [[tool.hatch.metadata.hooks.fancy-pypi-readme.substitutions]]
67
+ pattern = '\]\((?!https?://|#)([^)]+)\)'
68
+ replacement = '](https://github.com/Manishmaurya89/repo-bug-hunter/blob/main/\1)'
69
+
70
+ [tool.pytest.ini_options]
71
+ testpaths = ["tests"]
File without changes
@@ -0,0 +1,176 @@
1
+ """The agent loop: ask Claude for tool calls, run them, repeat until submit or a limit."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import time
6
+ from typing import Callable
7
+
8
+ import anthropic
9
+
10
+ from .tools import Toolbox
11
+
12
+ # USD per million tokens: input, output, 5-minute cache write, cache read.
13
+ # List prices as of September 2026; check https://claude.com/pricing before quoting costs.
14
+ PRICES = {
15
+ "claude-opus-5-5": (4.00, 20.00, 5.00, 0.20),
16
+ "claude-sonnet-5-5": (2.00, 10.00, 2.50, 0.20),
17
+ "claude-haiku-4-5": (1.00, 5.00, 1.25, 0.10),
18
+ "claude-fable-5-1": (10.00, 50.00, 12.50, 0.25),
19
+ # Possible refusal-fallback targets, so a fallback turn is still priced.
20
+ "claude-opus-5": (5.00, 25.00, 6.25, 0.50),
21
+ "claude-opus-4-8": (5.00, 25.00, 6.25, 0.50),
22
+ "claude-sonnet-5": (2.00, 10.00, 2.50, 0.20),
23
+ }
24
+ # Models that take `fallbacks: "default"` (re-run a classifier refusal on another model).
25
+ FALLBACK_MODELS = {"claude-fable-5-1", "claude-opus-5-5", "claude-opus-5", "claude-sonnet-5-5"}
26
+
27
+ SYSTEM = """\
28
+ You are an autonomous software engineer fixing an issue in a Python repository. The repository \
29
+ is checked out at {workdir} with its dependencies installed. There is no network access, so tests \
30
+ that need the internet fail whatever you change.
31
+
32
+ Resolve the issue by changing the library's source code. Hidden tests written by the project's \
33
+ maintainers will check your change, so fix the underlying behavior the issue describes rather \
34
+ than special-casing its example. Changes to existing test files are discarded before grading.
35
+
36
+ Use the tools to explore the code, edit files, and run code and tests. Prefer edit_file and \
37
+ write_file over editing files with shell commands. Nobody will answer questions, so keep working \
38
+ until the fix is complete, then call submit."""
39
+
40
+ TEST_FIRST = """
41
+
42
+ Work test-first. Before changing any source file, write a test that reproduces the issue and \
43
+ fails on the current code. A standalone script such as {workdir}/repro_test.py that exits \
44
+ non-zero while the bug is present works for any project. Register it with record_failing_test, \
45
+ which runs it and checks that it fails; source files stay locked until then. When you call \
46
+ submit, the harness runs it again and it must pass."""
47
+
48
+
49
+ def system_prompt(test_first: bool, workdir: str = "/testbed") -> str:
50
+ # The path must be where the repository really is: told /testbed while working elsewhere,
51
+ # a model goes looking for it across the whole machine.
52
+ return (SYSTEM + (TEST_FIRST if test_first else "")).format(workdir=workdir)
53
+
54
+
55
+ def task_prompt(task: dict) -> str:
56
+ return f"<issue>\n{task['problem_statement'].strip()}\n</issue>\n\nRepository: {task['repo']}"
57
+
58
+
59
+ def cost_of(model: str, usage) -> float:
60
+ # A provider that reports what a request cost (OpenRouter does) is taken at its word.
61
+ reported = getattr(usage, "cost_usd", None)
62
+ if reported is not None:
63
+ return float(reported)
64
+ # Models not in the table (local models, free tiers) are counted as free.
65
+ p_in, p_out, p_write, p_read = PRICES.get(model, (0, 0, 0, 0))
66
+ get = lambda f: getattr(usage, f, 0) or 0
67
+ return (get("input_tokens") * p_in + get("output_tokens") * p_out
68
+ + get("cache_creation_input_tokens") * p_write
69
+ + get("cache_read_input_tokens") * p_read) / 1e6
70
+
71
+
72
+ def claude(model: str, effort: str, system: str, tools: list[dict]) -> Callable[[list], object]:
73
+ """Returns query(messages) -> Message, with caching and fallbacks configured."""
74
+ client = anthropic.Anthropic(max_retries=8)
75
+ params = dict(model=model, max_tokens=64_000, system=system, tools=tools,
76
+ cache_control={"type": "ephemeral"}) # moves to the end of the history each turn
77
+ betas = []
78
+ if not model.startswith("claude-haiku"): # Haiku 4.5 predates adaptive thinking and effort
79
+ params["thinking"] = {"type": "adaptive", "display": "summarized"}
80
+ params["output_config"] = {"effort": effort}
81
+ if model in FALLBACK_MODELS:
82
+ # A classifier refusal is re-run on another model. Steps record the serving model,
83
+ # so a run that fell back is visible in the results.
84
+ params["fallbacks"] = "default"
85
+ betas.append("server-side-fallback-2026-07-01")
86
+
87
+ def query(messages: list) -> object:
88
+ with client.beta.messages.stream(messages=messages, betas=betas, **params) as stream:
89
+ return stream.get_final_message()
90
+
91
+ return query
92
+
93
+
94
+ def run_agent(task: dict, env, query: Callable[[list], object], *, model: str,
95
+ test_first: bool = False, max_steps: int = 50, max_cost: float = 2.0,
96
+ log: Callable[[str], None] = lambda s: None) -> dict:
97
+ tools = Toolbox(env, test_first=test_first)
98
+ messages: list = [{"role": "user", "content": task_prompt(task)}]
99
+ traj = {"instance_id": task["instance_id"], "variant": "test_first" if test_first else "baseline",
100
+ "model": model, "problem_statement": task["problem_statement"], "steps": [],
101
+ "exit_status": "step_limit", "cost": 0.0}
102
+ started = time.time()
103
+
104
+ for i in range(1, max_steps + 1):
105
+ try:
106
+ resp = query(messages)
107
+ except Exception as e: # after the client's own retries; keep the trajectory so far
108
+ traj["exit_status"], traj["error"] = "api_error", f"{type(e).__name__}: {e}"
109
+ break
110
+ served_by = resp.model if resp.model in PRICES else model
111
+ step = {"i": i, "model": resp.model, "stop_reason": resp.stop_reason,
112
+ "cost": cost_of(served_by, resp.usage), "thinking": [], "text": [], "actions": [],
113
+ "usage": {k: getattr(resp.usage, k, 0) or 0 for k in (
114
+ "input_tokens", "output_tokens", "cache_creation_input_tokens",
115
+ "cache_read_input_tokens")}}
116
+ traj["steps"].append(step)
117
+ traj["cost"] += step["cost"]
118
+ # Append the content unchanged: thinking blocks must be passed back as-is.
119
+ messages.append({"role": "assistant", "content": resp.content})
120
+
121
+ for b in resp.content:
122
+ if b.type == "thinking" and b.thinking:
123
+ step["thinking"].append(b.thinking)
124
+ elif b.type == "text" and b.text.strip():
125
+ step["text"].append(b.text)
126
+ if resp.stop_reason == "refusal":
127
+ details = getattr(resp, "stop_details", None)
128
+ traj["exit_status"], traj["error"] = "refusal", str(getattr(details, "category", None))
129
+ break
130
+
131
+ calls = [b for b in resp.content if b.type == "tool_use"]
132
+ if not calls:
133
+ messages.append({"role": "user", "content": "No tool was called, so nothing ran. Writing a "
134
+ "call out as text does not run it: use the tools to keep working, and call "
135
+ "submit when the fix is complete."})
136
+ log(f"step {i} ${traj['cost']:.2f} (no tool call)")
137
+ continue
138
+
139
+ results = []
140
+ for b in calls:
141
+ t0 = time.time()
142
+ if resp.stop_reason == "max_tokens":
143
+ out, is_error = "The response hit max_tokens before this call was complete; it was not run.", True
144
+ else:
145
+ out, is_error = tools.call(b.name, b.input)
146
+ step["actions"].append({"tool": b.name, "input": b.input, "output": out,
147
+ "is_error": is_error, "seconds": round(time.time() - t0, 2)})
148
+ results.append({"type": "tool_result", "tool_use_id": b.id, "content": out,
149
+ "is_error": is_error})
150
+ log(f"step {i} ${traj['cost']:.2f} {b.name}: {_brief(b.input)}")
151
+ if tools.submitted:
152
+ break
153
+ messages.append({"role": "user", "content": results})
154
+
155
+ if tools.submitted:
156
+ traj["exit_status"] = "submitted"
157
+ break
158
+ if traj["cost"] >= max_cost:
159
+ traj["exit_status"] = "cost_limit"
160
+ break
161
+
162
+ traj["patch"] = tools.patch()
163
+ traj["n_steps"] = len(traj["steps"])
164
+ traj["n_tool_calls"] = sum(len(s["actions"]) for s in traj["steps"])
165
+ traj["seconds"] = round(time.time() - started, 1)
166
+ traj["fallback_used"] = any(s["model"] != model for s in traj["steps"])
167
+ if test_first:
168
+ traj["repro"] = {"status": tools.repro_status, "command": tools.repro_command,
169
+ "attempts": tools.repro_attempts,
170
+ "passed_at_submit": tools.repro_passed_at_submit}
171
+ return traj
172
+
173
+
174
+ def _brief(args: dict) -> str:
175
+ first = next(iter(args.values()), "") if args else ""
176
+ return str(first).replace("\n", " ")[:80]