overkill 0.4.0__tar.gz → 0.5.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {overkill-0.4.0 → overkill-0.5.0}/.github/workflows/test.yml +1 -1
- {overkill-0.4.0 → overkill-0.5.0}/AGENTS.md +1 -1
- {overkill-0.4.0 → overkill-0.5.0}/PKG-INFO +4 -8
- {overkill-0.4.0 → overkill-0.5.0}/README.md +1 -6
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-refactor-full.prompt.md +4 -4
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-refactor-layer.prompt.md +4 -4
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-refactor-micro.prompt.md +3 -3
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-refactor-module.prompt.md +3 -3
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-review.prompt.md +2 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/codex-refactor-full.prompt.md +4 -4
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/codex-refactor-layer.prompt.md +4 -4
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/codex-refactor-micro.prompt.md +3 -3
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/codex-refactor-module.prompt.md +3 -3
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/codex-review.prompt.md +2 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/gemini-refactor-full.prompt.md +4 -4
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/gemini-refactor-layer.prompt.md +4 -4
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/gemini-refactor-micro.prompt.md +3 -3
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/gemini-refactor-module.prompt.md +3 -3
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/gemini-review.prompt.md +2 -0
- {overkill-0.4.0 → overkill-0.5.0}/pyproject.toml +5 -4
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/agents.py +10 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/classify.py +1 -4
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/cli.py +17 -2
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/loop_engine.py +2 -4
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/models.py +14 -4
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/refactor_suggest.py +2 -3
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/retry.py +1 -4
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/review_loop.py +1 -5
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/self_review.py +0 -1
- {overkill-0.4.0 → overkill-0.5.0}/uv.lock +96 -3
- overkill-0.4.0/bin/lib/check-claude-limit.sh +0 -256
- overkill-0.4.0/bin/lib/check-codex-limit.sh +0 -202
- overkill-0.4.0/bin/lib/common.sh +0 -511
- overkill-0.4.0/bin/lib/retry.sh +0 -306
- overkill-0.4.0/bin/lib/self-review.sh +0 -527
- overkill-0.4.0/bin/refactor-suggest.sh +0 -748
- overkill-0.4.0/bin/review-loop.sh +0 -593
- overkill-0.4.0/test/refactor-suggest.bats +0 -159
- overkill-0.4.0/test/test_helper.bash +0 -117
- {overkill-0.4.0 → overkill-0.5.0}/.github/workflows/publish.yml +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/.gitignore +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/.overkillrc.example +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/.python-version +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/.refactorsuggestrc.example +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/CLAUDE.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/GEMINI.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/LICENSE +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/install.sh +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-fix-execute.prompt.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-fix.prompt.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-refactor-fix-execute.prompt.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-refactor-fix.prompt.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/active/claude-self-review.prompt.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/reference/claude-code-review-plugin.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/reference/claude-security-review.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/prompts/reference/codex-review-original.md +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/__init__.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/__main__.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/budget/__init__.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/budget/claude.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/budget/codex.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/budget/gemini.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/budget_report.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/data/__init__.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/git_ops.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/init.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/json_extract.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/reporting.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/resume.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/time_utils.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/src/mr_overkill/two_step_fix.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/__init__.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/conftest.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_agents.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_budget_claude.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_budget_codex.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_budget_gemini.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_budget_policy.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_budget_report.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_classify.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_cli.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_git_ops.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_init.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_integration.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_json_extract.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_loop_engine.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_refactor_suggest.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_reporting.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_resume.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_retry.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_review_loop.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_self_review.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_time_utils.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/tests/test_two_step_fix.py +0 -0
- {overkill-0.4.0 → overkill-0.5.0}/uninstall.sh +0 -0
|
@@ -9,7 +9,7 @@
|
|
|
9
9
|
|
|
10
10
|
## Pull Request Rules
|
|
11
11
|
|
|
12
|
-
Every PR must pass the review loop (`review-loop
|
|
12
|
+
Every PR must pass the review loop (`overkill review-loop --dry-run`) before merging. No exceptions. We eat our own dog food — if Mr. Overkill can't approve it, neither can you.
|
|
13
13
|
|
|
14
14
|
## Branch Rules
|
|
15
15
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: overkill
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.5.0
|
|
4
4
|
Summary: AI-powered code review loop — automates Codex review + Claude fix cycles
|
|
5
5
|
Project-URL: Repository, https://github.com/modocai/mr-overkill
|
|
6
6
|
Project-URL: Issues, https://github.com/modocai/mr-overkill/issues
|
|
@@ -14,10 +14,11 @@ Classifier: Intended Audience :: Developers
|
|
|
14
14
|
Classifier: License :: OSI Approved :: MIT License
|
|
15
15
|
Classifier: Operating System :: OS Independent
|
|
16
16
|
Classifier: Programming Language :: Python :: 3
|
|
17
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
17
18
|
Classifier: Programming Language :: Python :: 3.12
|
|
18
19
|
Classifier: Programming Language :: Python :: 3.13
|
|
19
20
|
Classifier: Topic :: Software Development :: Quality Assurance
|
|
20
|
-
Requires-Python: >=3.
|
|
21
|
+
Requires-Python: >=3.11
|
|
21
22
|
Description-Content-Type: text/markdown
|
|
22
23
|
|
|
23
24
|
# :tophat: Mr. Overkill
|
|
@@ -72,7 +73,7 @@ You need these to participate in the madness:
|
|
|
72
73
|
|
|
73
74
|
**Runtime**:
|
|
74
75
|
|
|
75
|
-
- [Python](https://www.python.org/) 3.
|
|
76
|
+
- [Python](https://www.python.org/) 3.11+ — for the `overkill` CLI
|
|
76
77
|
- [Node.js](https://nodejs.org/) v18+ — Codex and Claude Code CLI are npm packages, so yes, you need this
|
|
77
78
|
- A fast credit card — essential
|
|
78
79
|
|
|
@@ -392,12 +393,7 @@ uv run mypy src/
|
|
|
392
393
|
## Testing
|
|
393
394
|
|
|
394
395
|
```bash
|
|
395
|
-
# Python tests
|
|
396
396
|
uv run pytest --tb=short
|
|
397
|
-
|
|
398
|
-
# Bash integration tests (requires bats-core)
|
|
399
|
-
brew install bats-core # one-time setup
|
|
400
|
-
bats test/ # run all tests
|
|
401
397
|
```
|
|
402
398
|
|
|
403
399
|
## License
|
|
@@ -50,7 +50,7 @@ You need these to participate in the madness:
|
|
|
50
50
|
|
|
51
51
|
**Runtime**:
|
|
52
52
|
|
|
53
|
-
- [Python](https://www.python.org/) 3.
|
|
53
|
+
- [Python](https://www.python.org/) 3.11+ — for the `overkill` CLI
|
|
54
54
|
- [Node.js](https://nodejs.org/) v18+ — Codex and Claude Code CLI are npm packages, so yes, you need this
|
|
55
55
|
- A fast credit card — essential
|
|
56
56
|
|
|
@@ -370,12 +370,7 @@ uv run mypy src/
|
|
|
370
370
|
## Testing
|
|
371
371
|
|
|
372
372
|
```bash
|
|
373
|
-
# Python tests
|
|
374
373
|
uv run pytest --tb=short
|
|
375
|
-
|
|
376
|
-
# Bash integration tests (requires bats-core)
|
|
377
|
-
brew install bats-core # one-time setup
|
|
378
|
-
bats test/ # run all tests
|
|
379
374
|
```
|
|
380
375
|
|
|
381
376
|
## License
|
|
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
|
|
|
35
35
|
|
|
36
36
|
```json
|
|
37
37
|
{
|
|
38
|
-
"title": "[P1] Invert dependency:
|
|
39
|
-
"body": "Currently `
|
|
38
|
+
"title": "[P1] Invert dependency: loop_engine hardcodes AI provider details",
|
|
39
|
+
"body": "Currently `src/mr_overkill/loop_engine.py` directly calls AI CLI tools with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a Protocol-based `AIProvider` interface with `call()` that encapsulates provider selection, API calls, and retries. `loop_engine.py` calls only the Protocol and doesn't know which provider is behind it.\n\n**Rollback**: If the Protocol abstraction causes issues, revert the single file and restore inline calls — the orchestration logic in loop_engine.py doesn't change.",
|
|
40
40
|
"confidence_score": 0.8,
|
|
41
41
|
"priority": 1,
|
|
42
|
-
"code_location": { "file_path": "
|
|
42
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 150, "end": 220} }
|
|
43
43
|
}
|
|
44
44
|
```
|
|
45
45
|
|
|
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
|
|
|
51
51
|
"body": "The codebase would benefit from separating concerns into Model-View-Controller layers for better organization.",
|
|
52
52
|
"confidence_score": 0.5,
|
|
53
53
|
"priority": 2,
|
|
54
|
-
"code_location": { "file_path": "
|
|
54
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
|
|
55
55
|
}
|
|
56
56
|
```
|
|
57
57
|
Why this is bad: no concrete impact statement, no evidence of current pain, suggests a pattern without demonstrating need, no phased migration.
|
|
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
|
|
|
35
35
|
|
|
36
36
|
```json
|
|
37
37
|
{
|
|
38
|
-
"title": "[P1] Unify error
|
|
39
|
-
"body": "Error
|
|
38
|
+
"title": "[P1] Unify error handling pattern across orchestration modules",
|
|
39
|
+
"body": "Error handling is done 3 different ways:\n1. `src/mr_overkill/review_loop.py` raises `SystemExit` (lines 30-35) with stderr logging\n2. `src/mr_overkill/refactor_suggest.py` uses `logger.error` + bare `sys.exit(1)` (lines 88, 142, 201)\n3. `src/mr_overkill/retry.py` raises `RuntimeError` directly (lines 45, 78)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared exception hierarchy with cleanup hooks would prevent resource leaks across all modules.",
|
|
40
40
|
"confidence_score": 0.85,
|
|
41
41
|
"priority": 1,
|
|
42
|
-
"code_location": { "file_path": "
|
|
42
|
+
"code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 30, "end": 35} }
|
|
43
43
|
}
|
|
44
44
|
```
|
|
45
45
|
|
|
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
|
|
|
51
51
|
"body": "The codebase could benefit from more consistent error handling patterns.",
|
|
52
52
|
"confidence_score": 0.5,
|
|
53
53
|
"priority": 2,
|
|
54
|
-
"code_location": { "file_path": "
|
|
54
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
|
|
55
55
|
}
|
|
56
56
|
```
|
|
57
57
|
Why this is bad: no specific examples of inconsistency, no file/line citations, doesn't explain why a coordinated change is needed.
|
|
@@ -37,10 +37,10 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
|
|
|
37
37
|
```json
|
|
38
38
|
{
|
|
39
39
|
"title": "[P1] Extract duplicated validation into shared helper in process_input()",
|
|
40
|
-
"body": "Lines
|
|
40
|
+
"body": "Lines 55-70 and 130-145 of `src/mr_overkill/loop_engine.py` contain identical iteration-result validation logic (same 3 conditions, same error messages). Extracting into a `_validate_iteration()` helper eliminates the duplication and ensures future validation changes are applied consistently.",
|
|
41
41
|
"confidence_score": 0.9,
|
|
42
42
|
"priority": 1,
|
|
43
|
-
"code_location": { "file_path": "
|
|
43
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 55, "end": 70} }
|
|
44
44
|
}
|
|
45
45
|
```
|
|
46
46
|
|
|
@@ -52,7 +52,7 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
|
|
|
52
52
|
"body": "This code could benefit from the Strategy pattern for better flexibility.",
|
|
53
53
|
"confidence_score": 0.5,
|
|
54
54
|
"priority": 3,
|
|
55
|
-
"code_location": { "file_path": "
|
|
55
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 200} }
|
|
56
56
|
}
|
|
57
57
|
```
|
|
58
58
|
Why this is bad: vague, no specific code reference, no measurable benefit, suggests an architecture-level change.
|
|
@@ -35,10 +35,10 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
|
|
|
35
35
|
```json
|
|
36
36
|
{
|
|
37
37
|
"title": "[P1] Extract repeated JSON validation into shared validate_json()",
|
|
38
|
-
"body": "The same
|
|
38
|
+
"body": "The same JSON validation logic appears in `src/example/ingest.py` (lines 12-27), `src/example/export.py` (lines 50-65), and `src/example/validate.py` (lines 8-20). All three copies parse review JSON, extract `findings`, and handle parse errors — but the error messages have already diverged. Extracting to `src/example/json_utils.py:parse_review_json()` eliminates the duplication and unifies error handling.",
|
|
39
39
|
"confidence_score": 0.85,
|
|
40
40
|
"priority": 1,
|
|
41
|
-
"code_location": { "file_path": "
|
|
41
|
+
"code_location": { "file_path": "src/example/ingest.py", "line_range": {"start": 12, "end": 27} }
|
|
42
42
|
}
|
|
43
43
|
```
|
|
44
44
|
|
|
@@ -50,7 +50,7 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
|
|
|
50
50
|
"body": "Several files handle errors similarly. Consider creating a shared error handler.",
|
|
51
51
|
"confidence_score": 0.6,
|
|
52
52
|
"priority": 2,
|
|
53
|
-
"code_location": { "file_path": "
|
|
53
|
+
"code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 1, "end": 500} }
|
|
54
54
|
}
|
|
55
55
|
```
|
|
56
56
|
Why this is bad: doesn't specify which files, doesn't cite line numbers, doesn't explain what's duplicated or how copies have diverged.
|
|
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
|
|
|
35
35
|
|
|
36
36
|
```json
|
|
37
37
|
{
|
|
38
|
-
"title": "[P1] Invert dependency:
|
|
39
|
-
"body": "Currently `
|
|
38
|
+
"title": "[P1] Invert dependency: loop_engine hardcodes AI provider details",
|
|
39
|
+
"body": "Currently `src/mr_overkill/loop_engine.py` directly calls AI CLI tools with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a Protocol-based `AIProvider` interface with `call()` that encapsulates provider selection, API calls, and retries. `loop_engine.py` calls only the Protocol and doesn't know which provider is behind it.\n\n**Rollback**: If the Protocol abstraction causes issues, revert the single file and restore inline calls — the orchestration logic in loop_engine.py doesn't change.",
|
|
40
40
|
"confidence_score": 0.8,
|
|
41
41
|
"priority": 1,
|
|
42
|
-
"code_location": { "file_path": "
|
|
42
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 150, "end": 220} }
|
|
43
43
|
}
|
|
44
44
|
```
|
|
45
45
|
|
|
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
|
|
|
51
51
|
"body": "The codebase would benefit from separating concerns into Model-View-Controller layers for better organization.",
|
|
52
52
|
"confidence_score": 0.5,
|
|
53
53
|
"priority": 2,
|
|
54
|
-
"code_location": { "file_path": "
|
|
54
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
|
|
55
55
|
}
|
|
56
56
|
```
|
|
57
57
|
Why this is bad: no concrete impact statement, no evidence of current pain, suggests a pattern without demonstrating need, no phased migration.
|
|
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
|
|
|
35
35
|
|
|
36
36
|
```json
|
|
37
37
|
{
|
|
38
|
-
"title": "[P1] Unify error
|
|
39
|
-
"body": "Error
|
|
38
|
+
"title": "[P1] Unify error handling pattern across orchestration modules",
|
|
39
|
+
"body": "Error handling is done 3 different ways:\n1. `src/mr_overkill/review_loop.py` raises `SystemExit` (lines 30-35) with stderr logging\n2. `src/mr_overkill/refactor_suggest.py` uses `logger.error` + bare `sys.exit(1)` (lines 88, 142, 201)\n3. `src/mr_overkill/retry.py` raises `RuntimeError` directly (lines 45, 78)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared exception hierarchy with cleanup hooks would prevent resource leaks across all modules.",
|
|
40
40
|
"confidence_score": 0.85,
|
|
41
41
|
"priority": 1,
|
|
42
|
-
"code_location": { "file_path": "
|
|
42
|
+
"code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 30, "end": 35} }
|
|
43
43
|
}
|
|
44
44
|
```
|
|
45
45
|
|
|
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
|
|
|
51
51
|
"body": "The codebase could benefit from more consistent error handling patterns.",
|
|
52
52
|
"confidence_score": 0.5,
|
|
53
53
|
"priority": 2,
|
|
54
|
-
"code_location": { "file_path": "
|
|
54
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
|
|
55
55
|
}
|
|
56
56
|
```
|
|
57
57
|
Why this is bad: no specific examples of inconsistency, no file/line citations, doesn't explain why a coordinated change is needed.
|
|
@@ -37,10 +37,10 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
|
|
|
37
37
|
```json
|
|
38
38
|
{
|
|
39
39
|
"title": "[P1] Extract duplicated validation into shared helper in process_input()",
|
|
40
|
-
"body": "Lines
|
|
40
|
+
"body": "Lines 55-70 and 130-145 of `src/mr_overkill/loop_engine.py` contain identical iteration-result validation logic (same 3 conditions, same error messages). Extracting into a `_validate_iteration()` helper eliminates the duplication and ensures future validation changes are applied consistently.",
|
|
41
41
|
"confidence_score": 0.9,
|
|
42
42
|
"priority": 1,
|
|
43
|
-
"code_location": { "file_path": "
|
|
43
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 55, "end": 70} }
|
|
44
44
|
}
|
|
45
45
|
```
|
|
46
46
|
|
|
@@ -52,7 +52,7 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
|
|
|
52
52
|
"body": "This code could benefit from the Strategy pattern for better flexibility.",
|
|
53
53
|
"confidence_score": 0.5,
|
|
54
54
|
"priority": 3,
|
|
55
|
-
"code_location": { "file_path": "
|
|
55
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 200} }
|
|
56
56
|
}
|
|
57
57
|
```
|
|
58
58
|
Why this is bad: vague, no specific code reference, no measurable benefit, suggests an architecture-level change.
|
|
@@ -35,10 +35,10 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
|
|
|
35
35
|
```json
|
|
36
36
|
{
|
|
37
37
|
"title": "[P1] Extract repeated JSON validation into shared validate_json()",
|
|
38
|
-
"body": "The same
|
|
38
|
+
"body": "The same JSON validation logic appears in `src/example/ingest.py` (lines 12-27), `src/example/export.py` (lines 50-65), and `src/example/validate.py` (lines 8-20). All three copies parse review JSON, extract `findings`, and handle parse errors — but the error messages have already diverged. Extracting to `src/example/json_utils.py:parse_review_json()` eliminates the duplication and unifies error handling.",
|
|
39
39
|
"confidence_score": 0.85,
|
|
40
40
|
"priority": 1,
|
|
41
|
-
"code_location": { "file_path": "
|
|
41
|
+
"code_location": { "file_path": "src/example/ingest.py", "line_range": {"start": 12, "end": 27} }
|
|
42
42
|
}
|
|
43
43
|
```
|
|
44
44
|
|
|
@@ -50,7 +50,7 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
|
|
|
50
50
|
"body": "Several files handle errors similarly. Consider creating a shared error handler.",
|
|
51
51
|
"confidence_score": 0.6,
|
|
52
52
|
"priority": 2,
|
|
53
|
-
"code_location": { "file_path": "
|
|
53
|
+
"code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 1, "end": 500} }
|
|
54
54
|
}
|
|
55
55
|
```
|
|
56
56
|
Why this is bad: doesn't specify which files, doesn't cite line numbers, doesn't explain what's duplicated or how copies have diverged.
|
|
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
|
|
|
35
35
|
|
|
36
36
|
```json
|
|
37
37
|
{
|
|
38
|
-
"title": "[P1] Invert dependency:
|
|
39
|
-
"body": "Currently `
|
|
38
|
+
"title": "[P1] Invert dependency: loop_engine hardcodes AI provider details",
|
|
39
|
+
"body": "Currently `src/mr_overkill/loop_engine.py` directly calls AI CLI tools with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a Protocol-based `AIProvider` interface with `call()` that encapsulates provider selection, API calls, and retries. `loop_engine.py` calls only the Protocol and doesn't know which provider is behind it.\n\n**Rollback**: If the Protocol abstraction causes issues, revert the single file and restore inline calls — the orchestration logic in loop_engine.py doesn't change.",
|
|
40
40
|
"confidence_score": 0.8,
|
|
41
41
|
"priority": 1,
|
|
42
|
-
"code_location": { "file_path": "
|
|
42
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 150, "end": 220} }
|
|
43
43
|
}
|
|
44
44
|
```
|
|
45
45
|
|
|
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
|
|
|
51
51
|
"body": "The codebase would benefit from separating concerns into Model-View-Controller layers for better organization.",
|
|
52
52
|
"confidence_score": 0.5,
|
|
53
53
|
"priority": 2,
|
|
54
|
-
"code_location": { "file_path": "
|
|
54
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
|
|
55
55
|
}
|
|
56
56
|
```
|
|
57
57
|
Why this is bad: no concrete impact statement, no evidence of current pain, suggests a pattern without demonstrating need, no phased migration.
|
|
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
|
|
|
35
35
|
|
|
36
36
|
```json
|
|
37
37
|
{
|
|
38
|
-
"title": "[P1] Unify error
|
|
39
|
-
"body": "Error
|
|
38
|
+
"title": "[P1] Unify error handling pattern across orchestration modules",
|
|
39
|
+
"body": "Error handling is done 3 different ways:\n1. `src/mr_overkill/review_loop.py` raises `SystemExit` (lines 30-35) with stderr logging\n2. `src/mr_overkill/refactor_suggest.py` uses `logger.error` + bare `sys.exit(1)` (lines 88, 142, 201)\n3. `src/mr_overkill/retry.py` raises `RuntimeError` directly (lines 45, 78)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared exception hierarchy with cleanup hooks would prevent resource leaks across all modules.",
|
|
40
40
|
"confidence_score": 0.85,
|
|
41
41
|
"priority": 1,
|
|
42
|
-
"code_location": { "file_path": "
|
|
42
|
+
"code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 30, "end": 35} }
|
|
43
43
|
}
|
|
44
44
|
```
|
|
45
45
|
|
|
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
|
|
|
51
51
|
"body": "The codebase could benefit from more consistent error handling patterns.",
|
|
52
52
|
"confidence_score": 0.5,
|
|
53
53
|
"priority": 2,
|
|
54
|
-
"code_location": { "file_path": "
|
|
54
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
|
|
55
55
|
}
|
|
56
56
|
```
|
|
57
57
|
Why this is bad: no specific examples of inconsistency, no file/line citations, doesn't explain why a coordinated change is needed.
|
|
@@ -37,10 +37,10 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
|
|
|
37
37
|
```json
|
|
38
38
|
{
|
|
39
39
|
"title": "[P1] Extract duplicated validation into shared helper in process_input()",
|
|
40
|
-
"body": "Lines
|
|
40
|
+
"body": "Lines 55-70 and 130-145 of `src/mr_overkill/loop_engine.py` contain identical iteration-result validation logic (same 3 conditions, same error messages). Extracting into a `_validate_iteration()` helper eliminates the duplication and ensures future validation changes are applied consistently.",
|
|
41
41
|
"confidence_score": 0.9,
|
|
42
42
|
"priority": 1,
|
|
43
|
-
"code_location": { "file_path": "
|
|
43
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 55, "end": 70} }
|
|
44
44
|
}
|
|
45
45
|
```
|
|
46
46
|
|
|
@@ -52,7 +52,7 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
|
|
|
52
52
|
"body": "This code could benefit from the Strategy pattern for better flexibility.",
|
|
53
53
|
"confidence_score": 0.5,
|
|
54
54
|
"priority": 3,
|
|
55
|
-
"code_location": { "file_path": "
|
|
55
|
+
"code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 200} }
|
|
56
56
|
}
|
|
57
57
|
```
|
|
58
58
|
Why this is bad: vague, no specific code reference, no measurable benefit, suggests an architecture-level change.
|
|
@@ -35,10 +35,10 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
|
|
|
35
35
|
```json
|
|
36
36
|
{
|
|
37
37
|
"title": "[P1] Extract repeated JSON validation into shared validate_json()",
|
|
38
|
-
"body": "The same
|
|
38
|
+
"body": "The same JSON validation logic appears in `src/example/ingest.py` (lines 12-27), `src/example/export.py` (lines 50-65), and `src/example/validate.py` (lines 8-20). All three copies parse review JSON, extract `findings`, and handle parse errors — but the error messages have already diverged. Extracting to `src/example/json_utils.py:parse_review_json()` eliminates the duplication and unifies error handling.",
|
|
39
39
|
"confidence_score": 0.85,
|
|
40
40
|
"priority": 1,
|
|
41
|
-
"code_location": { "file_path": "
|
|
41
|
+
"code_location": { "file_path": "src/example/ingest.py", "line_range": {"start": 12, "end": 27} }
|
|
42
42
|
}
|
|
43
43
|
```
|
|
44
44
|
|
|
@@ -50,7 +50,7 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
|
|
|
50
50
|
"body": "Several files handle errors similarly. Consider creating a shared error handler.",
|
|
51
51
|
"confidence_score": 0.6,
|
|
52
52
|
"priority": 2,
|
|
53
|
-
"code_location": { "file_path": "
|
|
53
|
+
"code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 1, "end": 500} }
|
|
54
54
|
}
|
|
55
55
|
```
|
|
56
56
|
Why this is bad: doesn't specify which files, doesn't cite line numbers, doesn't explain what's duplicated or how copies have diverged.
|
|
@@ -1,10 +1,10 @@
|
|
|
1
1
|
[project]
|
|
2
2
|
name = "overkill"
|
|
3
|
-
version = "0.
|
|
3
|
+
version = "0.5.0"
|
|
4
4
|
description = "AI-powered code review loop — automates Codex review + Claude fix cycles"
|
|
5
5
|
readme = "README.md"
|
|
6
6
|
license = { text = "MIT" }
|
|
7
|
-
requires-python = ">=3.
|
|
7
|
+
requires-python = ">=3.11"
|
|
8
8
|
authors = [{ name = "ModocAI" }]
|
|
9
9
|
keywords = ["code-review", "refactoring", "ai", "claude", "codex", "automation"]
|
|
10
10
|
classifiers = [
|
|
@@ -14,6 +14,7 @@ classifiers = [
|
|
|
14
14
|
"License :: OSI Approved :: MIT License",
|
|
15
15
|
"Operating System :: OS Independent",
|
|
16
16
|
"Programming Language :: Python :: 3",
|
|
17
|
+
"Programming Language :: Python :: 3.11",
|
|
17
18
|
"Programming Language :: Python :: 3.12",
|
|
18
19
|
"Programming Language :: Python :: 3.13",
|
|
19
20
|
"Topic :: Software Development :: Quality Assurance",
|
|
@@ -44,14 +45,14 @@ testpaths = ["tests"]
|
|
|
44
45
|
pythonpath = ["src"]
|
|
45
46
|
|
|
46
47
|
[tool.ruff]
|
|
47
|
-
target-version = "
|
|
48
|
+
target-version = "py311"
|
|
48
49
|
src = ["src", "tests"]
|
|
49
50
|
|
|
50
51
|
[tool.ruff.lint]
|
|
51
52
|
select = ["E", "F", "I", "UP", "B", "SIM", "RUF"]
|
|
52
53
|
|
|
53
54
|
[tool.mypy]
|
|
54
|
-
python_version = "3.
|
|
55
|
+
python_version = "3.11"
|
|
55
56
|
strict = true
|
|
56
57
|
warn_return_any = true
|
|
57
58
|
warn_unused_configs = true
|
|
@@ -38,6 +38,13 @@ from mr_overkill.two_step_fix import claude_two_step_fix
|
|
|
38
38
|
logger = logging.getLogger(__name__)
|
|
39
39
|
|
|
40
40
|
|
|
41
|
+
def _format_reviewer_context(raw: str) -> str:
|
|
42
|
+
"""Wrap non-empty reviewer context in a markdown section."""
|
|
43
|
+
if not raw:
|
|
44
|
+
return ""
|
|
45
|
+
return f"## Author Context\n\n{raw}"
|
|
46
|
+
|
|
47
|
+
|
|
41
48
|
# ── Budget / retry helpers (moved from review_loop.py) ───────────────
|
|
42
49
|
|
|
43
50
|
|
|
@@ -167,6 +174,7 @@ class CodexReviewAgent(ReviewAgent):
|
|
|
167
174
|
"CURRENT_BRANCH": config.current_branch,
|
|
168
175
|
"TARGET_BRANCH": config.target_branch,
|
|
169
176
|
"ITERATION": str(iteration),
|
|
177
|
+
"REVIEWER_CONTEXT": _format_reviewer_context(config.reviewer_context),
|
|
170
178
|
})
|
|
171
179
|
|
|
172
180
|
if not self._budget_fn("codex", config.budget_scope, 0):
|
|
@@ -268,6 +276,7 @@ class ClaudeReviewAgent(ReviewAgent):
|
|
|
268
276
|
"CURRENT_BRANCH": config.current_branch,
|
|
269
277
|
"TARGET_BRANCH": config.target_branch,
|
|
270
278
|
"ITERATION": str(iteration),
|
|
279
|
+
"REVIEWER_CONTEXT": _format_reviewer_context(config.reviewer_context),
|
|
271
280
|
})
|
|
272
281
|
|
|
273
282
|
if not self._budget_fn("claude", config.budget_scope, 0):
|
|
@@ -365,6 +374,7 @@ class GeminiReviewAgent(ReviewAgent):
|
|
|
365
374
|
"CURRENT_BRANCH": config.current_branch,
|
|
366
375
|
"TARGET_BRANCH": config.target_branch,
|
|
367
376
|
"ITERATION": str(iteration),
|
|
377
|
+
"REVIEWER_CONTEXT": _format_reviewer_context(config.reviewer_context),
|
|
368
378
|
})
|
|
369
379
|
|
|
370
380
|
if not self._budget_fn("gemini", config.budget_scope, 0):
|
|
@@ -108,13 +108,14 @@ def _load_rc_file(rc_name: str) -> dict[str, str]:
|
|
|
108
108
|
"RETRY_INITIAL_WAIT", "BUDGET_SCOPE", "DIAGNOSTIC_LOG",
|
|
109
109
|
"SCOPE", "AUTO_APPROVE", "CREATE_PR", "WITH_REVIEW",
|
|
110
110
|
"REVIEW_LOOPS", "FIX_NITS", "REVIEWER_BACKEND",
|
|
111
|
+
"REVIEWER_CONTEXT",
|
|
111
112
|
}
|
|
112
113
|
boolean_keys = {
|
|
113
114
|
"DRY_RUN", "AUTO_COMMIT", "DIAGNOSTIC_LOG",
|
|
114
115
|
"AUTO_APPROVE", "CREATE_PR", "WITH_REVIEW", "FIX_NITS",
|
|
115
116
|
}
|
|
116
117
|
kv_re = re.compile(
|
|
117
|
-
r"^\s*(\w+)=[
|
|
118
|
+
r"""^\s*(\w+)=(?:"([^"]*)"|'([^']*)'|(.*?))\s*$"""
|
|
118
119
|
)
|
|
119
120
|
values: dict[str, str] = {}
|
|
120
121
|
|
|
@@ -124,7 +125,7 @@ def _load_rc_file(rc_name: str) -> dict[str, str]:
|
|
|
124
125
|
continue
|
|
125
126
|
m = kv_re.match(line)
|
|
126
127
|
if m and m.group(1) in allowed_keys:
|
|
127
|
-
key, val = m.group(1), m.group(2).strip()
|
|
128
|
+
key, val = m.group(1), (m.group(2) or m.group(3) or m.group(4) or "").strip()
|
|
128
129
|
if key in boolean_keys and val.lower() not in ("true", "false"):
|
|
129
130
|
msg = f"{rc_path.name}: {key} must be 'true' or 'false', got '{val}'."
|
|
130
131
|
raise SystemExit(f"Error: {msg}")
|
|
@@ -299,6 +300,11 @@ def parse_review_loop_args(
|
|
|
299
300
|
choices=["claude", "codex", "gemini"],
|
|
300
301
|
help="Backend for code review (default: codex)",
|
|
301
302
|
)
|
|
303
|
+
parser.add_argument(
|
|
304
|
+
"--context",
|
|
305
|
+
default=None,
|
|
306
|
+
help="Additional context for the reviewer (e.g. design intent, constraints)",
|
|
307
|
+
)
|
|
302
308
|
|
|
303
309
|
args = parser.parse_args(argv)
|
|
304
310
|
|
|
@@ -394,6 +400,10 @@ def parse_review_loop_args(
|
|
|
394
400
|
saved = log_dir / "reviewer-backend.txt"
|
|
395
401
|
if saved.is_file():
|
|
396
402
|
args.reviewer_backend = saved.read_text().strip()
|
|
403
|
+
if args.context is None:
|
|
404
|
+
saved = log_dir / "reviewer-context.txt"
|
|
405
|
+
if saved.is_file():
|
|
406
|
+
args.context = saved.read_text().strip()
|
|
397
407
|
|
|
398
408
|
if max_loop is not None and max_loop < 1:
|
|
399
409
|
parser.error("--max-loop must be a positive integer")
|
|
@@ -405,6 +415,10 @@ def parse_review_loop_args(
|
|
|
405
415
|
f" got {reviewer_backend!r}"
|
|
406
416
|
)
|
|
407
417
|
|
|
418
|
+
reviewer_context = (
|
|
419
|
+
args.context if args.context is not None else rc.get("REVIEWER_CONTEXT", "")
|
|
420
|
+
)
|
|
421
|
+
|
|
408
422
|
return LoopConfig(
|
|
409
423
|
current_branch=current_branch,
|
|
410
424
|
target_branch=target,
|
|
@@ -425,6 +439,7 @@ def parse_review_loop_args(
|
|
|
425
439
|
prompts_dir=prompts_dir,
|
|
426
440
|
pr_number=pr_number,
|
|
427
441
|
reviewer_backend=reviewer_backend,
|
|
442
|
+
reviewer_context=reviewer_context,
|
|
428
443
|
)
|
|
429
444
|
|
|
430
445
|
|
|
@@ -1,8 +1,5 @@
|
|
|
1
1
|
"""Unified review-fix loop engine.
|
|
2
2
|
|
|
3
|
-
Consolidates the common loop pattern from review-loop.sh,
|
|
4
|
-
refactor-suggest.sh, and self-review.sh into a single reusable engine.
|
|
5
|
-
|
|
6
3
|
Uses Protocol-based DI for external operations (fix, review, budget)
|
|
7
4
|
and imports Wave 1 modules directly for JSON parsing, git ops, and reporting.
|
|
8
5
|
"""
|
|
@@ -613,7 +610,7 @@ def _no_diff(target: str, current: str, cwd: Path | None) -> bool:
|
|
|
613
610
|
def _clean_stale_logs(log_dir: Path) -> None:
|
|
614
611
|
"""Remove iteration artifacts from prior runs.
|
|
615
612
|
|
|
616
|
-
|
|
613
|
+
Cleanup stale log files from previous runs so that fresh runs
|
|
617
614
|
do not mix stale review/fix/summary files into new results.
|
|
618
615
|
"""
|
|
619
616
|
patterns = [
|
|
@@ -638,6 +635,7 @@ def _save_metadata(config: LoopConfig, cwd: Path | None) -> None:
|
|
|
638
635
|
(log_dir / "target-branch.txt").write_text(config.target_branch)
|
|
639
636
|
(log_dir / "max-loop.txt").write_text(str(config.max_loop))
|
|
640
637
|
(log_dir / "reviewer-backend.txt").write_text(config.reviewer_backend)
|
|
638
|
+
(log_dir / "reviewer-context.txt").write_text(config.reviewer_context)
|
|
641
639
|
if config.scope:
|
|
642
640
|
(log_dir / "scope.txt").write_text(config.scope)
|
|
643
641
|
|