overkill 0.4.0__tar.gz → 0.4.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (95) hide show
  1. {overkill-0.4.0 → overkill-0.4.1}/AGENTS.md +1 -1
  2. {overkill-0.4.0 → overkill-0.4.1}/PKG-INFO +1 -6
  3. {overkill-0.4.0 → overkill-0.4.1}/README.md +0 -5
  4. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-refactor-full.prompt.md +4 -4
  5. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-refactor-layer.prompt.md +4 -4
  6. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-refactor-micro.prompt.md +3 -3
  7. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-refactor-module.prompt.md +3 -3
  8. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/codex-refactor-full.prompt.md +4 -4
  9. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/codex-refactor-layer.prompt.md +4 -4
  10. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/codex-refactor-micro.prompt.md +3 -3
  11. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/codex-refactor-module.prompt.md +3 -3
  12. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/gemini-refactor-full.prompt.md +4 -4
  13. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/gemini-refactor-layer.prompt.md +4 -4
  14. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/gemini-refactor-micro.prompt.md +3 -3
  15. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/gemini-refactor-module.prompt.md +3 -3
  16. {overkill-0.4.0 → overkill-0.4.1}/pyproject.toml +1 -1
  17. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/classify.py +1 -4
  18. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/loop_engine.py +1 -4
  19. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/refactor_suggest.py +2 -3
  20. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/retry.py +1 -4
  21. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/review_loop.py +1 -5
  22. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/self_review.py +0 -1
  23. overkill-0.4.0/bin/lib/check-claude-limit.sh +0 -256
  24. overkill-0.4.0/bin/lib/check-codex-limit.sh +0 -202
  25. overkill-0.4.0/bin/lib/common.sh +0 -511
  26. overkill-0.4.0/bin/lib/retry.sh +0 -306
  27. overkill-0.4.0/bin/lib/self-review.sh +0 -527
  28. overkill-0.4.0/bin/refactor-suggest.sh +0 -748
  29. overkill-0.4.0/bin/review-loop.sh +0 -593
  30. overkill-0.4.0/test/refactor-suggest.bats +0 -159
  31. overkill-0.4.0/test/test_helper.bash +0 -117
  32. {overkill-0.4.0 → overkill-0.4.1}/.github/workflows/publish.yml +0 -0
  33. {overkill-0.4.0 → overkill-0.4.1}/.github/workflows/test.yml +0 -0
  34. {overkill-0.4.0 → overkill-0.4.1}/.gitignore +0 -0
  35. {overkill-0.4.0 → overkill-0.4.1}/.overkillrc.example +0 -0
  36. {overkill-0.4.0 → overkill-0.4.1}/.python-version +0 -0
  37. {overkill-0.4.0 → overkill-0.4.1}/.refactorsuggestrc.example +0 -0
  38. {overkill-0.4.0 → overkill-0.4.1}/CLAUDE.md +0 -0
  39. {overkill-0.4.0 → overkill-0.4.1}/GEMINI.md +0 -0
  40. {overkill-0.4.0 → overkill-0.4.1}/LICENSE +0 -0
  41. {overkill-0.4.0 → overkill-0.4.1}/install.sh +0 -0
  42. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-fix-execute.prompt.md +0 -0
  43. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-fix.prompt.md +0 -0
  44. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-refactor-fix-execute.prompt.md +0 -0
  45. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-refactor-fix.prompt.md +0 -0
  46. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-review.prompt.md +0 -0
  47. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/claude-self-review.prompt.md +0 -0
  48. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/codex-review.prompt.md +0 -0
  49. {overkill-0.4.0 → overkill-0.4.1}/prompts/active/gemini-review.prompt.md +0 -0
  50. {overkill-0.4.0 → overkill-0.4.1}/prompts/reference/claude-code-review-plugin.md +0 -0
  51. {overkill-0.4.0 → overkill-0.4.1}/prompts/reference/claude-security-review.md +0 -0
  52. {overkill-0.4.0 → overkill-0.4.1}/prompts/reference/codex-review-original.md +0 -0
  53. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/__init__.py +0 -0
  54. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/__main__.py +0 -0
  55. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/agents.py +0 -0
  56. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/budget/__init__.py +0 -0
  57. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/budget/claude.py +0 -0
  58. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/budget/codex.py +0 -0
  59. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/budget/gemini.py +0 -0
  60. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/budget_report.py +0 -0
  61. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/cli.py +0 -0
  62. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/data/__init__.py +0 -0
  63. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/git_ops.py +0 -0
  64. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/init.py +0 -0
  65. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/json_extract.py +0 -0
  66. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/models.py +0 -0
  67. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/reporting.py +0 -0
  68. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/resume.py +0 -0
  69. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/time_utils.py +0 -0
  70. {overkill-0.4.0 → overkill-0.4.1}/src/mr_overkill/two_step_fix.py +0 -0
  71. {overkill-0.4.0 → overkill-0.4.1}/tests/__init__.py +0 -0
  72. {overkill-0.4.0 → overkill-0.4.1}/tests/conftest.py +0 -0
  73. {overkill-0.4.0 → overkill-0.4.1}/tests/test_agents.py +0 -0
  74. {overkill-0.4.0 → overkill-0.4.1}/tests/test_budget_claude.py +0 -0
  75. {overkill-0.4.0 → overkill-0.4.1}/tests/test_budget_codex.py +0 -0
  76. {overkill-0.4.0 → overkill-0.4.1}/tests/test_budget_gemini.py +0 -0
  77. {overkill-0.4.0 → overkill-0.4.1}/tests/test_budget_policy.py +0 -0
  78. {overkill-0.4.0 → overkill-0.4.1}/tests/test_budget_report.py +0 -0
  79. {overkill-0.4.0 → overkill-0.4.1}/tests/test_classify.py +0 -0
  80. {overkill-0.4.0 → overkill-0.4.1}/tests/test_cli.py +0 -0
  81. {overkill-0.4.0 → overkill-0.4.1}/tests/test_git_ops.py +0 -0
  82. {overkill-0.4.0 → overkill-0.4.1}/tests/test_init.py +0 -0
  83. {overkill-0.4.0 → overkill-0.4.1}/tests/test_integration.py +0 -0
  84. {overkill-0.4.0 → overkill-0.4.1}/tests/test_json_extract.py +0 -0
  85. {overkill-0.4.0 → overkill-0.4.1}/tests/test_loop_engine.py +0 -0
  86. {overkill-0.4.0 → overkill-0.4.1}/tests/test_refactor_suggest.py +0 -0
  87. {overkill-0.4.0 → overkill-0.4.1}/tests/test_reporting.py +0 -0
  88. {overkill-0.4.0 → overkill-0.4.1}/tests/test_resume.py +0 -0
  89. {overkill-0.4.0 → overkill-0.4.1}/tests/test_retry.py +0 -0
  90. {overkill-0.4.0 → overkill-0.4.1}/tests/test_review_loop.py +0 -0
  91. {overkill-0.4.0 → overkill-0.4.1}/tests/test_self_review.py +0 -0
  92. {overkill-0.4.0 → overkill-0.4.1}/tests/test_time_utils.py +0 -0
  93. {overkill-0.4.0 → overkill-0.4.1}/tests/test_two_step_fix.py +0 -0
  94. {overkill-0.4.0 → overkill-0.4.1}/uninstall.sh +0 -0
  95. {overkill-0.4.0 → overkill-0.4.1}/uv.lock +0 -0
@@ -9,7 +9,7 @@
9
9
 
10
10
  ## Pull Request Rules
11
11
 
12
- Every PR must pass the review loop (`review-loop.sh --dry-run`) before merging. No exceptions. We eat our own dog food — if Mr. Overkill can't approve it, neither can you.
12
+ Every PR must pass the review loop (`overkill review-loop --dry-run`) before merging. No exceptions. We eat our own dog food — if Mr. Overkill can't approve it, neither can you.
13
13
 
14
14
  ## Branch Rules
15
15
 
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: overkill
3
- Version: 0.4.0
3
+ Version: 0.4.1
4
4
  Summary: AI-powered code review loop — automates Codex review + Claude fix cycles
5
5
  Project-URL: Repository, https://github.com/modocai/mr-overkill
6
6
  Project-URL: Issues, https://github.com/modocai/mr-overkill/issues
@@ -392,12 +392,7 @@ uv run mypy src/
392
392
  ## Testing
393
393
 
394
394
  ```bash
395
- # Python tests
396
395
  uv run pytest --tb=short
397
-
398
- # Bash integration tests (requires bats-core)
399
- brew install bats-core # one-time setup
400
- bats test/ # run all tests
401
396
  ```
402
397
 
403
398
  ## License
@@ -370,12 +370,7 @@ uv run mypy src/
370
370
  ## Testing
371
371
 
372
372
  ```bash
373
- # Python tests
374
373
  uv run pytest --tb=short
375
-
376
- # Bash integration tests (requires bats-core)
377
- brew install bats-core # one-time setup
378
- bats test/ # run all tests
379
374
  ```
380
375
 
381
376
  ## License
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
35
35
 
36
36
  ```json
37
37
  {
38
- "title": "[P1] Invert dependency: review-loop.sh hardcodes AI provider details",
39
- "body": "Currently `bin/review-loop.sh` directly calls OpenAI/Claude APIs with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a `lib/ai-provider.sh` interface with `call_ai()` that encapsulates provider selection, API calls, and retries. `review-loop.sh` calls only `call_ai()` and doesn't know which provider is behind it.\n\n**Rollback**: If `lib/ai-provider.sh` causes issues, revert the single file and restore inline calls — the orchestration logic in review-loop.sh doesn't change.",
38
+ "title": "[P1] Invert dependency: loop_engine hardcodes AI provider details",
39
+ "body": "Currently `src/mr_overkill/loop_engine.py` directly calls AI CLI tools with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a Protocol-based `AIProvider` interface with `call()` that encapsulates provider selection, API calls, and retries. `loop_engine.py` calls only the Protocol and doesn't know which provider is behind it.\n\n**Rollback**: If the Protocol abstraction causes issues, revert the single file and restore inline calls — the orchestration logic in loop_engine.py doesn't change.",
40
40
  "confidence_score": 0.8,
41
41
  "priority": 1,
42
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 150, "end": 220} }
42
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 150, "end": 220} }
43
43
  }
44
44
  ```
45
45
 
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
51
51
  "body": "The codebase would benefit from separating concerns into Model-View-Controller layers for better organization.",
52
52
  "confidence_score": 0.5,
53
53
  "priority": 2,
54
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 700} }
54
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
55
55
  }
56
56
  ```
57
57
  Why this is bad: no concrete impact statement, no evidence of current pain, suggests a pattern without demonstrating need, no phased migration.
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
35
35
 
36
36
  ```json
37
37
  {
38
- "title": "[P1] Unify error exit pattern across all bin/ scripts",
39
- "body": "Error exits are handled 3 different ways:\n1. `bin/review-loop.sh` uses `die()` (lines 25-28) which logs to stderr and exits 1\n2. `bin/refactor-suggest.sh` uses `log_error` + bare `exit 1` (lines 88, 142, 201)\n3. `bin/apply-fix.sh` calls `echo \"ERROR: ...\" >&2` directly (lines 33, 67)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared `die()` that includes cleanup would prevent resource leaks across all scripts.",
38
+ "title": "[P1] Unify error handling pattern across orchestration modules",
39
+ "body": "Error handling is done 3 different ways:\n1. `src/mr_overkill/review_loop.py` raises `SystemExit` (lines 30-35) with stderr logging\n2. `src/mr_overkill/refactor_suggest.py` uses `logger.error` + bare `sys.exit(1)` (lines 88, 142, 201)\n3. `src/mr_overkill/retry.py` raises `RuntimeError` directly (lines 45, 78)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared exception hierarchy with cleanup hooks would prevent resource leaks across all modules.",
40
40
  "confidence_score": 0.85,
41
41
  "priority": 1,
42
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 25, "end": 28} }
42
+ "code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 30, "end": 35} }
43
43
  }
44
44
  ```
45
45
 
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
51
51
  "body": "The codebase could benefit from more consistent error handling patterns.",
52
52
  "confidence_score": 0.5,
53
53
  "priority": 2,
54
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 700} }
54
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
55
55
  }
56
56
  ```
57
57
  Why this is bad: no specific examples of inconsistency, no file/line citations, doesn't explain why a coordinated change is needed.
@@ -37,10 +37,10 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
37
37
  ```json
38
38
  {
39
39
  "title": "[P1] Extract duplicated validation into shared helper in process_input()",
40
- "body": "Lines 42-58 and 103-119 of `bin/review-loop.sh` contain identical input validation logic (same 3 conditions, same error messages). Extracting into a `validate_input()` function eliminates the duplication and ensures future validation changes are applied consistently.",
40
+ "body": "Lines 55-70 and 130-145 of `src/mr_overkill/loop_engine.py` contain identical iteration-result validation logic (same 3 conditions, same error messages). Extracting into a `_validate_iteration()` helper eliminates the duplication and ensures future validation changes are applied consistently.",
41
41
  "confidence_score": 0.9,
42
42
  "priority": 1,
43
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 42, "end": 58} }
43
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 55, "end": 70} }
44
44
  }
45
45
  ```
46
46
 
@@ -52,7 +52,7 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
52
52
  "body": "This code could benefit from the Strategy pattern for better flexibility.",
53
53
  "confidence_score": 0.5,
54
54
  "priority": 3,
55
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 200} }
55
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 200} }
56
56
  }
57
57
  ```
58
58
  Why this is bad: vague, no specific code reference, no measurable benefit, suggests an architecture-level change.
@@ -35,10 +35,10 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
35
35
  ```json
36
36
  {
37
37
  "title": "[P1] Extract repeated JSON validation into shared validate_json()",
38
- "body": "The same jq-based JSON validation logic appears in `bin/review-loop.sh` (lines 120-135), `bin/refactor-suggest.sh` (lines 45-60), and `bin/apply-fix.sh` (lines 30-42). All three copies check for valid JSON, extract `.findings`, and handle parse errors — but the error messages have already diverged (review-loop prints to stderr, the others use `log_error`). Extracting to `lib/json-utils.sh:validate_json()` eliminates the duplication and unifies error handling.",
38
+ "body": "The same JSON validation logic appears in `src/example/ingest.py` (lines 12-27), `src/example/export.py` (lines 50-65), and `src/example/validate.py` (lines 8-20). All three copies parse review JSON, extract `findings`, and handle parse errors — but the error messages have already diverged. Extracting to `src/example/json_utils.py:parse_review_json()` eliminates the duplication and unifies error handling.",
39
39
  "confidence_score": 0.85,
40
40
  "priority": 1,
41
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 120, "end": 135} }
41
+ "code_location": { "file_path": "src/example/ingest.py", "line_range": {"start": 12, "end": 27} }
42
42
  }
43
43
  ```
44
44
 
@@ -50,7 +50,7 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
50
50
  "body": "Several files handle errors similarly. Consider creating a shared error handler.",
51
51
  "confidence_score": 0.6,
52
52
  "priority": 2,
53
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 500} }
53
+ "code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 1, "end": 500} }
54
54
  }
55
55
  ```
56
56
  Why this is bad: doesn't specify which files, doesn't cite line numbers, doesn't explain what's duplicated or how copies have diverged.
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
35
35
 
36
36
  ```json
37
37
  {
38
- "title": "[P1] Invert dependency: review-loop.sh hardcodes AI provider details",
39
- "body": "Currently `bin/review-loop.sh` directly calls OpenAI/Claude APIs with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a `lib/ai-provider.sh` interface with `call_ai()` that encapsulates provider selection, API calls, and retries. `review-loop.sh` calls only `call_ai()` and doesn't know which provider is behind it.\n\n**Rollback**: If `lib/ai-provider.sh` causes issues, revert the single file and restore inline calls — the orchestration logic in review-loop.sh doesn't change.",
38
+ "title": "[P1] Invert dependency: loop_engine hardcodes AI provider details",
39
+ "body": "Currently `src/mr_overkill/loop_engine.py` directly calls AI CLI tools with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a Protocol-based `AIProvider` interface with `call()` that encapsulates provider selection, API calls, and retries. `loop_engine.py` calls only the Protocol and doesn't know which provider is behind it.\n\n**Rollback**: If the Protocol abstraction causes issues, revert the single file and restore inline calls — the orchestration logic in loop_engine.py doesn't change.",
40
40
  "confidence_score": 0.8,
41
41
  "priority": 1,
42
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 150, "end": 220} }
42
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 150, "end": 220} }
43
43
  }
44
44
  ```
45
45
 
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
51
51
  "body": "The codebase would benefit from separating concerns into Model-View-Controller layers for better organization.",
52
52
  "confidence_score": 0.5,
53
53
  "priority": 2,
54
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 700} }
54
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
55
55
  }
56
56
  ```
57
57
  Why this is bad: no concrete impact statement, no evidence of current pain, suggests a pattern without demonstrating need, no phased migration.
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
35
35
 
36
36
  ```json
37
37
  {
38
- "title": "[P1] Unify error exit pattern across all bin/ scripts",
39
- "body": "Error exits are handled 3 different ways:\n1. `bin/review-loop.sh` uses `die()` (lines 25-28) which logs to stderr and exits 1\n2. `bin/refactor-suggest.sh` uses `log_error` + bare `exit 1` (lines 88, 142, 201)\n3. `bin/apply-fix.sh` calls `echo \"ERROR: ...\" >&2` directly (lines 33, 67)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared `die()` that includes cleanup would prevent resource leaks across all scripts.",
38
+ "title": "[P1] Unify error handling pattern across orchestration modules",
39
+ "body": "Error handling is done 3 different ways:\n1. `src/mr_overkill/review_loop.py` raises `SystemExit` (lines 30-35) with stderr logging\n2. `src/mr_overkill/refactor_suggest.py` uses `logger.error` + bare `sys.exit(1)` (lines 88, 142, 201)\n3. `src/mr_overkill/retry.py` raises `RuntimeError` directly (lines 45, 78)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared exception hierarchy with cleanup hooks would prevent resource leaks across all modules.",
40
40
  "confidence_score": 0.85,
41
41
  "priority": 1,
42
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 25, "end": 28} }
42
+ "code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 30, "end": 35} }
43
43
  }
44
44
  ```
45
45
 
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
51
51
  "body": "The codebase could benefit from more consistent error handling patterns.",
52
52
  "confidence_score": 0.5,
53
53
  "priority": 2,
54
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 700} }
54
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
55
55
  }
56
56
  ```
57
57
  Why this is bad: no specific examples of inconsistency, no file/line citations, doesn't explain why a coordinated change is needed.
@@ -37,10 +37,10 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
37
37
  ```json
38
38
  {
39
39
  "title": "[P1] Extract duplicated validation into shared helper in process_input()",
40
- "body": "Lines 42-58 and 103-119 of `bin/review-loop.sh` contain identical input validation logic (same 3 conditions, same error messages). Extracting into a `validate_input()` function eliminates the duplication and ensures future validation changes are applied consistently.",
40
+ "body": "Lines 55-70 and 130-145 of `src/mr_overkill/loop_engine.py` contain identical iteration-result validation logic (same 3 conditions, same error messages). Extracting into a `_validate_iteration()` helper eliminates the duplication and ensures future validation changes are applied consistently.",
41
41
  "confidence_score": 0.9,
42
42
  "priority": 1,
43
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 42, "end": 58} }
43
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 55, "end": 70} }
44
44
  }
45
45
  ```
46
46
 
@@ -52,7 +52,7 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
52
52
  "body": "This code could benefit from the Strategy pattern for better flexibility.",
53
53
  "confidence_score": 0.5,
54
54
  "priority": 3,
55
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 200} }
55
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 200} }
56
56
  }
57
57
  ```
58
58
  Why this is bad: vague, no specific code reference, no measurable benefit, suggests an architecture-level change.
@@ -35,10 +35,10 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
35
35
  ```json
36
36
  {
37
37
  "title": "[P1] Extract repeated JSON validation into shared validate_json()",
38
- "body": "The same jq-based JSON validation logic appears in `bin/review-loop.sh` (lines 120-135), `bin/refactor-suggest.sh` (lines 45-60), and `bin/apply-fix.sh` (lines 30-42). All three copies check for valid JSON, extract `.findings`, and handle parse errors — but the error messages have already diverged (review-loop prints to stderr, the others use `log_error`). Extracting to `lib/json-utils.sh:validate_json()` eliminates the duplication and unifies error handling.",
38
+ "body": "The same JSON validation logic appears in `src/example/ingest.py` (lines 12-27), `src/example/export.py` (lines 50-65), and `src/example/validate.py` (lines 8-20). All three copies parse review JSON, extract `findings`, and handle parse errors — but the error messages have already diverged. Extracting to `src/example/json_utils.py:parse_review_json()` eliminates the duplication and unifies error handling.",
39
39
  "confidence_score": 0.85,
40
40
  "priority": 1,
41
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 120, "end": 135} }
41
+ "code_location": { "file_path": "src/example/ingest.py", "line_range": {"start": 12, "end": 27} }
42
42
  }
43
43
  ```
44
44
 
@@ -50,7 +50,7 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
50
50
  "body": "Several files handle errors similarly. Consider creating a shared error handler.",
51
51
  "confidence_score": 0.6,
52
52
  "priority": 2,
53
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 500} }
53
+ "code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 1, "end": 500} }
54
54
  }
55
55
  ```
56
56
  Why this is bad: doesn't specify which files, doesn't cite line numbers, doesn't explain what's duplicated or how copies have diverged.
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
35
35
 
36
36
  ```json
37
37
  {
38
- "title": "[P1] Invert dependency: review-loop.sh hardcodes AI provider details",
39
- "body": "Currently `bin/review-loop.sh` directly calls OpenAI/Claude APIs with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a `lib/ai-provider.sh` interface with `call_ai()` that encapsulates provider selection, API calls, and retries. `review-loop.sh` calls only `call_ai()` and doesn't know which provider is behind it.\n\n**Rollback**: If `lib/ai-provider.sh` causes issues, revert the single file and restore inline calls — the orchestration logic in review-loop.sh doesn't change.",
38
+ "title": "[P1] Invert dependency: loop_engine hardcodes AI provider details",
39
+ "body": "Currently `src/mr_overkill/loop_engine.py` directly calls AI CLI tools with provider-specific logic scattered across lines 150-220, 340-380, and 450-490. This means:\n- Adding a new AI provider requires modifying 3 sections of a 700-line file\n- Testing with a mock provider is impossible without editing production code\n- Provider-specific retry/error logic is interleaved with review orchestration\n\n**Suggested structure**: Extract a Protocol-based `AIProvider` interface with `call()` that encapsulates provider selection, API calls, and retries. `loop_engine.py` calls only the Protocol and doesn't know which provider is behind it.\n\n**Rollback**: If the Protocol abstraction causes issues, revert the single file and restore inline calls — the orchestration logic in loop_engine.py doesn't change.",
40
40
  "confidence_score": 0.8,
41
41
  "priority": 1,
42
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 150, "end": 220} }
42
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 150, "end": 220} }
43
43
  }
44
44
  ```
45
45
 
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **architecture-le
51
51
  "body": "The codebase would benefit from separating concerns into Model-View-Controller layers for better organization.",
52
52
  "confidence_score": 0.5,
53
53
  "priority": 2,
54
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 700} }
54
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
55
55
  }
56
56
  ```
57
57
  Why this is bad: no concrete impact statement, no evidence of current pain, suggests a pattern without demonstrating need, no phased migration.
@@ -35,11 +35,11 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
35
35
 
36
36
  ```json
37
37
  {
38
- "title": "[P1] Unify error exit pattern across all bin/ scripts",
39
- "body": "Error exits are handled 3 different ways:\n1. `bin/review-loop.sh` uses `die()` (lines 25-28) which logs to stderr and exits 1\n2. `bin/refactor-suggest.sh` uses `log_error` + bare `exit 1` (lines 88, 142, 201)\n3. `bin/apply-fix.sh` calls `echo \"ERROR: ...\" >&2` directly (lines 33, 67)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared `die()` that includes cleanup would prevent resource leaks across all scripts.",
38
+ "title": "[P1] Unify error handling pattern across orchestration modules",
39
+ "body": "Error handling is done 3 different ways:\n1. `src/mr_overkill/review_loop.py` raises `SystemExit` (lines 30-35) with stderr logging\n2. `src/mr_overkill/refactor_suggest.py` uses `logger.error` + bare `sys.exit(1)` (lines 88, 142, 201)\n3. `src/mr_overkill/retry.py` raises `RuntimeError` directly (lines 45, 78)\n\nThis makes it easy to miss cleanup (temp file removal) on error paths. A coordinated change to use a shared exception hierarchy with cleanup hooks would prevent resource leaks across all modules.",
40
40
  "confidence_score": 0.85,
41
41
  "priority": 1,
42
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 25, "end": 28} }
42
+ "code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 30, "end": 35} }
43
43
  }
44
44
  ```
45
45
 
@@ -51,7 +51,7 @@ You are a refactoring advisor analyzing an entire codebase for **layer-level** (
51
51
  "body": "The codebase could benefit from more consistent error handling patterns.",
52
52
  "confidence_score": 0.5,
53
53
  "priority": 2,
54
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 700} }
54
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 700} }
55
55
  }
56
56
  ```
57
57
  Why this is bad: no specific examples of inconsistency, no file/line citations, doesn't explain why a coordinated change is needed.
@@ -37,10 +37,10 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
37
37
  ```json
38
38
  {
39
39
  "title": "[P1] Extract duplicated validation into shared helper in process_input()",
40
- "body": "Lines 42-58 and 103-119 of `bin/review-loop.sh` contain identical input validation logic (same 3 conditions, same error messages). Extracting into a `validate_input()` function eliminates the duplication and ensures future validation changes are applied consistently.",
40
+ "body": "Lines 55-70 and 130-145 of `src/mr_overkill/loop_engine.py` contain identical iteration-result validation logic (same 3 conditions, same error messages). Extracting into a `_validate_iteration()` helper eliminates the duplication and ensures future validation changes are applied consistently.",
41
41
  "confidence_score": 0.9,
42
42
  "priority": 1,
43
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 42, "end": 58} }
43
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 55, "end": 70} }
44
44
  }
45
45
  ```
46
46
 
@@ -52,7 +52,7 @@ You are a refactoring advisor analyzing an entire codebase for **micro-level** i
52
52
  "body": "This code could benefit from the Strategy pattern for better flexibility.",
53
53
  "confidence_score": 0.5,
54
54
  "priority": 3,
55
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 200} }
55
+ "code_location": { "file_path": "src/mr_overkill/loop_engine.py", "line_range": {"start": 1, "end": 200} }
56
56
  }
57
57
  ```
58
58
  Why this is bad: vague, no specific code reference, no measurable benefit, suggests an architecture-level change.
@@ -35,10 +35,10 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
35
35
  ```json
36
36
  {
37
37
  "title": "[P1] Extract repeated JSON validation into shared validate_json()",
38
- "body": "The same jq-based JSON validation logic appears in `bin/review-loop.sh` (lines 120-135), `bin/refactor-suggest.sh` (lines 45-60), and `bin/apply-fix.sh` (lines 30-42). All three copies check for valid JSON, extract `.findings`, and handle parse errors — but the error messages have already diverged (review-loop prints to stderr, the others use `log_error`). Extracting to `lib/json-utils.sh:validate_json()` eliminates the duplication and unifies error handling.",
38
+ "body": "The same JSON validation logic appears in `src/example/ingest.py` (lines 12-27), `src/example/export.py` (lines 50-65), and `src/example/validate.py` (lines 8-20). All three copies parse review JSON, extract `findings`, and handle parse errors — but the error messages have already diverged. Extracting to `src/example/json_utils.py:parse_review_json()` eliminates the duplication and unifies error handling.",
39
39
  "confidence_score": 0.85,
40
40
  "priority": 1,
41
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 120, "end": 135} }
41
+ "code_location": { "file_path": "src/example/ingest.py", "line_range": {"start": 12, "end": 27} }
42
42
  }
43
43
  ```
44
44
 
@@ -50,7 +50,7 @@ You are a refactoring advisor analyzing an entire codebase for **module-level**
50
50
  "body": "Several files handle errors similarly. Consider creating a shared error handler.",
51
51
  "confidence_score": 0.6,
52
52
  "priority": 2,
53
- "code_location": { "file_path": "bin/review-loop.sh", "line_range": {"start": 1, "end": 500} }
53
+ "code_location": { "file_path": "src/mr_overkill/review_loop.py", "line_range": {"start": 1, "end": 500} }
54
54
  }
55
55
  ```
56
56
  Why this is bad: doesn't specify which files, doesn't cite line numbers, doesn't explain what's duplicated or how copies have diverged.
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "overkill"
3
- version = "0.4.0"
3
+ version = "0.4.1"
4
4
  description = "AI-powered code review loop — automates Codex review + Claude fix cycles"
5
5
  readme = "README.md"
6
6
  license = { text = "MIT" }
@@ -1,7 +1,4 @@
1
- """Error classification for CLI command failures.
2
-
3
- Ports ``_classify_cli_error`` from ``bin/lib/retry.sh`` to Python.
4
- """
1
+ """Error classification for CLI command failures."""
5
2
 
6
3
  from __future__ import annotations
7
4
 
@@ -1,8 +1,5 @@
1
1
  """Unified review-fix loop engine.
2
2
 
3
- Consolidates the common loop pattern from review-loop.sh,
4
- refactor-suggest.sh, and self-review.sh into a single reusable engine.
5
-
6
3
  Uses Protocol-based DI for external operations (fix, review, budget)
7
4
  and imports Wave 1 modules directly for JSON parsing, git ops, and reporting.
8
5
  """
@@ -613,7 +610,7 @@ def _no_diff(target: str, current: str, cwd: Path | None) -> bool:
613
610
  def _clean_stale_logs(log_dir: Path) -> None:
614
611
  """Remove iteration artifacts from prior runs.
615
612
 
616
- Mirrors the cleanup in ``bin/review-loop.sh`` so that fresh runs
613
+ Cleanup stale log files from previous runs so that fresh runs
617
614
  do not mix stale review/fix/summary files into new results.
618
615
  """
619
616
  patterns = [
@@ -1,8 +1,7 @@
1
1
  """Refactor-suggest entry point.
2
2
 
3
- Ports ``bin/refactor-suggest.sh`` to Python: budget-aware scope resolution,
4
- branch creation, analysis loop, draft PR creation, and optional review-loop
5
- chaining.
3
+ Budget-aware scope resolution, branch creation, analysis loop, draft PR
4
+ creation, and optional review-loop chaining.
6
5
  """
7
6
 
8
7
  from __future__ import annotations
@@ -1,7 +1,4 @@
1
- """Retry-with-backoff wrappers for Claude/Codex CLI calls.
2
-
3
- Ports retry logic from ``bin/lib/retry.sh`` to Python.
4
- """
1
+ """Retry-with-backoff wrappers for Claude/Codex CLI calls."""
5
2
 
6
3
  from __future__ import annotations
7
4
 
@@ -1,8 +1,4 @@
1
- """Review-loop entry point — wires Protocol implementations to loop_engine.
2
-
3
- Ports the argument parsing and orchestration from ``bin/review-loop.sh``,
4
- delegating the actual loop to :func:`loop_engine.review_fix_loop`.
5
- """
1
+ """Review-loop entry point — wires Protocol implementations to loop_engine."""
6
2
 
7
3
  from __future__ import annotations
8
4
 
@@ -1,6 +1,5 @@
1
1
  """Self-review sub-loop for verifying and re-fixing Claude's changes.
2
2
 
3
- Ports ``_self_review_subloop`` from ``bin/lib/self-review.sh``.
4
3
  Implements the :class:`SelfReviewFn` Protocol from ``loop_engine``.
5
4
  """
6
5