overkill 0.6.1__tar.gz → 0.8.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {overkill-0.6.1 → overkill-0.8.0}/.overkillrc.example +8 -0
- {overkill-0.6.1 → overkill-0.8.0}/PKG-INFO +7 -1
- {overkill-0.6.1 → overkill-0.8.0}/README.md +6 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-review.prompt.md +1 -1
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/codex-review.prompt.md +1 -1
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/gemini-review.prompt.md +1 -1
- {overkill-0.6.1 → overkill-0.8.0}/pyproject.toml +1 -1
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/agents.py +56 -11
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/cli.py +30 -1
- overkill-0.8.0/src/mr_overkill/data/review.schema.json +76 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/git_ops.py +43 -6
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/json_extract.py +30 -18
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/loop_engine.py +50 -1
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/models.py +9 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/retry.py +12 -5
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/review_loop.py +13 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_agents.py +68 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_cli.py +97 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_git_ops.py +75 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_json_extract.py +44 -2
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_loop_engine.py +120 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_retry.py +11 -0
- overkill-0.8.0/tests/test_review_loop.py +225 -0
- {overkill-0.6.1 → overkill-0.8.0}/uv.lock +1 -1
- overkill-0.6.1/tests/test_review_loop.py +0 -66
- {overkill-0.6.1 → overkill-0.8.0}/.github/workflows/publish.yml +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/.github/workflows/test.yml +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/.gitignore +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/.python-version +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/.refactorsuggestrc.example +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/AGENTS.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/CLAUDE.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/GEMINI.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/LICENSE +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/install.sh +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-fix-execute.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-fix.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-refactor-fix-execute.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-refactor-fix.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-refactor-full.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-refactor-layer.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-refactor-micro.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-refactor-module.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/claude-self-review.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/codex-refactor-full.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/codex-refactor-layer.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/codex-refactor-micro.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/codex-refactor-module.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/gemini-refactor-full.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/gemini-refactor-layer.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/gemini-refactor-micro.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/active/gemini-refactor-module.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/reference/claude-code-review-plugin.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/reference/claude-security-review.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/prompts/reference/codex-review-original.md +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/__init__.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/__main__.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/budget/__init__.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/budget/claude.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/budget/codex.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/budget/gemini.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/budget_report.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/classify.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/data/__init__.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/init.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/refactor_suggest.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/reporting.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/resume.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/self_review.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/time_utils.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/src/mr_overkill/two_step_fix.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/__init__.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/conftest.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_budget_claude.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_budget_codex.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_budget_gemini.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_budget_policy.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_budget_report.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_classify.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_init.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_integration.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_refactor_suggest.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_reporting.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_resume.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_self_review.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_time_utils.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/tests/test_two_step_fix.py +0 -0
- {overkill-0.6.1 → overkill-0.8.0}/uninstall.sh +0 -0
|
@@ -17,6 +17,14 @@
|
|
|
17
17
|
# Enable auto-commit of fixes (default: true)
|
|
18
18
|
# AUTO_COMMIT=true
|
|
19
19
|
|
|
20
|
+
# CI trigger policy for iteration commits (default: last-only)
|
|
21
|
+
# last-only — append "[skip ci]" to each iteration commit; push a single
|
|
22
|
+
# empty "chore: trigger CI" commit only when the loop ends
|
|
23
|
+
# with all_clear (saves CI cost on long iteration runs)
|
|
24
|
+
# every — each iteration commit triggers CI (pre-0.3 behaviour)
|
|
25
|
+
# none — append "[skip ci]"; no trigger commit (forks / CI-less repos)
|
|
26
|
+
# CI_TRIGGER_MODE="last-only"
|
|
27
|
+
|
|
20
28
|
# Path to active prompt templates (default: <script_dir>/../prompts/active)
|
|
21
29
|
# PROMPTS_DIR="./custom-prompts"
|
|
22
30
|
|
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: overkill
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.8.0
|
|
4
4
|
Summary: AI-powered code review loop — automates Codex/Gemini review + Claude fix cycles
|
|
5
5
|
Project-URL: Repository, https://github.com/modocai/mr-overkill
|
|
6
6
|
Project-URL: Issues, https://github.com/modocai/mr-overkill/issues
|
|
@@ -125,6 +125,11 @@ Options:
|
|
|
125
125
|
--no-auto-commit Fix but do not commit/push (single iteration)
|
|
126
126
|
--resume Resume from a previously interrupted run (reuses existing logs)
|
|
127
127
|
--reviewer-backend <be> Reviewer backend: claude|codex (default: codex)
|
|
128
|
+
--ci-trigger-mode <m> CI trigger policy: every|last-only|none (default: last-only).
|
|
129
|
+
'last-only' tags each iteration commit with [skip ci]
|
|
130
|
+
and pushes a single empty trigger commit on PASS —
|
|
131
|
+
CI runs once instead of once per iteration.
|
|
132
|
+
Use 'every' to restore pre-0.3 per-commit CI.
|
|
128
133
|
--diagnostic-log Save full Claude event stream to sidecar files
|
|
129
134
|
|
|
130
135
|
Examples:
|
|
@@ -134,6 +139,7 @@ Examples:
|
|
|
134
139
|
overkill review-loop -n 3 --no-self-review # disable self-review sub-loop
|
|
135
140
|
overkill review-loop --resume # resume an interrupted run
|
|
136
141
|
overkill review-loop -n 2 --reviewer-backend claude # use Claude as reviewer
|
|
142
|
+
overkill review-loop -n 10 --ci-trigger-mode last-only # CI fires once on PASS
|
|
137
143
|
```
|
|
138
144
|
|
|
139
145
|
## Usage: overkill refactor-suggest
|
|
@@ -102,6 +102,11 @@ Options:
|
|
|
102
102
|
--no-auto-commit Fix but do not commit/push (single iteration)
|
|
103
103
|
--resume Resume from a previously interrupted run (reuses existing logs)
|
|
104
104
|
--reviewer-backend <be> Reviewer backend: claude|codex (default: codex)
|
|
105
|
+
--ci-trigger-mode <m> CI trigger policy: every|last-only|none (default: last-only).
|
|
106
|
+
'last-only' tags each iteration commit with [skip ci]
|
|
107
|
+
and pushes a single empty trigger commit on PASS —
|
|
108
|
+
CI runs once instead of once per iteration.
|
|
109
|
+
Use 'every' to restore pre-0.3 per-commit CI.
|
|
105
110
|
--diagnostic-log Save full Claude event stream to sidecar files
|
|
106
111
|
|
|
107
112
|
Examples:
|
|
@@ -111,6 +116,7 @@ Examples:
|
|
|
111
116
|
overkill review-loop -n 3 --no-self-review # disable self-review sub-loop
|
|
112
117
|
overkill review-loop --resume # resume an interrupted run
|
|
113
118
|
overkill review-loop -n 2 --reviewer-backend claude # use Claude as reviewer
|
|
119
|
+
overkill review-loop -n 10 --ci-trigger-mode last-only # CI fires once on PASS
|
|
114
120
|
```
|
|
115
121
|
|
|
116
122
|
## Usage: overkill refactor-suggest
|
|
@@ -62,7 +62,7 @@ We only want findings where you are confident the issue is real. **If you are no
|
|
|
62
62
|
|
|
63
63
|
## Output Format
|
|
64
64
|
|
|
65
|
-
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
|
|
65
|
+
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
|
|
66
66
|
|
|
67
67
|
{
|
|
68
68
|
"findings": [
|
|
@@ -53,7 +53,7 @@ Output all findings that the original author would fix if they knew about them.
|
|
|
53
53
|
|
|
54
54
|
## Output Format
|
|
55
55
|
|
|
56
|
-
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
|
|
56
|
+
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
|
|
57
57
|
|
|
58
58
|
{
|
|
59
59
|
"findings": [
|
|
@@ -62,7 +62,7 @@ We only want findings where you are confident the issue is real. **If you are no
|
|
|
62
62
|
|
|
63
63
|
## Output Format
|
|
64
64
|
|
|
65
|
-
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
|
|
65
|
+
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
|
|
66
66
|
|
|
67
67
|
{
|
|
68
68
|
"findings": [
|
|
@@ -13,6 +13,7 @@ import logging
|
|
|
13
13
|
import string
|
|
14
14
|
import subprocess
|
|
15
15
|
from abc import ABC, abstractmethod
|
|
16
|
+
from importlib.resources import as_file, files
|
|
16
17
|
from pathlib import Path
|
|
17
18
|
|
|
18
19
|
from mr_overkill.budget.claude import claude_budget_sufficient
|
|
@@ -45,6 +46,41 @@ def _format_reviewer_context(raw: str) -> str:
|
|
|
45
46
|
return f"## Author Context\n\n{raw}"
|
|
46
47
|
|
|
47
48
|
|
|
49
|
+
# ── Review schema (single source of truth for structured output) ─────
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _review_schema_text() -> str:
|
|
53
|
+
"""Read the bundled review JSON Schema as a string."""
|
|
54
|
+
return files("mr_overkill.data").joinpath("review.schema.json").read_text(
|
|
55
|
+
encoding="utf-8"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _unwrap_claude_structured_output(output_path: Path) -> None:
|
|
60
|
+
"""Replace Claude's ``--output-format json`` wrapper with its inner schema object.
|
|
61
|
+
|
|
62
|
+
Claude CLI emits ``{"type":"result", ..., "structured_output": {...}, ...}``
|
|
63
|
+
when both ``--output-format json`` and ``--json-schema`` are set. The
|
|
64
|
+
schema-conforming object lives under ``structured_output``. Downstream
|
|
65
|
+
consumers expect the file to contain that object directly. If the file does
|
|
66
|
+
not match the wrapper shape (older Claude versions, error responses), it is
|
|
67
|
+
left untouched so the parser's fallback tiers can still try.
|
|
68
|
+
"""
|
|
69
|
+
try:
|
|
70
|
+
raw = output_path.read_text(encoding="utf-8")
|
|
71
|
+
except OSError:
|
|
72
|
+
return
|
|
73
|
+
try:
|
|
74
|
+
wrapper = json.loads(raw)
|
|
75
|
+
except json.JSONDecodeError:
|
|
76
|
+
return
|
|
77
|
+
if not isinstance(wrapper, dict):
|
|
78
|
+
return
|
|
79
|
+
inner = wrapper.get("structured_output")
|
|
80
|
+
if isinstance(inner, dict):
|
|
81
|
+
output_path.write_text(json.dumps(inner), encoding="utf-8")
|
|
82
|
+
|
|
83
|
+
|
|
48
84
|
# ── Budget / retry helpers (moved from review_loop.py) ───────────────
|
|
49
85
|
|
|
50
86
|
|
|
@@ -183,16 +219,20 @@ class CodexReviewAgent(ReviewAgent):
|
|
|
183
219
|
)
|
|
184
220
|
|
|
185
221
|
stderr_path = output_path.with_suffix(".stderr")
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
"
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
222
|
+
with as_file(
|
|
223
|
+
files("mr_overkill.data").joinpath("review.schema.json")
|
|
224
|
+
) as schema_path:
|
|
225
|
+
return retry_codex_cmd(
|
|
226
|
+
stderr_path,
|
|
227
|
+
"Codex review",
|
|
228
|
+
[
|
|
229
|
+
"codex", "exec", "--sandbox", "read-only",
|
|
230
|
+
"--output-schema", str(schema_path),
|
|
231
|
+
"-o", str(output_path), prompt_text,
|
|
232
|
+
],
|
|
233
|
+
max_wait=config.retry_max_wait,
|
|
234
|
+
initial_wait=config.retry_initial_wait,
|
|
235
|
+
)
|
|
196
236
|
|
|
197
237
|
|
|
198
238
|
class CodexRefactorReviewAgent(ReviewAgent):
|
|
@@ -284,15 +324,20 @@ class ClaudeReviewAgent(ReviewAgent):
|
|
|
284
324
|
f"Claude budget timeout (iteration {iteration})."
|
|
285
325
|
)
|
|
286
326
|
|
|
287
|
-
|
|
327
|
+
ok = self._retry_fn(
|
|
288
328
|
output_path,
|
|
289
329
|
"Claude review",
|
|
290
330
|
[
|
|
291
331
|
"claude", "-p", "-",
|
|
292
332
|
"--allowedTools", "Bash,Read,Glob,Grep",
|
|
333
|
+
"--output-format", "json",
|
|
334
|
+
"--json-schema", _review_schema_text(),
|
|
293
335
|
],
|
|
294
336
|
stdin=prompt_text,
|
|
295
337
|
)
|
|
338
|
+
if ok:
|
|
339
|
+
_unwrap_claude_structured_output(output_path)
|
|
340
|
+
return ok
|
|
296
341
|
|
|
297
342
|
|
|
298
343
|
class ClaudeRefactorReviewAgent(ReviewAgent):
|
|
@@ -108,7 +108,7 @@ def _load_rc_file(rc_name: str) -> dict[str, str]:
|
|
|
108
108
|
"RETRY_INITIAL_WAIT", "BUDGET_SCOPE", "DIAGNOSTIC_LOG",
|
|
109
109
|
"SCOPE", "AUTO_APPROVE", "CREATE_PR", "WITH_REVIEW",
|
|
110
110
|
"REVIEW_LOOPS", "FIX_NITS", "REVIEWER_BACKEND",
|
|
111
|
-
"REVIEWER_CONTEXT",
|
|
111
|
+
"REVIEWER_CONTEXT", "CI_TRIGGER_MODE",
|
|
112
112
|
}
|
|
113
113
|
boolean_keys = {
|
|
114
114
|
"DRY_RUN", "AUTO_COMMIT", "DIAGNOSTIC_LOG",
|
|
@@ -306,6 +306,18 @@ def parse_review_loop_args(
|
|
|
306
306
|
default=None,
|
|
307
307
|
help="Additional context for the reviewer (e.g. design intent, constraints)",
|
|
308
308
|
)
|
|
309
|
+
parser.add_argument(
|
|
310
|
+
"--ci-trigger-mode",
|
|
311
|
+
default=None,
|
|
312
|
+
choices=["every", "last-only", "none"],
|
|
313
|
+
help=(
|
|
314
|
+
"CI trigger policy for iteration commits. "
|
|
315
|
+
"'last-only' (default): append [skip ci] to iteration commits and "
|
|
316
|
+
"push a single empty 'chore: trigger CI' commit only on PASS. "
|
|
317
|
+
"'every': every commit triggers CI. "
|
|
318
|
+
"'none': append [skip ci] with no trigger commit."
|
|
319
|
+
),
|
|
320
|
+
)
|
|
309
321
|
|
|
310
322
|
args = parser.parse_args(argv)
|
|
311
323
|
|
|
@@ -405,6 +417,10 @@ def parse_review_loop_args(
|
|
|
405
417
|
saved = log_dir / "reviewer-context.txt"
|
|
406
418
|
if saved.is_file():
|
|
407
419
|
args.context = saved.read_text().strip()
|
|
420
|
+
if args.ci_trigger_mode is None:
|
|
421
|
+
saved = log_dir / "ci-trigger-mode.txt"
|
|
422
|
+
if saved.is_file():
|
|
423
|
+
args.ci_trigger_mode = saved.read_text().strip()
|
|
408
424
|
|
|
409
425
|
if max_loop is not None and max_loop < 1:
|
|
410
426
|
parser.error("--max-loop must be a positive integer")
|
|
@@ -420,6 +436,17 @@ def parse_review_loop_args(
|
|
|
420
436
|
args.context if args.context is not None else rc.get("REVIEWER_CONTEXT", "")
|
|
421
437
|
)
|
|
422
438
|
|
|
439
|
+
ci_trigger_mode = (
|
|
440
|
+
args.ci_trigger_mode
|
|
441
|
+
if args.ci_trigger_mode is not None
|
|
442
|
+
else rc.get("CI_TRIGGER_MODE", "last-only")
|
|
443
|
+
)
|
|
444
|
+
if ci_trigger_mode not in ("every", "last-only", "none"):
|
|
445
|
+
parser.error(
|
|
446
|
+
f"CI_TRIGGER_MODE must be 'every', 'last-only', or 'none',"
|
|
447
|
+
f" got {ci_trigger_mode!r}"
|
|
448
|
+
)
|
|
449
|
+
|
|
423
450
|
return LoopConfig(
|
|
424
451
|
current_branch=current_branch,
|
|
425
452
|
target_branch=target,
|
|
@@ -441,6 +468,7 @@ def parse_review_loop_args(
|
|
|
441
468
|
pr_number=pr_number,
|
|
442
469
|
reviewer_backend=reviewer_backend,
|
|
443
470
|
reviewer_context=reviewer_context,
|
|
471
|
+
ci_trigger_mode=ci_trigger_mode,
|
|
444
472
|
)
|
|
445
473
|
|
|
446
474
|
|
|
@@ -713,6 +741,7 @@ def parse_refactor_suggest_args(
|
|
|
713
741
|
prompts_dir=prompts_dir,
|
|
714
742
|
scope=scope,
|
|
715
743
|
reviewer_backend=reviewer_backend,
|
|
744
|
+
ci_trigger_mode="every",
|
|
716
745
|
)
|
|
717
746
|
|
|
718
747
|
extra = _RefactorExtra(
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "ReviewResult",
|
|
4
|
+
"type": "object",
|
|
5
|
+
"required": [
|
|
6
|
+
"findings",
|
|
7
|
+
"overall_correctness",
|
|
8
|
+
"overall_explanation",
|
|
9
|
+
"overall_confidence_score"
|
|
10
|
+
],
|
|
11
|
+
"additionalProperties": false,
|
|
12
|
+
"properties": {
|
|
13
|
+
"findings": {
|
|
14
|
+
"type": "array",
|
|
15
|
+
"items": {
|
|
16
|
+
"type": "object",
|
|
17
|
+
"required": ["title", "body", "confidence_score", "priority", "code_location"],
|
|
18
|
+
"additionalProperties": false,
|
|
19
|
+
"properties": {
|
|
20
|
+
"title": {
|
|
21
|
+
"type": "string",
|
|
22
|
+
"maxLength": 80,
|
|
23
|
+
"description": "P-tag + imperative description (max 80 chars)"
|
|
24
|
+
},
|
|
25
|
+
"body": {
|
|
26
|
+
"type": "string",
|
|
27
|
+
"description": "Markdown explanation; cite files/lines/functions"
|
|
28
|
+
},
|
|
29
|
+
"confidence_score": {
|
|
30
|
+
"type": "number",
|
|
31
|
+
"minimum": 0,
|
|
32
|
+
"maximum": 1
|
|
33
|
+
},
|
|
34
|
+
"priority": {
|
|
35
|
+
"type": "integer",
|
|
36
|
+
"minimum": 0,
|
|
37
|
+
"maximum": 3
|
|
38
|
+
},
|
|
39
|
+
"code_location": {
|
|
40
|
+
"type": "object",
|
|
41
|
+
"required": ["file_path", "line_range"],
|
|
42
|
+
"additionalProperties": false,
|
|
43
|
+
"properties": {
|
|
44
|
+
"file_path": {
|
|
45
|
+
"type": "string",
|
|
46
|
+
"description": "Repo-relative file path"
|
|
47
|
+
},
|
|
48
|
+
"line_range": {
|
|
49
|
+
"type": "object",
|
|
50
|
+
"required": ["start", "end"],
|
|
51
|
+
"additionalProperties": false,
|
|
52
|
+
"properties": {
|
|
53
|
+
"start": {"type": "integer", "minimum": 0},
|
|
54
|
+
"end": {"type": "integer", "minimum": 0}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
},
|
|
62
|
+
"overall_correctness": {
|
|
63
|
+
"type": "string",
|
|
64
|
+
"enum": ["patch is correct", "patch is incorrect"]
|
|
65
|
+
},
|
|
66
|
+
"overall_explanation": {
|
|
67
|
+
"type": "string",
|
|
68
|
+
"description": "1-3 sentence justification"
|
|
69
|
+
},
|
|
70
|
+
"overall_confidence_score": {
|
|
71
|
+
"type": "number",
|
|
72
|
+
"minimum": 0,
|
|
73
|
+
"maximum": 1
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
}
|
|
@@ -234,7 +234,12 @@ def commit_and_push(
|
|
|
234
234
|
|
|
235
235
|
logger.info("Committed.")
|
|
236
236
|
|
|
237
|
-
|
|
237
|
+
_push_current_branch(branch=branch, cwd=cwd)
|
|
238
|
+
return True
|
|
239
|
+
|
|
240
|
+
|
|
241
|
+
def _push_current_branch(branch: str = "", cwd: Path | None = None) -> None:
|
|
242
|
+
"""Push the current branch, setting upstream on first push if needed."""
|
|
238
243
|
upstream_check = _run(
|
|
239
244
|
["git", "rev-parse", "--abbrev-ref", "--symbolic-full-name", "@{u}"],
|
|
240
245
|
cwd=cwd,
|
|
@@ -244,7 +249,8 @@ def commit_and_push(
|
|
|
244
249
|
if push_result.returncode != 0:
|
|
245
250
|
raise RuntimeError(f"git push failed: {push_result.stderr.strip()}")
|
|
246
251
|
logger.info("Pushed.")
|
|
247
|
-
|
|
252
|
+
return
|
|
253
|
+
if branch:
|
|
248
254
|
remote_check = _run(["git", "remote"], cwd=cwd)
|
|
249
255
|
remotes = remote_check.stdout.strip().splitlines()
|
|
250
256
|
remote = "origin" if "origin" in remotes else (remotes[0] if remotes else "")
|
|
@@ -253,9 +259,40 @@ def commit_and_push(
|
|
|
253
259
|
if push_result.returncode != 0:
|
|
254
260
|
raise RuntimeError(f"git push failed: {push_result.stderr.strip()}")
|
|
255
261
|
logger.info("Pushed (upstream set).")
|
|
256
|
-
|
|
257
|
-
|
|
258
|
-
|
|
259
|
-
|
|
262
|
+
return
|
|
263
|
+
logger.info("No upstream/remote set — skipping push.")
|
|
264
|
+
return
|
|
265
|
+
logger.info("No upstream set — skipping push.")
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def push_trigger_commit(branch: str = "", cwd: Path | None = None) -> bool:
|
|
269
|
+
"""Push an empty ``chore: trigger CI`` commit.
|
|
270
|
+
|
|
271
|
+
Used by ``--ci-trigger-mode last-only`` to fire CI once after a run of
|
|
272
|
+
iteration commits that were each tagged with ``[skip ci]``. Returns
|
|
273
|
+
``True`` on success, ``False`` if the empty commit could not be created
|
|
274
|
+
(e.g. repo has no HEAD yet). Raises ``RuntimeError`` if ``git push`` fails.
|
|
275
|
+
|
|
276
|
+
Idempotent: if HEAD already lacks ``[skip ci]`` (a prior call succeeded
|
|
277
|
+
locally but the push may not have reached the remote), skip creating
|
|
278
|
+
another empty commit and just retry the push.
|
|
279
|
+
"""
|
|
280
|
+
head_msg = _run(["git", "log", "-1", "--pretty=%B"], cwd=cwd).stdout
|
|
281
|
+
if head_msg and "[skip ci]" not in head_msg:
|
|
282
|
+
logger.info("HEAD already triggers CI — retrying push only.")
|
|
283
|
+
_push_current_branch(branch=branch, cwd=cwd)
|
|
284
|
+
return True
|
|
260
285
|
|
|
286
|
+
result = _run(
|
|
287
|
+
["git", "commit", "--allow-empty", "-m", "chore: trigger CI"],
|
|
288
|
+
cwd=cwd,
|
|
289
|
+
)
|
|
290
|
+
if result.returncode != 0:
|
|
291
|
+
logger.warning(
|
|
292
|
+
"trigger commit failed: %s",
|
|
293
|
+
result.stderr.strip() or result.stdout.strip(),
|
|
294
|
+
)
|
|
295
|
+
return False
|
|
296
|
+
logger.info("Created CI trigger commit.")
|
|
297
|
+
_push_current_branch(branch=branch, cwd=cwd)
|
|
261
298
|
return True
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""JSON extraction utilities ported from common.sh and self-review.sh.
|
|
2
2
|
|
|
3
3
|
Provides a 3-tier extraction pipeline (direct parse -> fenced code block ->
|
|
4
|
-
|
|
4
|
+
balanced-brace scan) that mirrors _extract_json_from_file in the shell codebase,
|
|
5
5
|
plus helpers for path normalisation and refactoring-plan injection.
|
|
6
6
|
"""
|
|
7
7
|
|
|
@@ -24,7 +24,9 @@ def extract_json_from_file(path: Path) -> dict[str, Any] | None:
|
|
|
24
24
|
Tier 2: Look for a fenced code block (`` ```json ... ``` ``) and parse its
|
|
25
25
|
content.
|
|
26
26
|
|
|
27
|
-
Tier 3:
|
|
27
|
+
Tier 3: Scan every ``{`` position with ``json.JSONDecoder().raw_decode()``
|
|
28
|
+
and return the largest valid dict found. Robust to prose with embedded
|
|
29
|
+
braces (e.g. preambles that mention schemas).
|
|
28
30
|
|
|
29
31
|
Returns
|
|
30
32
|
-------
|
|
@@ -53,8 +55,8 @@ def extract_json_from_file(path: Path) -> dict[str, Any] | None:
|
|
|
53
55
|
if result is not None:
|
|
54
56
|
return result
|
|
55
57
|
|
|
56
|
-
# ── Tier 3:
|
|
57
|
-
return
|
|
58
|
+
# ── Tier 3: balanced-brace scan ───────────────────────────────────
|
|
59
|
+
return _try_balanced_scan(content)
|
|
58
60
|
|
|
59
61
|
|
|
60
62
|
def parse_review_json(
|
|
@@ -188,18 +190,28 @@ def _try_fenced_block(content: str) -> dict[str, Any] | None:
|
|
|
188
190
|
return None
|
|
189
191
|
|
|
190
192
|
|
|
191
|
-
|
|
193
|
+
def _try_balanced_scan(content: str) -> dict[str, Any] | None:
|
|
194
|
+
"""Tier 3: try ``raw_decode`` from each ``{`` position, keep the largest dict.
|
|
192
195
|
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
196
|
+
Greedy regex (``\\{.*\\}``) breaks when prose preceding the JSON contains
|
|
197
|
+
its own ``{`` — the match starts at the prose brace and consumes the rest
|
|
198
|
+
of the file as one invalid string. ``raw_decode`` walks the JSON grammar
|
|
199
|
+
and stops at the matching brace, so each candidate is parsed correctly.
|
|
200
|
+
"""
|
|
201
|
+
decoder = json.JSONDecoder()
|
|
202
|
+
best: dict[str, Any] | None = None
|
|
203
|
+
best_len = 0
|
|
204
|
+
idx = 0
|
|
205
|
+
while True:
|
|
206
|
+
idx = content.find("{", idx)
|
|
207
|
+
if idx < 0:
|
|
208
|
+
break
|
|
209
|
+
try:
|
|
210
|
+
obj, end = decoder.raw_decode(content, idx)
|
|
211
|
+
except json.JSONDecodeError:
|
|
212
|
+
idx += 1
|
|
213
|
+
continue
|
|
214
|
+
if isinstance(obj, dict) and (end - idx) > best_len:
|
|
215
|
+
best, best_len = obj, end - idx
|
|
216
|
+
idx = end if end > idx else idx + 1
|
|
217
|
+
return best
|
|
@@ -218,10 +218,20 @@ def review_fix_loop(
|
|
|
218
218
|
# Already-completed runs can short-circuit after branch validation
|
|
219
219
|
if state.status == "completed":
|
|
220
220
|
resolved_status = FinalStatus(state.prev_status or "max_iterations_reached")
|
|
221
|
+
# Re-attempt the CI trigger push: a prior run may have written
|
|
222
|
+
# the terminal summary but failed (or been interrupted) before
|
|
223
|
+
# pushing the trigger commit, leaving the remote on a [skip ci]
|
|
224
|
+
# commit. push_trigger_commit is idempotent.
|
|
225
|
+
needs_trigger = (
|
|
226
|
+
resolved_status == FinalStatus.ALL_CLEAR
|
|
227
|
+
and config.ci_trigger_mode in ("last-only", "none")
|
|
228
|
+
and _has_skipped_fix_commit(commit_pattern, log_dir, cwd)
|
|
229
|
+
)
|
|
221
230
|
return LoopResult(
|
|
222
231
|
final_status=resolved_status,
|
|
223
232
|
iterations_run=0,
|
|
224
233
|
summary_path=_generate_summary_safe(config, resolved_status),
|
|
234
|
+
made_skipped_fix_commit=needs_trigger,
|
|
225
235
|
)
|
|
226
236
|
|
|
227
237
|
resume_from = state.resume_from
|
|
@@ -285,6 +295,14 @@ def review_fix_loop(
|
|
|
285
295
|
allowed_stashed = False
|
|
286
296
|
had_findings = False
|
|
287
297
|
fix_committed = False
|
|
298
|
+
# On resume, prior iteration commits already exist; in last-only/none
|
|
299
|
+
# modes those carry [skip ci], so a final trigger commit is still needed
|
|
300
|
+
# even if no new fix commit is made in this process.
|
|
301
|
+
made_skipped_fix_commit = (
|
|
302
|
+
config.resume
|
|
303
|
+
and resume_from > 1
|
|
304
|
+
and config.ci_trigger_mode in ("last-only", "none")
|
|
305
|
+
)
|
|
288
306
|
|
|
289
307
|
for i in range(1, config.max_loop + 1):
|
|
290
308
|
logger.info("── Iteration %d / %d ──", i, config.max_loop)
|
|
@@ -489,8 +507,11 @@ def review_fix_loop(
|
|
|
489
507
|
|
|
490
508
|
# j. Commit & push
|
|
491
509
|
if config.auto_commit:
|
|
510
|
+
subject = f"{commit_pattern} {i} fixes"
|
|
511
|
+
if config.ci_trigger_mode in ("last-only", "none"):
|
|
512
|
+
subject += " [skip ci]"
|
|
492
513
|
commit_msg = (
|
|
493
|
-
f"{
|
|
514
|
+
f"{subject}\n\n"
|
|
494
515
|
f"Auto-generated by review loop (iteration {i}/{config.max_loop})"
|
|
495
516
|
)
|
|
496
517
|
if self_review_summary:
|
|
@@ -509,6 +530,8 @@ def review_fix_loop(
|
|
|
509
530
|
final_status = FinalStatus.COMMIT_PUSH_ERROR
|
|
510
531
|
iterations_run = i
|
|
511
532
|
break
|
|
533
|
+
if fix_committed and config.ci_trigger_mode in ("last-only", "none"):
|
|
534
|
+
made_skipped_fix_commit = True
|
|
512
535
|
else:
|
|
513
536
|
logger.info("AUTO_COMMIT is disabled — skipping commit and push.")
|
|
514
537
|
finally:
|
|
@@ -552,12 +575,37 @@ def review_fix_loop(
|
|
|
552
575
|
final_status=final_status,
|
|
553
576
|
iterations_run=iterations_run,
|
|
554
577
|
summary_path=summary_path,
|
|
578
|
+
made_skipped_fix_commit=made_skipped_fix_commit,
|
|
555
579
|
)
|
|
556
580
|
|
|
557
581
|
|
|
558
582
|
# ── Private helpers ──────────────────────────────────────────────────
|
|
559
583
|
|
|
560
584
|
|
|
585
|
+
def _has_skipped_fix_commit(
|
|
586
|
+
commit_pattern: str, log_dir: Path, cwd: Path | None
|
|
587
|
+
) -> bool:
|
|
588
|
+
"""Return True if any [skip ci] fix commit exists since the run started."""
|
|
589
|
+
start_file = log_dir / "start-commit.txt"
|
|
590
|
+
if not start_file.is_file():
|
|
591
|
+
return False
|
|
592
|
+
start = start_file.read_text().strip()
|
|
593
|
+
if not start:
|
|
594
|
+
return False
|
|
595
|
+
result = subprocess.run(
|
|
596
|
+
[
|
|
597
|
+
"git", "log", "--fixed-strings",
|
|
598
|
+
f"--grep={commit_pattern}", "--grep=[skip ci]", "--all-match",
|
|
599
|
+
"--oneline", f"{start}..HEAD",
|
|
600
|
+
],
|
|
601
|
+
cwd=cwd,
|
|
602
|
+
capture_output=True,
|
|
603
|
+
text=True,
|
|
604
|
+
check=False,
|
|
605
|
+
)
|
|
606
|
+
return result.returncode == 0 and bool(result.stdout.strip())
|
|
607
|
+
|
|
608
|
+
|
|
561
609
|
def _validate_target_branch(target: str, cwd: Path | None) -> bool:
|
|
562
610
|
"""Return True if *target* is a valid git ref."""
|
|
563
611
|
result = subprocess.run(
|
|
@@ -636,6 +684,7 @@ def _save_metadata(config: LoopConfig, cwd: Path | None) -> None:
|
|
|
636
684
|
(log_dir / "max-loop.txt").write_text(str(config.max_loop))
|
|
637
685
|
(log_dir / "reviewer-backend.txt").write_text(config.reviewer_backend)
|
|
638
686
|
(log_dir / "reviewer-context.txt").write_text(config.reviewer_context)
|
|
687
|
+
(log_dir / "ci-trigger-mode.txt").write_text(config.ci_trigger_mode)
|
|
639
688
|
if config.scope:
|
|
640
689
|
(log_dir / "scope.txt").write_text(config.scope)
|
|
641
690
|
|
|
@@ -162,6 +162,14 @@ class LoopConfig:
|
|
|
162
162
|
resume: bool = False
|
|
163
163
|
auto_approve: bool = False
|
|
164
164
|
|
|
165
|
+
# CI trigger policy for iteration commits:
|
|
166
|
+
# "last-only" — default: append "[skip ci]" to iteration commits; push a
|
|
167
|
+
# single empty "chore: trigger CI" commit only on ALL_CLEAR.
|
|
168
|
+
# "every" — each commit triggers CI (pre-0.3 behaviour).
|
|
169
|
+
# "none" — append "[skip ci]" to iteration commits; never emit a
|
|
170
|
+
# trigger commit (forks / CI-less repos).
|
|
171
|
+
ci_trigger_mode: str = "last-only"
|
|
172
|
+
|
|
165
173
|
# Retry / budget
|
|
166
174
|
retry_max_wait: int = 7200
|
|
167
175
|
retry_initial_wait: int = 30
|
|
@@ -193,6 +201,7 @@ class LoopResult:
|
|
|
193
201
|
final_status: FinalStatus
|
|
194
202
|
iterations_run: int
|
|
195
203
|
summary_path: Path | None = None
|
|
204
|
+
made_skipped_fix_commit: bool = False
|
|
196
205
|
|
|
197
206
|
|
|
198
207
|
# ── Protocols (DI contracts) ─────────────────────────────────────────
|
|
@@ -26,9 +26,13 @@ BUDGET_POLL_MAX = 1200
|
|
|
26
26
|
|
|
27
27
|
|
|
28
28
|
def extract_result_from_stream(stream_path: Path) -> str:
|
|
29
|
-
"""Extract final result
|
|
29
|
+
"""Extract final result from a Claude stream-json event log.
|
|
30
30
|
|
|
31
|
-
|
|
31
|
+
When the ``result`` event includes a ``structured_output`` object
|
|
32
|
+
(set by ``--json-schema``), the JSON-encoded structured object is
|
|
33
|
+
returned so downstream parsers receive schema-conforming JSON.
|
|
34
|
+
Otherwise the plain ``result`` text is returned. Returns empty
|
|
35
|
+
string if the file is empty or no result event is found.
|
|
32
36
|
"""
|
|
33
37
|
if not stream_path.is_file() or stream_path.stat().st_size == 0:
|
|
34
38
|
return ""
|
|
@@ -39,10 +43,13 @@ def extract_result_from_stream(stream_path: Path) -> str:
|
|
|
39
43
|
continue
|
|
40
44
|
try:
|
|
41
45
|
data = json.loads(line)
|
|
42
|
-
|
|
43
|
-
last_result = data["result"]
|
|
44
|
-
except (json.JSONDecodeError, KeyError):
|
|
46
|
+
except json.JSONDecodeError:
|
|
45
47
|
continue
|
|
48
|
+
structured = data.get("structured_output")
|
|
49
|
+
if isinstance(structured, dict):
|
|
50
|
+
last_result = json.dumps(structured)
|
|
51
|
+
elif data.get("result"):
|
|
52
|
+
last_result = data["result"]
|
|
46
53
|
return last_result
|
|
47
54
|
|
|
48
55
|
|