overkill 0.6.1__tar.gz → 0.7.0__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- {overkill-0.6.1 → overkill-0.7.0}/PKG-INFO +1 -1
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-review.prompt.md +1 -1
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-review.prompt.md +1 -1
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-review.prompt.md +1 -1
- {overkill-0.6.1 → overkill-0.7.0}/pyproject.toml +1 -1
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/agents.py +56 -11
- overkill-0.7.0/src/mr_overkill/data/review.schema.json +76 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/json_extract.py +30 -18
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/retry.py +12 -5
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_agents.py +68 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_json_extract.py +44 -2
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_retry.py +11 -0
- {overkill-0.6.1 → overkill-0.7.0}/uv.lock +1 -1
- {overkill-0.6.1 → overkill-0.7.0}/.github/workflows/publish.yml +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/.github/workflows/test.yml +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/.gitignore +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/.overkillrc.example +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/.python-version +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/.refactorsuggestrc.example +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/AGENTS.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/CLAUDE.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/GEMINI.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/LICENSE +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/README.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/install.sh +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-fix-execute.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-fix.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-fix-execute.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-fix.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-full.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-layer.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-micro.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-module.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-self-review.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-refactor-full.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-refactor-layer.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-refactor-micro.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-refactor-module.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-refactor-full.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-refactor-layer.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-refactor-micro.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-refactor-module.prompt.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/reference/claude-code-review-plugin.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/reference/claude-security-review.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/prompts/reference/codex-review-original.md +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/__init__.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/__main__.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget/__init__.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget/claude.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget/codex.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget/gemini.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget_report.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/classify.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/cli.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/data/__init__.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/git_ops.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/init.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/loop_engine.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/models.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/refactor_suggest.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/reporting.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/resume.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/review_loop.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/self_review.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/time_utils.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/two_step_fix.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/__init__.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/conftest.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_claude.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_codex.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_gemini.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_policy.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_report.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_classify.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_cli.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_git_ops.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_init.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_integration.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_loop_engine.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_refactor_suggest.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_reporting.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_resume.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_review_loop.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_self_review.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_time_utils.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/tests/test_two_step_fix.py +0 -0
- {overkill-0.6.1 → overkill-0.7.0}/uninstall.sh +0 -0
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
Metadata-Version: 2.4
|
|
2
2
|
Name: overkill
|
|
3
|
-
Version: 0.
|
|
3
|
+
Version: 0.7.0
|
|
4
4
|
Summary: AI-powered code review loop — automates Codex/Gemini review + Claude fix cycles
|
|
5
5
|
Project-URL: Repository, https://github.com/modocai/mr-overkill
|
|
6
6
|
Project-URL: Issues, https://github.com/modocai/mr-overkill/issues
|
|
@@ -62,7 +62,7 @@ We only want findings where you are confident the issue is real. **If you are no
|
|
|
62
62
|
|
|
63
63
|
## Output Format
|
|
64
64
|
|
|
65
|
-
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
|
|
65
|
+
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
|
|
66
66
|
|
|
67
67
|
{
|
|
68
68
|
"findings": [
|
|
@@ -53,7 +53,7 @@ Output all findings that the original author would fix if they knew about them.
|
|
|
53
53
|
|
|
54
54
|
## Output Format
|
|
55
55
|
|
|
56
|
-
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
|
|
56
|
+
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
|
|
57
57
|
|
|
58
58
|
{
|
|
59
59
|
"findings": [
|
|
@@ -62,7 +62,7 @@ We only want findings where you are confident the issue is real. **If you are no
|
|
|
62
62
|
|
|
63
63
|
## Output Format
|
|
64
64
|
|
|
65
|
-
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
|
|
65
|
+
Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
|
|
66
66
|
|
|
67
67
|
{
|
|
68
68
|
"findings": [
|
|
@@ -13,6 +13,7 @@ import logging
|
|
|
13
13
|
import string
|
|
14
14
|
import subprocess
|
|
15
15
|
from abc import ABC, abstractmethod
|
|
16
|
+
from importlib.resources import as_file, files
|
|
16
17
|
from pathlib import Path
|
|
17
18
|
|
|
18
19
|
from mr_overkill.budget.claude import claude_budget_sufficient
|
|
@@ -45,6 +46,41 @@ def _format_reviewer_context(raw: str) -> str:
|
|
|
45
46
|
return f"## Author Context\n\n{raw}"
|
|
46
47
|
|
|
47
48
|
|
|
49
|
+
# ── Review schema (single source of truth for structured output) ─────
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
def _review_schema_text() -> str:
|
|
53
|
+
"""Read the bundled review JSON Schema as a string."""
|
|
54
|
+
return files("mr_overkill.data").joinpath("review.schema.json").read_text(
|
|
55
|
+
encoding="utf-8"
|
|
56
|
+
)
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _unwrap_claude_structured_output(output_path: Path) -> None:
|
|
60
|
+
"""Replace Claude's ``--output-format json`` wrapper with its inner schema object.
|
|
61
|
+
|
|
62
|
+
Claude CLI emits ``{"type":"result", ..., "structured_output": {...}, ...}``
|
|
63
|
+
when both ``--output-format json`` and ``--json-schema`` are set. The
|
|
64
|
+
schema-conforming object lives under ``structured_output``. Downstream
|
|
65
|
+
consumers expect the file to contain that object directly. If the file does
|
|
66
|
+
not match the wrapper shape (older Claude versions, error responses), it is
|
|
67
|
+
left untouched so the parser's fallback tiers can still try.
|
|
68
|
+
"""
|
|
69
|
+
try:
|
|
70
|
+
raw = output_path.read_text(encoding="utf-8")
|
|
71
|
+
except OSError:
|
|
72
|
+
return
|
|
73
|
+
try:
|
|
74
|
+
wrapper = json.loads(raw)
|
|
75
|
+
except json.JSONDecodeError:
|
|
76
|
+
return
|
|
77
|
+
if not isinstance(wrapper, dict):
|
|
78
|
+
return
|
|
79
|
+
inner = wrapper.get("structured_output")
|
|
80
|
+
if isinstance(inner, dict):
|
|
81
|
+
output_path.write_text(json.dumps(inner), encoding="utf-8")
|
|
82
|
+
|
|
83
|
+
|
|
48
84
|
# ── Budget / retry helpers (moved from review_loop.py) ───────────────
|
|
49
85
|
|
|
50
86
|
|
|
@@ -183,16 +219,20 @@ class CodexReviewAgent(ReviewAgent):
|
|
|
183
219
|
)
|
|
184
220
|
|
|
185
221
|
stderr_path = output_path.with_suffix(".stderr")
|
|
186
|
-
|
|
187
|
-
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
"
|
|
192
|
-
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
222
|
+
with as_file(
|
|
223
|
+
files("mr_overkill.data").joinpath("review.schema.json")
|
|
224
|
+
) as schema_path:
|
|
225
|
+
return retry_codex_cmd(
|
|
226
|
+
stderr_path,
|
|
227
|
+
"Codex review",
|
|
228
|
+
[
|
|
229
|
+
"codex", "exec", "--sandbox", "read-only",
|
|
230
|
+
"--output-schema", str(schema_path),
|
|
231
|
+
"-o", str(output_path), prompt_text,
|
|
232
|
+
],
|
|
233
|
+
max_wait=config.retry_max_wait,
|
|
234
|
+
initial_wait=config.retry_initial_wait,
|
|
235
|
+
)
|
|
196
236
|
|
|
197
237
|
|
|
198
238
|
class CodexRefactorReviewAgent(ReviewAgent):
|
|
@@ -284,15 +324,20 @@ class ClaudeReviewAgent(ReviewAgent):
|
|
|
284
324
|
f"Claude budget timeout (iteration {iteration})."
|
|
285
325
|
)
|
|
286
326
|
|
|
287
|
-
|
|
327
|
+
ok = self._retry_fn(
|
|
288
328
|
output_path,
|
|
289
329
|
"Claude review",
|
|
290
330
|
[
|
|
291
331
|
"claude", "-p", "-",
|
|
292
332
|
"--allowedTools", "Bash,Read,Glob,Grep",
|
|
333
|
+
"--output-format", "json",
|
|
334
|
+
"--json-schema", _review_schema_text(),
|
|
293
335
|
],
|
|
294
336
|
stdin=prompt_text,
|
|
295
337
|
)
|
|
338
|
+
if ok:
|
|
339
|
+
_unwrap_claude_structured_output(output_path)
|
|
340
|
+
return ok
|
|
296
341
|
|
|
297
342
|
|
|
298
343
|
class ClaudeRefactorReviewAgent(ReviewAgent):
|
|
@@ -0,0 +1,76 @@
|
|
|
1
|
+
{
|
|
2
|
+
"$schema": "http://json-schema.org/draft-07/schema#",
|
|
3
|
+
"title": "ReviewResult",
|
|
4
|
+
"type": "object",
|
|
5
|
+
"required": [
|
|
6
|
+
"findings",
|
|
7
|
+
"overall_correctness",
|
|
8
|
+
"overall_explanation",
|
|
9
|
+
"overall_confidence_score"
|
|
10
|
+
],
|
|
11
|
+
"additionalProperties": false,
|
|
12
|
+
"properties": {
|
|
13
|
+
"findings": {
|
|
14
|
+
"type": "array",
|
|
15
|
+
"items": {
|
|
16
|
+
"type": "object",
|
|
17
|
+
"required": ["title", "body", "confidence_score", "priority", "code_location"],
|
|
18
|
+
"additionalProperties": false,
|
|
19
|
+
"properties": {
|
|
20
|
+
"title": {
|
|
21
|
+
"type": "string",
|
|
22
|
+
"maxLength": 80,
|
|
23
|
+
"description": "P-tag + imperative description (max 80 chars)"
|
|
24
|
+
},
|
|
25
|
+
"body": {
|
|
26
|
+
"type": "string",
|
|
27
|
+
"description": "Markdown explanation; cite files/lines/functions"
|
|
28
|
+
},
|
|
29
|
+
"confidence_score": {
|
|
30
|
+
"type": "number",
|
|
31
|
+
"minimum": 0,
|
|
32
|
+
"maximum": 1
|
|
33
|
+
},
|
|
34
|
+
"priority": {
|
|
35
|
+
"type": "integer",
|
|
36
|
+
"minimum": 0,
|
|
37
|
+
"maximum": 3
|
|
38
|
+
},
|
|
39
|
+
"code_location": {
|
|
40
|
+
"type": "object",
|
|
41
|
+
"required": ["file_path", "line_range"],
|
|
42
|
+
"additionalProperties": false,
|
|
43
|
+
"properties": {
|
|
44
|
+
"file_path": {
|
|
45
|
+
"type": "string",
|
|
46
|
+
"description": "Repo-relative file path"
|
|
47
|
+
},
|
|
48
|
+
"line_range": {
|
|
49
|
+
"type": "object",
|
|
50
|
+
"required": ["start", "end"],
|
|
51
|
+
"additionalProperties": false,
|
|
52
|
+
"properties": {
|
|
53
|
+
"start": {"type": "integer", "minimum": 0},
|
|
54
|
+
"end": {"type": "integer", "minimum": 0}
|
|
55
|
+
}
|
|
56
|
+
}
|
|
57
|
+
}
|
|
58
|
+
}
|
|
59
|
+
}
|
|
60
|
+
}
|
|
61
|
+
},
|
|
62
|
+
"overall_correctness": {
|
|
63
|
+
"type": "string",
|
|
64
|
+
"enum": ["patch is correct", "patch is incorrect"]
|
|
65
|
+
},
|
|
66
|
+
"overall_explanation": {
|
|
67
|
+
"type": "string",
|
|
68
|
+
"description": "1-3 sentence justification"
|
|
69
|
+
},
|
|
70
|
+
"overall_confidence_score": {
|
|
71
|
+
"type": "number",
|
|
72
|
+
"minimum": 0,
|
|
73
|
+
"maximum": 1
|
|
74
|
+
}
|
|
75
|
+
}
|
|
76
|
+
}
|
|
@@ -1,7 +1,7 @@
|
|
|
1
1
|
"""JSON extraction utilities ported from common.sh and self-review.sh.
|
|
2
2
|
|
|
3
3
|
Provides a 3-tier extraction pipeline (direct parse -> fenced code block ->
|
|
4
|
-
|
|
4
|
+
balanced-brace scan) that mirrors _extract_json_from_file in the shell codebase,
|
|
5
5
|
plus helpers for path normalisation and refactoring-plan injection.
|
|
6
6
|
"""
|
|
7
7
|
|
|
@@ -24,7 +24,9 @@ def extract_json_from_file(path: Path) -> dict[str, Any] | None:
|
|
|
24
24
|
Tier 2: Look for a fenced code block (`` ```json ... ``` ``) and parse its
|
|
25
25
|
content.
|
|
26
26
|
|
|
27
|
-
Tier 3:
|
|
27
|
+
Tier 3: Scan every ``{`` position with ``json.JSONDecoder().raw_decode()``
|
|
28
|
+
and return the largest valid dict found. Robust to prose with embedded
|
|
29
|
+
braces (e.g. preambles that mention schemas).
|
|
28
30
|
|
|
29
31
|
Returns
|
|
30
32
|
-------
|
|
@@ -53,8 +55,8 @@ def extract_json_from_file(path: Path) -> dict[str, Any] | None:
|
|
|
53
55
|
if result is not None:
|
|
54
56
|
return result
|
|
55
57
|
|
|
56
|
-
# ── Tier 3:
|
|
57
|
-
return
|
|
58
|
+
# ── Tier 3: balanced-brace scan ───────────────────────────────────
|
|
59
|
+
return _try_balanced_scan(content)
|
|
58
60
|
|
|
59
61
|
|
|
60
62
|
def parse_review_json(
|
|
@@ -188,18 +190,28 @@ def _try_fenced_block(content: str) -> dict[str, Any] | None:
|
|
|
188
190
|
return None
|
|
189
191
|
|
|
190
192
|
|
|
191
|
-
|
|
193
|
+
def _try_balanced_scan(content: str) -> dict[str, Any] | None:
|
|
194
|
+
"""Tier 3: try ``raw_decode`` from each ``{`` position, keep the largest dict.
|
|
192
195
|
|
|
193
|
-
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
|
|
197
|
-
|
|
198
|
-
|
|
199
|
-
|
|
200
|
-
|
|
201
|
-
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
196
|
+
Greedy regex (``\\{.*\\}``) breaks when prose preceding the JSON contains
|
|
197
|
+
its own ``{`` — the match starts at the prose brace and consumes the rest
|
|
198
|
+
of the file as one invalid string. ``raw_decode`` walks the JSON grammar
|
|
199
|
+
and stops at the matching brace, so each candidate is parsed correctly.
|
|
200
|
+
"""
|
|
201
|
+
decoder = json.JSONDecoder()
|
|
202
|
+
best: dict[str, Any] | None = None
|
|
203
|
+
best_len = 0
|
|
204
|
+
idx = 0
|
|
205
|
+
while True:
|
|
206
|
+
idx = content.find("{", idx)
|
|
207
|
+
if idx < 0:
|
|
208
|
+
break
|
|
209
|
+
try:
|
|
210
|
+
obj, end = decoder.raw_decode(content, idx)
|
|
211
|
+
except json.JSONDecodeError:
|
|
212
|
+
idx += 1
|
|
213
|
+
continue
|
|
214
|
+
if isinstance(obj, dict) and (end - idx) > best_len:
|
|
215
|
+
best, best_len = obj, end - idx
|
|
216
|
+
idx = end if end > idx else idx + 1
|
|
217
|
+
return best
|
|
@@ -26,9 +26,13 @@ BUDGET_POLL_MAX = 1200
|
|
|
26
26
|
|
|
27
27
|
|
|
28
28
|
def extract_result_from_stream(stream_path: Path) -> str:
|
|
29
|
-
"""Extract final result
|
|
29
|
+
"""Extract final result from a Claude stream-json event log.
|
|
30
30
|
|
|
31
|
-
|
|
31
|
+
When the ``result`` event includes a ``structured_output`` object
|
|
32
|
+
(set by ``--json-schema``), the JSON-encoded structured object is
|
|
33
|
+
returned so downstream parsers receive schema-conforming JSON.
|
|
34
|
+
Otherwise the plain ``result`` text is returned. Returns empty
|
|
35
|
+
string if the file is empty or no result event is found.
|
|
32
36
|
"""
|
|
33
37
|
if not stream_path.is_file() or stream_path.stat().st_size == 0:
|
|
34
38
|
return ""
|
|
@@ -39,10 +43,13 @@ def extract_result_from_stream(stream_path: Path) -> str:
|
|
|
39
43
|
continue
|
|
40
44
|
try:
|
|
41
45
|
data = json.loads(line)
|
|
42
|
-
|
|
43
|
-
last_result = data["result"]
|
|
44
|
-
except (json.JSONDecodeError, KeyError):
|
|
46
|
+
except json.JSONDecodeError:
|
|
45
47
|
continue
|
|
48
|
+
structured = data.get("structured_output")
|
|
49
|
+
if isinstance(structured, dict):
|
|
50
|
+
last_result = json.dumps(structured)
|
|
51
|
+
elif data.get("result"):
|
|
52
|
+
last_result = data["result"]
|
|
46
53
|
return last_result
|
|
47
54
|
|
|
48
55
|
|
|
@@ -60,6 +60,12 @@ class TestCodexReviewAgent:
|
|
|
60
60
|
assert "feat/test" in prompt_text
|
|
61
61
|
assert "develop" in prompt_text
|
|
62
62
|
|
|
63
|
+
# Structured output: schema flag must point at the bundled review schema.
|
|
64
|
+
assert "--output-schema" in cmd_args
|
|
65
|
+
schema_path = cmd_args[cmd_args.index("--output-schema") + 1]
|
|
66
|
+
assert Path(schema_path).name == "review.schema.json"
|
|
67
|
+
assert Path(schema_path).is_file()
|
|
68
|
+
|
|
63
69
|
def test_fails_on_missing_prompt(
|
|
64
70
|
self,
|
|
65
71
|
tmp_path: Path,
|
|
@@ -143,6 +149,68 @@ class TestClaudeReviewAgent:
|
|
|
143
149
|
assert "feat/test" in call_kw["stdin"]
|
|
144
150
|
assert "develop" in call_kw["stdin"]
|
|
145
151
|
|
|
152
|
+
# Structured output: --output-format json + --json-schema with the
|
|
153
|
+
# bundled review schema text inlined into argv.
|
|
154
|
+
cmd_args = mock_retry.call_args[0][2]
|
|
155
|
+
assert "--output-format" in cmd_args
|
|
156
|
+
assert cmd_args[cmd_args.index("--output-format") + 1] == "json"
|
|
157
|
+
assert "--json-schema" in cmd_args
|
|
158
|
+
schema_text = cmd_args[cmd_args.index("--json-schema") + 1]
|
|
159
|
+
schema = json.loads(schema_text)
|
|
160
|
+
assert schema["title"] == "ReviewResult"
|
|
161
|
+
assert "findings" in schema["properties"]
|
|
162
|
+
|
|
163
|
+
def test_unwraps_claude_structured_output(
|
|
164
|
+
self,
|
|
165
|
+
tmp_path: Path,
|
|
166
|
+
make_loop_config: Callable[..., LoopConfig],
|
|
167
|
+
) -> None:
|
|
168
|
+
"""Wrapper produced by ``--output-format json`` is unwrapped on disk.
|
|
169
|
+
|
|
170
|
+
Regression for the Claude CLI emitting
|
|
171
|
+
``{"type":"result", ..., "structured_output": {<review>}}`` — the
|
|
172
|
+
downstream parser expects the review object, not the wrapper.
|
|
173
|
+
"""
|
|
174
|
+
review = {
|
|
175
|
+
"findings": [],
|
|
176
|
+
"overall_correctness": "patch is correct",
|
|
177
|
+
"overall_confidence_score": 0.9,
|
|
178
|
+
}
|
|
179
|
+
wrapper = {
|
|
180
|
+
"type": "result",
|
|
181
|
+
"subtype": "success",
|
|
182
|
+
"is_error": False,
|
|
183
|
+
"result": "Done.",
|
|
184
|
+
"structured_output": review,
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
prompts = tmp_path / "prompts"
|
|
188
|
+
prompts.mkdir(exist_ok=True)
|
|
189
|
+
(prompts / "claude-review.prompt.md").write_text("prompt body")
|
|
190
|
+
config = make_loop_config(prompts_dir=prompts)
|
|
191
|
+
|
|
192
|
+
output = tmp_path / "review.json"
|
|
193
|
+
|
|
194
|
+
def fake_retry(
|
|
195
|
+
out_path: Path,
|
|
196
|
+
_label: str,
|
|
197
|
+
_argv: list[str],
|
|
198
|
+
**_kw: object,
|
|
199
|
+
) -> bool:
|
|
200
|
+
out_path.write_text(json.dumps(wrapper))
|
|
201
|
+
return True
|
|
202
|
+
|
|
203
|
+
with (
|
|
204
|
+
patch("mr_overkill.agents._make_budget_fn") as mb,
|
|
205
|
+
patch("mr_overkill.agents._make_retry_fn") as mr,
|
|
206
|
+
):
|
|
207
|
+
mb.return_value = MagicMock(return_value=True)
|
|
208
|
+
mr.return_value = fake_retry
|
|
209
|
+
agent = ClaudeReviewAgent(config)
|
|
210
|
+
assert agent(output, 1) is True
|
|
211
|
+
|
|
212
|
+
assert json.loads(output.read_text()) == review
|
|
213
|
+
|
|
146
214
|
def test_fails_on_missing_prompt(
|
|
147
215
|
self,
|
|
148
216
|
tmp_path: Path,
|
|
@@ -57,14 +57,56 @@ class TestExtractJsonFromFile:
|
|
|
57
57
|
p.write_text(content)
|
|
58
58
|
assert extract_json_from_file(p) == data
|
|
59
59
|
|
|
60
|
-
def
|
|
61
|
-
"""Tier 3: JSON embedded in free text is extracted
|
|
60
|
+
def test_balanced_scan_fallback(self, tmp_path: Path) -> None:
|
|
61
|
+
"""Tier 3: JSON embedded in free text is extracted by the balanced scan."""
|
|
62
62
|
data = {"found": True}
|
|
63
63
|
content = "Random preamble text\n" + json.dumps(data) + "\nmore text\n"
|
|
64
64
|
p = tmp_path / "mixed.txt"
|
|
65
65
|
p.write_text(content)
|
|
66
66
|
assert extract_json_from_file(p) == data
|
|
67
67
|
|
|
68
|
+
def test_balanced_scan_with_prose_braces(self, tmp_path: Path) -> None:
|
|
69
|
+
"""Tier 3: prose containing ``{`` does not derail extraction.
|
|
70
|
+
|
|
71
|
+
Regression for parse_error encountered when a Claude reviewer prefixed
|
|
72
|
+
its JSON with prose that mentioned schema fields like ``{user_id}``.
|
|
73
|
+
"""
|
|
74
|
+
review = {
|
|
75
|
+
"findings": [],
|
|
76
|
+
"overall_correctness": "patch is correct",
|
|
77
|
+
"overall_confidence_score": 0.88,
|
|
78
|
+
}
|
|
79
|
+
content = (
|
|
80
|
+
"Based on my analysis of the iteration 2 review of this diff, "
|
|
81
|
+
"I've examined:\n"
|
|
82
|
+
"- The schema additions like {user_id} and {created_at}\n"
|
|
83
|
+
"- The JSON shape `{findings: [], overall_correctness: ...}`\n"
|
|
84
|
+
"- IAM permissions for s3 access\n"
|
|
85
|
+
"\n"
|
|
86
|
+
"Here is the review:\n"
|
|
87
|
+
f"{json.dumps(review, indent=2)}\n"
|
|
88
|
+
)
|
|
89
|
+
p = tmp_path / "prose.json"
|
|
90
|
+
p.write_text(content)
|
|
91
|
+
assert extract_json_from_file(p) == review
|
|
92
|
+
|
|
93
|
+
def test_balanced_scan_picks_largest_dict(self, tmp_path: Path) -> None:
|
|
94
|
+
"""Tier 3: when several ``{...}`` candidates parse, prefer the largest."""
|
|
95
|
+
small = {"k": 1}
|
|
96
|
+
big = {
|
|
97
|
+
"findings": [],
|
|
98
|
+
"overall_correctness": "patch is correct",
|
|
99
|
+
"overall_confidence_score": 0.9,
|
|
100
|
+
}
|
|
101
|
+
content = (
|
|
102
|
+
f"first candidate: {json.dumps(small)}\n"
|
|
103
|
+
"more prose\n"
|
|
104
|
+
f"actual review: {json.dumps(big)}\n"
|
|
105
|
+
)
|
|
106
|
+
p = tmp_path / "multi.txt"
|
|
107
|
+
p.write_text(content)
|
|
108
|
+
assert extract_json_from_file(p) == big
|
|
109
|
+
|
|
68
110
|
def test_empty_file_returns_none(self, tmp_path: Path) -> None:
|
|
69
111
|
"""Empty (or whitespace-only) file returns None."""
|
|
70
112
|
p = tmp_path / "empty.json"
|
|
@@ -46,6 +46,17 @@ class TestExtractResultFromStream:
|
|
|
46
46
|
f.write_text(f"not json with result\n{ok_line}")
|
|
47
47
|
assert extract_result_from_stream(f) == "ok"
|
|
48
48
|
|
|
49
|
+
def test_prefers_structured_output(self, tmp_path: Path) -> None:
|
|
50
|
+
f = tmp_path / "stream.jsonl"
|
|
51
|
+
structured = {"findings": [], "overall_correctness": "patch is correct"}
|
|
52
|
+
line = json.dumps({
|
|
53
|
+
"type": "result",
|
|
54
|
+
"result": "plain text summary",
|
|
55
|
+
"structured_output": structured,
|
|
56
|
+
})
|
|
57
|
+
f.write_text(line)
|
|
58
|
+
assert json.loads(extract_result_from_stream(f)) == structured
|
|
59
|
+
|
|
49
60
|
|
|
50
61
|
class TestWaitForBudget:
|
|
51
62
|
def test_budget_ok_immediately(self) -> None:
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|
|
File without changes
|