overkill 0.6.1__tar.gz → 0.7.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (87) hide show
  1. {overkill-0.6.1 → overkill-0.7.0}/PKG-INFO +1 -1
  2. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-review.prompt.md +1 -1
  3. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-review.prompt.md +1 -1
  4. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-review.prompt.md +1 -1
  5. {overkill-0.6.1 → overkill-0.7.0}/pyproject.toml +1 -1
  6. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/agents.py +56 -11
  7. overkill-0.7.0/src/mr_overkill/data/review.schema.json +76 -0
  8. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/json_extract.py +30 -18
  9. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/retry.py +12 -5
  10. {overkill-0.6.1 → overkill-0.7.0}/tests/test_agents.py +68 -0
  11. {overkill-0.6.1 → overkill-0.7.0}/tests/test_json_extract.py +44 -2
  12. {overkill-0.6.1 → overkill-0.7.0}/tests/test_retry.py +11 -0
  13. {overkill-0.6.1 → overkill-0.7.0}/uv.lock +1 -1
  14. {overkill-0.6.1 → overkill-0.7.0}/.github/workflows/publish.yml +0 -0
  15. {overkill-0.6.1 → overkill-0.7.0}/.github/workflows/test.yml +0 -0
  16. {overkill-0.6.1 → overkill-0.7.0}/.gitignore +0 -0
  17. {overkill-0.6.1 → overkill-0.7.0}/.overkillrc.example +0 -0
  18. {overkill-0.6.1 → overkill-0.7.0}/.python-version +0 -0
  19. {overkill-0.6.1 → overkill-0.7.0}/.refactorsuggestrc.example +0 -0
  20. {overkill-0.6.1 → overkill-0.7.0}/AGENTS.md +0 -0
  21. {overkill-0.6.1 → overkill-0.7.0}/CLAUDE.md +0 -0
  22. {overkill-0.6.1 → overkill-0.7.0}/GEMINI.md +0 -0
  23. {overkill-0.6.1 → overkill-0.7.0}/LICENSE +0 -0
  24. {overkill-0.6.1 → overkill-0.7.0}/README.md +0 -0
  25. {overkill-0.6.1 → overkill-0.7.0}/install.sh +0 -0
  26. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-fix-execute.prompt.md +0 -0
  27. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-fix.prompt.md +0 -0
  28. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-fix-execute.prompt.md +0 -0
  29. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-fix.prompt.md +0 -0
  30. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-full.prompt.md +0 -0
  31. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-layer.prompt.md +0 -0
  32. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-micro.prompt.md +0 -0
  33. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-refactor-module.prompt.md +0 -0
  34. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/claude-self-review.prompt.md +0 -0
  35. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-refactor-full.prompt.md +0 -0
  36. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-refactor-layer.prompt.md +0 -0
  37. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-refactor-micro.prompt.md +0 -0
  38. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/codex-refactor-module.prompt.md +0 -0
  39. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-refactor-full.prompt.md +0 -0
  40. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-refactor-layer.prompt.md +0 -0
  41. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-refactor-micro.prompt.md +0 -0
  42. {overkill-0.6.1 → overkill-0.7.0}/prompts/active/gemini-refactor-module.prompt.md +0 -0
  43. {overkill-0.6.1 → overkill-0.7.0}/prompts/reference/claude-code-review-plugin.md +0 -0
  44. {overkill-0.6.1 → overkill-0.7.0}/prompts/reference/claude-security-review.md +0 -0
  45. {overkill-0.6.1 → overkill-0.7.0}/prompts/reference/codex-review-original.md +0 -0
  46. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/__init__.py +0 -0
  47. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/__main__.py +0 -0
  48. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget/__init__.py +0 -0
  49. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget/claude.py +0 -0
  50. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget/codex.py +0 -0
  51. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget/gemini.py +0 -0
  52. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/budget_report.py +0 -0
  53. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/classify.py +0 -0
  54. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/cli.py +0 -0
  55. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/data/__init__.py +0 -0
  56. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/git_ops.py +0 -0
  57. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/init.py +0 -0
  58. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/loop_engine.py +0 -0
  59. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/models.py +0 -0
  60. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/refactor_suggest.py +0 -0
  61. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/reporting.py +0 -0
  62. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/resume.py +0 -0
  63. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/review_loop.py +0 -0
  64. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/self_review.py +0 -0
  65. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/time_utils.py +0 -0
  66. {overkill-0.6.1 → overkill-0.7.0}/src/mr_overkill/two_step_fix.py +0 -0
  67. {overkill-0.6.1 → overkill-0.7.0}/tests/__init__.py +0 -0
  68. {overkill-0.6.1 → overkill-0.7.0}/tests/conftest.py +0 -0
  69. {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_claude.py +0 -0
  70. {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_codex.py +0 -0
  71. {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_gemini.py +0 -0
  72. {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_policy.py +0 -0
  73. {overkill-0.6.1 → overkill-0.7.0}/tests/test_budget_report.py +0 -0
  74. {overkill-0.6.1 → overkill-0.7.0}/tests/test_classify.py +0 -0
  75. {overkill-0.6.1 → overkill-0.7.0}/tests/test_cli.py +0 -0
  76. {overkill-0.6.1 → overkill-0.7.0}/tests/test_git_ops.py +0 -0
  77. {overkill-0.6.1 → overkill-0.7.0}/tests/test_init.py +0 -0
  78. {overkill-0.6.1 → overkill-0.7.0}/tests/test_integration.py +0 -0
  79. {overkill-0.6.1 → overkill-0.7.0}/tests/test_loop_engine.py +0 -0
  80. {overkill-0.6.1 → overkill-0.7.0}/tests/test_refactor_suggest.py +0 -0
  81. {overkill-0.6.1 → overkill-0.7.0}/tests/test_reporting.py +0 -0
  82. {overkill-0.6.1 → overkill-0.7.0}/tests/test_resume.py +0 -0
  83. {overkill-0.6.1 → overkill-0.7.0}/tests/test_review_loop.py +0 -0
  84. {overkill-0.6.1 → overkill-0.7.0}/tests/test_self_review.py +0 -0
  85. {overkill-0.6.1 → overkill-0.7.0}/tests/test_time_utils.py +0 -0
  86. {overkill-0.6.1 → overkill-0.7.0}/tests/test_two_step_fix.py +0 -0
  87. {overkill-0.6.1 → overkill-0.7.0}/uninstall.sh +0 -0
@@ -1,6 +1,6 @@
1
1
  Metadata-Version: 2.4
2
2
  Name: overkill
3
- Version: 0.6.1
3
+ Version: 0.7.0
4
4
  Summary: AI-powered code review loop — automates Codex/Gemini review + Claude fix cycles
5
5
  Project-URL: Repository, https://github.com/modocai/mr-overkill
6
6
  Project-URL: Issues, https://github.com/modocai/mr-overkill/issues
@@ -62,7 +62,7 @@ We only want findings where you are confident the issue is real. **If you are no
62
62
 
63
63
  ## Output Format
64
64
 
65
- Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
65
+ Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
66
66
 
67
67
  {
68
68
  "findings": [
@@ -53,7 +53,7 @@ Output all findings that the original author would fix if they knew about them.
53
53
 
54
54
  ## Output Format
55
55
 
56
- Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
56
+ Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
57
57
 
58
58
  {
59
59
  "findings": [
@@ -62,7 +62,7 @@ We only want findings where you are confident the issue is real. **If you are no
62
62
 
63
63
  ## Output Format
64
64
 
65
- Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON.
65
+ Output **only** valid JSON matching this schema exactly. Do NOT wrap in markdown fences or add any text outside the JSON. Begin your response with the `{` character — no prose, no commentary, no preamble.
66
66
 
67
67
  {
68
68
  "findings": [
@@ -1,6 +1,6 @@
1
1
  [project]
2
2
  name = "overkill"
3
- version = "0.6.1"
3
+ version = "0.7.0"
4
4
  description = "AI-powered code review loop — automates Codex/Gemini review + Claude fix cycles"
5
5
  readme = "README.md"
6
6
  license = { text = "MIT" }
@@ -13,6 +13,7 @@ import logging
13
13
  import string
14
14
  import subprocess
15
15
  from abc import ABC, abstractmethod
16
+ from importlib.resources import as_file, files
16
17
  from pathlib import Path
17
18
 
18
19
  from mr_overkill.budget.claude import claude_budget_sufficient
@@ -45,6 +46,41 @@ def _format_reviewer_context(raw: str) -> str:
45
46
  return f"## Author Context\n\n{raw}"
46
47
 
47
48
 
49
+ # ── Review schema (single source of truth for structured output) ─────
50
+
51
+
52
+ def _review_schema_text() -> str:
53
+ """Read the bundled review JSON Schema as a string."""
54
+ return files("mr_overkill.data").joinpath("review.schema.json").read_text(
55
+ encoding="utf-8"
56
+ )
57
+
58
+
59
+ def _unwrap_claude_structured_output(output_path: Path) -> None:
60
+ """Replace Claude's ``--output-format json`` wrapper with its inner schema object.
61
+
62
+ Claude CLI emits ``{"type":"result", ..., "structured_output": {...}, ...}``
63
+ when both ``--output-format json`` and ``--json-schema`` are set. The
64
+ schema-conforming object lives under ``structured_output``. Downstream
65
+ consumers expect the file to contain that object directly. If the file does
66
+ not match the wrapper shape (older Claude versions, error responses), it is
67
+ left untouched so the parser's fallback tiers can still try.
68
+ """
69
+ try:
70
+ raw = output_path.read_text(encoding="utf-8")
71
+ except OSError:
72
+ return
73
+ try:
74
+ wrapper = json.loads(raw)
75
+ except json.JSONDecodeError:
76
+ return
77
+ if not isinstance(wrapper, dict):
78
+ return
79
+ inner = wrapper.get("structured_output")
80
+ if isinstance(inner, dict):
81
+ output_path.write_text(json.dumps(inner), encoding="utf-8")
82
+
83
+
48
84
  # ── Budget / retry helpers (moved from review_loop.py) ───────────────
49
85
 
50
86
 
@@ -183,16 +219,20 @@ class CodexReviewAgent(ReviewAgent):
183
219
  )
184
220
 
185
221
  stderr_path = output_path.with_suffix(".stderr")
186
- return retry_codex_cmd(
187
- stderr_path,
188
- "Codex review",
189
- [
190
- "codex", "exec", "--sandbox", "read-only",
191
- "-o", str(output_path), prompt_text,
192
- ],
193
- max_wait=config.retry_max_wait,
194
- initial_wait=config.retry_initial_wait,
195
- )
222
+ with as_file(
223
+ files("mr_overkill.data").joinpath("review.schema.json")
224
+ ) as schema_path:
225
+ return retry_codex_cmd(
226
+ stderr_path,
227
+ "Codex review",
228
+ [
229
+ "codex", "exec", "--sandbox", "read-only",
230
+ "--output-schema", str(schema_path),
231
+ "-o", str(output_path), prompt_text,
232
+ ],
233
+ max_wait=config.retry_max_wait,
234
+ initial_wait=config.retry_initial_wait,
235
+ )
196
236
 
197
237
 
198
238
  class CodexRefactorReviewAgent(ReviewAgent):
@@ -284,15 +324,20 @@ class ClaudeReviewAgent(ReviewAgent):
284
324
  f"Claude budget timeout (iteration {iteration})."
285
325
  )
286
326
 
287
- return self._retry_fn(
327
+ ok = self._retry_fn(
288
328
  output_path,
289
329
  "Claude review",
290
330
  [
291
331
  "claude", "-p", "-",
292
332
  "--allowedTools", "Bash,Read,Glob,Grep",
333
+ "--output-format", "json",
334
+ "--json-schema", _review_schema_text(),
293
335
  ],
294
336
  stdin=prompt_text,
295
337
  )
338
+ if ok:
339
+ _unwrap_claude_structured_output(output_path)
340
+ return ok
296
341
 
297
342
 
298
343
  class ClaudeRefactorReviewAgent(ReviewAgent):
@@ -0,0 +1,76 @@
1
+ {
2
+ "$schema": "http://json-schema.org/draft-07/schema#",
3
+ "title": "ReviewResult",
4
+ "type": "object",
5
+ "required": [
6
+ "findings",
7
+ "overall_correctness",
8
+ "overall_explanation",
9
+ "overall_confidence_score"
10
+ ],
11
+ "additionalProperties": false,
12
+ "properties": {
13
+ "findings": {
14
+ "type": "array",
15
+ "items": {
16
+ "type": "object",
17
+ "required": ["title", "body", "confidence_score", "priority", "code_location"],
18
+ "additionalProperties": false,
19
+ "properties": {
20
+ "title": {
21
+ "type": "string",
22
+ "maxLength": 80,
23
+ "description": "P-tag + imperative description (max 80 chars)"
24
+ },
25
+ "body": {
26
+ "type": "string",
27
+ "description": "Markdown explanation; cite files/lines/functions"
28
+ },
29
+ "confidence_score": {
30
+ "type": "number",
31
+ "minimum": 0,
32
+ "maximum": 1
33
+ },
34
+ "priority": {
35
+ "type": "integer",
36
+ "minimum": 0,
37
+ "maximum": 3
38
+ },
39
+ "code_location": {
40
+ "type": "object",
41
+ "required": ["file_path", "line_range"],
42
+ "additionalProperties": false,
43
+ "properties": {
44
+ "file_path": {
45
+ "type": "string",
46
+ "description": "Repo-relative file path"
47
+ },
48
+ "line_range": {
49
+ "type": "object",
50
+ "required": ["start", "end"],
51
+ "additionalProperties": false,
52
+ "properties": {
53
+ "start": {"type": "integer", "minimum": 0},
54
+ "end": {"type": "integer", "minimum": 0}
55
+ }
56
+ }
57
+ }
58
+ }
59
+ }
60
+ }
61
+ },
62
+ "overall_correctness": {
63
+ "type": "string",
64
+ "enum": ["patch is correct", "patch is incorrect"]
65
+ },
66
+ "overall_explanation": {
67
+ "type": "string",
68
+ "description": "1-3 sentence justification"
69
+ },
70
+ "overall_confidence_score": {
71
+ "type": "number",
72
+ "minimum": 0,
73
+ "maximum": 1
74
+ }
75
+ }
76
+ }
@@ -1,7 +1,7 @@
1
1
  """JSON extraction utilities ported from common.sh and self-review.sh.
2
2
 
3
3
  Provides a 3-tier extraction pipeline (direct parse -> fenced code block ->
4
- regex fallback) that mirrors _extract_json_from_file in the shell codebase,
4
+ balanced-brace scan) that mirrors _extract_json_from_file in the shell codebase,
5
5
  plus helpers for path normalisation and refactoring-plan injection.
6
6
  """
7
7
 
@@ -24,7 +24,9 @@ def extract_json_from_file(path: Path) -> dict[str, Any] | None:
24
24
  Tier 2: Look for a fenced code block (`` ```json ... ``` ``) and parse its
25
25
  content.
26
26
 
27
- Tier 3: Use a greedy regex to extract the outermost ``{ ... }`` and parse.
27
+ Tier 3: Scan every ``{`` position with ``json.JSONDecoder().raw_decode()``
28
+ and return the largest valid dict found. Robust to prose with embedded
29
+ braces (e.g. preambles that mention schemas).
28
30
 
29
31
  Returns
30
32
  -------
@@ -53,8 +55,8 @@ def extract_json_from_file(path: Path) -> dict[str, Any] | None:
53
55
  if result is not None:
54
56
  return result
55
57
 
56
- # ── Tier 3: regex fallback ────────────────────────────────────────
57
- return _try_regex_fallback(content)
58
+ # ── Tier 3: balanced-brace scan ───────────────────────────────────
59
+ return _try_balanced_scan(content)
58
60
 
59
61
 
60
62
  def parse_review_json(
@@ -188,18 +190,28 @@ def _try_fenced_block(content: str) -> dict[str, Any] | None:
188
190
  return None
189
191
 
190
192
 
191
- _BRACE_RE = re.compile(r"\{.*\}", re.DOTALL)
193
+ def _try_balanced_scan(content: str) -> dict[str, Any] | None:
194
+ """Tier 3: try ``raw_decode`` from each ``{`` position, keep the largest dict.
192
195
 
193
-
194
- def _try_regex_fallback(content: str) -> dict[str, Any] | None:
195
- """Tier 3: greedy regex for the outermost ``{ ... }``."""
196
- match = _BRACE_RE.search(content)
197
- if match is None:
198
- return None
199
- try:
200
- obj = json.loads(match.group(0))
201
- if isinstance(obj, dict):
202
- return obj
203
- except json.JSONDecodeError:
204
- pass
205
- return None
196
+ Greedy regex (``\\{.*\\}``) breaks when prose preceding the JSON contains
197
+ its own ``{`` — the match starts at the prose brace and consumes the rest
198
+ of the file as one invalid string. ``raw_decode`` walks the JSON grammar
199
+ and stops at the matching brace, so each candidate is parsed correctly.
200
+ """
201
+ decoder = json.JSONDecoder()
202
+ best: dict[str, Any] | None = None
203
+ best_len = 0
204
+ idx = 0
205
+ while True:
206
+ idx = content.find("{", idx)
207
+ if idx < 0:
208
+ break
209
+ try:
210
+ obj, end = decoder.raw_decode(content, idx)
211
+ except json.JSONDecodeError:
212
+ idx += 1
213
+ continue
214
+ if isinstance(obj, dict) and (end - idx) > best_len:
215
+ best, best_len = obj, end - idx
216
+ idx = end if end > idx else idx + 1
217
+ return best
@@ -26,9 +26,13 @@ BUDGET_POLL_MAX = 1200
26
26
 
27
27
 
28
28
  def extract_result_from_stream(stream_path: Path) -> str:
29
- """Extract final result text from a Claude stream-json event log.
29
+ """Extract final result from a Claude stream-json event log.
30
30
 
31
- Returns empty string if the file is empty or no result event found.
31
+ When the ``result`` event includes a ``structured_output`` object
32
+ (set by ``--json-schema``), the JSON-encoded structured object is
33
+ returned so downstream parsers receive schema-conforming JSON.
34
+ Otherwise the plain ``result`` text is returned. Returns empty
35
+ string if the file is empty or no result event is found.
32
36
  """
33
37
  if not stream_path.is_file() or stream_path.stat().st_size == 0:
34
38
  return ""
@@ -39,10 +43,13 @@ def extract_result_from_stream(stream_path: Path) -> str:
39
43
  continue
40
44
  try:
41
45
  data = json.loads(line)
42
- if data.get("result"):
43
- last_result = data["result"]
44
- except (json.JSONDecodeError, KeyError):
46
+ except json.JSONDecodeError:
45
47
  continue
48
+ structured = data.get("structured_output")
49
+ if isinstance(structured, dict):
50
+ last_result = json.dumps(structured)
51
+ elif data.get("result"):
52
+ last_result = data["result"]
46
53
  return last_result
47
54
 
48
55
 
@@ -60,6 +60,12 @@ class TestCodexReviewAgent:
60
60
  assert "feat/test" in prompt_text
61
61
  assert "develop" in prompt_text
62
62
 
63
+ # Structured output: schema flag must point at the bundled review schema.
64
+ assert "--output-schema" in cmd_args
65
+ schema_path = cmd_args[cmd_args.index("--output-schema") + 1]
66
+ assert Path(schema_path).name == "review.schema.json"
67
+ assert Path(schema_path).is_file()
68
+
63
69
  def test_fails_on_missing_prompt(
64
70
  self,
65
71
  tmp_path: Path,
@@ -143,6 +149,68 @@ class TestClaudeReviewAgent:
143
149
  assert "feat/test" in call_kw["stdin"]
144
150
  assert "develop" in call_kw["stdin"]
145
151
 
152
+ # Structured output: --output-format json + --json-schema with the
153
+ # bundled review schema text inlined into argv.
154
+ cmd_args = mock_retry.call_args[0][2]
155
+ assert "--output-format" in cmd_args
156
+ assert cmd_args[cmd_args.index("--output-format") + 1] == "json"
157
+ assert "--json-schema" in cmd_args
158
+ schema_text = cmd_args[cmd_args.index("--json-schema") + 1]
159
+ schema = json.loads(schema_text)
160
+ assert schema["title"] == "ReviewResult"
161
+ assert "findings" in schema["properties"]
162
+
163
+ def test_unwraps_claude_structured_output(
164
+ self,
165
+ tmp_path: Path,
166
+ make_loop_config: Callable[..., LoopConfig],
167
+ ) -> None:
168
+ """Wrapper produced by ``--output-format json`` is unwrapped on disk.
169
+
170
+ Regression for the Claude CLI emitting
171
+ ``{"type":"result", ..., "structured_output": {<review>}}`` — the
172
+ downstream parser expects the review object, not the wrapper.
173
+ """
174
+ review = {
175
+ "findings": [],
176
+ "overall_correctness": "patch is correct",
177
+ "overall_confidence_score": 0.9,
178
+ }
179
+ wrapper = {
180
+ "type": "result",
181
+ "subtype": "success",
182
+ "is_error": False,
183
+ "result": "Done.",
184
+ "structured_output": review,
185
+ }
186
+
187
+ prompts = tmp_path / "prompts"
188
+ prompts.mkdir(exist_ok=True)
189
+ (prompts / "claude-review.prompt.md").write_text("prompt body")
190
+ config = make_loop_config(prompts_dir=prompts)
191
+
192
+ output = tmp_path / "review.json"
193
+
194
+ def fake_retry(
195
+ out_path: Path,
196
+ _label: str,
197
+ _argv: list[str],
198
+ **_kw: object,
199
+ ) -> bool:
200
+ out_path.write_text(json.dumps(wrapper))
201
+ return True
202
+
203
+ with (
204
+ patch("mr_overkill.agents._make_budget_fn") as mb,
205
+ patch("mr_overkill.agents._make_retry_fn") as mr,
206
+ ):
207
+ mb.return_value = MagicMock(return_value=True)
208
+ mr.return_value = fake_retry
209
+ agent = ClaudeReviewAgent(config)
210
+ assert agent(output, 1) is True
211
+
212
+ assert json.loads(output.read_text()) == review
213
+
146
214
  def test_fails_on_missing_prompt(
147
215
  self,
148
216
  tmp_path: Path,
@@ -57,14 +57,56 @@ class TestExtractJsonFromFile:
57
57
  p.write_text(content)
58
58
  assert extract_json_from_file(p) == data
59
59
 
60
- def test_regex_fallback(self, tmp_path: Path) -> None:
61
- """Tier 3: JSON embedded in free text is extracted via regex."""
60
+ def test_balanced_scan_fallback(self, tmp_path: Path) -> None:
61
+ """Tier 3: JSON embedded in free text is extracted by the balanced scan."""
62
62
  data = {"found": True}
63
63
  content = "Random preamble text\n" + json.dumps(data) + "\nmore text\n"
64
64
  p = tmp_path / "mixed.txt"
65
65
  p.write_text(content)
66
66
  assert extract_json_from_file(p) == data
67
67
 
68
+ def test_balanced_scan_with_prose_braces(self, tmp_path: Path) -> None:
69
+ """Tier 3: prose containing ``{`` does not derail extraction.
70
+
71
+ Regression for parse_error encountered when a Claude reviewer prefixed
72
+ its JSON with prose that mentioned schema fields like ``{user_id}``.
73
+ """
74
+ review = {
75
+ "findings": [],
76
+ "overall_correctness": "patch is correct",
77
+ "overall_confidence_score": 0.88,
78
+ }
79
+ content = (
80
+ "Based on my analysis of the iteration 2 review of this diff, "
81
+ "I've examined:\n"
82
+ "- The schema additions like {user_id} and {created_at}\n"
83
+ "- The JSON shape `{findings: [], overall_correctness: ...}`\n"
84
+ "- IAM permissions for s3 access\n"
85
+ "\n"
86
+ "Here is the review:\n"
87
+ f"{json.dumps(review, indent=2)}\n"
88
+ )
89
+ p = tmp_path / "prose.json"
90
+ p.write_text(content)
91
+ assert extract_json_from_file(p) == review
92
+
93
+ def test_balanced_scan_picks_largest_dict(self, tmp_path: Path) -> None:
94
+ """Tier 3: when several ``{...}`` candidates parse, prefer the largest."""
95
+ small = {"k": 1}
96
+ big = {
97
+ "findings": [],
98
+ "overall_correctness": "patch is correct",
99
+ "overall_confidence_score": 0.9,
100
+ }
101
+ content = (
102
+ f"first candidate: {json.dumps(small)}\n"
103
+ "more prose\n"
104
+ f"actual review: {json.dumps(big)}\n"
105
+ )
106
+ p = tmp_path / "multi.txt"
107
+ p.write_text(content)
108
+ assert extract_json_from_file(p) == big
109
+
68
110
  def test_empty_file_returns_none(self, tmp_path: Path) -> None:
69
111
  """Empty (or whitespace-only) file returns None."""
70
112
  p = tmp_path / "empty.json"
@@ -46,6 +46,17 @@ class TestExtractResultFromStream:
46
46
  f.write_text(f"not json with result\n{ok_line}")
47
47
  assert extract_result_from_stream(f) == "ok"
48
48
 
49
+ def test_prefers_structured_output(self, tmp_path: Path) -> None:
50
+ f = tmp_path / "stream.jsonl"
51
+ structured = {"findings": [], "overall_correctness": "patch is correct"}
52
+ line = json.dumps({
53
+ "type": "result",
54
+ "result": "plain text summary",
55
+ "structured_output": structured,
56
+ })
57
+ f.write_text(line)
58
+ assert json.loads(extract_result_from_stream(f)) == structured
59
+
49
60
 
50
61
  class TestWaitForBudget:
51
62
  def test_budget_ok_immediately(self) -> None:
@@ -247,7 +247,7 @@ wheels = [
247
247
 
248
248
  [[package]]
249
249
  name = "overkill"
250
- version = "0.6.1"
250
+ version = "0.7.0"
251
251
  source = { editable = "." }
252
252
 
253
253
  [package.dev-dependencies]
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes
File without changes