codemop 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- codemop/__init__.py +3 -0
- codemop/cli.py +233 -0
- codemop/config.py +66 -0
- codemop/github/__init__.py +1 -0
- codemop/github/client.py +106 -0
- codemop/providers/__init__.py +59 -0
- codemop/providers/anthropic.py +114 -0
- codemop/providers/base.py +84 -0
- codemop/providers/openai_compatible.py +187 -0
- codemop/providers/pricing.py +45 -0
- codemop/review/__init__.py +1 -0
- codemop/review/chunks.py +131 -0
- codemop/review/diff.py +123 -0
- codemop/review/pipeline.py +86 -0
- codemop/review/placement.py +44 -0
- codemop/review/prompt.py +54 -0
- codemop/review/schema.py +39 -0
- codemop-0.1.0.dist-info/METADATA +351 -0
- codemop-0.1.0.dist-info/RECORD +22 -0
- codemop-0.1.0.dist-info/WHEEL +4 -0
- codemop-0.1.0.dist-info/entry_points.txt +2 -0
- codemop-0.1.0.dist-info/licenses/LICENSE +21 -0
|
@@ -0,0 +1,187 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Any provider with an OpenAI-compatible chat completions API: Mistral, OpenAI, OpenRouter,
|
|
3
|
+
and local models through Ollama or vLLM.
|
|
4
|
+
|
|
5
|
+
The review schema is requested with response_format "json_schema". Not every provider
|
|
6
|
+
enforces it strictly, so the answer is validated here, and an invalid one gets a single
|
|
7
|
+
repair attempt with the validation errors before giving up.
|
|
8
|
+
"""
|
|
9
|
+
import asyncio
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
from dataclasses import dataclass
|
|
13
|
+
from typing import Optional
|
|
14
|
+
|
|
15
|
+
import httpx
|
|
16
|
+
import pydantic
|
|
17
|
+
|
|
18
|
+
from codemop.providers.base import DEFAULT_CHUNK_TOKENS, NoReview, Usage
|
|
19
|
+
from codemop.review.schema import ModelReview
|
|
20
|
+
|
|
21
|
+
|
|
22
|
+
@dataclass(frozen=True)
|
|
23
|
+
class Preset:
|
|
24
|
+
base_url: str
|
|
25
|
+
key_env: Optional[str] # None: no key needed
|
|
26
|
+
chunk_tokens: int = DEFAULT_CHUNK_TOKENS
|
|
27
|
+
max_output_tokens: int = 16000
|
|
28
|
+
context_hint: str = "raise the model's context window"
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
PRESETS = {
|
|
32
|
+
"openai": Preset("https://api.openai.com/v1", "OPENAI_API_KEY"),
|
|
33
|
+
"mistral": Preset("https://api.mistral.ai/v1", "MISTRAL_API_KEY"),
|
|
34
|
+
"openrouter": Preset("https://openrouter.ai/api/v1", "OPENROUTER_API_KEY"),
|
|
35
|
+
# Local models run with small context windows (Ollama defaults to 4096 tokens) and
|
|
36
|
+
# are slow on a CPU, so they get smaller chunks and a shorter review
|
|
37
|
+
"ollama": Preset(
|
|
38
|
+
"http://localhost:11434/v1", None, chunk_tokens=8_000, max_output_tokens=4096,
|
|
39
|
+
context_hint="set OLLAMA_CONTEXT_LENGTH for the Ollama server (16384 or more) and restart it",
|
|
40
|
+
),
|
|
41
|
+
}
|
|
42
|
+
|
|
43
|
+
# Fewer than this many characters per reported prompt token means the server dropped
|
|
44
|
+
# part of the input: real text and code run at about 3-4 characters per token
|
|
45
|
+
TRUNCATED_CHARS_PER_TOKEN = 6
|
|
46
|
+
# Below this size the ratio is too noisy to judge
|
|
47
|
+
TRUNCATION_CHECK_MIN_CHARS = 2_000
|
|
48
|
+
|
|
49
|
+
RETRY_STATUSES = {408, 409, 429, 500, 502, 503, 504}
|
|
50
|
+
REPAIR_INSTRUCTION = (
|
|
51
|
+
"That reply didn't match the required JSON schema:\n{errors}\n"
|
|
52
|
+
"Reply again with only the corrected JSON object."
|
|
53
|
+
)
|
|
54
|
+
|
|
55
|
+
|
|
56
|
+
class OpenAICompatibleModel:
|
|
57
|
+
"""A ReviewModel for any OpenAI-compatible chat completions API"""
|
|
58
|
+
|
|
59
|
+
def __init__(
|
|
60
|
+
self,
|
|
61
|
+
model: str,
|
|
62
|
+
*,
|
|
63
|
+
provider: str = "openai",
|
|
64
|
+
base_url: Optional[str] = None,
|
|
65
|
+
api_key: Optional[str] = None,
|
|
66
|
+
key_env: Optional[str] = None,
|
|
67
|
+
max_output_tokens: Optional[int] = None,
|
|
68
|
+
chunk_tokens: Optional[int] = None,
|
|
69
|
+
max_retries: int = 2,
|
|
70
|
+
timeout: float = 300.0,
|
|
71
|
+
transport: Optional[httpx.AsyncBaseTransport] = None,
|
|
72
|
+
):
|
|
73
|
+
preset = PRESETS.get(provider, Preset("", None))
|
|
74
|
+
self.model = model
|
|
75
|
+
self.provider = provider
|
|
76
|
+
self.base_url = (base_url or preset.base_url).rstrip("/")
|
|
77
|
+
if not self.base_url:
|
|
78
|
+
raise ValueError(f"Unknown provider {provider!r}: give a base_url")
|
|
79
|
+
self.key_env = key_env or preset.key_env
|
|
80
|
+
self.api_key = api_key or (os.environ.get(self.key_env) if self.key_env else None)
|
|
81
|
+
self.max_output_tokens = max_output_tokens or preset.max_output_tokens
|
|
82
|
+
self.chunk_tokens = chunk_tokens or preset.chunk_tokens
|
|
83
|
+
self.context_hint = preset.context_hint
|
|
84
|
+
self.max_retries = max_retries
|
|
85
|
+
self.timeout = timeout
|
|
86
|
+
self._transport = transport
|
|
87
|
+
|
|
88
|
+
@property
|
|
89
|
+
def name(self) -> str:
|
|
90
|
+
return f"{self.provider}/{self.model}"
|
|
91
|
+
|
|
92
|
+
def cost(self, usage: Usage) -> Optional[float]:
|
|
93
|
+
# Local models cost nothing per request; hosted prices vary too much to guess
|
|
94
|
+
return 0.0 if self.provider == "ollama" else None
|
|
95
|
+
|
|
96
|
+
def _key_hint(self) -> str:
|
|
97
|
+
return f"check {self.key_env} is set to a valid key" if self.key_env else "check the API key"
|
|
98
|
+
|
|
99
|
+
async def _complete(self, client: httpx.AsyncClient, messages: list) -> tuple[dict, Usage]:
|
|
100
|
+
body = {
|
|
101
|
+
"model": self.model,
|
|
102
|
+
"messages": messages,
|
|
103
|
+
"max_tokens": self.max_output_tokens,
|
|
104
|
+
"response_format": {
|
|
105
|
+
"type": "json_schema",
|
|
106
|
+
"json_schema": {"name": "review", "schema": ModelReview.model_json_schema()},
|
|
107
|
+
},
|
|
108
|
+
}
|
|
109
|
+
for attempt in range(self.max_retries + 1):
|
|
110
|
+
try:
|
|
111
|
+
response = await client.post("/chat/completions", json=body)
|
|
112
|
+
except httpx.TransportError:
|
|
113
|
+
if attempt < self.max_retries:
|
|
114
|
+
await asyncio.sleep(2 ** attempt)
|
|
115
|
+
continue
|
|
116
|
+
raise NoReview(f"Couldn't connect to {self.base_url}; check the network and base URL")
|
|
117
|
+
if response.status_code in RETRY_STATUSES and attempt < self.max_retries:
|
|
118
|
+
await asyncio.sleep(2 ** attempt)
|
|
119
|
+
continue
|
|
120
|
+
break
|
|
121
|
+
|
|
122
|
+
status = response.status_code
|
|
123
|
+
if status in (401, 403):
|
|
124
|
+
raise NoReview(f"{self.provider} rejected the API key: {self._key_hint()}", fatal=True)
|
|
125
|
+
if status == 404:
|
|
126
|
+
raise NoReview(f"Model {self.model} wasn't found at {self.base_url}", fatal=True)
|
|
127
|
+
if status == 429:
|
|
128
|
+
raise NoReview(f"{self.provider} rate limit reached (after retries); try again later")
|
|
129
|
+
if status == 400 and "response_format" in response.text:
|
|
130
|
+
raise NoReview(
|
|
131
|
+
f"{self.name} doesn't support structured output (response_format json_schema)",
|
|
132
|
+
fatal=True,
|
|
133
|
+
)
|
|
134
|
+
if status >= 400:
|
|
135
|
+
raise NoReview(f"{self.provider} returned an error ({status}): {response.text[:200]}")
|
|
136
|
+
|
|
137
|
+
data = response.json()
|
|
138
|
+
usage = data.get("usage") or {}
|
|
139
|
+
result = Usage(
|
|
140
|
+
input_tokens=usage.get("prompt_tokens", 0),
|
|
141
|
+
output_tokens=usage.get("completion_tokens", 0),
|
|
142
|
+
)
|
|
143
|
+
# Some servers (Ollama among them) silently drop input that doesn't fit the
|
|
144
|
+
# context window and review what's left; never pass that off as a review
|
|
145
|
+
sent = sum(len(message["content"]) for message in messages)
|
|
146
|
+
if sent >= TRUNCATION_CHECK_MIN_CHARS and 0 < result.input_tokens < sent / TRUNCATED_CHARS_PER_TOKEN:
|
|
147
|
+
raise NoReview(
|
|
148
|
+
f"{self.name} only read {result.input_tokens:,} tokens of a request of about "
|
|
149
|
+
f"{sent // 4:,}: its context window has to fit the request plus a review of up to "
|
|
150
|
+
f"{self.max_output_tokens:,} tokens. To fix it, {self.context_hint}, or lower --chunk-tokens",
|
|
151
|
+
fatal=True,
|
|
152
|
+
usage=result,
|
|
153
|
+
)
|
|
154
|
+
return data, result
|
|
155
|
+
|
|
156
|
+
async def review(self, instructions: str, diff_text: str) -> tuple[ModelReview, Usage]:
|
|
157
|
+
headers = {"Authorization": f"Bearer {self.api_key}"} if self.api_key else {}
|
|
158
|
+
messages = [
|
|
159
|
+
{"role": "system", "content": instructions},
|
|
160
|
+
{"role": "user", "content": diff_text},
|
|
161
|
+
]
|
|
162
|
+
total = Usage()
|
|
163
|
+
async with httpx.AsyncClient(
|
|
164
|
+
base_url=self.base_url, headers=headers, timeout=self.timeout, transport=self._transport
|
|
165
|
+
) as client:
|
|
166
|
+
for attempt in range(2): # the answer, then one repair attempt
|
|
167
|
+
data, usage = await self._complete(client, messages)
|
|
168
|
+
total += usage
|
|
169
|
+
choice = (data.get("choices") or [{}])[0]
|
|
170
|
+
message = choice.get("message") or {}
|
|
171
|
+
|
|
172
|
+
if message.get("refusal") or choice.get("finish_reason") == "content_filter":
|
|
173
|
+
raise NoReview.refused(self.name, total)
|
|
174
|
+
if choice.get("finish_reason") == "length":
|
|
175
|
+
raise NoReview.cut_off(self.name, self.max_output_tokens, total)
|
|
176
|
+
|
|
177
|
+
content = message.get("content") or ""
|
|
178
|
+
try:
|
|
179
|
+
return ModelReview.model_validate_json(content), total
|
|
180
|
+
except pydantic.ValidationError as e:
|
|
181
|
+
errors = json.dumps(e.errors(include_url=False, include_context=False), default=str)[:2000]
|
|
182
|
+
messages += [
|
|
183
|
+
{"role": "assistant", "content": content},
|
|
184
|
+
{"role": "user", "content": REPAIR_INSTRUCTION.format(errors=errors)},
|
|
185
|
+
]
|
|
186
|
+
|
|
187
|
+
raise NoReview(f"{self.name} returned a review that doesn't match the schema, even after a retry", usage=total)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Estimating what a review cost, from list prices.
|
|
3
|
+
|
|
4
|
+
Prices change, so these are dated and the result is always described as an estimate. A model
|
|
5
|
+
without a known price gets no estimate rather than a guess.
|
|
6
|
+
"""
|
|
7
|
+
from dataclasses import dataclass
|
|
8
|
+
from typing import Dict, Optional
|
|
9
|
+
|
|
10
|
+
from codemop.providers.base import Usage
|
|
11
|
+
|
|
12
|
+
PRICES_AS_OF = "2026-09-25"
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class Price:
|
|
17
|
+
"""US dollars per million tokens"""
|
|
18
|
+
input: float
|
|
19
|
+
output: float
|
|
20
|
+
cache_read: Optional[float] = None # default: a tenth of the input price
|
|
21
|
+
|
|
22
|
+
def cost(self, usage: Usage) -> float:
|
|
23
|
+
cache_read = self.cache_read if self.cache_read is not None else self.input / 10
|
|
24
|
+
return (
|
|
25
|
+
usage.input_tokens * self.input
|
|
26
|
+
+ usage.output_tokens * self.output
|
|
27
|
+
+ usage.cache_read_tokens * cache_read
|
|
28
|
+
+ usage.cache_write_tokens * self.input * 1.25 # five-minute cache writes
|
|
29
|
+
) / 1_000_000
|
|
30
|
+
|
|
31
|
+
|
|
32
|
+
# Anthropic list prices on the Claude API
|
|
33
|
+
ANTHROPIC_PRICES: Dict[str, Price] = {
|
|
34
|
+
"claude-fable-5-1": Price(10.00, 50.00, cache_read=0.25),
|
|
35
|
+
"claude-fable-5": Price(10.00, 50.00),
|
|
36
|
+
"claude-opus-5-5": Price(4.00, 20.00, cache_read=0.20),
|
|
37
|
+
"claude-opus-5": Price(5.00, 25.00),
|
|
38
|
+
"claude-opus-4-8": Price(5.00, 25.00),
|
|
39
|
+
"claude-opus-4-7": Price(5.00, 25.00),
|
|
40
|
+
"claude-opus-4-6": Price(5.00, 25.00),
|
|
41
|
+
"claude-sonnet-5-5": Price(2.00, 10.00, cache_read=0.20),
|
|
42
|
+
"claude-sonnet-5": Price(2.00, 10.00),
|
|
43
|
+
"claude-sonnet-4-6": Price(3.00, 15.00),
|
|
44
|
+
"claude-haiku-4-5": Price(1.00, 5.00),
|
|
45
|
+
}
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Turning a diff into validated review suggestions; independent of any model provider."""
|
codemop/review/chunks.py
ADDED
|
@@ -0,0 +1,131 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Splitting a diff into chunks that each fit in one model request.
|
|
3
|
+
|
|
4
|
+
Files are packed together up to a token budget. A file too big for one chunk is split by
|
|
5
|
+
hunk, and anything that still doesn't fit, or isn't worth reviewing (lock files, binaries,
|
|
6
|
+
deletions), is skipped with a reason, so a large PR is never silently cut short.
|
|
7
|
+
"""
|
|
8
|
+
import math
|
|
9
|
+
from dataclasses import dataclass, field, replace
|
|
10
|
+
from fnmatch import fnmatch
|
|
11
|
+
from typing import Callable, Iterable, List, Sequence
|
|
12
|
+
|
|
13
|
+
from codemop.review.diff import FileDiff, Hunk
|
|
14
|
+
from codemop.review.prompt import render_hunks
|
|
15
|
+
|
|
16
|
+
# Generated files that are rarely worth reviewing
|
|
17
|
+
DEFAULT_IGNORED_PATHS = (
|
|
18
|
+
"*.lock",
|
|
19
|
+
"package-lock.json",
|
|
20
|
+
"pnpm-lock.yaml",
|
|
21
|
+
"go.sum",
|
|
22
|
+
"*.min.js",
|
|
23
|
+
"*.min.css",
|
|
24
|
+
"*.map",
|
|
25
|
+
"*.svg",
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
def estimate_tokens(text: str) -> int:
|
|
30
|
+
"""A deliberately high estimate (code averages more than 3 characters per token)"""
|
|
31
|
+
return math.ceil(len(text) / 3)
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
@dataclass
|
|
35
|
+
class Chunk:
|
|
36
|
+
"""Part of the diff, rendered for the model, and the files it shows (with only the hunks it shows)"""
|
|
37
|
+
files: List[FileDiff] = field(default_factory=list)
|
|
38
|
+
parts: List[str] = field(default_factory=list)
|
|
39
|
+
tokens: int = 0
|
|
40
|
+
|
|
41
|
+
@property
|
|
42
|
+
def text(self) -> str:
|
|
43
|
+
return "\n\n".join(self.parts)
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class Skipped:
|
|
48
|
+
path: str
|
|
49
|
+
reason: str
|
|
50
|
+
|
|
51
|
+
|
|
52
|
+
@dataclass
|
|
53
|
+
class ChunkPlan:
|
|
54
|
+
chunks: List[Chunk]
|
|
55
|
+
skipped: List[Skipped]
|
|
56
|
+
|
|
57
|
+
|
|
58
|
+
def is_ignored(path: str, patterns: Sequence[str]) -> bool:
|
|
59
|
+
name = path.rsplit("/", 1)[-1]
|
|
60
|
+
return any(fnmatch(path, pattern) or fnmatch(name, pattern) for pattern in patterns)
|
|
61
|
+
|
|
62
|
+
|
|
63
|
+
def _skip_reason(file: FileDiff, ignored_paths: Sequence[str]) -> str | None:
|
|
64
|
+
if file.status == "deleted":
|
|
65
|
+
return "deleted file"
|
|
66
|
+
if file.is_binary:
|
|
67
|
+
return "binary file"
|
|
68
|
+
if not file.hunks:
|
|
69
|
+
return "no changed lines" + (" (rename only)" if file.status == "renamed" else "")
|
|
70
|
+
if is_ignored(file.path, ignored_paths):
|
|
71
|
+
return "matches an ignored path pattern"
|
|
72
|
+
return None
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
def _pieces(
|
|
76
|
+
file: FileDiff, budget: int, count_tokens: Callable[[str], int], skipped: List[Skipped]
|
|
77
|
+
) -> Iterable[tuple[str, int, List[Hunk]]]:
|
|
78
|
+
"""The file as one rendered piece if it fits, otherwise its hunks packed into pieces"""
|
|
79
|
+
whole = render_hunks(file, file.hunks)
|
|
80
|
+
whole_tokens = count_tokens(whole)
|
|
81
|
+
if whole_tokens <= budget:
|
|
82
|
+
yield whole, whole_tokens, list(file.hunks)
|
|
83
|
+
return
|
|
84
|
+
|
|
85
|
+
group: List[Hunk] = []
|
|
86
|
+
for hunk in file.hunks:
|
|
87
|
+
if count_tokens(render_hunks(file, [hunk])) > budget:
|
|
88
|
+
skipped.append(Skipped(file.path, f"hunk too large to review: {hunk.header}"))
|
|
89
|
+
continue
|
|
90
|
+
if group and count_tokens(render_hunks(file, group + [hunk])) > budget:
|
|
91
|
+
rendered = render_hunks(file, group)
|
|
92
|
+
yield rendered, count_tokens(rendered), group
|
|
93
|
+
group = []
|
|
94
|
+
group.append(hunk)
|
|
95
|
+
if group:
|
|
96
|
+
rendered = render_hunks(file, group)
|
|
97
|
+
yield rendered, count_tokens(rendered), group
|
|
98
|
+
|
|
99
|
+
|
|
100
|
+
def plan_chunks(
|
|
101
|
+
files: Sequence[FileDiff],
|
|
102
|
+
budget_tokens: int,
|
|
103
|
+
count_tokens: Callable[[str], int] = estimate_tokens,
|
|
104
|
+
ignored_paths: Sequence[str] = DEFAULT_IGNORED_PATHS,
|
|
105
|
+
) -> ChunkPlan:
|
|
106
|
+
"""Pack the reviewable parts of a diff into chunks of at most `budget_tokens` each"""
|
|
107
|
+
chunks: List[Chunk] = []
|
|
108
|
+
skipped: List[Skipped] = []
|
|
109
|
+
current = Chunk()
|
|
110
|
+
|
|
111
|
+
for file in files:
|
|
112
|
+
reason = _skip_reason(file, ignored_paths)
|
|
113
|
+
if reason:
|
|
114
|
+
skipped.append(Skipped(file.path, reason))
|
|
115
|
+
continue
|
|
116
|
+
for rendered, tokens, hunks in _pieces(file, budget_tokens, count_tokens, skipped):
|
|
117
|
+
if current.parts and current.tokens + tokens > budget_tokens:
|
|
118
|
+
chunks.append(current)
|
|
119
|
+
current = Chunk()
|
|
120
|
+
current.parts.append(rendered)
|
|
121
|
+
current.tokens += tokens
|
|
122
|
+
# Only the hunks shown, so suggestions can only be placed on lines the model saw
|
|
123
|
+
shown = next((f for f in current.files if f.path == file.path), None)
|
|
124
|
+
if shown:
|
|
125
|
+
shown.hunks.extend(hunks)
|
|
126
|
+
else:
|
|
127
|
+
current.files.append(replace(file, hunks=list(hunks)))
|
|
128
|
+
|
|
129
|
+
if current.parts:
|
|
130
|
+
chunks.append(current)
|
|
131
|
+
return ChunkPlan(chunks=chunks, skipped=skipped)
|
codemop/review/diff.py
ADDED
|
@@ -0,0 +1,123 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Parsing unified diffs (as GitHub returns them) into files, hunks and lines.
|
|
3
|
+
|
|
4
|
+
Each line keeps its number in the old and new versions of the file, because GitHub only
|
|
5
|
+
accepts review comments on lines that appear in the diff.
|
|
6
|
+
"""
|
|
7
|
+
import re
|
|
8
|
+
from dataclasses import dataclass, field
|
|
9
|
+
from typing import List, Literal, Optional, Set
|
|
10
|
+
|
|
11
|
+
HUNK_HEADER = re.compile(r"^@@ -(\d+)(?:,\d+)? \+(\d+)(?:,\d+)? @@")
|
|
12
|
+
|
|
13
|
+
LineKind = Literal["added", "removed", "context"]
|
|
14
|
+
FileStatus = Literal["added", "deleted", "modified", "renamed"]
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
@dataclass(frozen=True)
|
|
18
|
+
class DiffLine:
|
|
19
|
+
kind: LineKind
|
|
20
|
+
text: str
|
|
21
|
+
old_number: Optional[int] # None for added lines
|
|
22
|
+
new_number: Optional[int] # None for removed lines
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
@dataclass
|
|
26
|
+
class Hunk:
|
|
27
|
+
header: str # the whole "@@ -a,b +c,d @@ context" line
|
|
28
|
+
lines: List[DiffLine] = field(default_factory=list)
|
|
29
|
+
|
|
30
|
+
|
|
31
|
+
@dataclass
|
|
32
|
+
class FileDiff:
|
|
33
|
+
old_path: Optional[str] # None for an added file
|
|
34
|
+
new_path: Optional[str] # None for a deleted file
|
|
35
|
+
status: FileStatus
|
|
36
|
+
hunks: List[Hunk] = field(default_factory=list)
|
|
37
|
+
is_binary: bool = False
|
|
38
|
+
|
|
39
|
+
@property
|
|
40
|
+
def path(self) -> str:
|
|
41
|
+
return self.new_path or self.old_path or ""
|
|
42
|
+
|
|
43
|
+
def commentable_lines(self) -> Set[int]:
|
|
44
|
+
"""New-file line numbers a review comment can go on: added or unchanged lines shown in the diff"""
|
|
45
|
+
return {
|
|
46
|
+
line.new_number
|
|
47
|
+
for hunk in self.hunks
|
|
48
|
+
for line in hunk.lines
|
|
49
|
+
if line.new_number is not None
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def _strip_prefix(path: str, prefix: str) -> Optional[str]:
|
|
54
|
+
if path == "/dev/null":
|
|
55
|
+
return None
|
|
56
|
+
return path[len(prefix):] if path.startswith(prefix) else path
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def _paths_from_git_header(line: str) -> tuple[Optional[str], Optional[str]]:
|
|
60
|
+
"""Paths from "diff --git a/x b/y", used when there are no ---/+++ lines (renames, binaries)"""
|
|
61
|
+
rest = line[len("diff --git "):]
|
|
62
|
+
split = rest.rfind(" b/")
|
|
63
|
+
if not rest.startswith("a/") or split == -1:
|
|
64
|
+
return None, None
|
|
65
|
+
return rest[2:split], rest[split + 3:]
|
|
66
|
+
|
|
67
|
+
|
|
68
|
+
def parse_diff(diff: str) -> List[FileDiff]:
|
|
69
|
+
"""Parse a unified diff, as `git diff` or the GitHub API produce, into files"""
|
|
70
|
+
files: List[FileDiff] = []
|
|
71
|
+
current: Optional[FileDiff] = None
|
|
72
|
+
hunk: Optional[Hunk] = None
|
|
73
|
+
old_number = new_number = 0
|
|
74
|
+
|
|
75
|
+
for line in diff.splitlines():
|
|
76
|
+
if line.startswith("diff --git "):
|
|
77
|
+
old_path, new_path = _paths_from_git_header(line)
|
|
78
|
+
current = FileDiff(old_path=old_path, new_path=new_path, status="modified")
|
|
79
|
+
files.append(current)
|
|
80
|
+
hunk = None
|
|
81
|
+
continue
|
|
82
|
+
if current is None:
|
|
83
|
+
continue
|
|
84
|
+
|
|
85
|
+
if hunk is None:
|
|
86
|
+
# File header lines, before the first hunk
|
|
87
|
+
if line.startswith("new file mode"):
|
|
88
|
+
current.status, current.old_path = "added", None
|
|
89
|
+
elif line.startswith("deleted file mode"):
|
|
90
|
+
current.status, current.new_path = "deleted", None
|
|
91
|
+
elif line.startswith("rename from "):
|
|
92
|
+
current.status, current.old_path = "renamed", line[len("rename from "):]
|
|
93
|
+
elif line.startswith("rename to "):
|
|
94
|
+
current.new_path = line[len("rename to "):]
|
|
95
|
+
elif line.startswith("Binary files ") or line.startswith("GIT binary patch"):
|
|
96
|
+
current.is_binary = True
|
|
97
|
+
elif line.startswith("--- "):
|
|
98
|
+
current.old_path = _strip_prefix(line[4:], "a/")
|
|
99
|
+
elif line.startswith("+++ "):
|
|
100
|
+
current.new_path = _strip_prefix(line[4:], "b/")
|
|
101
|
+
|
|
102
|
+
match = HUNK_HEADER.match(line)
|
|
103
|
+
if match:
|
|
104
|
+
old_number, new_number = int(match.group(1)), int(match.group(2))
|
|
105
|
+
hunk = Hunk(header=line)
|
|
106
|
+
current.hunks.append(hunk)
|
|
107
|
+
continue
|
|
108
|
+
if hunk is None or line.startswith("\\"): # ""
|
|
109
|
+
continue
|
|
110
|
+
|
|
111
|
+
if line.startswith("+"):
|
|
112
|
+
hunk.lines.append(DiffLine("added", line[1:], None, new_number))
|
|
113
|
+
new_number += 1
|
|
114
|
+
elif line.startswith("-"):
|
|
115
|
+
hunk.lines.append(DiffLine("removed", line[1:], old_number, None))
|
|
116
|
+
old_number += 1
|
|
117
|
+
else:
|
|
118
|
+
# Context lines start with a space; some tools drop it from empty lines
|
|
119
|
+
hunk.lines.append(DiffLine("context", line[1:], old_number, new_number))
|
|
120
|
+
old_number += 1
|
|
121
|
+
new_number += 1
|
|
122
|
+
|
|
123
|
+
return files
|
|
@@ -0,0 +1,86 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Reviewing a whole diff: chunk it, review the chunks concurrently, and gather one report
|
|
3
|
+
that says what was reviewed and what wasn't, and why.
|
|
4
|
+
"""
|
|
5
|
+
import asyncio
|
|
6
|
+
from dataclasses import dataclass, field
|
|
7
|
+
from typing import List, Literal, Optional, Sequence
|
|
8
|
+
|
|
9
|
+
from codemop.config import DEFAULT_MIN_CONFIDENCE
|
|
10
|
+
from codemop.providers.base import NoReview, NoReviewKind, ReviewModel, Usage
|
|
11
|
+
from codemop.review.chunks import DEFAULT_IGNORED_PATHS, Skipped, plan_chunks
|
|
12
|
+
from codemop.review.diff import parse_diff
|
|
13
|
+
from codemop.review.placement import Unplaced, place_suggestions
|
|
14
|
+
from codemop.review.prompt import SYSTEM_PROMPT
|
|
15
|
+
from codemop.review.schema import ModelSuggestion
|
|
16
|
+
|
|
17
|
+
DEFAULT_CONCURRENCY = 4
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class FailedChunk:
|
|
22
|
+
paths: List[str]
|
|
23
|
+
reason: str
|
|
24
|
+
kind: NoReviewKind | Literal["not_reviewed"] = "error" # not_reviewed: skipped after a fatal error
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
@dataclass
|
|
28
|
+
class ReviewReport:
|
|
29
|
+
model: str
|
|
30
|
+
suggestions: List[ModelSuggestion] = field(default_factory=list)
|
|
31
|
+
unplaced: List[Unplaced] = field(default_factory=list) # suggestions on lines GitHub can't comment on
|
|
32
|
+
below_confidence: int = 0 # suggestions dropped by min_confidence
|
|
33
|
+
skipped: List[Skipped] = field(default_factory=list) # parts of the diff not reviewed
|
|
34
|
+
failed: List[FailedChunk] = field(default_factory=list) # chunks the model gave no review for
|
|
35
|
+
stopped: Optional[str] = None # why the review stopped early, if it did
|
|
36
|
+
chunks: int = 0
|
|
37
|
+
usage: Usage = Usage()
|
|
38
|
+
cost: Optional[float] = None # estimated US dollars at list prices; None if unknown
|
|
39
|
+
|
|
40
|
+
@property
|
|
41
|
+
def complete(self) -> bool:
|
|
42
|
+
"""Every reviewable part of the diff got a review"""
|
|
43
|
+
return not self.failed and self.stopped is None
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
async def review_diff(
|
|
47
|
+
diff: str,
|
|
48
|
+
model: ReviewModel,
|
|
49
|
+
*,
|
|
50
|
+
chunk_tokens: Optional[int] = None,
|
|
51
|
+
concurrency: int = DEFAULT_CONCURRENCY,
|
|
52
|
+
ignored_paths: Sequence[str] = DEFAULT_IGNORED_PATHS,
|
|
53
|
+
min_confidence: float = DEFAULT_MIN_CONFIDENCE,
|
|
54
|
+
) -> ReviewReport:
|
|
55
|
+
"""Review a unified diff with `model` (chunk_tokens defaults to the model's own chunk size)"""
|
|
56
|
+
plan = plan_chunks(parse_diff(diff), chunk_tokens or model.chunk_tokens, ignored_paths=ignored_paths)
|
|
57
|
+
report = ReviewReport(model=model.name, skipped=list(plan.skipped), chunks=len(plan.chunks))
|
|
58
|
+
limit = asyncio.Semaphore(concurrency)
|
|
59
|
+
|
|
60
|
+
async def review_chunk(chunk):
|
|
61
|
+
async with limit:
|
|
62
|
+
paths = [file.path for file in chunk.files]
|
|
63
|
+
if report.stopped:
|
|
64
|
+
report.failed.append(FailedChunk(paths, f"not reviewed: {report.stopped}", "not_reviewed"))
|
|
65
|
+
return
|
|
66
|
+
try:
|
|
67
|
+
review, usage = await model.review(SYSTEM_PROMPT, chunk.text)
|
|
68
|
+
except NoReview as e:
|
|
69
|
+
report.usage += e.usage
|
|
70
|
+
report.failed.append(FailedChunk(paths, e.reason, e.kind))
|
|
71
|
+
if e.fatal and not report.stopped:
|
|
72
|
+
report.stopped = e.reason
|
|
73
|
+
return
|
|
74
|
+
report.usage += usage
|
|
75
|
+
placement = place_suggestions(review.suggestions, chunk.files)
|
|
76
|
+
report.unplaced.extend(placement.unplaced)
|
|
77
|
+
for suggestion in placement.placed:
|
|
78
|
+
if suggestion.confidence >= min_confidence:
|
|
79
|
+
report.suggestions.append(suggestion)
|
|
80
|
+
else:
|
|
81
|
+
report.below_confidence += 1
|
|
82
|
+
|
|
83
|
+
await asyncio.gather(*(review_chunk(chunk) for chunk in plan.chunks))
|
|
84
|
+
report.suggestions.sort(key=lambda s: (s.file_path, s.line))
|
|
85
|
+
report.cost = model.cost(report.usage)
|
|
86
|
+
return report
|
|
@@ -0,0 +1,44 @@
|
|
|
1
|
+
"""
|
|
2
|
+
Checking that each suggestion points at lines GitHub can comment on.
|
|
3
|
+
|
|
4
|
+
A review comment has to be on an added or unchanged line shown in the diff; a model that
|
|
5
|
+
points elsewhere (a removed line, a line outside the hunks, a file it wasn't shown) has
|
|
6
|
+
made a mistake, and that suggestion is set aside rather than posted in the wrong place.
|
|
7
|
+
"""
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from typing import Dict, List, Sequence, Set
|
|
10
|
+
|
|
11
|
+
from codemop.review.diff import FileDiff
|
|
12
|
+
from codemop.review.schema import ModelSuggestion
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
@dataclass(frozen=True)
|
|
16
|
+
class Unplaced:
|
|
17
|
+
suggestion: ModelSuggestion
|
|
18
|
+
reason: str
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
@dataclass
|
|
22
|
+
class Placement:
|
|
23
|
+
placed: List[ModelSuggestion]
|
|
24
|
+
unplaced: List[Unplaced]
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
def place_suggestions(suggestions: Sequence[ModelSuggestion], files: Sequence[FileDiff]) -> Placement:
|
|
28
|
+
"""Split suggestions into those on commentable lines and those that aren't"""
|
|
29
|
+
commentable: Dict[str, Set[int]] = {file.path: file.commentable_lines() for file in files}
|
|
30
|
+
placed: List[ModelSuggestion] = []
|
|
31
|
+
unplaced: List[Unplaced] = []
|
|
32
|
+
|
|
33
|
+
for suggestion in suggestions:
|
|
34
|
+
lines = commentable.get(suggestion.file_path)
|
|
35
|
+
end_line = suggestion.end_line or suggestion.line
|
|
36
|
+
if lines is None:
|
|
37
|
+
unplaced.append(Unplaced(suggestion, "file isn't in the reviewed diff"))
|
|
38
|
+
elif end_line < suggestion.line:
|
|
39
|
+
unplaced.append(Unplaced(suggestion, "end_line is before line"))
|
|
40
|
+
elif not set(range(suggestion.line, end_line + 1)) <= lines:
|
|
41
|
+
unplaced.append(Unplaced(suggestion, "lines aren't added or unchanged lines in the diff"))
|
|
42
|
+
else:
|
|
43
|
+
placed.append(suggestion)
|
|
44
|
+
return Placement(placed=placed, unplaced=unplaced)
|
codemop/review/prompt.py
ADDED
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
"""
|
|
2
|
+
The instructions given to every model, and how a diff is shown to it.
|
|
3
|
+
|
|
4
|
+
Each line is shown with its new-file line number, so the model can say exactly which line
|
|
5
|
+
an issue is on; those numbers are what GitHub review comments use.
|
|
6
|
+
"""
|
|
7
|
+
from typing import Iterable
|
|
8
|
+
|
|
9
|
+
from codemop.review.diff import FileDiff, Hunk
|
|
10
|
+
|
|
11
|
+
SYSTEM_PROMPT = """\
|
|
12
|
+
You are reviewing a pull request. Report real problems in the changed code: bugs, security
|
|
13
|
+
issues, performance problems, and code that is hard to maintain in a way that will cause bugs.
|
|
14
|
+
Don't report style preferences, naming, formatting or missing comments.
|
|
15
|
+
|
|
16
|
+
The diff shows each file's changes in hunks. Every line has a marker and a line number:
|
|
17
|
+
"+ 12 | ..." is an added line, number 12 in the new version of the file
|
|
18
|
+
" 12 | ..." is an unchanged line, number 12 in the new version
|
|
19
|
+
"- | ..." is a removed line; it has no new line number, so never point at it
|
|
20
|
+
|
|
21
|
+
For each problem:
|
|
22
|
+
- file_path is the path exactly as it appears after "###"
|
|
23
|
+
- line (and end_line, for several lines) are new-file numbers of added or unchanged lines
|
|
24
|
+
shown in the diff. Prefer pointing at added lines
|
|
25
|
+
- suggested_code, if you have a concrete fix, replaces lines line..end_line exactly: the
|
|
26
|
+
complete new code for those lines, with no "+"/"-" markers and no line numbers
|
|
27
|
+
- confidence is how sure you are the problem is real
|
|
28
|
+
|
|
29
|
+
Only report problems you are confident about. If the changes look fine, return no suggestions.\
|
|
30
|
+
"""
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def render_hunks(file: FileDiff, hunks: Iterable[Hunk]) -> str:
|
|
34
|
+
"""A file's hunks with markers and new-file line numbers, as the model sees them"""
|
|
35
|
+
hunks = list(hunks)
|
|
36
|
+
width = max(
|
|
37
|
+
(len(str(line.new_number)) for hunk in hunks for line in hunk.lines if line.new_number),
|
|
38
|
+
default=1,
|
|
39
|
+
)
|
|
40
|
+
title = f"### {file.path} ({file.status}"
|
|
41
|
+
if file.status == "renamed" and file.old_path:
|
|
42
|
+
title += f" from {file.old_path}"
|
|
43
|
+
parts = [title + ")"]
|
|
44
|
+
for hunk in hunks:
|
|
45
|
+
parts.append(hunk.header)
|
|
46
|
+
for line in hunk.lines:
|
|
47
|
+
marker = {"added": "+", "removed": "-", "context": " "}[line.kind]
|
|
48
|
+
number = str(line.new_number).rjust(width) if line.new_number else " " * width
|
|
49
|
+
parts.append(f"{marker} {number} | {line.text}")
|
|
50
|
+
return "\n".join(parts)
|
|
51
|
+
|
|
52
|
+
|
|
53
|
+
def render_file(file: FileDiff) -> str:
|
|
54
|
+
return render_hunks(file, file.hunks)
|