open-code-review-toolkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ocr_toolkit/__init__.py +1 -0
- ocr_toolkit/_version.py +24 -0
- ocr_toolkit/cli.py +56 -0
- ocr_toolkit/common/__init__.py +1 -0
- ocr_toolkit/common/language.py +60 -0
- ocr_toolkit/common/markdown.py +214 -0
- ocr_toolkit/common/redaction.py +324 -0
- ocr_toolkit/config_writer.py +106 -0
- ocr_toolkit/configure.py +194 -0
- ocr_toolkit/context/__init__.py +1 -0
- ocr_toolkit/context/__main__.py +8 -0
- ocr_toolkit/context/ansible.py +561 -0
- ocr_toolkit/context/categorize.py +162 -0
- ocr_toolkit/context/instructions.py +269 -0
- ocr_toolkit/context/manifests.py +459 -0
- ocr_toolkit/context/planner.py +227 -0
- ocr_toolkit/context/render.py +955 -0
- ocr_toolkit/context/repo.py +672 -0
- ocr_toolkit/context/settings.py +97 -0
- ocr_toolkit/mcp_config.py +257 -0
- ocr_toolkit/posting/__init__.py +1 -0
- ocr_toolkit/posting/__main__.py +8 -0
- ocr_toolkit/posting/comments.py +87 -0
- ocr_toolkit/posting/formatting.py +755 -0
- ocr_toolkit/posting/gitlab.py +853 -0
- ocr_toolkit/posting/markers.py +284 -0
- ocr_toolkit/posting/payloads.py +141 -0
- ocr_toolkit/posting/result.py +116 -0
- ocr_toolkit/posting/settings.py +181 -0
- ocr_toolkit/posting/snapshot.py +468 -0
- ocr_toolkit/posting/workflow.py +873 -0
- ocr_toolkit/preflight.py +395 -0
- ocr_toolkit/py.typed +1 -0
- open_code_review_toolkit-0.1.0.dist-info/METADATA +283 -0
- open_code_review_toolkit-0.1.0.dist-info/RECORD +38 -0
- open_code_review_toolkit-0.1.0.dist-info/WHEEL +4 -0
- open_code_review_toolkit-0.1.0.dist-info/entry_points.txt +2 -0
- open_code_review_toolkit-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,755 @@
|
|
|
1
|
+
"""Markdown formatting for OCR findings and summary notes."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections.abc import Mapping, Sequence
|
|
7
|
+
from typing import Any
|
|
8
|
+
|
|
9
|
+
from ocr_toolkit.common.markdown import (
|
|
10
|
+
escape_control_chars,
|
|
11
|
+
markdown_code_block,
|
|
12
|
+
neutralize_quick_actions,
|
|
13
|
+
neutralize_suggestion_fences,
|
|
14
|
+
)
|
|
15
|
+
from ocr_toolkit.common.markdown import (
|
|
16
|
+
inline_code as _inline_code,
|
|
17
|
+
)
|
|
18
|
+
from ocr_toolkit.posting.comments import (
|
|
19
|
+
clean_text,
|
|
20
|
+
code_text,
|
|
21
|
+
comment_line,
|
|
22
|
+
compact_control_text,
|
|
23
|
+
compact_escaped_text,
|
|
24
|
+
line_number,
|
|
25
|
+
)
|
|
26
|
+
from ocr_toolkit.posting.payloads import truncate_code_text, truncate_note_body
|
|
27
|
+
from ocr_toolkit.posting.settings import (
|
|
28
|
+
FALLBACK_NOTE_CHUNK_BUDGET,
|
|
29
|
+
MAX_FALLBACK_CODE_DETAILS_CHARS,
|
|
30
|
+
MAX_REVIEWER_GUIDE_COMMENTS,
|
|
31
|
+
MAX_REVIEWER_GUIDE_LABEL_CHARS,
|
|
32
|
+
MAX_REVIEWER_GUIDE_LOCATION_CHARS,
|
|
33
|
+
MAX_REVIEWER_GUIDE_TEXT_CHARS,
|
|
34
|
+
MAX_SUGGESTION_CODE_CHARS,
|
|
35
|
+
MAX_SUGGESTION_SPAN_LINES,
|
|
36
|
+
MAX_TOOL_CALL_NAME_CHARS,
|
|
37
|
+
MAX_TOOL_CALL_SUMMARY_TOOLS,
|
|
38
|
+
SUGGESTION_HEADER,
|
|
39
|
+
post_mode,
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
OCR_FINDING_CATEGORIES = {
|
|
43
|
+
"bug",
|
|
44
|
+
"security",
|
|
45
|
+
"performance",
|
|
46
|
+
"maintainability",
|
|
47
|
+
"test",
|
|
48
|
+
"style",
|
|
49
|
+
"documentation",
|
|
50
|
+
"other",
|
|
51
|
+
}
|
|
52
|
+
|
|
53
|
+
OCR_FINDING_SEVERITIES = {"critical", "high", "medium", "low"}
|
|
54
|
+
OCR_FINDING_SEVERITY_ORDER = ("critical", "high", "medium", "low")
|
|
55
|
+
OCR_FINDING_CATEGORY_ORDER = (
|
|
56
|
+
"security",
|
|
57
|
+
"bug",
|
|
58
|
+
"performance",
|
|
59
|
+
"maintainability",
|
|
60
|
+
"test",
|
|
61
|
+
"documentation",
|
|
62
|
+
"style",
|
|
63
|
+
"other",
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def suggestion_range_suffix(comment: dict[str, Any]) -> str:
|
|
68
|
+
"""Return a GitLab suggestion range suffix."""
|
|
69
|
+
|
|
70
|
+
end_line = line_number(comment.get("end_line") or comment.get("line"))
|
|
71
|
+
start_line = line_number(comment.get("start_line") or comment.get("line") or end_line)
|
|
72
|
+
|
|
73
|
+
if start_line <= 0 or end_line <= 0 or start_line > end_line:
|
|
74
|
+
return ""
|
|
75
|
+
|
|
76
|
+
span = end_line - start_line
|
|
77
|
+
if span > MAX_SUGGESTION_SPAN_LINES:
|
|
78
|
+
return ""
|
|
79
|
+
|
|
80
|
+
return f"-0+{span}"
|
|
81
|
+
|
|
82
|
+
|
|
83
|
+
def inline_code(value: str) -> str:
|
|
84
|
+
"""Return a Markdown inline-code representation safe for backticks."""
|
|
85
|
+
|
|
86
|
+
return _inline_code(value, escape_controls=True)
|
|
87
|
+
|
|
88
|
+
|
|
89
|
+
def normalized_ocr_metadata(value: Any, allowed_values: set[str]) -> str:
|
|
90
|
+
"""Return a whitelisted OCR metadata value suitable for display."""
|
|
91
|
+
|
|
92
|
+
text = clean_text(value).casefold()
|
|
93
|
+
return text if text in allowed_values else ""
|
|
94
|
+
|
|
95
|
+
|
|
96
|
+
def finding_metadata(comment: dict[str, Any]) -> tuple[str, str]:
|
|
97
|
+
"""Return structured OCR category/severity metadata from a finding."""
|
|
98
|
+
|
|
99
|
+
severity = normalized_ocr_metadata(
|
|
100
|
+
comment.get("severity"), OCR_FINDING_SEVERITIES
|
|
101
|
+
) or normalized_ocr_metadata(comment.get("priority"), OCR_FINDING_SEVERITIES)
|
|
102
|
+
category = normalized_ocr_metadata(comment.get("category"), OCR_FINDING_CATEGORIES)
|
|
103
|
+
return severity, category
|
|
104
|
+
|
|
105
|
+
|
|
106
|
+
def format_finding_tags(comment: dict[str, Any]) -> str:
|
|
107
|
+
"""Return GitLab-visible tags for structured OCR finding metadata."""
|
|
108
|
+
|
|
109
|
+
severity, category = finding_metadata(comment)
|
|
110
|
+
tags = []
|
|
111
|
+
if severity:
|
|
112
|
+
tags.append(inline_code(f"severity:{severity}"))
|
|
113
|
+
if category:
|
|
114
|
+
tags.append(inline_code(f"category:{category}"))
|
|
115
|
+
return f"**OCR tags:** {' '.join(tags)}" if tags else ""
|
|
116
|
+
|
|
117
|
+
|
|
118
|
+
def format_suggestion_block(comment: dict[str, Any]) -> str:
|
|
119
|
+
"""Return a GitLab suggestion block if OCR supplied replacement code."""
|
|
120
|
+
|
|
121
|
+
suggestion = code_text(comment.get("suggestion_code"))
|
|
122
|
+
if not suggestion.strip():
|
|
123
|
+
return ""
|
|
124
|
+
|
|
125
|
+
if "```" in suggestion:
|
|
126
|
+
return ""
|
|
127
|
+
|
|
128
|
+
if any(line.lstrip().startswith("/") for line in suggestion.splitlines()):
|
|
129
|
+
return ""
|
|
130
|
+
|
|
131
|
+
if len(suggestion) > MAX_SUGGESTION_CODE_CHARS:
|
|
132
|
+
return (
|
|
133
|
+
"\n\nSuggestion block was omitted because the generated replacement "
|
|
134
|
+
"was too large to publish safely."
|
|
135
|
+
)
|
|
136
|
+
|
|
137
|
+
range_suffix = suggestion_range_suffix(comment)
|
|
138
|
+
if not range_suffix:
|
|
139
|
+
return ""
|
|
140
|
+
|
|
141
|
+
return f"\n\n{SUGGESTION_HEADER}\n```suggestion:{range_suffix}\n{suggestion}\n```"
|
|
142
|
+
|
|
143
|
+
|
|
144
|
+
def format_inline_comment(comment: dict[str, Any], include_suggestion: bool = True) -> str:
|
|
145
|
+
"""Format one OCR comment as Markdown for an inline GitLab discussion."""
|
|
146
|
+
|
|
147
|
+
raw_content = clean_text(comment.get("content")) or "Open Code Review reported an issue here."
|
|
148
|
+
content = neutralize_suggestion_fences(neutralize_quick_actions(raw_content))
|
|
149
|
+
content = "\n".join(escape_control_chars(line) for line in content.split("\n"))
|
|
150
|
+
tags = format_finding_tags(comment)
|
|
151
|
+
body = f"{tags}\n\n{content}" if tags else content
|
|
152
|
+
if include_suggestion:
|
|
153
|
+
body += format_suggestion_block(comment)
|
|
154
|
+
|
|
155
|
+
return body
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
def format_fallback_comment(comment: dict[str, Any]) -> str:
|
|
159
|
+
"""Format an OCR comment for a fallback non-inline MR note."""
|
|
160
|
+
|
|
161
|
+
path = clean_text(comment.get("path")) or "unknown"
|
|
162
|
+
# Parse line fields as integers. Without this a malformed OCR
|
|
163
|
+
# payload (`"line": "10\n/quickaction"`) would inject a new
|
|
164
|
+
# Markdown heading line into the fallback note.
|
|
165
|
+
start_line = line_number(comment.get("start_line") or comment.get("line"))
|
|
166
|
+
end_line = line_number(comment.get("end_line") or comment.get("line"))
|
|
167
|
+
|
|
168
|
+
location = ""
|
|
169
|
+
if start_line > 0 and end_line > 0 and start_line <= end_line:
|
|
170
|
+
location = f" L{start_line}" if start_line == end_line else f" L{start_line}-L{end_line}"
|
|
171
|
+
elif end_line > 0:
|
|
172
|
+
location = f" L{end_line}"
|
|
173
|
+
elif start_line > 0:
|
|
174
|
+
location = f" L{start_line}"
|
|
175
|
+
|
|
176
|
+
safe_path = _inline_code(path, escape_controls=True)
|
|
177
|
+
body = (
|
|
178
|
+
f"### {safe_path}{location}\n\n{format_inline_comment(comment, include_suggestion=False)}"
|
|
179
|
+
)
|
|
180
|
+
|
|
181
|
+
existing = code_text(comment.get("existing_code"))
|
|
182
|
+
suggestion = code_text(comment.get("suggestion_code"))
|
|
183
|
+
|
|
184
|
+
if existing.strip() and suggestion.strip():
|
|
185
|
+
body += "\n\n<details><summary>Suggested change details</summary>\n\n"
|
|
186
|
+
body += "**Before:**\n"
|
|
187
|
+
body += neutralize_quick_actions(
|
|
188
|
+
markdown_code_block(
|
|
189
|
+
"text", truncate_code_text(existing, MAX_FALLBACK_CODE_DETAILS_CHARS)
|
|
190
|
+
)
|
|
191
|
+
)
|
|
192
|
+
body += "\n\n"
|
|
193
|
+
|
|
194
|
+
body += "**After:**\n"
|
|
195
|
+
body += neutralize_quick_actions(
|
|
196
|
+
markdown_code_block(
|
|
197
|
+
"text", truncate_code_text(suggestion, MAX_FALLBACK_CODE_DETAILS_CHARS)
|
|
198
|
+
)
|
|
199
|
+
)
|
|
200
|
+
body += "\n\n"
|
|
201
|
+
body += "</details>"
|
|
202
|
+
|
|
203
|
+
return body
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def format_fallback_comment_chunks(comments: Sequence[dict[str, Any]]) -> list[str]:
|
|
207
|
+
"""Split fallback comments into safe chunks before publishing MR notes."""
|
|
208
|
+
|
|
209
|
+
chunks: list[str] = []
|
|
210
|
+
current = ""
|
|
211
|
+
|
|
212
|
+
for comment in comments:
|
|
213
|
+
item = truncate_note_body(
|
|
214
|
+
format_fallback_comment(comment), max_chars=FALLBACK_NOTE_CHUNK_BUDGET
|
|
215
|
+
)
|
|
216
|
+
separator = "\n\n---\n\n" if current else ""
|
|
217
|
+
|
|
218
|
+
if current and len(current) + len(separator) + len(item) > FALLBACK_NOTE_CHUNK_BUDGET:
|
|
219
|
+
chunks.append(current)
|
|
220
|
+
current = item
|
|
221
|
+
else:
|
|
222
|
+
current += separator + item
|
|
223
|
+
|
|
224
|
+
if current:
|
|
225
|
+
chunks.append(current)
|
|
226
|
+
|
|
227
|
+
return chunks
|
|
228
|
+
|
|
229
|
+
|
|
230
|
+
def format_omitted_comments_summary(
|
|
231
|
+
publishable_total: int, publish_limit: int, omitted: int
|
|
232
|
+
) -> str:
|
|
233
|
+
"""Return a bounded note body for comments omitted by the publish cap."""
|
|
234
|
+
|
|
235
|
+
return (
|
|
236
|
+
"**Open Code Review omitted comments**\n\n"
|
|
237
|
+
f"After reviewer skip filters, Open Code Review has "
|
|
238
|
+
f"{publishable_total} publishable comment(s). This CI job publishes "
|
|
239
|
+
f"at most {publish_limit} comment(s) per run. The remaining {omitted} "
|
|
240
|
+
"comment(s) were "
|
|
241
|
+
"omitted to avoid excessive GitLab API calls and MR noise. Raise "
|
|
242
|
+
"`OCR_MAX_POST_COMMENTS` deliberately and rerun if more comments need "
|
|
243
|
+
"to be expanded in the MR."
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
|
|
247
|
+
def format_metadata_counts(
|
|
248
|
+
comments: Sequence[dict[str, Any]], field: str, ordered_values: Sequence[str]
|
|
249
|
+
) -> str:
|
|
250
|
+
"""Return compact counts for structured OCR finding metadata."""
|
|
251
|
+
|
|
252
|
+
allowed = set(ordered_values)
|
|
253
|
+
counts: dict[str, int] = {}
|
|
254
|
+
for comment in comments:
|
|
255
|
+
severity, category = finding_metadata(comment)
|
|
256
|
+
value = severity if field == "severity" else category
|
|
257
|
+
if value in allowed:
|
|
258
|
+
counts[value] = counts.get(value, 0) + 1
|
|
259
|
+
|
|
260
|
+
parts = [
|
|
261
|
+
f"{inline_code(value)}: {counts[value]}"
|
|
262
|
+
for value in ordered_values
|
|
263
|
+
if counts.get(value, 0) > 0
|
|
264
|
+
]
|
|
265
|
+
return ", ".join(parts)
|
|
266
|
+
|
|
267
|
+
|
|
268
|
+
def nonnegative_int(value: Any) -> int | None:
|
|
269
|
+
"""Parse a non-negative integer from OCR JSON, ignoring malformed values."""
|
|
270
|
+
|
|
271
|
+
if isinstance(value, bool) or value is None:
|
|
272
|
+
return None
|
|
273
|
+
|
|
274
|
+
if isinstance(value, int):
|
|
275
|
+
return value if value >= 0 else None
|
|
276
|
+
|
|
277
|
+
if isinstance(value, float):
|
|
278
|
+
if value.is_integer() and value >= 0:
|
|
279
|
+
return int(value)
|
|
280
|
+
return None
|
|
281
|
+
|
|
282
|
+
if isinstance(value, str):
|
|
283
|
+
try:
|
|
284
|
+
parsed = int(value.strip())
|
|
285
|
+
except ValueError:
|
|
286
|
+
return None
|
|
287
|
+
return parsed if parsed >= 0 else None
|
|
288
|
+
|
|
289
|
+
return None
|
|
290
|
+
|
|
291
|
+
|
|
292
|
+
def truncate_tool_call_name(name: str) -> str:
|
|
293
|
+
"""Return a compact tool name for one-line MR summaries."""
|
|
294
|
+
|
|
295
|
+
if len(name) <= MAX_TOOL_CALL_NAME_CHARS:
|
|
296
|
+
return name
|
|
297
|
+
|
|
298
|
+
return name[: MAX_TOOL_CALL_NAME_CHARS - 3].rstrip() + "..."
|
|
299
|
+
|
|
300
|
+
|
|
301
|
+
def tool_call_name(value: Any) -> str:
|
|
302
|
+
"""Extract a displayable tool name from common OCR tool-call shapes."""
|
|
303
|
+
|
|
304
|
+
if isinstance(value, str):
|
|
305
|
+
return clean_text(value)
|
|
306
|
+
|
|
307
|
+
if not isinstance(value, dict):
|
|
308
|
+
return ""
|
|
309
|
+
|
|
310
|
+
for key in ("name", "tool", "tool_name"):
|
|
311
|
+
name = clean_text(value.get(key))
|
|
312
|
+
if name:
|
|
313
|
+
return name
|
|
314
|
+
|
|
315
|
+
function_value = value.get("function")
|
|
316
|
+
if isinstance(function_value, dict):
|
|
317
|
+
return clean_text(function_value.get("name"))
|
|
318
|
+
|
|
319
|
+
return ""
|
|
320
|
+
|
|
321
|
+
|
|
322
|
+
def tool_call_counts_from_items(
|
|
323
|
+
items: list[Any],
|
|
324
|
+
) -> tuple[int | None, list[tuple[str, int]]]:
|
|
325
|
+
"""Summarize a list-style OCR tool_calls payload."""
|
|
326
|
+
|
|
327
|
+
counts: dict[str, int] = {}
|
|
328
|
+
for item in items:
|
|
329
|
+
name = tool_call_name(item)
|
|
330
|
+
if not name:
|
|
331
|
+
continue
|
|
332
|
+
counts[name] = counts.get(name, 0) + 1
|
|
333
|
+
|
|
334
|
+
total = sum(counts.values())
|
|
335
|
+
if total == 0 and items:
|
|
336
|
+
return None, []
|
|
337
|
+
|
|
338
|
+
return total, list(counts.items())
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def format_tool_calls_summary(tool_calls: Any) -> str:
|
|
342
|
+
"""Return one bounded MR summary line for OCR tool-call statistics."""
|
|
343
|
+
|
|
344
|
+
entries: list[tuple[str, int]]
|
|
345
|
+
total: int | None
|
|
346
|
+
scalar_total = nonnegative_int(tool_calls)
|
|
347
|
+
if scalar_total is not None:
|
|
348
|
+
total = scalar_total
|
|
349
|
+
entries = []
|
|
350
|
+
elif isinstance(tool_calls, list):
|
|
351
|
+
total, entries = tool_call_counts_from_items(tool_calls)
|
|
352
|
+
elif isinstance(tool_calls, dict):
|
|
353
|
+
by_tool_value = tool_calls.get("by_tool")
|
|
354
|
+
entries = []
|
|
355
|
+
by_tool_total = 0
|
|
356
|
+
valid_by_tool_count = False
|
|
357
|
+
|
|
358
|
+
if isinstance(by_tool_value, dict):
|
|
359
|
+
for raw_name, raw_count in by_tool_value.items():
|
|
360
|
+
count = nonnegative_int(raw_count)
|
|
361
|
+
if count is None:
|
|
362
|
+
continue
|
|
363
|
+
name = clean_text(raw_name)
|
|
364
|
+
if not name:
|
|
365
|
+
continue
|
|
366
|
+
valid_by_tool_count = True
|
|
367
|
+
by_tool_total += count
|
|
368
|
+
if count > 0:
|
|
369
|
+
entries.append((name, count))
|
|
370
|
+
|
|
371
|
+
calls_value = tool_calls.get("calls")
|
|
372
|
+
if valid_by_tool_count:
|
|
373
|
+
list_total = None
|
|
374
|
+
elif isinstance(calls_value, list):
|
|
375
|
+
list_total, entries = tool_call_counts_from_items(calls_value)
|
|
376
|
+
else:
|
|
377
|
+
list_total = None
|
|
378
|
+
|
|
379
|
+
total = nonnegative_int(tool_calls.get("total"))
|
|
380
|
+
if total is None:
|
|
381
|
+
if valid_by_tool_count:
|
|
382
|
+
total = by_tool_total
|
|
383
|
+
elif list_total is not None:
|
|
384
|
+
total = list_total
|
|
385
|
+
elif by_tool_value == {}:
|
|
386
|
+
total = 0
|
|
387
|
+
else:
|
|
388
|
+
return ""
|
|
389
|
+
else:
|
|
390
|
+
return ""
|
|
391
|
+
|
|
392
|
+
if total is None:
|
|
393
|
+
return ""
|
|
394
|
+
|
|
395
|
+
line = f"- tool calls: {total} total"
|
|
396
|
+
if not entries:
|
|
397
|
+
return line
|
|
398
|
+
|
|
399
|
+
entries.sort(key=lambda item: (-item[1], item[0]))
|
|
400
|
+
shown_entries = entries[:MAX_TOOL_CALL_SUMMARY_TOOLS]
|
|
401
|
+
detail_parts = [
|
|
402
|
+
f"{inline_code(truncate_tool_call_name(name))}: {count}" for name, count in shown_entries
|
|
403
|
+
]
|
|
404
|
+
omitted_entries = len(entries) - len(shown_entries)
|
|
405
|
+
if omitted_entries > 0:
|
|
406
|
+
detail_parts.append(f"+{omitted_entries} more")
|
|
407
|
+
|
|
408
|
+
return f"{line} ({', '.join(detail_parts)})"
|
|
409
|
+
|
|
410
|
+
|
|
411
|
+
TOKEN_USAGE_KEYS = (
|
|
412
|
+
"usage",
|
|
413
|
+
"token_usage",
|
|
414
|
+
"tokenUsage",
|
|
415
|
+
"token_usage_summary",
|
|
416
|
+
"tokenUsageSummary",
|
|
417
|
+
)
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
TOKEN_USAGE_CONTAINER_KEYS = (
|
|
421
|
+
*TOKEN_USAGE_KEYS,
|
|
422
|
+
"summary",
|
|
423
|
+
"project_summary",
|
|
424
|
+
"metadata",
|
|
425
|
+
"stats",
|
|
426
|
+
"statistics",
|
|
427
|
+
)
|
|
428
|
+
|
|
429
|
+
|
|
430
|
+
TOKEN_TOTAL_KEYS = ("total_tokens", "totalTokens", "tokens", "total")
|
|
431
|
+
|
|
432
|
+
|
|
433
|
+
TOKEN_EXPLICIT_TOTAL_KEYS = ("total_tokens", "totalTokens", "tokens")
|
|
434
|
+
|
|
435
|
+
|
|
436
|
+
TOKEN_PROMPT_KEYS = (
|
|
437
|
+
"prompt_tokens",
|
|
438
|
+
"input_tokens",
|
|
439
|
+
"promptTokens",
|
|
440
|
+
"inputTokens",
|
|
441
|
+
"prompt",
|
|
442
|
+
"input",
|
|
443
|
+
)
|
|
444
|
+
|
|
445
|
+
|
|
446
|
+
TOKEN_COMPLETION_KEYS = (
|
|
447
|
+
"completion_tokens",
|
|
448
|
+
"output_tokens",
|
|
449
|
+
"completionTokens",
|
|
450
|
+
"outputTokens",
|
|
451
|
+
"completion",
|
|
452
|
+
"output",
|
|
453
|
+
)
|
|
454
|
+
|
|
455
|
+
|
|
456
|
+
TOKEN_CACHED_KEYS = (
|
|
457
|
+
"cached_tokens",
|
|
458
|
+
"cache_read_input_tokens",
|
|
459
|
+
"cachedInputTokens",
|
|
460
|
+
"cacheReadInputTokens",
|
|
461
|
+
)
|
|
462
|
+
|
|
463
|
+
|
|
464
|
+
def first_nonnegative_int(mapping: dict[str, Any], keys: Sequence[str]) -> int | None:
|
|
465
|
+
"""Return the first non-negative integer from known OCR usage keys."""
|
|
466
|
+
|
|
467
|
+
for key in keys:
|
|
468
|
+
value = nonnegative_int(mapping.get(key))
|
|
469
|
+
if value is not None:
|
|
470
|
+
return value
|
|
471
|
+
return None
|
|
472
|
+
|
|
473
|
+
|
|
474
|
+
def token_usage_mapping(
|
|
475
|
+
value: Any, max_depth: int = 8, *, explicit_container: bool = False
|
|
476
|
+
) -> dict[str, Any] | None:
|
|
477
|
+
"""Find the first dict-shaped token usage object in OCR result metadata."""
|
|
478
|
+
|
|
479
|
+
if max_depth <= 0:
|
|
480
|
+
return None
|
|
481
|
+
|
|
482
|
+
if isinstance(value, dict):
|
|
483
|
+
if any(
|
|
484
|
+
key in value
|
|
485
|
+
for key in (TOKEN_TOTAL_KEYS if explicit_container else TOKEN_EXPLICIT_TOTAL_KEYS)
|
|
486
|
+
+ TOKEN_PROMPT_KEYS
|
|
487
|
+
+ TOKEN_COMPLETION_KEYS
|
|
488
|
+
+ TOKEN_CACHED_KEYS
|
|
489
|
+
):
|
|
490
|
+
return value
|
|
491
|
+
for key in TOKEN_USAGE_CONTAINER_KEYS:
|
|
492
|
+
nested = value.get(key)
|
|
493
|
+
if isinstance(nested, dict):
|
|
494
|
+
found = token_usage_mapping(
|
|
495
|
+
nested,
|
|
496
|
+
max_depth=max_depth - 1,
|
|
497
|
+
explicit_container=key in TOKEN_USAGE_KEYS,
|
|
498
|
+
)
|
|
499
|
+
if found is not None:
|
|
500
|
+
return found
|
|
501
|
+
return None
|
|
502
|
+
|
|
503
|
+
|
|
504
|
+
def format_token_usage_summary(result: dict[str, Any]) -> str:
|
|
505
|
+
"""Return one bounded MR summary line for structured OCR token usage."""
|
|
506
|
+
|
|
507
|
+
usage = token_usage_mapping(result)
|
|
508
|
+
if usage is None:
|
|
509
|
+
return ""
|
|
510
|
+
|
|
511
|
+
total = first_nonnegative_int(usage, TOKEN_TOTAL_KEYS)
|
|
512
|
+
prompt = first_nonnegative_int(usage, TOKEN_PROMPT_KEYS)
|
|
513
|
+
completion = first_nonnegative_int(usage, TOKEN_COMPLETION_KEYS)
|
|
514
|
+
cached = first_nonnegative_int(usage, TOKEN_CACHED_KEYS)
|
|
515
|
+
|
|
516
|
+
if total is None and (prompt is not None or completion is not None):
|
|
517
|
+
total = (prompt or 0) + (completion or 0)
|
|
518
|
+
if total is None and cached is not None:
|
|
519
|
+
total = cached
|
|
520
|
+
if total is None:
|
|
521
|
+
return ""
|
|
522
|
+
|
|
523
|
+
details: list[str] = []
|
|
524
|
+
if prompt is not None:
|
|
525
|
+
details.append(f"prompt: {prompt}")
|
|
526
|
+
if completion is not None:
|
|
527
|
+
details.append(f"completion: {completion}")
|
|
528
|
+
if cached is not None:
|
|
529
|
+
details.append(f"cached: {cached}")
|
|
530
|
+
|
|
531
|
+
line = f"- token usage: {total} total"
|
|
532
|
+
if details:
|
|
533
|
+
line += f" ({', '.join(details)})"
|
|
534
|
+
return line
|
|
535
|
+
|
|
536
|
+
|
|
537
|
+
SECURITY_SIGNAL_RE = re.compile(
|
|
538
|
+
r"(?i)\b("
|
|
539
|
+
r"security|credential|secret|token|password|private[_ -]?token|"
|
|
540
|
+
r"client[_ -]?secret|api[_ -]?key|authorization|auth|"
|
|
541
|
+
r"injection|xss|csrf|ssrf|rce|path traversal|host header|"
|
|
542
|
+
r"privilege|permission|access control|vault|leak"
|
|
543
|
+
r")\b"
|
|
544
|
+
)
|
|
545
|
+
|
|
546
|
+
|
|
547
|
+
HIGH_SIGNAL_RE = re.compile(r"(?i)\b(high|critical|blocker|security)\b")
|
|
548
|
+
|
|
549
|
+
|
|
550
|
+
def comment_signal_text(comment: dict[str, Any]) -> str:
|
|
551
|
+
"""Return fields useful for conservative guide classification."""
|
|
552
|
+
|
|
553
|
+
values = [
|
|
554
|
+
clean_text(comment.get("priority") or comment.get("severity")),
|
|
555
|
+
clean_text(comment.get("category")),
|
|
556
|
+
clean_text(comment.get("rule_id")),
|
|
557
|
+
clean_text(comment.get("content")),
|
|
558
|
+
]
|
|
559
|
+
return "\n".join(value for value in values if value)
|
|
560
|
+
|
|
561
|
+
|
|
562
|
+
def comment_has_security_signal(comment: dict[str, Any]) -> bool:
|
|
563
|
+
"""Return true only when OCR text explicitly carries security wording."""
|
|
564
|
+
|
|
565
|
+
return bool(SECURITY_SIGNAL_RE.search(comment_signal_text(comment)))
|
|
566
|
+
|
|
567
|
+
|
|
568
|
+
def comment_is_high_signal(comment: dict[str, Any]) -> bool:
|
|
569
|
+
"""Return true for explicit high/critical/security OCR metadata."""
|
|
570
|
+
|
|
571
|
+
return bool(HIGH_SIGNAL_RE.search(comment_signal_text(comment)))
|
|
572
|
+
|
|
573
|
+
|
|
574
|
+
def estimate_review_effort(comments: Sequence[dict[str, Any]], omitted_count: int) -> int:
|
|
575
|
+
"""Estimate review effort from visible OCR findings without inventing context."""
|
|
576
|
+
|
|
577
|
+
total = len(comments) + omitted_count
|
|
578
|
+
if total <= 0:
|
|
579
|
+
return 1
|
|
580
|
+
|
|
581
|
+
security_count = sum(1 for comment in comments if comment_has_security_signal(comment))
|
|
582
|
+
high_count = sum(1 for comment in comments if comment_is_high_signal(comment))
|
|
583
|
+
|
|
584
|
+
if total <= 2 and security_count == 0 and high_count == 0:
|
|
585
|
+
return 1
|
|
586
|
+
if total <= 5 and high_count <= 1 and security_count <= 1:
|
|
587
|
+
return 2
|
|
588
|
+
if total <= 10:
|
|
589
|
+
return 3
|
|
590
|
+
if total <= 25:
|
|
591
|
+
return 4
|
|
592
|
+
return 5
|
|
593
|
+
|
|
594
|
+
|
|
595
|
+
def guide_comment_label(comment: dict[str, Any]) -> str:
|
|
596
|
+
"""Return compact severity/category label for the reviewer guide."""
|
|
597
|
+
|
|
598
|
+
parts = [
|
|
599
|
+
clean_text(comment.get("priority") or comment.get("severity")),
|
|
600
|
+
clean_text(comment.get("category")),
|
|
601
|
+
]
|
|
602
|
+
parts = [part for part in parts if part]
|
|
603
|
+
if not parts:
|
|
604
|
+
return "review finding"
|
|
605
|
+
return _inline_code(
|
|
606
|
+
compact_control_text(", ".join(parts[:2]), MAX_REVIEWER_GUIDE_LABEL_CHARS),
|
|
607
|
+
)
|
|
608
|
+
|
|
609
|
+
|
|
610
|
+
def guide_comment_location(comment: dict[str, Any]) -> str:
|
|
611
|
+
"""Return a compact path/line label for the reviewer guide."""
|
|
612
|
+
|
|
613
|
+
path = clean_text(comment.get("path")) or "unknown"
|
|
614
|
+
line = comment_line(comment)
|
|
615
|
+
location = f"{path}:L{line}" if line > 0 else path
|
|
616
|
+
location = compact_control_text(location, MAX_REVIEWER_GUIDE_LOCATION_CHARS)
|
|
617
|
+
return _inline_code(location)
|
|
618
|
+
|
|
619
|
+
|
|
620
|
+
def guide_comment_snippet(comment: dict[str, Any]) -> str:
|
|
621
|
+
"""Return one Markdown-neutral summary snippet for an OCR finding."""
|
|
622
|
+
|
|
623
|
+
content = neutralize_suggestion_fences(
|
|
624
|
+
neutralize_quick_actions(clean_text(comment.get("content")))
|
|
625
|
+
)
|
|
626
|
+
excerpt = compact_escaped_text(content, MAX_REVIEWER_GUIDE_TEXT_CHARS)
|
|
627
|
+
if not excerpt:
|
|
628
|
+
return "Open Code Review reported an issue here."
|
|
629
|
+
|
|
630
|
+
return excerpt
|
|
631
|
+
|
|
632
|
+
|
|
633
|
+
def format_reviewer_guide(comments: Sequence[dict[str, Any]], omitted_count: int) -> str:
|
|
634
|
+
"""Build a bounded reviewer guide from already published OCR findings."""
|
|
635
|
+
|
|
636
|
+
if not comments and omitted_count <= 0:
|
|
637
|
+
return ""
|
|
638
|
+
|
|
639
|
+
effort = estimate_review_effort(comments, omitted_count)
|
|
640
|
+
security_comments = [comment for comment in comments if comment_has_security_signal(comment)]
|
|
641
|
+
|
|
642
|
+
lines = [""]
|
|
643
|
+
if security_comments:
|
|
644
|
+
lines.append("## Security review focus")
|
|
645
|
+
lines.append(
|
|
646
|
+
f"- **Security signal:** {len(security_comments)} published OCR finding(s) explicitly mention security-sensitive terms."
|
|
647
|
+
)
|
|
648
|
+
lines.append(
|
|
649
|
+
"- Prioritize these findings before ordinary reliability or maintainability items."
|
|
650
|
+
)
|
|
651
|
+
|
|
652
|
+
lines.extend(
|
|
653
|
+
[
|
|
654
|
+
"",
|
|
655
|
+
"## Reviewer guide",
|
|
656
|
+
f"- Estimated effort to review: {effort}/5",
|
|
657
|
+
]
|
|
658
|
+
)
|
|
659
|
+
|
|
660
|
+
if not security_comments and omitted_count:
|
|
661
|
+
lines.append(
|
|
662
|
+
"- Security signal: none in published OCR findings; omitted findings were not inspected for this guide."
|
|
663
|
+
)
|
|
664
|
+
elif not security_comments:
|
|
665
|
+
lines.append(
|
|
666
|
+
"- Security signal: no security-sensitive findings detected in the published OCR comments."
|
|
667
|
+
)
|
|
668
|
+
|
|
669
|
+
if omitted_count:
|
|
670
|
+
lines.append(
|
|
671
|
+
f"- Visibility: {omitted_count} finding(s) omitted by `OCR_MAX_POST_COMMENTS`; rerun with a higher limit if needed."
|
|
672
|
+
)
|
|
673
|
+
|
|
674
|
+
guide_comments = list(comments[:MAX_REVIEWER_GUIDE_COMMENTS])
|
|
675
|
+
if guide_comments:
|
|
676
|
+
lines.append("")
|
|
677
|
+
lines.append("### Recommended focus areas")
|
|
678
|
+
lines.append(
|
|
679
|
+
"These are abbreviated navigation snippets; read the posted inline or fallback OCR discussions for the full findings."
|
|
680
|
+
)
|
|
681
|
+
for comment in guide_comments:
|
|
682
|
+
lines.append(
|
|
683
|
+
f"- {guide_comment_label(comment)}; affected path: "
|
|
684
|
+
f"{guide_comment_location(comment)}; snippet: {guide_comment_snippet(comment)}"
|
|
685
|
+
)
|
|
686
|
+
|
|
687
|
+
remaining = len(comments) - len(guide_comments)
|
|
688
|
+
if remaining > 0:
|
|
689
|
+
lines.append(f"- ... and {remaining} more published finding(s).")
|
|
690
|
+
|
|
691
|
+
return "\n".join(lines)
|
|
692
|
+
|
|
693
|
+
|
|
694
|
+
def summarize_result(
|
|
695
|
+
total: int,
|
|
696
|
+
inline_count: int,
|
|
697
|
+
fallback_count: int,
|
|
698
|
+
warning_count: int,
|
|
699
|
+
*,
|
|
700
|
+
comments: Sequence[dict[str, Any]] = (),
|
|
701
|
+
omitted_count: int = 0,
|
|
702
|
+
tool_calls_summary: str = "",
|
|
703
|
+
token_usage_summary: str = "",
|
|
704
|
+
reviewer_guide: str = "",
|
|
705
|
+
fallback_reasons: Mapping[str, int] | None = None,
|
|
706
|
+
reviewed_sha: str = "",
|
|
707
|
+
mr_head_sha: str = "",
|
|
708
|
+
) -> str:
|
|
709
|
+
"""Build a compact summary note for the MR."""
|
|
710
|
+
|
|
711
|
+
lines = [
|
|
712
|
+
f"**Open Code Review** found **{total}** issue(s).",
|
|
713
|
+
f"- posting mode: `{post_mode()}`",
|
|
714
|
+
f"- {inline_count} posted as inline discussion(s)",
|
|
715
|
+
f"- {fallback_count} posted as fallback summary item(s)",
|
|
716
|
+
]
|
|
717
|
+
|
|
718
|
+
if reviewed_sha:
|
|
719
|
+
lines.append(f"- reviewed SHA: {_inline_code(reviewed_sha)}")
|
|
720
|
+
if mr_head_sha and mr_head_sha != reviewed_sha:
|
|
721
|
+
lines.append(f"- MR head SHA: {_inline_code(mr_head_sha)}")
|
|
722
|
+
|
|
723
|
+
severity_counts = format_metadata_counts(comments, "severity", OCR_FINDING_SEVERITY_ORDER)
|
|
724
|
+
if severity_counts:
|
|
725
|
+
lines.append(f"- severity tags: {severity_counts}")
|
|
726
|
+
|
|
727
|
+
category_counts = format_metadata_counts(comments, "category", OCR_FINDING_CATEGORY_ORDER)
|
|
728
|
+
if category_counts:
|
|
729
|
+
lines.append(f"- category tags: {category_counts}")
|
|
730
|
+
|
|
731
|
+
if fallback_reasons:
|
|
732
|
+
reason_parts = [
|
|
733
|
+
f"{_inline_code(reason)}: {count}"
|
|
734
|
+
for reason, count in sorted(fallback_reasons.items())
|
|
735
|
+
if count > 0
|
|
736
|
+
]
|
|
737
|
+
if reason_parts:
|
|
738
|
+
lines.append(f"- fallback reasons: {', '.join(reason_parts)}")
|
|
739
|
+
|
|
740
|
+
if omitted_count:
|
|
741
|
+
lines.append(f"- {omitted_count} omitted by `OCR_MAX_POST_COMMENTS`")
|
|
742
|
+
|
|
743
|
+
if warning_count:
|
|
744
|
+
lines.append(f"- {warning_count} warning(s) reported by OCR")
|
|
745
|
+
|
|
746
|
+
if tool_calls_summary:
|
|
747
|
+
lines.append(tool_calls_summary)
|
|
748
|
+
|
|
749
|
+
if token_usage_summary:
|
|
750
|
+
lines.append(token_usage_summary)
|
|
751
|
+
|
|
752
|
+
if reviewer_guide:
|
|
753
|
+
lines.append(reviewer_guide)
|
|
754
|
+
|
|
755
|
+
return "\n".join(lines)
|