open-code-review-toolkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. ocr_toolkit/__init__.py +1 -0
  2. ocr_toolkit/_version.py +24 -0
  3. ocr_toolkit/cli.py +56 -0
  4. ocr_toolkit/common/__init__.py +1 -0
  5. ocr_toolkit/common/language.py +60 -0
  6. ocr_toolkit/common/markdown.py +214 -0
  7. ocr_toolkit/common/redaction.py +324 -0
  8. ocr_toolkit/config_writer.py +106 -0
  9. ocr_toolkit/configure.py +194 -0
  10. ocr_toolkit/context/__init__.py +1 -0
  11. ocr_toolkit/context/__main__.py +8 -0
  12. ocr_toolkit/context/ansible.py +561 -0
  13. ocr_toolkit/context/categorize.py +162 -0
  14. ocr_toolkit/context/instructions.py +269 -0
  15. ocr_toolkit/context/manifests.py +459 -0
  16. ocr_toolkit/context/planner.py +227 -0
  17. ocr_toolkit/context/render.py +955 -0
  18. ocr_toolkit/context/repo.py +672 -0
  19. ocr_toolkit/context/settings.py +97 -0
  20. ocr_toolkit/mcp_config.py +257 -0
  21. ocr_toolkit/posting/__init__.py +1 -0
  22. ocr_toolkit/posting/__main__.py +8 -0
  23. ocr_toolkit/posting/comments.py +87 -0
  24. ocr_toolkit/posting/formatting.py +755 -0
  25. ocr_toolkit/posting/gitlab.py +853 -0
  26. ocr_toolkit/posting/markers.py +284 -0
  27. ocr_toolkit/posting/payloads.py +141 -0
  28. ocr_toolkit/posting/result.py +116 -0
  29. ocr_toolkit/posting/settings.py +181 -0
  30. ocr_toolkit/posting/snapshot.py +468 -0
  31. ocr_toolkit/posting/workflow.py +873 -0
  32. ocr_toolkit/preflight.py +395 -0
  33. ocr_toolkit/py.typed +1 -0
  34. open_code_review_toolkit-0.1.0.dist-info/METADATA +283 -0
  35. open_code_review_toolkit-0.1.0.dist-info/RECORD +38 -0
  36. open_code_review_toolkit-0.1.0.dist-info/WHEEL +4 -0
  37. open_code_review_toolkit-0.1.0.dist-info/entry_points.txt +2 -0
  38. open_code_review_toolkit-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,162 @@
1
+ """Changed-file categorization for OCR context."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Mapping, Sequence
7
+
8
+ from ocr_toolkit.context.ansible import is_root_ansible_playbook
9
+
10
+ CATEGORY_ORDER = (
11
+ "ansible_playbooks",
12
+ "ocr_integration",
13
+ "ci",
14
+ "dependency_manifests",
15
+ "molecule_tests",
16
+ "systemd_units",
17
+ "containers",
18
+ "shell",
19
+ "ansible_roles",
20
+ "ansible_inventory",
21
+ "templates",
22
+ "terraform_hcl",
23
+ "python",
24
+ "go",
25
+ "php",
26
+ "javascript_typescript",
27
+ "sql",
28
+ "docs",
29
+ "other",
30
+ )
31
+
32
+
33
+ DEPENDENCY_MANIFEST_PATTERN = re.compile(
34
+ r"(^|/)(requirements[^/]*\.(?:txt|in)|constraints[^/]*\.(?:txt|in)|requirements/[^/]+\.(?:txt|in)|requirements\.ya?ml|pyproject\.toml|"
35
+ r"poetry\.lock|uv\.lock|Pipfile(\.lock)?|package(-lock)?\.json|pnpm-lock\.yaml|"
36
+ r"yarn\.lock|composer\.(json|lock)|go\.(mod|sum)|Cargo\.(toml|lock)|"
37
+ r"Gemfile(\.lock)?|pom\.xml|build\.gradle(\.kts)?|gradle\.lockfile)$",
38
+ re.I,
39
+ )
40
+
41
+
42
+ PYTHON_MANIFEST_PATTERN = re.compile(
43
+ r"(^|/)(requirements[^/]*\.(?:txt|in)|constraints[^/]*\.(?:txt|in)|requirements/[^/]+\.(?:txt|in)|pyproject\.toml|poetry\.lock|uv\.lock|Pipfile(\.lock)?)$",
44
+ re.I,
45
+ )
46
+
47
+
48
+ CATEGORY_PATTERNS: list[tuple[str, re.Pattern[str]]] = [
49
+ (
50
+ "ocr_integration",
51
+ re.compile(r"(^|/)\.opencodereview/", re.I),
52
+ ),
53
+ ("ci", re.compile(r"(^|/)\.gitlab-ci\.ya?ml$|(^|/)\.github/", re.I)),
54
+ (
55
+ "dependency_manifests",
56
+ DEPENDENCY_MANIFEST_PATTERN,
57
+ ),
58
+ (
59
+ "molecule_tests",
60
+ re.compile(r"(^|/)roles/[^/]+/molecule/", re.I),
61
+ ),
62
+ (
63
+ "systemd_units",
64
+ re.compile(
65
+ r"\.(service|timer|socket|device|mount|automount|path|target|slice|swap|scope)$|"
66
+ r"\.(service|timer|socket|device|mount|automount|path|target|slice|swap|scope)\.d/[^/]+\.conf$",
67
+ re.I,
68
+ ),
69
+ ),
70
+ (
71
+ "containers",
72
+ re.compile(
73
+ r"(^|/)(Dockerfile(?:\.[^/]+)?|Containerfile(?:\.[^/]+)?|"
74
+ r"[^/]+\.Dockerfile|docker-compose.*\.ya?ml|compose(?:\.[^/]+)?\.ya?ml)$|"
75
+ r"\.dockerfile$",
76
+ re.I,
77
+ ),
78
+ ),
79
+ ("shell", re.compile(r"(^|/)apb$|\.(sh|bash|zsh)$", re.I)),
80
+ (
81
+ "ansible_roles",
82
+ re.compile(r"(^|/)roles/", re.I),
83
+ ),
84
+ (
85
+ "ansible_inventory",
86
+ re.compile(
87
+ r"(^|/)(inventory|inventories|group_vars|host_vars)(/|$)|"
88
+ r"(^|/)(hosts|inventory)\.(ya?ml|ini|cfg|json)$",
89
+ re.I,
90
+ ),
91
+ ),
92
+ ("templates", re.compile(r"\.(j2|jinja|jinja2|tpl|tmpl)$", re.I)),
93
+ ("terraform_hcl", re.compile(r"\.(tf|tfvars|hcl)$", re.I)),
94
+ ("python", re.compile(r"\.(py|pyi)$", re.I)),
95
+ ("go", re.compile(r"\.go$", re.I)),
96
+ ("php", re.compile(r"\.php$", re.I)),
97
+ ("javascript_typescript", re.compile(r"\.(js|jsx|ts|tsx|mjs|cjs)$", re.I)),
98
+ ("sql", re.compile(r"\.sql$", re.I)),
99
+ ("docs", re.compile(r"\.(md|rst|adoc)$", re.I)),
100
+ ]
101
+
102
+
103
+ PROVIDER_MANIFEST_PATTERNS: dict[str, re.Pattern[str]] = {
104
+ "ansible": re.compile(r"(^|/)(ansible\.cfg|requirements\.ya?ml|galaxy\.ya?ml)$", re.I),
105
+ "python": PYTHON_MANIFEST_PATTERN,
106
+ "go": re.compile(r"(^|/)go\.(mod|sum)$", re.I),
107
+ "php": re.compile(r"(^|/)composer\.(json|lock)$", re.I),
108
+ "javascript": re.compile(
109
+ r"(^|/)(package\.json|package-lock\.json|pnpm-lock\.yaml|yarn\.lock)$",
110
+ re.I,
111
+ ),
112
+ }
113
+
114
+
115
+ def active_context_providers(
116
+ files: Sequence[str], categories: Mapping[str, Sequence[str]]
117
+ ) -> set[str]:
118
+ """Return generic ecosystem providers activated by changed paths."""
119
+
120
+ active_categories = set(categories)
121
+ providers: set[str] = set()
122
+ if active_categories & {
123
+ "ansible_playbooks",
124
+ "ansible_roles",
125
+ "ansible_inventory",
126
+ "molecule_tests",
127
+ }:
128
+ providers.add("ansible")
129
+ if "python" in active_categories:
130
+ providers.add("python")
131
+ if "go" in active_categories:
132
+ providers.add("go")
133
+ if "php" in active_categories:
134
+ providers.add("php")
135
+ if "javascript_typescript" in active_categories:
136
+ providers.add("javascript")
137
+ for file_path in files:
138
+ for provider, pattern in PROVIDER_MANIFEST_PATTERNS.items():
139
+ if pattern.search(file_path):
140
+ providers.add(provider)
141
+ return providers
142
+
143
+
144
+ def categorize_files(files: Sequence[str]) -> dict[str, list[str]]:
145
+ """Categorize changed files by technology or operational area."""
146
+
147
+ categorized: dict[str, list[str]] = {}
148
+ for file_path in files:
149
+ matched = False
150
+ if is_root_ansible_playbook(file_path):
151
+ categorized.setdefault("ansible_playbooks", []).append(file_path)
152
+ matched = True
153
+ for category, pattern in CATEGORY_PATTERNS:
154
+ if pattern.search(file_path):
155
+ categorized.setdefault(category, []).append(file_path)
156
+ matched = True
157
+ if not matched:
158
+ categorized.setdefault("other", []).append(file_path)
159
+
160
+ return {
161
+ category: categorized[category] for category in CATEGORY_ORDER if category in categorized
162
+ }
@@ -0,0 +1,269 @@
1
+ """Trusted project guidance handling for OCR review context."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Sequence
7
+ from pathlib import Path
8
+
9
+ from ocr_toolkit.common.redaction import redact_sensitive, redact_url_userinfo
10
+ from ocr_toolkit.context.repo import read_text, trusted_guidance_file
11
+ from ocr_toolkit.context.settings import MAX_INSTRUCTION_BYTES
12
+
13
+ GUIDANCE_KEYWORDS = (
14
+ "ocr",
15
+ "open code review",
16
+ "review",
17
+ "ci",
18
+ "gitlab",
19
+ "python",
20
+ "secret",
21
+ "redact",
22
+ "security",
23
+ "token",
24
+ "markdown",
25
+ "ansible",
26
+ )
27
+
28
+
29
+ def _sanitize_guidance_text(text: str) -> str:
30
+ return redact_sensitive(redact_url_userinfo(text)).strip()
31
+
32
+
33
+ def read_project_instructions(
34
+ limit_bytes: int = MAX_INSTRUCTION_BYTES,
35
+ changed_paths: Sequence[str] | None = None,
36
+ ) -> list[tuple[str, str]]:
37
+ """Read bounded excerpts from stable project AI instruction files.
38
+
39
+ Instruction files changed by the same MR are intentionally ignored. This
40
+ prevents a merge request from changing reviewer guidance and using that new
41
+ guidance to influence the review of itself.
42
+ """
43
+
44
+ instruction_files = [
45
+ "PR_REVIEW.md",
46
+ "AGENTS.md",
47
+ "AGENTS.MD",
48
+ "CLAUDE.md",
49
+ "CLAUDE.MD",
50
+ ".cursorrules",
51
+ ".github/copilot-instructions.md",
52
+ ]
53
+
54
+ changed_set = {path.casefold() for path in changed_paths or []}
55
+ excerpts: list[tuple[str, str]] = []
56
+ seen_instruction_files: set[tuple[int, int] | Path] = set()
57
+ remaining = limit_bytes
58
+
59
+ for rel_path in instruction_files:
60
+ if rel_path.casefold() in changed_set:
61
+ continue
62
+
63
+ safe_path = trusted_guidance_file(rel_path)
64
+ if safe_path is None or remaining <= 0:
65
+ continue
66
+ try:
67
+ stat_result = safe_path.stat()
68
+ instruction_key: tuple[int, int] | Path = (
69
+ stat_result.st_dev,
70
+ stat_result.st_ino,
71
+ )
72
+ except OSError:
73
+ instruction_key = safe_path
74
+ if instruction_key in seen_instruction_files:
75
+ continue
76
+ seen_instruction_files.add(instruction_key)
77
+
78
+ scan_budget = max(MAX_INSTRUCTION_BYTES, min(128_000, limit_bytes * 4))
79
+ raw_text = _sanitize_guidance_text(read_text(safe_path, max_bytes=scan_budget))
80
+ if not raw_text:
81
+ continue
82
+
83
+ text = selected_instruction_excerpt(
84
+ rel_path,
85
+ raw_text,
86
+ max_bytes=min(remaining, 8_000),
87
+ )
88
+ if not text:
89
+ continue
90
+
91
+ excerpts.append((rel_path, text))
92
+ remaining -= len(text.encode("utf-8", errors="ignore"))
93
+
94
+ return excerpts
95
+
96
+
97
+ def selected_instruction_excerpt(rel_path: str, text: str, max_bytes: int) -> str:
98
+ """Return a useful bounded excerpt from arbitrary trusted guidance text."""
99
+
100
+ if max_bytes <= 0 or not text.strip():
101
+ return ""
102
+ if Path(rel_path).name.lower() not in {"agents.md", "claude.md"}:
103
+ return read_text_slice(text, max_bytes=max_bytes)
104
+ if len(text.encode("utf-8")) <= max_bytes:
105
+ return text.strip()
106
+
107
+ pieces: list[str] = []
108
+ used: set[str] = set()
109
+ added_relevant_candidate = False
110
+ first_slice = read_text_slice(text, max_bytes=max(1, min(1_000, max_bytes // 4)))
111
+ if first_slice:
112
+ pieces.append(first_slice)
113
+ used.add(first_slice)
114
+
115
+ for candidate in _rank_guidance_candidates(text):
116
+ if not candidate or candidate in used:
117
+ continue
118
+ current = _join_instruction_pieces(pieces)
119
+ separator = "\n\n---\n\n" if current else ""
120
+ remaining = max_bytes - len((current + separator).encode("utf-8"))
121
+ if remaining <= 0:
122
+ break
123
+ clipped = read_text_slice(candidate, max_bytes=remaining)
124
+ if not clipped:
125
+ continue
126
+ pieces.append(clipped)
127
+ used.add(candidate)
128
+ added_relevant_candidate = True
129
+ if len(_join_instruction_pieces(pieces).encode("utf-8")) >= max_bytes:
130
+ break
131
+
132
+ if not added_relevant_candidate:
133
+ return read_text_slice(text, max_bytes=max_bytes)
134
+ return read_text_slice(
135
+ _join_instruction_pieces(pieces) if pieces else text, max_bytes=max_bytes
136
+ )
137
+
138
+
139
+ def _join_instruction_pieces(pieces: Sequence[str]) -> str:
140
+ return "\n\n---\n\n".join(piece.strip() for piece in pieces if piece.strip())
141
+
142
+
143
+ def _rank_guidance_candidates(text: str) -> list[str]:
144
+ sections = _heading_sections(text)
145
+ if sections:
146
+ preamble = text[: text.find(sections[0][1])].strip()
147
+ scored_sections = [
148
+ (_guidance_score(title, body), index, body)
149
+ for index, (title, body) in enumerate(sections)
150
+ ]
151
+ if preamble:
152
+ scored_sections.append((_guidance_score("", preamble), -1, preamble))
153
+ return [
154
+ _heading_guidance_candidate(body)
155
+ for score, _index, body in sorted(scored_sections, key=lambda item: (-item[0], item[1]))
156
+ if score > 0
157
+ ]
158
+
159
+ paragraphs = [part.strip() for part in re.split(r"\n\s*\n", text) if part.strip()]
160
+ scored_paragraphs = [
161
+ (_guidance_score("", paragraph), index, _keyword_window(paragraph))
162
+ for index, paragraph in enumerate(paragraphs)
163
+ ]
164
+ return [
165
+ paragraph
166
+ for score, _index, paragraph in sorted(
167
+ scored_paragraphs, key=lambda item: (-item[0], item[1])
168
+ )
169
+ if score > 0
170
+ ]
171
+
172
+
173
+ def _heading_guidance_candidate(text: str) -> str:
174
+ """Keep heading-section context while preserving deep keyword rules."""
175
+
176
+ window = _keyword_window(text)
177
+ prefix = read_text_slice(text, max_bytes=40).strip()
178
+ if not prefix or window == text or text.startswith(window):
179
+ return text
180
+ return _join_instruction_pieces([prefix, window])
181
+
182
+
183
+ def _keyword_window(text: str, radius: int = 500) -> str:
184
+ """Return a bounded window around the first guidance keyword."""
185
+
186
+ positions: list[int] = []
187
+ for keyword in GUIDANCE_KEYWORDS:
188
+ if " " in keyword:
189
+ match = re.search(re.escape(keyword), text, flags=re.IGNORECASE)
190
+ if match:
191
+ positions.append(match.start())
192
+ continue
193
+ match = re.search(
194
+ rf"(?<![a-z0-9]){re.escape(keyword)}(?![a-z0-9])",
195
+ text,
196
+ flags=re.IGNORECASE,
197
+ )
198
+ if match:
199
+ positions.append(match.start())
200
+ if not positions:
201
+ return text
202
+ position = min(positions)
203
+ newline_position = text.rfind("\n", 0, position)
204
+ start = newline_position + 1 if newline_position >= 0 else max(0, position - radius)
205
+ end = min(len(text), position + radius)
206
+ return text[start:end].strip()
207
+
208
+
209
+ def _heading_sections(text: str) -> list[tuple[str, str]]:
210
+ """Split Markdown-ish guidance by any heading level."""
211
+
212
+ matches = list(re.finditer(r"(?m)^(#{1,6})\s+(.+?)\s*$", text))
213
+ sections: list[tuple[str, str]] = []
214
+ for index, match in enumerate(matches):
215
+ title = match.group(2).strip()
216
+ end = matches[index + 1].start() if index + 1 < len(matches) else len(text)
217
+ sections.append((title, text[match.start() : end].strip()))
218
+ return sections
219
+
220
+
221
+ def _guidance_score(title: str, body: str) -> int:
222
+ haystack = f"{title}\n{body}".casefold()
223
+ score = 0
224
+ for keyword in GUIDANCE_KEYWORDS:
225
+ keyword_text = keyword.casefold()
226
+ if " " in keyword_text:
227
+ score += haystack.count(keyword_text)
228
+ continue
229
+ score += len(
230
+ re.findall(
231
+ rf"(?<![a-z0-9]){re.escape(keyword_text)}(?![a-z0-9])",
232
+ haystack,
233
+ )
234
+ )
235
+ return score
236
+
237
+
238
+ def read_text_slice(text: str, max_bytes: int) -> str:
239
+ """Return a UTF-8 byte-bounded text slice without adding notices."""
240
+
241
+ if max_bytes <= 0:
242
+ return ""
243
+ encoded = text.encode("utf-8")
244
+ if len(encoded) <= max_bytes:
245
+ return text
246
+ return encoded[:max_bytes].decode("utf-8", errors="ignore").rstrip()
247
+
248
+
249
+ def read_accepted_decisions(
250
+ changed_paths: Sequence[str] | None = None,
251
+ max_bytes: int = MAX_INSTRUCTION_BYTES,
252
+ ) -> str:
253
+ """Return the contents of `.opencodereview/accepted-decisions.md`.
254
+
255
+ This file lists project decisions the reviewer should not raise as
256
+ new issues (e.g. a known but accepted security tradeoff). If the
257
+ file is part of the current merge request's changed paths, it is
258
+ ignored — otherwise an MR could whitelist its own findings.
259
+ """
260
+
261
+ rel_path = ".opencodereview/accepted-decisions.md"
262
+ if rel_path.casefold() in {path.casefold() for path in changed_paths or []}:
263
+ return ""
264
+
265
+ path = trusted_guidance_file(rel_path)
266
+ if path is None:
267
+ return ""
268
+
269
+ return _sanitize_guidance_text(read_text(path, max_bytes=max_bytes))