open-code-review-toolkit 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. ocr_toolkit/__init__.py +1 -0
  2. ocr_toolkit/_version.py +24 -0
  3. ocr_toolkit/cli.py +56 -0
  4. ocr_toolkit/common/__init__.py +1 -0
  5. ocr_toolkit/common/language.py +60 -0
  6. ocr_toolkit/common/markdown.py +214 -0
  7. ocr_toolkit/common/redaction.py +324 -0
  8. ocr_toolkit/config_writer.py +106 -0
  9. ocr_toolkit/configure.py +194 -0
  10. ocr_toolkit/context/__init__.py +1 -0
  11. ocr_toolkit/context/__main__.py +8 -0
  12. ocr_toolkit/context/ansible.py +561 -0
  13. ocr_toolkit/context/categorize.py +162 -0
  14. ocr_toolkit/context/instructions.py +269 -0
  15. ocr_toolkit/context/manifests.py +459 -0
  16. ocr_toolkit/context/planner.py +227 -0
  17. ocr_toolkit/context/render.py +955 -0
  18. ocr_toolkit/context/repo.py +672 -0
  19. ocr_toolkit/context/settings.py +97 -0
  20. ocr_toolkit/mcp_config.py +257 -0
  21. ocr_toolkit/posting/__init__.py +1 -0
  22. ocr_toolkit/posting/__main__.py +8 -0
  23. ocr_toolkit/posting/comments.py +87 -0
  24. ocr_toolkit/posting/formatting.py +755 -0
  25. ocr_toolkit/posting/gitlab.py +853 -0
  26. ocr_toolkit/posting/markers.py +284 -0
  27. ocr_toolkit/posting/payloads.py +141 -0
  28. ocr_toolkit/posting/result.py +116 -0
  29. ocr_toolkit/posting/settings.py +181 -0
  30. ocr_toolkit/posting/snapshot.py +468 -0
  31. ocr_toolkit/posting/workflow.py +873 -0
  32. ocr_toolkit/preflight.py +395 -0
  33. ocr_toolkit/py.typed +1 -0
  34. open_code_review_toolkit-0.1.0.dist-info/METADATA +283 -0
  35. open_code_review_toolkit-0.1.0.dist-info/RECORD +38 -0
  36. open_code_review_toolkit-0.1.0.dist-info/WHEEL +4 -0
  37. open_code_review_toolkit-0.1.0.dist-info/entry_points.txt +2 -0
  38. open_code_review_toolkit-0.1.0.dist-info/licenses/LICENSE +202 -0
@@ -0,0 +1,561 @@
1
+ """Ansible-specific context discovery for OCR."""
2
+
3
+ from __future__ import annotations
4
+
5
+ import re
6
+ from collections.abc import Sequence
7
+ from pathlib import Path
8
+
9
+ from ocr_toolkit.common.redaction import redact_sensitive, redact_url_userinfo
10
+ from ocr_toolkit.context import repo as context_repo
11
+ from ocr_toolkit.context.settings import (
12
+ DEFAULT_MAX_FILE_BYTES,
13
+ MAX_BACKGROUND_SECTION_ITEMS,
14
+ is_env_file,
15
+ )
16
+
17
+ VERSION_PIN_BACKGROUND_SKIP_PARTS = {
18
+ ".cache",
19
+ ".git",
20
+ ".molecule",
21
+ ".review-context",
22
+ ".terraform",
23
+ ".venv",
24
+ "__pycache__",
25
+ "build",
26
+ "dist",
27
+ "fixtures",
28
+ "node_modules",
29
+ "site-packages",
30
+ "testdata",
31
+ "tests",
32
+ "vendor",
33
+ "venv",
34
+ }
35
+
36
+
37
+ def _is_background_version_fixture_path(rel_path: str) -> bool:
38
+ """Return whether auto-discovery should skip a version-pin candidate."""
39
+
40
+ return any(part in VERSION_PIN_BACKGROUND_SKIP_PARTS for part in Path(rel_path).parts)
41
+
42
+
43
+ def _nested_image_pins(text: str) -> tuple[list[str], set[int]]:
44
+ """Extract image.name or image.repository plus tag/digest mappings."""
45
+
46
+ pins: list[str] = []
47
+ consumed_lines: set[int] = set()
48
+ image_indent: int | None = None
49
+ fields: dict[str, str] = {}
50
+ image_line = -1
51
+
52
+ def flush() -> None:
53
+ name = fields.get("name") or fields.get("repository")
54
+ if name:
55
+ if fields.get("digest"):
56
+ pins.append(f"{name}@{fields['digest']}")
57
+ elif fields.get("tag"):
58
+ pins.append(f"{name}:{fields['tag']}")
59
+ else:
60
+ pins.append(name)
61
+ fields.clear()
62
+
63
+ source_lines = text.splitlines()
64
+ for line_number, raw in enumerate([*source_lines, ""]):
65
+ line = raw.strip()
66
+ indent = len(raw) - len(raw.lstrip())
67
+ at_eof = line_number == len(source_lines)
68
+ if image_indent is not None and (
69
+ at_eof or (line and not line.startswith("#") and indent <= image_indent)
70
+ ):
71
+ flush()
72
+ image_indent = None
73
+ if not line or line.startswith("#"):
74
+ continue
75
+ if image_indent is None:
76
+ if re.match(r"(?i)^image\s*:\s*$", line):
77
+ image_indent = indent
78
+ image_line = line_number
79
+ continue
80
+ field = re.match(
81
+ r"(?i)^(name|repository|tag|digest)\s*:\s*[\"']?([^\"'#\s]+)",
82
+ line,
83
+ )
84
+ if field:
85
+ fields[field.group(1).lower()] = redact_url_userinfo(field.group(2).strip())
86
+ consumed_lines.add(line_number)
87
+ consumed_lines.add(image_line)
88
+ return pins, consumed_lines
89
+
90
+
91
+ def _ansible_requirement_items(path: Path, limit: int) -> list[tuple[str, str | None]]:
92
+ """Parse bounded Galaxy requirement items without interpreting nested lists."""
93
+
94
+ items: list[tuple[str, str | None]] = []
95
+ item_indent: int | None = None
96
+ field_indent: int | None = None
97
+ fields: dict[str, str] = {}
98
+
99
+ def flush() -> None:
100
+ if len(items) >= limit:
101
+ fields.clear()
102
+ return
103
+ identifier = fields.get("name") or fields.get("src")
104
+ if identifier:
105
+ items.append((redact_url_userinfo(identifier), fields.get("version")))
106
+ fields.clear()
107
+
108
+ for raw in context_repo.read_text(path).splitlines():
109
+ line = raw.strip()
110
+ if not line or line.startswith("#"):
111
+ continue
112
+ indent = len(raw) - len(raw.lstrip(" "))
113
+ list_item = re.match(r"-\s*(name|src|version):\s*[\"']?([^\"'#]+)", line)
114
+ if line.startswith("-") and (item_indent is None or indent <= item_indent):
115
+ if item_indent is not None:
116
+ flush()
117
+ if len(items) >= limit:
118
+ break
119
+ item_indent = indent if list_item else None
120
+ field_indent = None
121
+ if list_item:
122
+ fields[list_item.group(1)] = list_item.group(2).strip()
123
+ continue
124
+ if item_indent is None or line.startswith("-") or indent <= item_indent:
125
+ continue
126
+ if field_indent is None:
127
+ field_indent = indent
128
+ if indent != field_indent:
129
+ continue
130
+ field = re.match(r"(name|src|version):\s*[\"']?([^\"'#]+)", line)
131
+ if field:
132
+ fields[field.group(1)] = field.group(2).strip()
133
+
134
+ if item_indent is not None and len(items) < limit:
135
+ flush()
136
+ return items
137
+
138
+
139
+ def parse_ansible_requirements(path: Path) -> list[str]:
140
+ """Extract common collection/role names and versions from requirements.yml.
141
+
142
+ This is a lightweight parser that avoids adding PyYAML to the CI image.
143
+ """
144
+
145
+ items: list[str] = []
146
+ for identifier, version in _ansible_requirement_items(path, MAX_BACKGROUND_SECTION_ITEMS):
147
+ items.append(f"{identifier}: {redact_url_userinfo(version)}" if version else identifier)
148
+ if len(items) >= MAX_BACKGROUND_SECTION_ITEMS:
149
+ break
150
+ return items
151
+
152
+
153
+ def parse_ansible_requirement_version_pins(path: Path, limit: int = 80) -> list[str]:
154
+ """Extract collection/role version pins with their names for context output."""
155
+
156
+ items: list[str] = []
157
+ rel_path = path.relative_to(context_repo.ROOT)
158
+ for identifier, version in _ansible_requirement_items(path, limit):
159
+ if version:
160
+ items.append(f"{rel_path}: {identifier}={redact_url_userinfo(version)}")
161
+ if len(items) >= limit:
162
+ break
163
+ return items
164
+
165
+
166
+ def is_root_ansible_playbook(rel_path: str) -> bool:
167
+ """Return whether a root YAML file looks like an Ansible playbook.
168
+
169
+ An Ansible playbook is a list of plays, so the top level must be a
170
+ sequence (``- ...``) and at least one play must contain a ``hosts``
171
+ key. A bare ``hosts:`` at the top level belongs to inventory/Compose/
172
+ etc. and is intentionally rejected to keep the background context
173
+ free of false positives.
174
+ """
175
+
176
+ path = Path(rel_path)
177
+ if len(path.parts) != 1 or path.suffix.lower() not in {".yml", ".yaml"}:
178
+ return False
179
+
180
+ text = context_repo.read_text(context_repo.ROOT / rel_path, max_bytes=64_000)
181
+ if not text:
182
+ return False
183
+
184
+ hosts_key = re.compile(r"^[\"']?hosts[\"']?\s*:")
185
+ import_playbook_key = re.compile(r"^[\"']?import_playbook[\"']?\s*:")
186
+ top_level_list_item = False
187
+ play_child_indent: int | None = None
188
+
189
+ for raw in text.splitlines():
190
+ line = raw.rstrip()
191
+ stripped = line.strip()
192
+
193
+ if not stripped or stripped.startswith("#") or stripped in {"---", "..."}:
194
+ continue
195
+
196
+ indent = len(line) - len(line.lstrip(" "))
197
+
198
+ if indent == 0 and stripped.startswith("-"):
199
+ top_level_list_item = True
200
+ play_child_indent = None
201
+ item = stripped[1:].lstrip()
202
+ if hosts_key.match(item) or import_playbook_key.match(item):
203
+ return True
204
+ continue
205
+
206
+ if indent == 0:
207
+ # A top-level mapping rules the file out: playbooks must start
208
+ # with a sequence. Reset list-item tracking and move on.
209
+ top_level_list_item = False
210
+ play_child_indent = None
211
+ continue
212
+
213
+ if not top_level_list_item:
214
+ continue
215
+
216
+ if play_child_indent is None:
217
+ play_child_indent = indent
218
+
219
+ if indent == play_child_indent and (
220
+ hosts_key.match(stripped) or import_playbook_key.match(stripped)
221
+ ):
222
+ return True
223
+
224
+ return False
225
+
226
+
227
+ def detect_root_ansible_playbooks(
228
+ limit: int = MAX_BACKGROUND_SECTION_ITEMS,
229
+ ) -> list[str]:
230
+ """Detect root playbook entrypoints without classifying every root YAML file."""
231
+
232
+ candidates = context_repo.rel_glob_files(["*.yml", "*.yaml"], limit=500)
233
+ playbooks = [rel_path for rel_path in candidates if is_root_ansible_playbook(rel_path)]
234
+ return sorted(playbooks)[:limit]
235
+
236
+
237
+ def is_inventory_topology_file(rel_path: str) -> bool:
238
+ """Return whether a path is likely an Ansible inventory topology file."""
239
+
240
+ path = Path(rel_path)
241
+ if any(part.startswith(".") for part in path.parts):
242
+ return False
243
+
244
+ parts = {part.lower() for part in path.parts}
245
+ name = path.name.lower()
246
+
247
+ if "group_vars" in parts or "host_vars" in parts:
248
+ return False
249
+
250
+ if name in {
251
+ "hosts",
252
+ "hosts.ini",
253
+ "hosts.yml",
254
+ "hosts.yaml",
255
+ "inventory",
256
+ "inventory.ini",
257
+ "inventory.yml",
258
+ "inventory.yaml",
259
+ }:
260
+ return True
261
+
262
+ if "inventory" in parts or "inventories" in parts:
263
+ return Path(rel_path).suffix.lower() in {
264
+ "",
265
+ ".ini",
266
+ ".cfg",
267
+ ".conf",
268
+ ".yml",
269
+ ".yaml",
270
+ }
271
+
272
+ return False
273
+
274
+
275
+ def extract_inventory_topology(paths: Sequence[str], limit: int = 120) -> list[str]:
276
+ """Extract a bounded list of Ansible inventory group-like names.
277
+
278
+ The parser is intentionally conservative. INI groups are detected from
279
+ `[group]` headers. YAML inventory groups are detected only under `children`
280
+ sections or as the top-level `all` group, avoiding hostnames under `hosts`.
281
+ """
282
+
283
+ groups: set[str] = set()
284
+
285
+ for rel_path in paths:
286
+ path = context_repo.ROOT / rel_path
287
+ if context_repo.resolve_repo_file(path) is None:
288
+ continue
289
+
290
+ # The indentation step inside `children:` is whatever the first
291
+ # nested key uses (commonly 2, but inventories with 4-space indent
292
+ # exist). Lock it in on first sighting so we only pick up direct
293
+ # children, not deeper descendants such as `hosts:` entries.
294
+ yaml_children_indent: int | None = None
295
+ yaml_child_step: int | None = None
296
+
297
+ for raw in context_repo.read_text(path, max_bytes=128_000).splitlines():
298
+ if not raw.strip() or raw.lstrip().startswith("#"):
299
+ continue
300
+
301
+ stripped = raw.strip()
302
+ indent = len(raw) - len(raw.lstrip(" "))
303
+
304
+ ini_match = re.match(r"^\[([A-Za-z0-9_.-]+)(?::(?:children|vars))?\]$", stripped)
305
+ if ini_match:
306
+ groups.add(ini_match.group(1))
307
+ if len(groups) >= limit:
308
+ break
309
+ continue
310
+
311
+ key_match = re.match(r"^([A-Za-z0-9_.-]+):\s*$", stripped)
312
+ if not key_match:
313
+ continue
314
+
315
+ key = key_match.group(1)
316
+
317
+ if key == "all" and indent == 0:
318
+ groups.add(key)
319
+ elif key == "children":
320
+ yaml_children_indent = indent
321
+ yaml_child_step = None
322
+ elif yaml_children_indent is not None and indent > yaml_children_indent:
323
+ if yaml_child_step is None:
324
+ yaml_child_step = indent - yaml_children_indent
325
+ if indent == yaml_children_indent + yaml_child_step and key not in {
326
+ "vars",
327
+ "hosts",
328
+ "children",
329
+ }:
330
+ groups.add(key)
331
+ elif yaml_children_indent is not None and indent <= yaml_children_indent:
332
+ yaml_children_indent = None
333
+ yaml_child_step = None
334
+
335
+ if len(groups) >= limit:
336
+ break
337
+
338
+ return sorted(groups)[:limit]
339
+
340
+
341
+ def extract_application_versions(
342
+ files: Sequence[str], limit: int = 160, *, include_discovered: bool = True
343
+ ) -> list[str]:
344
+ """Extract likely application/runtime version pins from common config files.
345
+
346
+ The scanner is heuristic and intentionally conservative. It looks for
347
+ version-like keys in configuration/manifests while avoiding ordinary Ansible
348
+ `tags` and arbitrary source-code variables.
349
+ """
350
+
351
+ candidate_patterns = [
352
+ "*.yml",
353
+ "*.yaml",
354
+ "*.json",
355
+ "*.toml",
356
+ "*.tf",
357
+ "*.tfvars",
358
+ "*.hcl",
359
+ "Dockerfile",
360
+ "Dockerfile.*",
361
+ "Containerfile",
362
+ "Containerfile.*",
363
+ "docker-compose*.yml",
364
+ "docker-compose*.yaml",
365
+ "Chart.yaml",
366
+ "values*.yaml",
367
+ "values*.yml",
368
+ "**/Chart.yaml",
369
+ "**/Dockerfile",
370
+ "**/Dockerfile.*",
371
+ "**/Containerfile",
372
+ "**/Containerfile.*",
373
+ "**/values*.yaml",
374
+ "**/values*.yml",
375
+ "**/defaults/**/*.yml",
376
+ "**/defaults/**/*.yaml",
377
+ "**/vars/**/*.yml",
378
+ "**/vars/**/*.yaml",
379
+ ]
380
+
381
+ candidate_suffixes = (
382
+ ".yml",
383
+ ".yaml",
384
+ ".json",
385
+ ".toml",
386
+ ".tf",
387
+ ".tfvars",
388
+ ".hcl",
389
+ )
390
+ candidate_basenames = {
391
+ "Dockerfile",
392
+ "Containerfile",
393
+ "Chart.yaml",
394
+ "docker-compose.yml",
395
+ "docker-compose.yaml",
396
+ }
397
+
398
+ candidate_files = {
399
+ file_path
400
+ for file_path in files
401
+ if not is_env_file(file_path)
402
+ and not _is_background_version_fixture_path(file_path)
403
+ and (
404
+ Path(file_path).name in candidate_basenames
405
+ or Path(file_path).name.lower().startswith(("dockerfile.", "containerfile."))
406
+ or file_path.endswith(candidate_suffixes)
407
+ )
408
+ }
409
+
410
+ if include_discovered:
411
+ auto_discovery_limit = max(80, limit * 4)
412
+ discovered_files: set[str] = set()
413
+ for pattern in candidate_patterns:
414
+ discovered_files.update(
415
+ rel_path
416
+ for rel_path in context_repo.rel_glob_files(
417
+ [pattern],
418
+ limit=auto_discovery_limit,
419
+ exclude_dirs=context_repo.DEFAULT_EXCLUDE_DIRS
420
+ | VERSION_PIN_BACKGROUND_SKIP_PARTS,
421
+ )
422
+ if not _is_background_version_fixture_path(rel_path)
423
+ )
424
+ for rel_path in sorted(discovered_files):
425
+ candidate_files.add(rel_path)
426
+ if len(candidate_files) >= auto_discovery_limit:
427
+ break
428
+
429
+ results: list[str] = []
430
+ seen: set[str] = set()
431
+
432
+ version_line = re.compile(
433
+ r"(?i)^\s*-?\s*[\"']?("
434
+ r"[A-Za-z0-9_.-]*(?:version|image|chart|app[_-]?version)[A-Za-z0-9_.-]*"
435
+ r"|tag"
436
+ r")[\"']?\s*[:=]\s*[\"']?([^\"'#\s]+)"
437
+ )
438
+
439
+ image_line = re.compile(r"(?i)^\s*-?\s*[\"']?image[\"']?\s*[:=]\s*[\"']?([^\"'#\s]+)")
440
+
441
+ changed_candidate_files = {file_path for file_path in files if file_path in candidate_files}
442
+ for rel_path in sorted(
443
+ candidate_files,
444
+ key=lambda item: (item not in changed_candidate_files, item),
445
+ ):
446
+ if Path(rel_path).name in {"requirements.yml", "requirements.yaml"}:
447
+ continue
448
+ if is_env_file(rel_path):
449
+ continue
450
+
451
+ path = context_repo.ROOT / rel_path
452
+ safe_path = context_repo.resolve_repo_file(path)
453
+ if safe_path is None:
454
+ continue
455
+
456
+ try:
457
+ if safe_path.stat().st_size > DEFAULT_MAX_FILE_BYTES:
458
+ continue
459
+ except OSError:
460
+ continue
461
+
462
+ text = context_repo.read_text(safe_path, max_bytes=128_000)
463
+ is_containerfile = Path(rel_path).name.lower().startswith(("dockerfile", "containerfile"))
464
+ nested_pins, consumed_lines = _nested_image_pins(text)
465
+ scan_lines = [f"image: {image_ref}" for image_ref in nested_pins]
466
+ scan_lines.extend(
467
+ raw
468
+ for line_number, raw in enumerate(text.splitlines())
469
+ if line_number not in consumed_lines
470
+ )
471
+ for raw in scan_lines:
472
+ line = raw.strip()
473
+ if not line or line.startswith("#"):
474
+ continue
475
+
476
+ if is_containerfile:
477
+ from_match = re.match(r"(?i)^FROM\s+([^\s]+)", line)
478
+ if not from_match:
479
+ continue
480
+ from_parts = line.split()[1:]
481
+ while from_parts and from_parts[0].startswith("--"):
482
+ from_parts.pop(0)
483
+ if not from_parts:
484
+ continue
485
+ key = "image"
486
+ value = redact_url_userinfo(from_parts[0])
487
+ else:
488
+ # Skip ordinary Ansible tags, but allow singular `tag: 1.2.3`.
489
+ if re.match(r"(?i)^\s*-?\s*[\"']?tags[\"']?\s*:", line):
490
+ continue
491
+
492
+ version_match = version_line.search(line)
493
+ image_match = image_line.search(line)
494
+
495
+ if not version_match and not image_match:
496
+ continue
497
+
498
+ if not re.search(
499
+ r"(?i)(version|image|chart|app[_-]?version|tag)\s*[:=]",
500
+ line,
501
+ ):
502
+ continue
503
+
504
+ if image_match:
505
+ key = "image"
506
+ value = redact_url_userinfo(image_match.group(1).strip())
507
+ elif version_match:
508
+ key = version_match.group(1).strip()
509
+ normalized_key = re.sub(r"(?<!^)(?=[A-Z])", "_", key).lower().replace("-", "_")
510
+ if normalized_key in {
511
+ "api_version",
512
+ "config_version",
513
+ "format_version",
514
+ "kind",
515
+ "kind_version",
516
+ "schema",
517
+ "schema_version",
518
+ "spec_version",
519
+ "$schema",
520
+ }:
521
+ continue
522
+ value = redact_url_userinfo(version_match.group(2).strip())
523
+ else:
524
+ continue
525
+
526
+ key_lower = key.lower()
527
+ is_image_key = key_lower == "image" or key_lower.endswith("image")
528
+ normalized_value = value.strip().lower()
529
+ image_ref = normalized_value.rsplit("/", 1)[-1]
530
+ image_name, _, _image_digest = image_ref.partition("@")
531
+ has_digest = "@" in normalized_value
532
+ image_tag = image_name.rsplit(":", 1)[-1] if ":" in image_name else ""
533
+ if (
534
+ not value
535
+ or normalized_value
536
+ in {
537
+ "true",
538
+ "false",
539
+ "yes",
540
+ "no",
541
+ "null",
542
+ "none",
543
+ "latest",
544
+ }
545
+ or (is_image_key and not has_digest and image_tag in {"", "latest"})
546
+ or any(marker in value for marker in ("{{", "{%", "${"))
547
+ or (not is_image_key and not re.search(r"\d", value))
548
+ ):
549
+ continue
550
+
551
+ item = redact_sensitive(f"{rel_path}: {key}={value}")
552
+ if item in seen:
553
+ continue
554
+
555
+ seen.add(item)
556
+ results.append(item)
557
+
558
+ if len(results) >= limit:
559
+ return results[:limit]
560
+
561
+ return results[:limit]