open-code-review-toolkit 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ocr_toolkit/__init__.py +1 -0
- ocr_toolkit/_version.py +24 -0
- ocr_toolkit/cli.py +56 -0
- ocr_toolkit/common/__init__.py +1 -0
- ocr_toolkit/common/language.py +60 -0
- ocr_toolkit/common/markdown.py +214 -0
- ocr_toolkit/common/redaction.py +324 -0
- ocr_toolkit/config_writer.py +106 -0
- ocr_toolkit/configure.py +194 -0
- ocr_toolkit/context/__init__.py +1 -0
- ocr_toolkit/context/__main__.py +8 -0
- ocr_toolkit/context/ansible.py +561 -0
- ocr_toolkit/context/categorize.py +162 -0
- ocr_toolkit/context/instructions.py +269 -0
- ocr_toolkit/context/manifests.py +459 -0
- ocr_toolkit/context/planner.py +227 -0
- ocr_toolkit/context/render.py +955 -0
- ocr_toolkit/context/repo.py +672 -0
- ocr_toolkit/context/settings.py +97 -0
- ocr_toolkit/mcp_config.py +257 -0
- ocr_toolkit/posting/__init__.py +1 -0
- ocr_toolkit/posting/__main__.py +8 -0
- ocr_toolkit/posting/comments.py +87 -0
- ocr_toolkit/posting/formatting.py +755 -0
- ocr_toolkit/posting/gitlab.py +853 -0
- ocr_toolkit/posting/markers.py +284 -0
- ocr_toolkit/posting/payloads.py +141 -0
- ocr_toolkit/posting/result.py +116 -0
- ocr_toolkit/posting/settings.py +181 -0
- ocr_toolkit/posting/snapshot.py +468 -0
- ocr_toolkit/posting/workflow.py +873 -0
- ocr_toolkit/preflight.py +395 -0
- ocr_toolkit/py.typed +1 -0
- open_code_review_toolkit-0.1.0.dist-info/METADATA +283 -0
- open_code_review_toolkit-0.1.0.dist-info/RECORD +38 -0
- open_code_review_toolkit-0.1.0.dist-info/WHEEL +4 -0
- open_code_review_toolkit-0.1.0.dist-info/entry_points.txt +2 -0
- open_code_review_toolkit-0.1.0.dist-info/licenses/LICENSE +202 -0
|
@@ -0,0 +1,561 @@
|
|
|
1
|
+
"""Ansible-specific context discovery for OCR."""
|
|
2
|
+
|
|
3
|
+
from __future__ import annotations
|
|
4
|
+
|
|
5
|
+
import re
|
|
6
|
+
from collections.abc import Sequence
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
|
|
9
|
+
from ocr_toolkit.common.redaction import redact_sensitive, redact_url_userinfo
|
|
10
|
+
from ocr_toolkit.context import repo as context_repo
|
|
11
|
+
from ocr_toolkit.context.settings import (
|
|
12
|
+
DEFAULT_MAX_FILE_BYTES,
|
|
13
|
+
MAX_BACKGROUND_SECTION_ITEMS,
|
|
14
|
+
is_env_file,
|
|
15
|
+
)
|
|
16
|
+
|
|
17
|
+
VERSION_PIN_BACKGROUND_SKIP_PARTS = {
|
|
18
|
+
".cache",
|
|
19
|
+
".git",
|
|
20
|
+
".molecule",
|
|
21
|
+
".review-context",
|
|
22
|
+
".terraform",
|
|
23
|
+
".venv",
|
|
24
|
+
"__pycache__",
|
|
25
|
+
"build",
|
|
26
|
+
"dist",
|
|
27
|
+
"fixtures",
|
|
28
|
+
"node_modules",
|
|
29
|
+
"site-packages",
|
|
30
|
+
"testdata",
|
|
31
|
+
"tests",
|
|
32
|
+
"vendor",
|
|
33
|
+
"venv",
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
|
|
37
|
+
def _is_background_version_fixture_path(rel_path: str) -> bool:
|
|
38
|
+
"""Return whether auto-discovery should skip a version-pin candidate."""
|
|
39
|
+
|
|
40
|
+
return any(part in VERSION_PIN_BACKGROUND_SKIP_PARTS for part in Path(rel_path).parts)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _nested_image_pins(text: str) -> tuple[list[str], set[int]]:
|
|
44
|
+
"""Extract image.name or image.repository plus tag/digest mappings."""
|
|
45
|
+
|
|
46
|
+
pins: list[str] = []
|
|
47
|
+
consumed_lines: set[int] = set()
|
|
48
|
+
image_indent: int | None = None
|
|
49
|
+
fields: dict[str, str] = {}
|
|
50
|
+
image_line = -1
|
|
51
|
+
|
|
52
|
+
def flush() -> None:
|
|
53
|
+
name = fields.get("name") or fields.get("repository")
|
|
54
|
+
if name:
|
|
55
|
+
if fields.get("digest"):
|
|
56
|
+
pins.append(f"{name}@{fields['digest']}")
|
|
57
|
+
elif fields.get("tag"):
|
|
58
|
+
pins.append(f"{name}:{fields['tag']}")
|
|
59
|
+
else:
|
|
60
|
+
pins.append(name)
|
|
61
|
+
fields.clear()
|
|
62
|
+
|
|
63
|
+
source_lines = text.splitlines()
|
|
64
|
+
for line_number, raw in enumerate([*source_lines, ""]):
|
|
65
|
+
line = raw.strip()
|
|
66
|
+
indent = len(raw) - len(raw.lstrip())
|
|
67
|
+
at_eof = line_number == len(source_lines)
|
|
68
|
+
if image_indent is not None and (
|
|
69
|
+
at_eof or (line and not line.startswith("#") and indent <= image_indent)
|
|
70
|
+
):
|
|
71
|
+
flush()
|
|
72
|
+
image_indent = None
|
|
73
|
+
if not line or line.startswith("#"):
|
|
74
|
+
continue
|
|
75
|
+
if image_indent is None:
|
|
76
|
+
if re.match(r"(?i)^image\s*:\s*$", line):
|
|
77
|
+
image_indent = indent
|
|
78
|
+
image_line = line_number
|
|
79
|
+
continue
|
|
80
|
+
field = re.match(
|
|
81
|
+
r"(?i)^(name|repository|tag|digest)\s*:\s*[\"']?([^\"'#\s]+)",
|
|
82
|
+
line,
|
|
83
|
+
)
|
|
84
|
+
if field:
|
|
85
|
+
fields[field.group(1).lower()] = redact_url_userinfo(field.group(2).strip())
|
|
86
|
+
consumed_lines.add(line_number)
|
|
87
|
+
consumed_lines.add(image_line)
|
|
88
|
+
return pins, consumed_lines
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _ansible_requirement_items(path: Path, limit: int) -> list[tuple[str, str | None]]:
|
|
92
|
+
"""Parse bounded Galaxy requirement items without interpreting nested lists."""
|
|
93
|
+
|
|
94
|
+
items: list[tuple[str, str | None]] = []
|
|
95
|
+
item_indent: int | None = None
|
|
96
|
+
field_indent: int | None = None
|
|
97
|
+
fields: dict[str, str] = {}
|
|
98
|
+
|
|
99
|
+
def flush() -> None:
|
|
100
|
+
if len(items) >= limit:
|
|
101
|
+
fields.clear()
|
|
102
|
+
return
|
|
103
|
+
identifier = fields.get("name") or fields.get("src")
|
|
104
|
+
if identifier:
|
|
105
|
+
items.append((redact_url_userinfo(identifier), fields.get("version")))
|
|
106
|
+
fields.clear()
|
|
107
|
+
|
|
108
|
+
for raw in context_repo.read_text(path).splitlines():
|
|
109
|
+
line = raw.strip()
|
|
110
|
+
if not line or line.startswith("#"):
|
|
111
|
+
continue
|
|
112
|
+
indent = len(raw) - len(raw.lstrip(" "))
|
|
113
|
+
list_item = re.match(r"-\s*(name|src|version):\s*[\"']?([^\"'#]+)", line)
|
|
114
|
+
if line.startswith("-") and (item_indent is None or indent <= item_indent):
|
|
115
|
+
if item_indent is not None:
|
|
116
|
+
flush()
|
|
117
|
+
if len(items) >= limit:
|
|
118
|
+
break
|
|
119
|
+
item_indent = indent if list_item else None
|
|
120
|
+
field_indent = None
|
|
121
|
+
if list_item:
|
|
122
|
+
fields[list_item.group(1)] = list_item.group(2).strip()
|
|
123
|
+
continue
|
|
124
|
+
if item_indent is None or line.startswith("-") or indent <= item_indent:
|
|
125
|
+
continue
|
|
126
|
+
if field_indent is None:
|
|
127
|
+
field_indent = indent
|
|
128
|
+
if indent != field_indent:
|
|
129
|
+
continue
|
|
130
|
+
field = re.match(r"(name|src|version):\s*[\"']?([^\"'#]+)", line)
|
|
131
|
+
if field:
|
|
132
|
+
fields[field.group(1)] = field.group(2).strip()
|
|
133
|
+
|
|
134
|
+
if item_indent is not None and len(items) < limit:
|
|
135
|
+
flush()
|
|
136
|
+
return items
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
def parse_ansible_requirements(path: Path) -> list[str]:
|
|
140
|
+
"""Extract common collection/role names and versions from requirements.yml.
|
|
141
|
+
|
|
142
|
+
This is a lightweight parser that avoids adding PyYAML to the CI image.
|
|
143
|
+
"""
|
|
144
|
+
|
|
145
|
+
items: list[str] = []
|
|
146
|
+
for identifier, version in _ansible_requirement_items(path, MAX_BACKGROUND_SECTION_ITEMS):
|
|
147
|
+
items.append(f"{identifier}: {redact_url_userinfo(version)}" if version else identifier)
|
|
148
|
+
if len(items) >= MAX_BACKGROUND_SECTION_ITEMS:
|
|
149
|
+
break
|
|
150
|
+
return items
|
|
151
|
+
|
|
152
|
+
|
|
153
|
+
def parse_ansible_requirement_version_pins(path: Path, limit: int = 80) -> list[str]:
|
|
154
|
+
"""Extract collection/role version pins with their names for context output."""
|
|
155
|
+
|
|
156
|
+
items: list[str] = []
|
|
157
|
+
rel_path = path.relative_to(context_repo.ROOT)
|
|
158
|
+
for identifier, version in _ansible_requirement_items(path, limit):
|
|
159
|
+
if version:
|
|
160
|
+
items.append(f"{rel_path}: {identifier}={redact_url_userinfo(version)}")
|
|
161
|
+
if len(items) >= limit:
|
|
162
|
+
break
|
|
163
|
+
return items
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def is_root_ansible_playbook(rel_path: str) -> bool:
|
|
167
|
+
"""Return whether a root YAML file looks like an Ansible playbook.
|
|
168
|
+
|
|
169
|
+
An Ansible playbook is a list of plays, so the top level must be a
|
|
170
|
+
sequence (``- ...``) and at least one play must contain a ``hosts``
|
|
171
|
+
key. A bare ``hosts:`` at the top level belongs to inventory/Compose/
|
|
172
|
+
etc. and is intentionally rejected to keep the background context
|
|
173
|
+
free of false positives.
|
|
174
|
+
"""
|
|
175
|
+
|
|
176
|
+
path = Path(rel_path)
|
|
177
|
+
if len(path.parts) != 1 or path.suffix.lower() not in {".yml", ".yaml"}:
|
|
178
|
+
return False
|
|
179
|
+
|
|
180
|
+
text = context_repo.read_text(context_repo.ROOT / rel_path, max_bytes=64_000)
|
|
181
|
+
if not text:
|
|
182
|
+
return False
|
|
183
|
+
|
|
184
|
+
hosts_key = re.compile(r"^[\"']?hosts[\"']?\s*:")
|
|
185
|
+
import_playbook_key = re.compile(r"^[\"']?import_playbook[\"']?\s*:")
|
|
186
|
+
top_level_list_item = False
|
|
187
|
+
play_child_indent: int | None = None
|
|
188
|
+
|
|
189
|
+
for raw in text.splitlines():
|
|
190
|
+
line = raw.rstrip()
|
|
191
|
+
stripped = line.strip()
|
|
192
|
+
|
|
193
|
+
if not stripped or stripped.startswith("#") or stripped in {"---", "..."}:
|
|
194
|
+
continue
|
|
195
|
+
|
|
196
|
+
indent = len(line) - len(line.lstrip(" "))
|
|
197
|
+
|
|
198
|
+
if indent == 0 and stripped.startswith("-"):
|
|
199
|
+
top_level_list_item = True
|
|
200
|
+
play_child_indent = None
|
|
201
|
+
item = stripped[1:].lstrip()
|
|
202
|
+
if hosts_key.match(item) or import_playbook_key.match(item):
|
|
203
|
+
return True
|
|
204
|
+
continue
|
|
205
|
+
|
|
206
|
+
if indent == 0:
|
|
207
|
+
# A top-level mapping rules the file out: playbooks must start
|
|
208
|
+
# with a sequence. Reset list-item tracking and move on.
|
|
209
|
+
top_level_list_item = False
|
|
210
|
+
play_child_indent = None
|
|
211
|
+
continue
|
|
212
|
+
|
|
213
|
+
if not top_level_list_item:
|
|
214
|
+
continue
|
|
215
|
+
|
|
216
|
+
if play_child_indent is None:
|
|
217
|
+
play_child_indent = indent
|
|
218
|
+
|
|
219
|
+
if indent == play_child_indent and (
|
|
220
|
+
hosts_key.match(stripped) or import_playbook_key.match(stripped)
|
|
221
|
+
):
|
|
222
|
+
return True
|
|
223
|
+
|
|
224
|
+
return False
|
|
225
|
+
|
|
226
|
+
|
|
227
|
+
def detect_root_ansible_playbooks(
|
|
228
|
+
limit: int = MAX_BACKGROUND_SECTION_ITEMS,
|
|
229
|
+
) -> list[str]:
|
|
230
|
+
"""Detect root playbook entrypoints without classifying every root YAML file."""
|
|
231
|
+
|
|
232
|
+
candidates = context_repo.rel_glob_files(["*.yml", "*.yaml"], limit=500)
|
|
233
|
+
playbooks = [rel_path for rel_path in candidates if is_root_ansible_playbook(rel_path)]
|
|
234
|
+
return sorted(playbooks)[:limit]
|
|
235
|
+
|
|
236
|
+
|
|
237
|
+
def is_inventory_topology_file(rel_path: str) -> bool:
|
|
238
|
+
"""Return whether a path is likely an Ansible inventory topology file."""
|
|
239
|
+
|
|
240
|
+
path = Path(rel_path)
|
|
241
|
+
if any(part.startswith(".") for part in path.parts):
|
|
242
|
+
return False
|
|
243
|
+
|
|
244
|
+
parts = {part.lower() for part in path.parts}
|
|
245
|
+
name = path.name.lower()
|
|
246
|
+
|
|
247
|
+
if "group_vars" in parts or "host_vars" in parts:
|
|
248
|
+
return False
|
|
249
|
+
|
|
250
|
+
if name in {
|
|
251
|
+
"hosts",
|
|
252
|
+
"hosts.ini",
|
|
253
|
+
"hosts.yml",
|
|
254
|
+
"hosts.yaml",
|
|
255
|
+
"inventory",
|
|
256
|
+
"inventory.ini",
|
|
257
|
+
"inventory.yml",
|
|
258
|
+
"inventory.yaml",
|
|
259
|
+
}:
|
|
260
|
+
return True
|
|
261
|
+
|
|
262
|
+
if "inventory" in parts or "inventories" in parts:
|
|
263
|
+
return Path(rel_path).suffix.lower() in {
|
|
264
|
+
"",
|
|
265
|
+
".ini",
|
|
266
|
+
".cfg",
|
|
267
|
+
".conf",
|
|
268
|
+
".yml",
|
|
269
|
+
".yaml",
|
|
270
|
+
}
|
|
271
|
+
|
|
272
|
+
return False
|
|
273
|
+
|
|
274
|
+
|
|
275
|
+
def extract_inventory_topology(paths: Sequence[str], limit: int = 120) -> list[str]:
|
|
276
|
+
"""Extract a bounded list of Ansible inventory group-like names.
|
|
277
|
+
|
|
278
|
+
The parser is intentionally conservative. INI groups are detected from
|
|
279
|
+
`[group]` headers. YAML inventory groups are detected only under `children`
|
|
280
|
+
sections or as the top-level `all` group, avoiding hostnames under `hosts`.
|
|
281
|
+
"""
|
|
282
|
+
|
|
283
|
+
groups: set[str] = set()
|
|
284
|
+
|
|
285
|
+
for rel_path in paths:
|
|
286
|
+
path = context_repo.ROOT / rel_path
|
|
287
|
+
if context_repo.resolve_repo_file(path) is None:
|
|
288
|
+
continue
|
|
289
|
+
|
|
290
|
+
# The indentation step inside `children:` is whatever the first
|
|
291
|
+
# nested key uses (commonly 2, but inventories with 4-space indent
|
|
292
|
+
# exist). Lock it in on first sighting so we only pick up direct
|
|
293
|
+
# children, not deeper descendants such as `hosts:` entries.
|
|
294
|
+
yaml_children_indent: int | None = None
|
|
295
|
+
yaml_child_step: int | None = None
|
|
296
|
+
|
|
297
|
+
for raw in context_repo.read_text(path, max_bytes=128_000).splitlines():
|
|
298
|
+
if not raw.strip() or raw.lstrip().startswith("#"):
|
|
299
|
+
continue
|
|
300
|
+
|
|
301
|
+
stripped = raw.strip()
|
|
302
|
+
indent = len(raw) - len(raw.lstrip(" "))
|
|
303
|
+
|
|
304
|
+
ini_match = re.match(r"^\[([A-Za-z0-9_.-]+)(?::(?:children|vars))?\]$", stripped)
|
|
305
|
+
if ini_match:
|
|
306
|
+
groups.add(ini_match.group(1))
|
|
307
|
+
if len(groups) >= limit:
|
|
308
|
+
break
|
|
309
|
+
continue
|
|
310
|
+
|
|
311
|
+
key_match = re.match(r"^([A-Za-z0-9_.-]+):\s*$", stripped)
|
|
312
|
+
if not key_match:
|
|
313
|
+
continue
|
|
314
|
+
|
|
315
|
+
key = key_match.group(1)
|
|
316
|
+
|
|
317
|
+
if key == "all" and indent == 0:
|
|
318
|
+
groups.add(key)
|
|
319
|
+
elif key == "children":
|
|
320
|
+
yaml_children_indent = indent
|
|
321
|
+
yaml_child_step = None
|
|
322
|
+
elif yaml_children_indent is not None and indent > yaml_children_indent:
|
|
323
|
+
if yaml_child_step is None:
|
|
324
|
+
yaml_child_step = indent - yaml_children_indent
|
|
325
|
+
if indent == yaml_children_indent + yaml_child_step and key not in {
|
|
326
|
+
"vars",
|
|
327
|
+
"hosts",
|
|
328
|
+
"children",
|
|
329
|
+
}:
|
|
330
|
+
groups.add(key)
|
|
331
|
+
elif yaml_children_indent is not None and indent <= yaml_children_indent:
|
|
332
|
+
yaml_children_indent = None
|
|
333
|
+
yaml_child_step = None
|
|
334
|
+
|
|
335
|
+
if len(groups) >= limit:
|
|
336
|
+
break
|
|
337
|
+
|
|
338
|
+
return sorted(groups)[:limit]
|
|
339
|
+
|
|
340
|
+
|
|
341
|
+
def extract_application_versions(
|
|
342
|
+
files: Sequence[str], limit: int = 160, *, include_discovered: bool = True
|
|
343
|
+
) -> list[str]:
|
|
344
|
+
"""Extract likely application/runtime version pins from common config files.
|
|
345
|
+
|
|
346
|
+
The scanner is heuristic and intentionally conservative. It looks for
|
|
347
|
+
version-like keys in configuration/manifests while avoiding ordinary Ansible
|
|
348
|
+
`tags` and arbitrary source-code variables.
|
|
349
|
+
"""
|
|
350
|
+
|
|
351
|
+
candidate_patterns = [
|
|
352
|
+
"*.yml",
|
|
353
|
+
"*.yaml",
|
|
354
|
+
"*.json",
|
|
355
|
+
"*.toml",
|
|
356
|
+
"*.tf",
|
|
357
|
+
"*.tfvars",
|
|
358
|
+
"*.hcl",
|
|
359
|
+
"Dockerfile",
|
|
360
|
+
"Dockerfile.*",
|
|
361
|
+
"Containerfile",
|
|
362
|
+
"Containerfile.*",
|
|
363
|
+
"docker-compose*.yml",
|
|
364
|
+
"docker-compose*.yaml",
|
|
365
|
+
"Chart.yaml",
|
|
366
|
+
"values*.yaml",
|
|
367
|
+
"values*.yml",
|
|
368
|
+
"**/Chart.yaml",
|
|
369
|
+
"**/Dockerfile",
|
|
370
|
+
"**/Dockerfile.*",
|
|
371
|
+
"**/Containerfile",
|
|
372
|
+
"**/Containerfile.*",
|
|
373
|
+
"**/values*.yaml",
|
|
374
|
+
"**/values*.yml",
|
|
375
|
+
"**/defaults/**/*.yml",
|
|
376
|
+
"**/defaults/**/*.yaml",
|
|
377
|
+
"**/vars/**/*.yml",
|
|
378
|
+
"**/vars/**/*.yaml",
|
|
379
|
+
]
|
|
380
|
+
|
|
381
|
+
candidate_suffixes = (
|
|
382
|
+
".yml",
|
|
383
|
+
".yaml",
|
|
384
|
+
".json",
|
|
385
|
+
".toml",
|
|
386
|
+
".tf",
|
|
387
|
+
".tfvars",
|
|
388
|
+
".hcl",
|
|
389
|
+
)
|
|
390
|
+
candidate_basenames = {
|
|
391
|
+
"Dockerfile",
|
|
392
|
+
"Containerfile",
|
|
393
|
+
"Chart.yaml",
|
|
394
|
+
"docker-compose.yml",
|
|
395
|
+
"docker-compose.yaml",
|
|
396
|
+
}
|
|
397
|
+
|
|
398
|
+
candidate_files = {
|
|
399
|
+
file_path
|
|
400
|
+
for file_path in files
|
|
401
|
+
if not is_env_file(file_path)
|
|
402
|
+
and not _is_background_version_fixture_path(file_path)
|
|
403
|
+
and (
|
|
404
|
+
Path(file_path).name in candidate_basenames
|
|
405
|
+
or Path(file_path).name.lower().startswith(("dockerfile.", "containerfile."))
|
|
406
|
+
or file_path.endswith(candidate_suffixes)
|
|
407
|
+
)
|
|
408
|
+
}
|
|
409
|
+
|
|
410
|
+
if include_discovered:
|
|
411
|
+
auto_discovery_limit = max(80, limit * 4)
|
|
412
|
+
discovered_files: set[str] = set()
|
|
413
|
+
for pattern in candidate_patterns:
|
|
414
|
+
discovered_files.update(
|
|
415
|
+
rel_path
|
|
416
|
+
for rel_path in context_repo.rel_glob_files(
|
|
417
|
+
[pattern],
|
|
418
|
+
limit=auto_discovery_limit,
|
|
419
|
+
exclude_dirs=context_repo.DEFAULT_EXCLUDE_DIRS
|
|
420
|
+
| VERSION_PIN_BACKGROUND_SKIP_PARTS,
|
|
421
|
+
)
|
|
422
|
+
if not _is_background_version_fixture_path(rel_path)
|
|
423
|
+
)
|
|
424
|
+
for rel_path in sorted(discovered_files):
|
|
425
|
+
candidate_files.add(rel_path)
|
|
426
|
+
if len(candidate_files) >= auto_discovery_limit:
|
|
427
|
+
break
|
|
428
|
+
|
|
429
|
+
results: list[str] = []
|
|
430
|
+
seen: set[str] = set()
|
|
431
|
+
|
|
432
|
+
version_line = re.compile(
|
|
433
|
+
r"(?i)^\s*-?\s*[\"']?("
|
|
434
|
+
r"[A-Za-z0-9_.-]*(?:version|image|chart|app[_-]?version)[A-Za-z0-9_.-]*"
|
|
435
|
+
r"|tag"
|
|
436
|
+
r")[\"']?\s*[:=]\s*[\"']?([^\"'#\s]+)"
|
|
437
|
+
)
|
|
438
|
+
|
|
439
|
+
image_line = re.compile(r"(?i)^\s*-?\s*[\"']?image[\"']?\s*[:=]\s*[\"']?([^\"'#\s]+)")
|
|
440
|
+
|
|
441
|
+
changed_candidate_files = {file_path for file_path in files if file_path in candidate_files}
|
|
442
|
+
for rel_path in sorted(
|
|
443
|
+
candidate_files,
|
|
444
|
+
key=lambda item: (item not in changed_candidate_files, item),
|
|
445
|
+
):
|
|
446
|
+
if Path(rel_path).name in {"requirements.yml", "requirements.yaml"}:
|
|
447
|
+
continue
|
|
448
|
+
if is_env_file(rel_path):
|
|
449
|
+
continue
|
|
450
|
+
|
|
451
|
+
path = context_repo.ROOT / rel_path
|
|
452
|
+
safe_path = context_repo.resolve_repo_file(path)
|
|
453
|
+
if safe_path is None:
|
|
454
|
+
continue
|
|
455
|
+
|
|
456
|
+
try:
|
|
457
|
+
if safe_path.stat().st_size > DEFAULT_MAX_FILE_BYTES:
|
|
458
|
+
continue
|
|
459
|
+
except OSError:
|
|
460
|
+
continue
|
|
461
|
+
|
|
462
|
+
text = context_repo.read_text(safe_path, max_bytes=128_000)
|
|
463
|
+
is_containerfile = Path(rel_path).name.lower().startswith(("dockerfile", "containerfile"))
|
|
464
|
+
nested_pins, consumed_lines = _nested_image_pins(text)
|
|
465
|
+
scan_lines = [f"image: {image_ref}" for image_ref in nested_pins]
|
|
466
|
+
scan_lines.extend(
|
|
467
|
+
raw
|
|
468
|
+
for line_number, raw in enumerate(text.splitlines())
|
|
469
|
+
if line_number not in consumed_lines
|
|
470
|
+
)
|
|
471
|
+
for raw in scan_lines:
|
|
472
|
+
line = raw.strip()
|
|
473
|
+
if not line or line.startswith("#"):
|
|
474
|
+
continue
|
|
475
|
+
|
|
476
|
+
if is_containerfile:
|
|
477
|
+
from_match = re.match(r"(?i)^FROM\s+([^\s]+)", line)
|
|
478
|
+
if not from_match:
|
|
479
|
+
continue
|
|
480
|
+
from_parts = line.split()[1:]
|
|
481
|
+
while from_parts and from_parts[0].startswith("--"):
|
|
482
|
+
from_parts.pop(0)
|
|
483
|
+
if not from_parts:
|
|
484
|
+
continue
|
|
485
|
+
key = "image"
|
|
486
|
+
value = redact_url_userinfo(from_parts[0])
|
|
487
|
+
else:
|
|
488
|
+
# Skip ordinary Ansible tags, but allow singular `tag: 1.2.3`.
|
|
489
|
+
if re.match(r"(?i)^\s*-?\s*[\"']?tags[\"']?\s*:", line):
|
|
490
|
+
continue
|
|
491
|
+
|
|
492
|
+
version_match = version_line.search(line)
|
|
493
|
+
image_match = image_line.search(line)
|
|
494
|
+
|
|
495
|
+
if not version_match and not image_match:
|
|
496
|
+
continue
|
|
497
|
+
|
|
498
|
+
if not re.search(
|
|
499
|
+
r"(?i)(version|image|chart|app[_-]?version|tag)\s*[:=]",
|
|
500
|
+
line,
|
|
501
|
+
):
|
|
502
|
+
continue
|
|
503
|
+
|
|
504
|
+
if image_match:
|
|
505
|
+
key = "image"
|
|
506
|
+
value = redact_url_userinfo(image_match.group(1).strip())
|
|
507
|
+
elif version_match:
|
|
508
|
+
key = version_match.group(1).strip()
|
|
509
|
+
normalized_key = re.sub(r"(?<!^)(?=[A-Z])", "_", key).lower().replace("-", "_")
|
|
510
|
+
if normalized_key in {
|
|
511
|
+
"api_version",
|
|
512
|
+
"config_version",
|
|
513
|
+
"format_version",
|
|
514
|
+
"kind",
|
|
515
|
+
"kind_version",
|
|
516
|
+
"schema",
|
|
517
|
+
"schema_version",
|
|
518
|
+
"spec_version",
|
|
519
|
+
"$schema",
|
|
520
|
+
}:
|
|
521
|
+
continue
|
|
522
|
+
value = redact_url_userinfo(version_match.group(2).strip())
|
|
523
|
+
else:
|
|
524
|
+
continue
|
|
525
|
+
|
|
526
|
+
key_lower = key.lower()
|
|
527
|
+
is_image_key = key_lower == "image" or key_lower.endswith("image")
|
|
528
|
+
normalized_value = value.strip().lower()
|
|
529
|
+
image_ref = normalized_value.rsplit("/", 1)[-1]
|
|
530
|
+
image_name, _, _image_digest = image_ref.partition("@")
|
|
531
|
+
has_digest = "@" in normalized_value
|
|
532
|
+
image_tag = image_name.rsplit(":", 1)[-1] if ":" in image_name else ""
|
|
533
|
+
if (
|
|
534
|
+
not value
|
|
535
|
+
or normalized_value
|
|
536
|
+
in {
|
|
537
|
+
"true",
|
|
538
|
+
"false",
|
|
539
|
+
"yes",
|
|
540
|
+
"no",
|
|
541
|
+
"null",
|
|
542
|
+
"none",
|
|
543
|
+
"latest",
|
|
544
|
+
}
|
|
545
|
+
or (is_image_key and not has_digest and image_tag in {"", "latest"})
|
|
546
|
+
or any(marker in value for marker in ("{{", "{%", "${"))
|
|
547
|
+
or (not is_image_key and not re.search(r"\d", value))
|
|
548
|
+
):
|
|
549
|
+
continue
|
|
550
|
+
|
|
551
|
+
item = redact_sensitive(f"{rel_path}: {key}={value}")
|
|
552
|
+
if item in seen:
|
|
553
|
+
continue
|
|
554
|
+
|
|
555
|
+
seen.add(item)
|
|
556
|
+
results.append(item)
|
|
557
|
+
|
|
558
|
+
if len(results) >= limit:
|
|
559
|
+
return results[:limit]
|
|
560
|
+
|
|
561
|
+
return results[:limit]
|