dockerls 1.0.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (230) hide show
  1. dockerls/__init__.py +31 -0
  2. dockerls/application/__init__.py +0 -0
  3. dockerls/application/dto/__init__.py +3 -0
  4. dockerls/application/dto/analysis.py +257 -0
  5. dockerls/application/services/__init__.py +0 -0
  6. dockerls/application/services/alternatives_lookup.py +167 -0
  7. dockerls/application/services/composite_repository.py +106 -0
  8. dockerls/application/services/cross_validation.py +233 -0
  9. dockerls/application/services/ecosystems.py +350 -0
  10. dockerls/application/services/fallback_scanner.py +97 -0
  11. dockerls/application/services/hardening_analysis.py +174 -0
  12. dockerls/application/services/migration.py +297 -0
  13. dockerls/application/services/progress.py +56 -0
  14. dockerls/application/services/remediation.py +321 -0
  15. dockerls/application/services/scan_history_store.py +88 -0
  16. dockerls/application/services/scanner_factory.py +88 -0
  17. dockerls/application/services/source_registry.py +151 -0
  18. dockerls/application/services/tag_history_store.py +76 -0
  19. dockerls/application/services/teardown.py +50 -0
  20. dockerls/application/services/verdict.py +298 -0
  21. dockerls/application/services/version_discovery.py +108 -0
  22. dockerls/application/use_cases/__init__.py +0 -0
  23. dockerls/application/use_cases/analyze_dockerfile.py +103 -0
  24. dockerls/application/use_cases/analyze_image.py +173 -0
  25. dockerls/application/use_cases/build_image.py +1795 -0
  26. dockerls/application/use_cases/compare_images.py +91 -0
  27. dockerls/application/use_cases/fleet_scan.py +240 -0
  28. dockerls/application/use_cases/recommend_images.py +1078 -0
  29. dockerls/application/use_cases/registry_audit.py +133 -0
  30. dockerls/application/use_cases/search_images.py +23 -0
  31. dockerls/application/use_cases/upgrade_base.py +167 -0
  32. dockerls/cache/__init__.py +0 -0
  33. dockerls/cache/sqlite_cache.py +184 -0
  34. dockerls/cli/__init__.py +0 -0
  35. dockerls/cli/analysis_baseline.py +98 -0
  36. dockerls/cli/app.py +294 -0
  37. dockerls/cli/commands/__init__.py +0 -0
  38. dockerls/cli/commands/advisor.py +262 -0
  39. dockerls/cli/commands/alternatives.py +291 -0
  40. dockerls/cli/commands/analyze.py +429 -0
  41. dockerls/cli/commands/analyze_dockerfile.py +104 -0
  42. dockerls/cli/commands/base_cmd.py +244 -0
  43. dockerls/cli/commands/base_image.py +551 -0
  44. dockerls/cli/commands/build.py +1300 -0
  45. dockerls/cli/commands/cache_cmd.py +104 -0
  46. dockerls/cli/commands/compare.py +177 -0
  47. dockerls/cli/commands/controls.py +110 -0
  48. dockerls/cli/commands/doctor.py +566 -0
  49. dockerls/cli/commands/export.py +81 -0
  50. dockerls/cli/commands/fleet.py +159 -0
  51. dockerls/cli/commands/health.py +84 -0
  52. dockerls/cli/commands/login.py +53 -0
  53. dockerls/cli/commands/policy_cmd.py +111 -0
  54. dockerls/cli/commands/provenance_cmd.py +162 -0
  55. dockerls/cli/commands/recommend.py +761 -0
  56. dockerls/cli/commands/registry_audit_cmd.py +103 -0
  57. dockerls/cli/commands/sbom.py +144 -0
  58. dockerls/cli/commands/search.py +86 -0
  59. dockerls/cli/commands/verify.py +115 -0
  60. dockerls/cli/commands/version.py +12 -0
  61. dockerls/cli/commands/vex_cmd.py +117 -0
  62. dockerls/cli/dependencies.py +530 -0
  63. dockerls/cli/image_names.py +79 -0
  64. dockerls/cli/options.py +42 -0
  65. dockerls/cli/progress.py +145 -0
  66. dockerls/cli/publish_prompt.py +123 -0
  67. dockerls/cli/rendering.py +214 -0
  68. dockerls/cli/runtime.py +65 -0
  69. dockerls/cli/scan_failure.py +71 -0
  70. dockerls/cli/text.py +39 -0
  71. dockerls/cli/validators.py +35 -0
  72. dockerls/cli/vulnerability_view.py +154 -0
  73. dockerls/domain/__init__.py +0 -0
  74. dockerls/domain/entities/__init__.py +75 -0
  75. dockerls/domain/entities/declared_metadata.py +147 -0
  76. dockerls/domain/entities/dockerfile_analysis.py +318 -0
  77. dockerls/domain/entities/image.py +109 -0
  78. dockerls/domain/entities/image_facts.py +137 -0
  79. dockerls/domain/entities/recommendation.py +33 -0
  80. dockerls/domain/entities/scan_result.py +129 -0
  81. dockerls/domain/entities/vulnerability.py +232 -0
  82. dockerls/domain/interfaces/__init__.py +17 -0
  83. dockerls/domain/interfaces/cache_store.py +18 -0
  84. dockerls/domain/interfaces/dockerfile_validator.py +99 -0
  85. dockerls/domain/interfaces/eol_checker.py +11 -0
  86. dockerls/domain/interfaces/image_repository.py +15 -0
  87. dockerls/domain/interfaces/scanner.py +15 -0
  88. dockerls/domain/security_controls.py +362 -0
  89. dockerls/domain/value_objects/__init__.py +49 -0
  90. dockerls/domain/value_objects/attack_surface.py +198 -0
  91. dockerls/domain/value_objects/base_recipe.py +600 -0
  92. dockerls/domain/value_objects/base_upgrade.py +292 -0
  93. dockerls/domain/value_objects/build_labels.py +99 -0
  94. dockerls/domain/value_objects/build_policy.py +412 -0
  95. dockerls/domain/value_objects/confidence.py +156 -0
  96. dockerls/domain/value_objects/fleet.py +174 -0
  97. dockerls/domain/value_objects/gate.py +327 -0
  98. dockerls/domain/value_objects/hardening.py +303 -0
  99. dockerls/domain/value_objects/image_reference.py +90 -0
  100. dockerls/domain/value_objects/inheritance.py +352 -0
  101. dockerls/domain/value_objects/network_policy.py +280 -0
  102. dockerls/domain/value_objects/production_readiness.py +145 -0
  103. dockerls/domain/value_objects/provenance.py +211 -0
  104. dockerls/domain/value_objects/recipe_diff.py +188 -0
  105. dockerls/domain/value_objects/registry_audit.py +195 -0
  106. dockerls/domain/value_objects/registry_target.py +225 -0
  107. dockerls/domain/value_objects/remediation_score.py +62 -0
  108. dockerls/domain/value_objects/scan_history.py +183 -0
  109. dockerls/domain/value_objects/scan_plan.py +193 -0
  110. dockerls/domain/value_objects/scanner_db.py +143 -0
  111. dockerls/domain/value_objects/security_score.py +160 -0
  112. dockerls/domain/value_objects/security_tier.py +122 -0
  113. dockerls/domain/value_objects/tag_history.py +180 -0
  114. dockerls/domain/value_objects/tool_release.py +253 -0
  115. dockerls/domain/value_objects/tristate.py +47 -0
  116. dockerls/domain/value_objects/vex.py +249 -0
  117. dockerls/exit_codes.py +23 -0
  118. dockerls/exporters/__init__.py +0 -0
  119. dockerls/exporters/base.py +17 -0
  120. dockerls/exporters/csv_exporter.py +85 -0
  121. dockerls/exporters/factory.py +31 -0
  122. dockerls/exporters/html_exporter.py +105 -0
  123. dockerls/exporters/json_exporter.py +19 -0
  124. dockerls/exporters/markdown_exporter.py +84 -0
  125. dockerls/exporters/sarif_exporter.py +245 -0
  126. dockerls/infrastructure/__init__.py +0 -0
  127. dockerls/infrastructure/config/__init__.py +0 -0
  128. dockerls/infrastructure/config/policy_file.py +165 -0
  129. dockerls/infrastructure/config/settings.py +197 -0
  130. dockerls/infrastructure/database/__init__.py +0 -0
  131. dockerls/infrastructure/database/models.py +78 -0
  132. dockerls/infrastructure/dockerfile_validator.py +1899 -0
  133. dockerls/infrastructure/evidence.py +99 -0
  134. dockerls/infrastructure/hashing.py +165 -0
  135. dockerls/infrastructure/logging/__init__.py +0 -0
  136. dockerls/infrastructure/logging/setup.py +98 -0
  137. dockerls/infrastructure/network/__init__.py +0 -0
  138. dockerls/infrastructure/network/guarded_client.py +107 -0
  139. dockerls/infrastructure/network/host_guard.py +117 -0
  140. dockerls/infrastructure/redaction.py +135 -0
  141. dockerls/infrastructure/templates/hardening/alpine.dockerfile +44 -0
  142. dockerls/infrastructure/templates/hardening/debian.dockerfile +45 -0
  143. dockerls/infrastructure/templates/hardening/distroless.dockerfile +34 -0
  144. dockerls/infrastructure/templates/hardening/go-alpine.dockerfile +52 -0
  145. dockerls/infrastructure/templates/hardening/go-debian.dockerfile +54 -0
  146. dockerls/infrastructure/templates/hardening/go-distroless.dockerfile +43 -0
  147. dockerls/infrastructure/templates/hardening/go-scratch.dockerfile +48 -0
  148. dockerls/infrastructure/templates/hardening/go.dockerfile +50 -0
  149. dockerls/infrastructure/templates/hardening/gradle-alpine.dockerfile +51 -0
  150. dockerls/infrastructure/templates/hardening/gradle.dockerfile +52 -0
  151. dockerls/infrastructure/templates/hardening/java-alpine.dockerfile +50 -0
  152. dockerls/infrastructure/templates/hardening/java-debian.dockerfile +50 -0
  153. dockerls/infrastructure/templates/hardening/java-distroless.dockerfile +39 -0
  154. dockerls/infrastructure/templates/hardening/java-ubuntu.dockerfile +54 -0
  155. dockerls/infrastructure/templates/hardening/java.dockerfile +60 -0
  156. dockerls/infrastructure/templates/hardening/maven-alpine.dockerfile +56 -0
  157. dockerls/infrastructure/templates/hardening/maven.dockerfile +57 -0
  158. dockerls/infrastructure/templates/hardening/node-alpine.dockerfile +47 -0
  159. dockerls/infrastructure/templates/hardening/node-debian.dockerfile +54 -0
  160. dockerls/infrastructure/templates/hardening/node-distroless.dockerfile +46 -0
  161. dockerls/infrastructure/templates/hardening/node-ubuntu.dockerfile +63 -0
  162. dockerls/infrastructure/templates/hardening/node.dockerfile +61 -0
  163. dockerls/infrastructure/templates/hardening/php-alpine.dockerfile +45 -0
  164. dockerls/infrastructure/templates/hardening/php-debian.dockerfile +45 -0
  165. dockerls/infrastructure/templates/hardening/php-ubuntu.dockerfile +49 -0
  166. dockerls/infrastructure/templates/hardening/php.dockerfile +44 -0
  167. dockerls/infrastructure/templates/hardening/python-alpine.dockerfile +51 -0
  168. dockerls/infrastructure/templates/hardening/python-debian.dockerfile +54 -0
  169. dockerls/infrastructure/templates/hardening/python-distroless.dockerfile +51 -0
  170. dockerls/infrastructure/templates/hardening/python-ubuntu.dockerfile +60 -0
  171. dockerls/infrastructure/templates/hardening/python.dockerfile +58 -0
  172. dockerls/infrastructure/templates/hardening/ruby-alpine.dockerfile +48 -0
  173. dockerls/infrastructure/templates/hardening/ruby-debian.dockerfile +50 -0
  174. dockerls/infrastructure/templates/hardening/rust-alpine.dockerfile +50 -0
  175. dockerls/infrastructure/templates/hardening/rust-debian.dockerfile +48 -0
  176. dockerls/infrastructure/templates/hardening/rust-scratch.dockerfile +44 -0
  177. dockerls/infrastructure/templates/hardening/rust.dockerfile +54 -0
  178. dockerls/infrastructure/templates/hardening/ubuntu.dockerfile +49 -0
  179. dockerls/infrastructure/toolchain/__init__.py +0 -0
  180. dockerls/infrastructure/toolchain/db_metadata.py +115 -0
  181. dockerls/infrastructure/toolchain/installer.py +435 -0
  182. dockerls/integrations/__init__.py +0 -0
  183. dockerls/integrations/dhi/__init__.py +0 -0
  184. dockerls/integrations/dhi/catalog.py +457 -0
  185. dockerls/integrations/dhi/definition.py +151 -0
  186. dockerls/integrations/dhi/repository.py +238 -0
  187. dockerls/integrations/dockerhub/__init__.py +0 -0
  188. dockerls/integrations/dockerhub/client.py +318 -0
  189. dockerls/integrations/dockerhub/urls.py +75 -0
  190. dockerls/integrations/endoflife/__init__.py +0 -0
  191. dockerls/integrations/endoflife/checker.py +216 -0
  192. dockerls/integrations/engine/__init__.py +0 -0
  193. dockerls/integrations/engine/batch.py +197 -0
  194. dockerls/integrations/engine/client.py +330 -0
  195. dockerls/integrations/engine/locator.py +96 -0
  196. dockerls/integrations/exploitdb/__init__.py +0 -0
  197. dockerls/integrations/exploitdb/client.py +271 -0
  198. dockerls/integrations/grype/__init__.py +0 -0
  199. dockerls/integrations/grype/scanner.py +336 -0
  200. dockerls/integrations/registry/__init__.py +0 -0
  201. dockerls/integrations/registry/hardened.py +284 -0
  202. dockerls/integrations/registry/inspector.py +420 -0
  203. dockerls/integrations/registry/oci.py +259 -0
  204. dockerls/integrations/registry/private.py +79 -0
  205. dockerls/integrations/registry/urls.py +36 -0
  206. dockerls/integrations/scan_errors.py +67 -0
  207. dockerls/integrations/scan_target.py +57 -0
  208. dockerls/integrations/signing/__init__.py +0 -0
  209. dockerls/integrations/signing/cosign.py +484 -0
  210. dockerls/integrations/threat_intel/__init__.py +0 -0
  211. dockerls/integrations/threat_intel/client.py +290 -0
  212. dockerls/integrations/trivy/__init__.py +0 -0
  213. dockerls/integrations/trivy/cache_pool.py +176 -0
  214. dockerls/integrations/trivy/scanner.py +447 -0
  215. dockerls/utils/__init__.py +0 -0
  216. dockerls/utils/auth.py +108 -0
  217. dockerls/utils/executables.py +39 -0
  218. dockerls/utils/ignore_file.py +130 -0
  219. dockerls/utils/rate_limit.py +130 -0
  220. dockerls/utils/resources.py +183 -0
  221. dockerls/utils/retry.py +31 -0
  222. dockerls/utils/safe_yaml.py +166 -0
  223. dockerls/utils/subprocess_runner.py +216 -0
  224. dockerls/utils/validation.py +72 -0
  225. dockerls-1.0.0.dist-info/METADATA +563 -0
  226. dockerls-1.0.0.dist-info/RECORD +230 -0
  227. dockerls-1.0.0.dist-info/WHEEL +5 -0
  228. dockerls-1.0.0.dist-info/entry_points.txt +2 -0
  229. dockerls-1.0.0.dist-info/licenses/LICENSE +21 -0
  230. dockerls-1.0.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,1078 @@
1
+ from __future__ import annotations
2
+
3
+ import asyncio
4
+ import hashlib
5
+ import re
6
+ from datetime import UTC, datetime
7
+ from typing import TYPE_CHECKING, Any
8
+
9
+ from loguru import logger
10
+ from pydantic import ValidationError
11
+
12
+ from dockerls import __version__
13
+ from dockerls.application.dto.analysis import (
14
+ AnalysisResult,
15
+ BaselineCriteria,
16
+ ImageAnalysis,
17
+ RunMetrics,
18
+ UnverifiedImage,
19
+ )
20
+ from dockerls.application.services.progress import NullObserver
21
+ from dockerls.application.services.teardown import close_quietly, sources_of
22
+ from dockerls.application.services.verdict import (
23
+ apply_facts,
24
+ cross_validation_agreed,
25
+ finalize_verdict,
26
+ rank,
27
+ )
28
+ from dockerls.domain.entities.recommendation import (
29
+ ActionType,
30
+ Recommendation,
31
+ RemediationStep,
32
+ )
33
+ from dockerls.domain.value_objects.remediation_score import RemediationScore
34
+ from dockerls.domain.value_objects.scan_plan import DEFAULT_SCAN_BUDGET, plan_scans
35
+ from dockerls.domain.value_objects.security_score import SecurityScore
36
+ from dockerls.domain.value_objects.security_tier import SecurityTier
37
+ from dockerls.domain.value_objects.tristate import Tristate
38
+ from dockerls.integrations.registry.urls import source_url
39
+ from dockerls.utils.ignore_file import active_ignored_cve_ids, load_ignore_rules
40
+ from dockerls.utils.validation import validate_threshold, validate_workers
41
+
42
+ if TYPE_CHECKING:
43
+ from collections.abc import Callable
44
+ from pathlib import Path
45
+
46
+ from dockerls.application.services.cross_validation import CrossValidator
47
+ from dockerls.application.services.hardening_analysis import HardeningAnalyzer
48
+ from dockerls.application.services.progress import ScanObserver
49
+ from dockerls.domain.entities.image import DockerImage
50
+ from dockerls.domain.interfaces.cache_store import CacheStoreInterface
51
+ from dockerls.domain.interfaces.eol_checker import EOLCheckerInterface
52
+ from dockerls.domain.interfaces.image_repository import ImageRepositoryInterface
53
+ from dockerls.domain.interfaces.scanner import ScannerInterface
54
+ from dockerls.infrastructure.evidence import EvidenceStore
55
+ from dockerls.integrations.exploitdb.client import ExploitDBClient, ExploitEntry
56
+ from dockerls.integrations.threat_intel.client import ThreatIntelClient
57
+
58
+ # How many ranked candidates are surfaced to the user.
59
+ TOP_N = 5
60
+
61
+
62
+ class UnverifiedRecommendationError(RuntimeError):
63
+ """Raised when an image without a proven successful scan would have been
64
+ presented as a recommendation. This is a programming error, not a user
65
+ error: it means a code path bypassed the verification gate."""
66
+
67
+
68
+ class RecommendImagesUseCase:
69
+ def __init__(
70
+ self,
71
+ repository: ImageRepositoryInterface,
72
+ scanner: ScannerInterface,
73
+ eol_checker: EOLCheckerInterface,
74
+ cache: CacheStoreInterface | None = None,
75
+ max_critical: int = 0,
76
+ max_high: int = 0,
77
+ max_medium: int = 5,
78
+ workers: int = 10,
79
+ ignore_path: Path | None = None,
80
+ threat_intel: ThreatIntelClient | None = None,
81
+ observer: ScanObserver | None = None,
82
+ cross_validator: CrossValidator | None = None,
83
+ evidence: EvidenceStore | None = None,
84
+ verify_hub_tags: bool = True,
85
+ log_file: Path | None = None,
86
+ cache_ttl_seconds: int = 86400,
87
+ hardening: HardeningAnalyzer | None = None,
88
+ resolve_digests: bool = True,
89
+ exploitdb: ExploitDBClient | None = None,
90
+ scan_budget: int = DEFAULT_SCAN_BUDGET,
91
+ ):
92
+ # Guarded at construction rather than only at the CLI boundary: the
93
+ # use case is the last place that can refuse a value which would
94
+ # otherwise deadlock the scan loop (`workers=0` blocks forever on a
95
+ # semaphore) or silently invert the baseline (a negative threshold
96
+ # can never be met). Any caller -- CLI, tests, a future API -- gets
97
+ # the same refusal.
98
+ self._repository = repository
99
+ self._scanner = scanner
100
+ self._eol_checker = eol_checker
101
+ # Quantas tags este run pode medir. 0 mede todas, que é o
102
+ # comportamento anterior e segue disponível por configuração.
103
+ self._scan_budget = max(0, scan_budget)
104
+ self._cache = cache
105
+ self._max_critical = validate_threshold(max_critical, "max_critical")
106
+ self._max_high = validate_threshold(max_high, "max_high")
107
+ self._max_medium = validate_threshold(max_medium, "max_medium")
108
+ self._workers = validate_workers(workers)
109
+ self._ignored_cves = active_ignored_cve_ids(load_ignore_rules(ignore_path))
110
+ self._threat_intel = threat_intel
111
+ self._exploitdb = exploitdb
112
+ self._observer: ScanObserver = observer or NullObserver()
113
+ self._cross_validator = cross_validator
114
+ self._evidence = evidence
115
+ self._verify_hub_tags = verify_hub_tags
116
+ self._log_file = log_file
117
+ self._cache_ttl_seconds = cache_ttl_seconds
118
+ self._hardening = hardening
119
+ self._resolve_digests = resolve_digests
120
+ # Filled in once the scanner has been asked who it is. Until then the
121
+ # fingerprint deliberately carries "unknown-scanner" rather than
122
+ # nothing: a run that could not identify its scanner must not share
123
+ # a cache namespace with one that could.
124
+ self._scanner_identity = "unknown-scanner"
125
+ self._analysis_fingerprint = self._compute_analysis_fingerprint()
126
+ self._metrics = RunMetrics()
127
+
128
+ def _compute_analysis_fingerprint(self) -> str:
129
+ """Identifica as entradas, fora a própria imagem, que mudam o
130
+ `ImageAnalysis` guardado em cache.
131
+
132
+ As regras de ignore e o enriquecimento de threat intel são aplicados
133
+ *antes* de cachear, mas a chave era só a referência da imagem. Um CVE
134
+ que deixava de ser ignorado -- porque a regra foi removida, ou porque
135
+ o `expires` dela venceu -- continuava suprimido até o TTL expirar
136
+ (24h no padrão). O arquivo de ignore promete que uma isenção vencida
137
+ deixa de valer; o cache desfazia essa promessa em silêncio.
138
+ """
139
+ material = "|".join(
140
+ [
141
+ ",".join(sorted(self._ignored_cves)),
142
+ "threat-intel" if self._threat_intel is not None else "no-threat-intel",
143
+ # Uma análise enriquecida com Exploit-DB carrega campos que a
144
+ # anterior não tinha; servir a antiga esconderia a coluna.
145
+ "exploitdb" if self._exploitdb is not None else "no-exploitdb",
146
+ # Which tool, at which version, produced the cached numbers.
147
+ # Without this the cache served a Trivy result to a run using
148
+ # Grype, and kept serving results from before a scanner
149
+ # upgrade -- a stale measurement presented as a current one,
150
+ # which is the same substitution this project refuses
151
+ # everywhere else, just slower.
152
+ self._scanner_identity,
153
+ # And which version of *this* tool produced them. A cached
154
+ # `ImageAnalysis` carries the score, the tier and the
155
+ # readiness verdict, all of which are computed by policy
156
+ # that lives here -- so a release that changes a penalty
157
+ # weight, a tier threshold or a blocking rule would keep
158
+ # serving verdicts decided under the previous rules until
159
+ # the TTL ran out. `CACHE_SCHEMA_VERSION` does not cover
160
+ # this: the payload's *shape* is unchanged, so validation
161
+ # accepts it and only the meaning has moved.
162
+ __version__,
163
+ ]
164
+ )
165
+ return hashlib.sha256(material.encode()).hexdigest()[:12]
166
+
167
+ async def _identify_scanner(self) -> None:
168
+ """Ask the scanner who it is, and re-key the cache accordingly.
169
+
170
+ Done once per run, before anything is read from or written to the
171
+ cache. A scanner that cannot answer leaves the identity as
172
+ "unknown-scanner", which is its own namespace: results whose
173
+ provenance is unknown are reused only by other runs in the same
174
+ situation.
175
+ """
176
+ version = getattr(self._scanner, "version", None)
177
+ name = type(self._scanner).__name__
178
+ if callable(version):
179
+ try:
180
+ reported = await version()
181
+ except Exception as e: # pragma: no cover - identity is best-effort
182
+ logger.debug(f"Could not identify {name}: {e}")
183
+ reported = ""
184
+ if isinstance(reported, str) and reported:
185
+ self._scanner_identity = reported
186
+ self._metrics.scanner_identity = self._scanner_identity
187
+ self._analysis_fingerprint = self._compute_analysis_fingerprint()
188
+ logger.info(f"Scanner identity for this run: {self._scanner_identity}")
189
+
190
+ def _cache_key(self, image: DockerImage) -> str:
191
+ """Chaveia a análise pelo **digest** do manifesto, não pela tag.
192
+
193
+ Tags são mutáveis: `node:22-alpine` de hoje não é a mesma imagem de
194
+ ontem. Uma entrada chaveada por tag continuava servindo o resultado
195
+ antigo por até 24h depois de um rebuild upstream -- ou seja, servia
196
+ um veredito de segurança sobre uma imagem que não existe mais. O
197
+ digest identifica bytes, então uma entrada só casa com a imagem que
198
+ de fato produziu aquele scan.
199
+
200
+ Sem digest (registries que listam só nomes de tag) a referência
201
+ continua sendo a chave, que é o melhor disponível.
202
+ """
203
+ identity = image.digest or image.full_reference
204
+ return f"analysis:{self._analysis_fingerprint}:{identity}"
205
+
206
+ async def execute(self, image_name: str, limit: int = 100) -> AnalysisResult:
207
+ try:
208
+ return await self._execute(image_name, limit)
209
+ finally:
210
+ await self._close_scanners()
211
+ await self._close_repositories()
212
+
213
+ @staticmethod
214
+ def _fallback_pool(analyses: list[ImageAnalysis]) -> list[ImageAnalysis]:
215
+ """Ranking apresentado quando nada atinge o baseline.
216
+
217
+ O filtro aqui era `critical_count == 0 and not is_eol` -- os mesmos
218
+ critérios duros que o baseline já havia acabado de rejeitar. Quando
219
+ toda tag candidata carregava um CRITICAL (o caso comum no Docker Hub),
220
+ as "alternativas" saíam vazias também e o usuário recebia
221
+ "No suitable images found" depois de esperar por uma centena de scans.
222
+ Isso descarta a informação mais útil que a execução produziu: qual das
223
+ imagens ruins é a menos ruim.
224
+
225
+ Agora nada é descartado. As candidatas são apenas *ordenadas* pelo que
226
+ importa nessa situação -- menos CRITICAL, menos HIGH, mais fácil de
227
+ remediar, menos MEDIUM, maior score -- e a camada de apresentação diz
228
+ com todas as letras que estão abaixo do alvo.
229
+ """
230
+ return sorted(
231
+ analyses,
232
+ key=lambda a: (
233
+ a.is_eol,
234
+ a.scan.critical_count,
235
+ a.scan.high_count,
236
+ -a.remediation_score,
237
+ a.scan.medium_count,
238
+ -a.security_score,
239
+ ),
240
+ )
241
+
242
+ def _baseline(self) -> BaselineCriteria:
243
+ return BaselineCriteria(
244
+ max_critical=self._max_critical,
245
+ max_high=self._max_high,
246
+ max_medium=self._max_medium,
247
+ )
248
+
249
+ async def _execute(self, image_name: str, limit: int = 100) -> AnalysisResult:
250
+ await self._identify_scanner()
251
+ self._observer.phase("Preparing vulnerability database")
252
+ setup_errors: list[str] = []
253
+ refresh_db = getattr(self._scanner, "refresh_db", None)
254
+ if callable(refresh_db) and not await refresh_db():
255
+ # O retorno era descartado. Sem a DB pronta, cada worker sai
256
+ # baixando a própria cópia em paralelo e o run inteiro reprova com
257
+ # `init error: DB error` -- uma vez por tag. Registrar a causa raiz
258
+ # uma única vez é o que transforma 93 linhas iguais num diagnóstico.
259
+ logger.warning("Vulnerability database is not ready; scans are likely to fail")
260
+ setup_errors.append(
261
+ "Vulnerability database could not be prepared -- scan failures below are "
262
+ "most likely a consequence of this, not of the images themselves"
263
+ )
264
+
265
+ self._observer.phase(f"Fetching tags for {image_name}")
266
+ tags = await self._repository.search_tags(image_name, limit=limit)
267
+ if not tags:
268
+ return AnalysisResult(
269
+ query=image_name,
270
+ total_tags_scanned=0,
271
+ baseline_met=False,
272
+ errors=["No tags found for image"],
273
+ log_file=str(self._log_file or ""),
274
+ baseline=self._baseline(),
275
+ )
276
+
277
+ self._observer.phase_result(
278
+ "Discovered tags",
279
+ [
280
+ ("found", str(len(tags))),
281
+ ("sources", ", ".join(_sources_of(tags)) or "none"),
282
+ ],
283
+ )
284
+
285
+ # Quem medir. Medir as 100 tags para mostrar cinco custa dois a
286
+ # quatro minutos, e 95 desses scans existem só para serem
287
+ # descartados no ranqueamento. O plano corta isso -- e declara o
288
+ # que cortou: uma tag adiada não é uma tag pior, é uma tag *não
289
+ # medida*, e ela aparece no resultado com o motivo.
290
+ plan = plan_scans(tags, self._scan_budget)
291
+ if plan.deferred:
292
+ self._observer.phase_result(
293
+ "Selected for measurement",
294
+ [
295
+ ("measuring", str(len(plan.selected))),
296
+ ("deferred", str(plan.deferred_count)),
297
+ ("budget", str(plan.budget)),
298
+ ],
299
+ )
300
+ tags = plan.selected
301
+
302
+ await self._pin_digests(tags)
303
+
304
+ analyses, unverified, errors = await self._scan_all(tags)
305
+ errors = [*setup_errors, *errors]
306
+ analyses.sort(key=lambda a: a.security_score, reverse=True)
307
+
308
+ # Reported after the fact rather than before: the cache-hit and
309
+ # scan counts are only known once the pass is done, and stating them
310
+ # up front would mean guessing at them.
311
+ self._observer.phase_result(
312
+ "Scanned candidates",
313
+ [
314
+ ("unique digests", str(self._metrics.unique_digests)),
315
+ ("duplicates collapsed", str(self._metrics.duplicates_collapsed)),
316
+ ("cache hits", str(self._metrics.cache_hits)),
317
+ ("scans performed", str(self._metrics.scans_performed)),
318
+ ("digests pinned", str(self._metrics.digests_resolved)),
319
+ ("unverified", str(len(unverified))),
320
+ ],
321
+ )
322
+
323
+ baseline_images = [
324
+ a
325
+ for a in analyses
326
+ if a.scan.critical_count <= self._max_critical
327
+ and a.scan.high_count <= self._max_high
328
+ and a.scan.medium_count <= self._max_medium
329
+ and not a.is_eol
330
+ ]
331
+
332
+ if baseline_images:
333
+ baseline_met = True
334
+ pool = baseline_images
335
+ else:
336
+ baseline_met = False
337
+ pool = self._fallback_pool(analyses)
338
+
339
+ selected = await self._finalize(pool, unverified)
340
+
341
+ result = AnalysisResult(
342
+ query=image_name,
343
+ total_tags_scanned=len(tags),
344
+ total_tags_analyzed=len(analyses),
345
+ baseline_met=baseline_met and bool(selected),
346
+ recommendations=selected if baseline_met else [],
347
+ alternatives=[] if baseline_met else selected,
348
+ errors=errors,
349
+ unverified=unverified,
350
+ log_file=str(self._log_file or ""),
351
+ baseline=self._baseline(),
352
+ sources_searched=_sources_of(tags),
353
+ metrics=self._metrics,
354
+ deferred=plan.deferred,
355
+ tags_discovered=plan.discovered,
356
+ )
357
+ result.evidence_manifest = await self._write_manifest(image_name, selected)
358
+ return result
359
+
360
+ async def _pin_digests(self, tags: list[DockerImage]) -> None:
361
+ """Resolve every candidate that arrived without a digest.
362
+
363
+ Deduplication keys on the digest, and a candidate with none is
364
+ keyed by its reference instead -- so the same manifest published
365
+ under `22`, `22-bookworm` and a hardened catalogue's alias is
366
+ scanned three times. One HEAD per unresolved tag replaces those
367
+ extra scans, and each scan costs orders of magnitude more than the
368
+ request that avoids it.
369
+
370
+ Failure is free: a registry that will not answer leaves the
371
+ candidate exactly as it arrived.
372
+ """
373
+ if self._hardening is None or not self._resolve_digests:
374
+ return
375
+ unresolved = [tag for tag in tags if not tag.digest_known]
376
+ if not unresolved:
377
+ return
378
+
379
+ self._observer.phase(f"Resolving digests for {len(unresolved)} tag(s)")
380
+ semaphore = asyncio.Semaphore(self._workers)
381
+
382
+ async def pin(image: DockerImage) -> None:
383
+ async with semaphore:
384
+ digest = await self._hardening.resolve_digest(image) if self._hardening else ""
385
+ if digest:
386
+ # Mutated in place because `tags` is the list the rest of
387
+ # the pipeline holds; replacing entries would leave the
388
+ # scan loop keyed on the unpinned copies.
389
+ image.digest = digest
390
+ self._metrics.digests_resolved += 1
391
+
392
+ await asyncio.gather(*[pin(image) for image in unresolved])
393
+ logger.info(
394
+ f"Resolved {self._metrics.digests_resolved}/{len(unresolved)} previously "
395
+ "unpinned tags to manifest digests"
396
+ )
397
+
398
+ async def _scan_all(
399
+ self, tags: list[DockerImage]
400
+ ) -> tuple[list[ImageAnalysis], list[UnverifiedImage], list[str]]:
401
+ semaphore = asyncio.Semaphore(self._workers)
402
+ errors: list[str] = []
403
+ unverified: list[UnverifiedImage] = []
404
+
405
+ # P0-3: dedupe scans by digest so tags sharing the same manifest
406
+ # digest are only scanned once and share the result.
407
+ scan_locks: dict[str, asyncio.Lock] = {}
408
+ scan_cache: dict[str, Any] = {}
409
+
410
+ def _dedup_key(image: DockerImage) -> str:
411
+ return image.digest or image.full_reference
412
+
413
+ self._metrics.tags_discovered = len(tags)
414
+ self._metrics.unique_digests = len({_dedup_key(tag) for tag in tags})
415
+ self._metrics.workers = self._workers
416
+
417
+ # Caminho em lote: quando a engine Go está disponível, todos os
418
+ # scans que faltam saem numa travessia de processo só, e
419
+ # `scan_cache` chega aqui já preenchido. `get_scan` abaixo então
420
+ # não dispara scan nenhum -- ele encontra tudo pela chave.
421
+ #
422
+ # `prefetched` são as análises que já estavam no cache do disco: o
423
+ # lote tem de perguntar por elas *antes* de medir, ou um run
424
+ # inteiramente cacheado voltaria a escanear cem imagens.
425
+ prefetched, batched = await self._prescan(tags, scan_cache, _dedup_key)
426
+
427
+ async def get_scan(image: DockerImage) -> Any:
428
+ key = _dedup_key(image)
429
+ lock = scan_locks.setdefault(key, asyncio.Lock())
430
+ async with lock:
431
+ if key in scan_cache:
432
+ return scan_cache[key]
433
+ async with semaphore:
434
+ scan = await self._scanner.scan(image.full_reference)
435
+ # Counted here rather than at the call site so a tag served
436
+ # from a sibling's digest is never counted as a scan.
437
+ self._metrics.scans_performed += 1
438
+ scan_cache[key] = scan
439
+ return scan
440
+
441
+ def _skip(image: DockerImage, status: str, reason: str, kind: str = "UNKNOWN") -> None:
442
+ logger.warning(f"Skipping {image.full_reference}: {status}/{kind} ({reason})")
443
+ unverified.append(
444
+ UnverifiedImage(
445
+ image_reference=image.full_reference,
446
+ status=status,
447
+ reason=reason or "no details",
448
+ kind=kind,
449
+ )
450
+ )
451
+ errors.append(f"{image.full_reference}: {status}/{kind} ({reason or 'no details'})")
452
+
453
+ async def analyze_tag(image: DockerImage) -> ImageAnalysis | None:
454
+ self._observer.scanning(image.full_reference)
455
+ analysis: ImageAnalysis | None = None
456
+ try:
457
+ # Já perguntado pelo lote; perguntar de novo seria uma
458
+ # segunda leitura do cache por imagem.
459
+ cached = (
460
+ prefetched.get(image.full_reference)
461
+ if batched
462
+ else await self._get_cached(image)
463
+ )
464
+ if cached:
465
+ self._metrics.cache_hits += 1
466
+ analysis = cached
467
+ return cached
468
+
469
+ scan = await get_scan(image)
470
+ # Single verification gate: anything short of a completed,
471
+ # parsed scan is reported as unverified and is never scored.
472
+ if not scan.is_verified:
473
+ _skip(image, scan.status.value, scan.error_message, scan.error_kind.value)
474
+ return None
475
+
476
+ if self._ignored_cves:
477
+ scan = _apply_ignore_rules(scan, self._ignored_cves)
478
+ if self._threat_intel is not None:
479
+ scan = await _enrich_with_threat_intel(
480
+ scan, self._threat_intel, self._exploitdb
481
+ )
482
+
483
+ product, version = _extract_product_version(image)
484
+ eol_status = await _eol_status(self._eol_checker, product, version)
485
+ is_eol = eol_status.is_true
486
+ is_lts = await self._eol_checker.is_lts(product, version)
487
+
488
+ score = SecurityScore(image, scan, is_eol=is_eol, is_lts=is_lts)
489
+ tier = SecurityTier(scan, score.value, is_eol=is_eol)
490
+ rem_score = RemediationScore(scan)
491
+
492
+ analysis = ImageAnalysis(
493
+ image=image,
494
+ scan=scan,
495
+ security_score=score.value,
496
+ tier=tier.tier.value,
497
+ remediation_score=rem_score.value,
498
+ is_eol=is_eol,
499
+ eol_status=eol_status,
500
+ is_lts=is_lts,
501
+ evidence_paths=(
502
+ {scan.scanner: scan.evidence_path} if scan.evidence_path else {}
503
+ ),
504
+ )
505
+
506
+ await self._set_cached(image, analysis)
507
+ return analysis
508
+ except Exception as e:
509
+ logger.warning(f"Failed to analyze {image.full_reference}: {e}")
510
+ _skip(image, "ERROR", str(e))
511
+ return None
512
+ finally:
513
+ self._observer.finished(image.full_reference, analysis is not None)
514
+
515
+ self._observer.start(len(tags))
516
+ results = await asyncio.gather(*[analyze_tag(tag) for tag in tags])
517
+ return [r for r in results if r is not None], unverified, errors
518
+
519
+ async def _prescan(
520
+ self,
521
+ tags: list[DockerImage],
522
+ scan_cache: dict[str, Any],
523
+ dedup_key: Callable[[DockerImage], str],
524
+ ) -> tuple[dict[str, ImageAnalysis], bool]:
525
+ """Mede o lote inteiro de uma vez, quando a engine Go existe.
526
+
527
+ O que muda em relação ao caminho de sempre não é o scan: o Trivy
528
+ continua sendo o Trivy e continua custando o que custa. O que muda
529
+ é o entorno -- criar e colher N processos, revezar o diretório de
530
+ cache, coordenar o dedup por digest -- que sai de N travessias
531
+ Python<->processo para uma.
532
+
533
+ Devolve `(análises já em cache, se o lote aconteceu)`. Quando o
534
+ lote não acontece -- engine ausente, versão incompatível, qualquer
535
+ falha -- devolve `({}, False)` e o pipeline segue exatamente como
536
+ antes. A engine é uma otimização, e uma otimização que pode
537
+ derrubar o comando não vale o ganho.
538
+ """
539
+ batch = getattr(self._scanner, "batch", None)
540
+ if batch is None:
541
+ return {}, False
542
+
543
+ # O cache do disco vem primeiro: um run inteiramente cacheado tem
544
+ # de continuar fazendo zero scans, e medir para depois descobrir
545
+ # que a resposta já estava guardada seria o pior dos dois mundos.
546
+ cached_analyses = await asyncio.gather(*[self._get_cached(tag) for tag in tags])
547
+ prefetched = {
548
+ tag.full_reference: analysis
549
+ for tag, analysis in zip(tags, cached_analyses, strict=True)
550
+ if analysis is not None
551
+ }
552
+
553
+ pending: list[tuple[str, str]] = []
554
+ seen: set[str] = set()
555
+ for tag in tags:
556
+ if tag.full_reference in prefetched:
557
+ continue
558
+ key = dedup_key(tag)
559
+ if key in seen:
560
+ continue
561
+ seen.add(key)
562
+ pending.append((tag.full_reference, key))
563
+
564
+ if not pending:
565
+ return prefetched, True
566
+
567
+ outcome = await batch.scan_batch(pending)
568
+ if outcome is None:
569
+ # A engine recusou o lote. As análises já lidas do cache não se
570
+ # perdem, mas o caminho individual precisa reler -- devolver
571
+ # `batched=True` aqui faria toda imagem não cacheada ser
572
+ # tratada como sem cache *e* sem scan.
573
+ return {}, False
574
+
575
+ for (_, key), result in zip(pending, outcome.results, strict=True):
576
+ scan_cache[key] = result
577
+ self._metrics.scans_performed += outcome.scans_performed
578
+ logger.info(
579
+ f"Go engine measured {len(pending)} targets in {outcome.wall_seconds:.1f}s "
580
+ f"({outcome.scans_performed} scans, {outcome.duplicates_collapsed} collapsed)"
581
+ )
582
+ return prefetched, True
583
+
584
+ async def _finalize(
585
+ self, pool: list[ImageAnalysis], unverified: list[UnverifiedImage]
586
+ ) -> list[ImageAnalysis]:
587
+ """Cross-validate, confirm Docker Hub tags, and enforce the
588
+ no-scan-no-recommendation invariant on the final candidate list."""
589
+ # Verify a wider slice than TOP_N so candidates dropped for a
590
+ # missing Hub tag can be backfilled from the next best ones.
591
+ candidates = pool[: TOP_N * 2]
592
+
593
+ # A verificação de tag vem primeiro, e a cross-validation só depois,
594
+ # sobre quem sobreviveu. Na ordem inversa, um candidato promovido
595
+ # para o top N no lugar de um descartado entrava na tabela sem nunca
596
+ # ter passado pelo segundo scanner -- ou seja, com a pontuação
597
+ # apresentada sem contestação justamente por não ter sido checada.
598
+ # De quebra, deixa de gastar um scan secundário em quem vai cair.
599
+ if self._verify_hub_tags and candidates:
600
+ self._observer.phase("Verifying tags in their source registries")
601
+ await self._verify_tags(candidates, unverified)
602
+ candidates = [c for c in candidates if c.hub_tag_verified is not False]
603
+
604
+ selected = candidates[:TOP_N]
605
+
606
+ if self._cross_validator is not None and self._cross_validator.enabled and selected:
607
+ self._observer.phase(f"Cross-validating top {len(selected)} candidates")
608
+ self._metrics.cross_validations = len(selected)
609
+ await self._cross_validator.validate(selected)
610
+
611
+ # Hardening evidence is gathered for the finalists only. Inspecting
612
+ # every discovered tag would cost two registry round-trips each --
613
+ # hundreds of requests to inform a decision between five images --
614
+ # and the candidates that reach this point are exactly the ones the
615
+ # decision is actually between.
616
+ await self._inspect(selected)
617
+
618
+ for analysis in selected:
619
+ finalize_verdict(analysis, cross_validated=cross_validation_agreed(analysis))
620
+
621
+ # The final ordering is the multi-source one: confidence first, then
622
+ # the measured vulnerability position, then hardening and surface.
623
+ # Up to here the pool was ordered by security score alone, which
624
+ # could not see the evidence that has just been gathered.
625
+ selected = rank(selected)
626
+
627
+ _assert_verified(selected)
628
+ for analysis in selected:
629
+ analysis.recommendation = build_recommendation(analysis)
630
+ return selected
631
+
632
+ async def _inspect(self, selected: list[ImageAnalysis]) -> None:
633
+ """Attach registry/catalogue/scanner evidence to each finalist."""
634
+ if self._hardening is None or not selected:
635
+ return
636
+ self._observer.phase(f"Inspecting {len(selected)} candidate image(s)")
637
+
638
+ async def inspect(analysis: ImageAnalysis) -> None:
639
+ if self._hardening is None:
640
+ return
641
+ digest, facts = await self._hardening.analyze(analysis.image, analysis.scan)
642
+ if digest and not analysis.image.digest_known:
643
+ analysis.image.digest = digest
644
+ apply_facts(analysis, facts)
645
+ if facts.config_verified:
646
+ self._metrics.images_inspected += 1
647
+
648
+ await asyncio.gather(*[inspect(a) for a in selected])
649
+
650
+ async def _verify_tags(
651
+ self, candidates: list[ImageAnalysis], unverified: list[UnverifiedImage]
652
+ ) -> None:
653
+ """Confirm each candidate tag against the registry that owns it.
654
+
655
+ Docker Hub tags are checked through the Hub API; hardened-source
656
+ tags are checked against that source's own listing. Either way the
657
+ answer comes from the registry, never from a constructed string.
658
+ """
659
+ checker = getattr(self._repository, "tag_exists", None)
660
+
661
+ async def check(analysis: ImageAnalysis) -> None:
662
+ analysis.hub_url = source_url(analysis.image.name, analysis.image.tag)
663
+ if not callable(checker):
664
+ return
665
+ exists = await checker(analysis.image.name, analysis.image.tag)
666
+ analysis.hub_tag_verified = exists
667
+ if exists is False:
668
+ logger.warning(
669
+ f"Dropping {analysis.image.full_reference}: "
670
+ f"tag not found in {analysis.image.source}"
671
+ )
672
+ unverified.append(
673
+ UnverifiedImage(
674
+ image_reference=analysis.image.full_reference,
675
+ status="TAG_NOT_FOUND",
676
+ reason=f"Tag does not exist in {analysis.image.source}",
677
+ )
678
+ )
679
+
680
+ await asyncio.gather(*[check(c) for c in candidates])
681
+
682
+ async def _write_manifest(self, query: str, selected: list[ImageAnalysis]) -> str:
683
+ if self._evidence is None or not selected:
684
+ return ""
685
+ entries = [
686
+ {
687
+ "image": a.image.full_reference,
688
+ "pinned_reference": a.image.pinned_reference,
689
+ "digest": a.image.digest,
690
+ "confidence": a.confidence.value,
691
+ "production_ready": a.production_ready,
692
+ "readiness_blockers": a.readiness_blockers,
693
+ "cross_validation": a.cross_validation,
694
+ "security_score": a.security_score,
695
+ "tier": a.tier,
696
+ "critical": a.scan.critical_count,
697
+ "high": a.scan.high_count,
698
+ "medium": a.scan.medium_count,
699
+ "scan_status": a.scan.status.value,
700
+ "scan_timestamp": a.scan.scan_timestamp,
701
+ "scan_divergence": a.scan_divergence,
702
+ "hub_url": a.hub_url,
703
+ "hub_tag_verified": a.hub_tag_verified,
704
+ "evidence": a.evidence_paths,
705
+ }
706
+ for a in selected
707
+ ]
708
+ return await self._evidence.record_manifest(
709
+ query,
710
+ entries,
711
+ provenance={
712
+ "dockerls_version": __version__,
713
+ "scanner": self._scanner_identity,
714
+ "resolved_at": datetime.now(tz=UTC).isoformat(),
715
+ "analysis_fingerprint": self._analysis_fingerprint,
716
+ },
717
+ )
718
+
719
+ async def _close_scanners(self) -> None:
720
+ secondary = self._cross_validator.scanner if self._cross_validator else None
721
+ await close_quietly(self._scanner, secondary, self._hardening)
722
+
723
+ async def _close_repositories(self) -> None:
724
+ """Release the HTTP connection pools the image sources hold.
725
+
726
+ The clients keep one `httpx.AsyncClient` alive for the whole run so
727
+ connections are reused; that makes closing them the caller's job.
728
+ """
729
+ await close_quietly(*sources_of(self._repository))
730
+
731
+ async def _get_cached(self, image: DockerImage) -> ImageAnalysis | None:
732
+ key = image.full_reference
733
+ if not self._cache:
734
+ return None
735
+ cache_key = self._cache_key(image)
736
+ try:
737
+ data = await self._cache.get(cache_key)
738
+ except Exception as e:
739
+ # An unreadable cache is a miss, not a scan failure.
740
+ logger.warning(f"Could not read cached analysis for {key}: {e}")
741
+ return None
742
+ if not (data and isinstance(data, dict)):
743
+ return None
744
+ try:
745
+ analysis: ImageAnalysis = ImageAnalysis.model_validate(data)
746
+ except ValidationError as e:
747
+ logger.warning(f"Discarding stale cache entry for {key}: {e}")
748
+ await self._discard(cache_key)
749
+ return None
750
+ # A cache hit is not proof of a successful scan: an entry written by
751
+ # an older build could carry a failed scan. Re-apply the gate.
752
+ if not analysis.scan.is_verified:
753
+ logger.warning(f"Discarding cache entry for {key}: cached scan is not verified")
754
+ await self._discard(cache_key)
755
+ return None
756
+ return analysis
757
+
758
+ async def _discard(self, cache_key: str) -> None:
759
+ """Best-effort eviction: failing to delete a bad entry must not
760
+ become a failure to analyze the image it belongs to."""
761
+ if not self._cache:
762
+ return
763
+ try:
764
+ await self._cache.delete(cache_key)
765
+ except Exception as e:
766
+ logger.warning(f"Could not evict cache entry {cache_key}: {e}")
767
+
768
+ async def _set_cached(self, image: DockerImage, analysis: ImageAnalysis) -> None:
769
+ key = image.full_reference
770
+ """Persist an analysis, treating a storage failure as a cache miss.
771
+
772
+ The cache is an optimisation, never a source of truth. Letting a
773
+ write error escape put it on the same path as a failed scan: the
774
+ exception unwound into `analyze_tag`'s handler, which reported a
775
+ fully-scanned, fully-scored image as `ERROR`/unverified. A locked
776
+ SQLite file -- ordinary under the concurrency this use case creates
777
+ -- was enough to make a clean image vanish from the results.
778
+ """
779
+ if not self._cache:
780
+ return
781
+ try:
782
+ await self._cache.set(
783
+ self._cache_key(image),
784
+ analysis.model_dump(),
785
+ ttl_seconds=self._cache_ttl_seconds,
786
+ )
787
+ except Exception as e:
788
+ logger.warning(f"Could not cache analysis for {key}: {e}")
789
+
790
+
791
+ def _sources_of(tags: list[DockerImage]) -> list[str]:
792
+ """Distinct catalogues that contributed a candidate, in first-seen
793
+ order, so the run can report what it actually looked at."""
794
+ seen: list[str] = []
795
+ for tag in tags:
796
+ if tag.source not in seen:
797
+ seen.append(tag.source)
798
+ return seen
799
+
800
+
801
+ def _assert_verified(analyses: list[ImageAnalysis]) -> None:
802
+ """Final gate before results leave the use case.
803
+
804
+ Nothing reaches the user's "Recommended Images" table without a scan
805
+ result that exists, completed successfully, and produced a timestamp.
806
+ """
807
+ offenders = [
808
+ a.image.full_reference for a in analyses if a.scan is None or not a.scan.is_verified
809
+ ]
810
+ if offenders:
811
+ raise UnverifiedRecommendationError(
812
+ f"Refusing to recommend images without a verified scan: {', '.join(offenders)}"
813
+ )
814
+
815
+
816
+ async def _exploitdb_lookup(
817
+ exploitdb: ExploitDBClient | None, cve_ids: list[str]
818
+ ) -> dict[str, list[ExploitEntry]]:
819
+ """Consulta o Exploit-DB sem deixar a falha dela derrubar o resto.
820
+
821
+ O cliente já degrada sozinho, mas este comando enriquece dezenas de tags
822
+ em paralelo e uma exceção inesperada aqui abortaria a análise inteira de
823
+ uma imagem por causa de uma fonte que é, por definição, opcional.
824
+ """
825
+ if exploitdb is None:
826
+ return {}
827
+ try:
828
+ return await exploitdb.exploits_for(cve_ids)
829
+ except Exception as e: # pragma: no cover - o cliente já trata o previsível
830
+ logger.warning(f"Exploit-DB lookup failed, exploit status stays UNKNOWN: {e}")
831
+ return {}
832
+
833
+
834
+ def _exploitdb_fields(entries: list[ExploitEntry] | None, *, available: bool) -> dict[str, Any]:
835
+ """Os três campos de explorabilidade, ou nada quando nada foi consultado.
836
+
837
+ Com a fonte indisponível os campos não são tocados: o default do modelo
838
+ é UNKNOWN, e escrever FALSE aqui transformaria uma consulta que não
839
+ aconteceu numa afirmação de que não existe exploit publicado.
840
+ """
841
+ if not available:
842
+ return {}
843
+ if not entries:
844
+ return {"exploitdb_status": Tristate.FALSE}
845
+ return {
846
+ "exploitdb_status": Tristate.TRUE,
847
+ "exploitdb_ids": [e.edb_id for e in entries],
848
+ # Um único exploit verificado já basta: a pergunta é se existe prova
849
+ # reproduzida, não se todas as entradas foram reproduzidas.
850
+ "exploitdb_verified": any(e.verified for e in entries),
851
+ }
852
+
853
+
854
+ async def _enrich_with_threat_intel(
855
+ scan: Any,
856
+ threat_intel: ThreatIntelClient,
857
+ exploitdb: ExploitDBClient | None = None,
858
+ ) -> Any:
859
+ """Tag CRITICAL/HIGH vulnerabilities with CISA KEV / EPSS / Exploit-DB signal.
860
+
861
+ The enrichment records *whether the feeds answered*, not just what they
862
+ said. With the KEV catalogue unreachable every lookup returns the empty
863
+ set, and marking each CVE `exploit_known=False` on that basis turned a
864
+ failed request into an affirmative safety claim -- the report went on to
865
+ state that the image had no known-exploited vulnerabilities. So a
866
+ CVE now carries `kev_status`: TRUE (listed), FALSE (catalogue answered
867
+ and does not list it), UNKNOWN (nothing was consulted).
868
+
869
+ Exploit-DB rides the same entry point rather than a second pass: it is
870
+ the same question about the same CVEs, and a separate flow would mean
871
+ two places to keep the "absent lookup is never a negative" rule in.
872
+ KEV and Exploit-DB are not redundant -- KEV means observed exploitation
873
+ in the wild, Exploit-DB means published exploit code -- so a CVE can
874
+ carry one and not the other.
875
+
876
+ Enrichment is attempted only for CRITICAL/HIGH findings, so anything
877
+ below stays UNKNOWN by construction -- which is correct: it was not
878
+ looked up.
879
+ """
880
+ notable_ids = [
881
+ v.cve_id
882
+ for v in scan.vulnerabilities
883
+ if v.severity.value in ("CRITICAL", "HIGH") and v.cve_id
884
+ ]
885
+ if not notable_ids:
886
+ return scan
887
+
888
+ # As três fontes respondem sobre o mesmo lote de CVEs e não dependem
889
+ # uma da outra -- pedi-las em sequência somava a latência das três num
890
+ # scan que já espera pelo scanner. Uma falha isolada não derruba as
891
+ # outras: cada chamada já degrada sozinha para o tri-state UNKNOWN.
892
+ kev_ids, epss, exploits = await asyncio.gather(
893
+ threat_intel.known_exploited(notable_ids),
894
+ threat_intel.epss_scores(notable_ids),
895
+ _exploitdb_lookup(exploitdb, notable_ids),
896
+ )
897
+ exploitdb_available = exploitdb is not None and bool(exploitdb.available)
898
+ kev_available = _answered(threat_intel.kev_available, bool(kev_ids))
899
+ epss_available = _answered(threat_intel.epss_available, bool(epss))
900
+ if not kev_available and not epss_available and not exploitdb_available:
901
+ # Nothing was learned. Returning the scan untouched leaves every
902
+ # `kev_status` at UNKNOWN, which is exactly what happened.
903
+ logger.warning(
904
+ "Threat intelligence unavailable: exploitation status stays UNKNOWN for "
905
+ f"{len(notable_ids)} finding(s) in {scan.image_reference}"
906
+ )
907
+ return scan
908
+
909
+ timestamp = datetime.now(tz=UTC).isoformat()
910
+ notable = set(notable_ids)
911
+ updated = []
912
+ for v in scan.vulnerabilities:
913
+ if v.cve_id not in notable:
914
+ updated.append(v)
915
+ continue
916
+ key = v.cve_id.upper()
917
+ listed = key in kev_ids
918
+ score = epss.get(key)
919
+ updated.append(
920
+ v.model_copy(
921
+ update={
922
+ "exploit_known": listed,
923
+ "kev_status": Tristate.of(listed) if kev_available else Tristate.UNKNOWN,
924
+ "epss_score": score if score is not None else v.epss_score,
925
+ "epss_known": epss_available and score is not None,
926
+ "epss_percentile": threat_intel.percentile_of(key),
927
+ "threat_intel_timestamp": timestamp,
928
+ **_exploitdb_fields(exploits.get(key), available=exploitdb_available),
929
+ }
930
+ )
931
+ )
932
+ return scan.model_copy(update={"vulnerabilities": updated})
933
+
934
+
935
+ async def _eol_status(checker: Any, product: str, version: str) -> Tristate:
936
+ """The three-valued lifecycle answer, from a checker that can give one.
937
+
938
+ Dispatched dynamically for the same reason `tag_exists` and `refresh_db`
939
+ are: `EOLCheckerInterface` predates the tri-state, and every test double
940
+ and alternative implementation in the wild implements the boolean. A
941
+ checker that only answers `is_eol` can distinguish TRUE from
942
+ "not TRUE", and "not TRUE" is honestly reported as FALSE only because
943
+ that is the entire content of what it said -- the richer checker is the
944
+ one that gets to say UNKNOWN.
945
+ """
946
+ status = getattr(checker, "eol_status", None)
947
+ if callable(status):
948
+ result = await status(product, version)
949
+ # The answer is validated, not assumed: `getattr` on a test double
950
+ # (or on any object with dynamic attributes) happily produces a
951
+ # callable that returns something else entirely, and a value that is
952
+ # not a Tristate must not be coerced into one -- `bool(mock)` is
953
+ # True, which would silently declare every image end-of-life.
954
+ if isinstance(result, Tristate):
955
+ return result
956
+ logger.debug(
957
+ f"{type(checker).__name__}.eol_status returned {type(result).__name__}, "
958
+ "not a Tristate; falling back to is_eol"
959
+ )
960
+ return Tristate.of(await checker.is_eol(product, version))
961
+
962
+
963
+ def _answered(reported: bool | None, produced_data: bool) -> bool:
964
+ """Whether a threat-intel source actually answered.
965
+
966
+ The client records this directly, and that record is authoritative in
967
+ both directions. `None` means nothing set it -- a source that was never
968
+ queried, or a substitute that does not track it -- and there the only
969
+ evidence available is the payload itself: data that came back proves the
970
+ source answered, while an empty result proves nothing either way and is
971
+ treated as "not consulted".
972
+
973
+ The asymmetry is the whole point. Erring towards UNKNOWN costs a
974
+ confidence level; erring towards "answered" would let a dead feed
975
+ produce the sentence "no known-exploited vulnerabilities".
976
+ """
977
+ if reported is not None:
978
+ return reported
979
+ return produced_data
980
+
981
+
982
+ def _apply_ignore_rules(scan: Any, ignored_cves: set[str]) -> Any:
983
+ """Return a copy of `scan` with vulnerabilities matching an active
984
+ .dockerls-ignore.yaml rule removed, so ignored CVEs never affect
985
+ scoring, tiering, or the baseline decision."""
986
+ filtered = [v for v in scan.vulnerabilities if v.cve_id.upper() not in ignored_cves]
987
+ if len(filtered) == len(scan.vulnerabilities):
988
+ return scan
989
+ return scan.model_copy(update={"vulnerabilities": filtered})
990
+
991
+
992
+ _LEADING_VERSION_RE = re.compile(r"^\d+(?:\.\d+){0,3}")
993
+
994
+
995
+ def _extract_product_version(image: DockerImage) -> tuple[str, str]:
996
+ name = image.name.split("/")[-1]
997
+ match = _LEADING_VERSION_RE.match(image.tag)
998
+ version = match.group(0) if match else ""
999
+ return name, version
1000
+
1001
+
1002
+ def build_recommendation(analysis: ImageAnalysis) -> Recommendation:
1003
+ steps: list[RemediationStep] = []
1004
+ step_num = 1
1005
+
1006
+ if analysis.scan.fixable_high_count > 0 or analysis.scan.fixable_critical_count > 0:
1007
+ fixable_pkgs = [
1008
+ v
1009
+ for v in analysis.scan.vulnerabilities
1010
+ if v.is_fixable and v.severity.value in ("CRITICAL", "HIGH")
1011
+ ]
1012
+ for vuln in fixable_pkgs[:5]:
1013
+ steps.append(
1014
+ RemediationStep(
1015
+ step_number=step_num,
1016
+ action=ActionType.UPDATE_PACKAGE,
1017
+ description=f"Update {vuln.package_name}",
1018
+ from_value=vuln.installed_version,
1019
+ to_value=vuln.fixed_version,
1020
+ expected_impact=f"Fix {vuln.severity.value} {vuln.cve_id}",
1021
+ )
1022
+ )
1023
+ step_num += 1
1024
+
1025
+ if not analysis.image.is_alpine and not analysis.image.is_distroless:
1026
+ steps.append(
1027
+ RemediationStep(
1028
+ step_number=step_num,
1029
+ action=ActionType.SWITCH_BASE,
1030
+ description="Consider switching to Alpine or Distroless variant",
1031
+ expected_impact="Reduced attack surface",
1032
+ )
1033
+ )
1034
+ step_num += 1
1035
+
1036
+ steps.append(
1037
+ RemediationStep(
1038
+ step_number=step_num,
1039
+ action=ActionType.REBUILD_IMAGE,
1040
+ description="Rebuild image to pick up latest base layer patches",
1041
+ )
1042
+ )
1043
+ step_num += 1
1044
+
1045
+ steps.append(
1046
+ RemediationStep(
1047
+ step_number=step_num,
1048
+ action=ActionType.RESCAN,
1049
+ description="Re-run vulnerability scan to verify fixes",
1050
+ )
1051
+ )
1052
+
1053
+ summary_parts = []
1054
+ if analysis.scan.critical_count == 0 and analysis.scan.high_count == 0:
1055
+ summary_parts.append("Image meets security baseline.")
1056
+ elif analysis.scan.critical_count == 0:
1057
+ summary_parts.append(
1058
+ f"Image has {analysis.scan.high_count} HIGH vulnerabilities "
1059
+ f"({analysis.scan.fixable_high_count} fixable)."
1060
+ )
1061
+ else:
1062
+ summary_parts.append(
1063
+ f"Image has {analysis.scan.critical_count} CRITICAL and "
1064
+ f"{analysis.scan.high_count} HIGH vulnerabilities."
1065
+ )
1066
+ if analysis.remediation_score == 100:
1067
+ summary_parts.append("All vulnerabilities have available fixes.")
1068
+ if analysis.scan_divergence:
1069
+ summary_parts.append(f"Scanner disagreement: {analysis.scan_divergence}.")
1070
+
1071
+ return Recommendation(
1072
+ image_reference=analysis.image.full_reference,
1073
+ security_score=analysis.security_score,
1074
+ tier=analysis.tier,
1075
+ remediation_score=analysis.remediation_score,
1076
+ steps=steps,
1077
+ summary=" ".join(summary_parts),
1078
+ )