suphm 0.3.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
suphm/__init__.py ADDED
@@ -0,0 +1,3 @@
1
+ """suphm: Supply chain health metrics for OSS packages."""
2
+
3
+ __version__ = "0.3.0"
@@ -0,0 +1,5 @@
1
+ """Analysis pipeline for suphm."""
2
+
3
+ from suphm.analysis.pipeline import AnalysisPipeline, AnalysisResult
4
+
5
+ __all__ = ["AnalysisPipeline", "AnalysisResult"]
@@ -0,0 +1,470 @@
1
+ """Full analysis pipeline for packages.
2
+
3
+ Orchestrates discovery, cloning, metrics collection, scanning, and scoring.
4
+ """
5
+
6
+ from __future__ import annotations
7
+
8
+ import asyncio
9
+ import logging
10
+ from collections.abc import Callable
11
+ from dataclasses import dataclass, field
12
+ from datetime import datetime
13
+ from pathlib import Path
14
+ from typing import Any
15
+
16
+ from suphm.cache import CacheManager
17
+ from suphm.config import get_config
18
+ from suphm.core.git import CloneResult, GitManager
19
+ from suphm.discovery import DiscoveryResult, PackageResolver, PURLParser
20
+ from suphm.metrics.git import GitMetricsAnalyzer, GitMetricsResult
21
+ from suphm.metrics.github import GitHubMetricsCollector, GitHubMetricsResult
22
+ from suphm.scanner import TarballScanner, TarballScanResult
23
+ from suphm.scoring.burnout import BurnoutScoreCalculator, BurnoutScoreResult
24
+ from suphm.scoring.health import HealthScoreCalculator, HealthScoreResult
25
+
26
+ log = logging.getLogger(__name__)
27
+
28
+
29
+ @dataclass
30
+ class AnalysisStep:
31
+ """Represents a step in the analysis pipeline."""
32
+
33
+ name: str
34
+ status: str = "pending" # pending, running, completed, skipped, failed
35
+ duration_seconds: float = 0.0
36
+ error: str | None = None
37
+
38
+
39
+ @dataclass
40
+ class AnalysisResult:
41
+ """Complete analysis result for a package."""
42
+
43
+ purl: str
44
+ analyzed_at: datetime
45
+ analysis_version: str = "1.0.0"
46
+
47
+ # Results from each step
48
+ discovery: DiscoveryResult | None = None
49
+ clone: CloneResult | None = None
50
+ git_metrics: GitMetricsResult | None = None
51
+ github_metrics: GitHubMetricsResult | None = None
52
+ tarball_scan: TarballScanResult | None = None
53
+ health_score: HealthScoreResult | None = None
54
+ burnout_score: BurnoutScoreResult | None = None
55
+
56
+ # Pipeline metadata
57
+ steps: list[AnalysisStep] = field(default_factory=list)
58
+ total_duration_seconds: float = 0.0
59
+ errors: list[str] = field(default_factory=list)
60
+
61
+ def to_dict(self) -> dict[str, Any]:
62
+ """Convert to unified JSON format matching the spec."""
63
+ result = {
64
+ "schema_version": "1.0.0",
65
+ "purl": self.purl,
66
+ "analyzed_at": self.analyzed_at.isoformat(),
67
+ "analysis_version": self.analysis_version,
68
+ "pipeline": {
69
+ "steps": [
70
+ {
71
+ "name": s.name,
72
+ "status": s.status,
73
+ "duration_seconds": round(s.duration_seconds, 2),
74
+ "error": s.error,
75
+ }
76
+ for s in self.steps
77
+ ],
78
+ "total_duration_seconds": round(self.total_duration_seconds, 2),
79
+ "errors": self.errors,
80
+ },
81
+ }
82
+
83
+ # Discovery
84
+ if self.discovery:
85
+ result["discovery"] = self.discovery.to_dict()
86
+
87
+ # Repository info
88
+ if self.clone:
89
+ result["repository"] = {
90
+ "url": self.clone.repo_url,
91
+ "local_path": str(self.clone.local_path),
92
+ "cloned": self.clone.success,
93
+ "last_commit": self.clone.last_commit_hash,
94
+ "last_commit_date": self.clone.last_commit_date.isoformat() if self.clone.last_commit_date else None,
95
+ }
96
+
97
+ # Git metrics
98
+ if self.git_metrics:
99
+ result["git_metrics"] = self.git_metrics.to_dict()
100
+
101
+ # GitHub metrics
102
+ if self.github_metrics:
103
+ result["github_metrics"] = self.github_metrics.to_dict()
104
+
105
+ # Tarball scan
106
+ if self.tarball_scan:
107
+ result["tarball_scan"] = self.tarball_scan.to_dict()
108
+
109
+ # Scores
110
+ if self.health_score:
111
+ result["health_score"] = self.health_score.to_dict()
112
+
113
+ if self.burnout_score:
114
+ result["burnout_score"] = self.burnout_score.to_dict()
115
+
116
+ # Unified summary
117
+ result["summary"] = self._build_summary()
118
+
119
+ return result
120
+
121
+ def _build_summary(self) -> dict[str, Any]:
122
+ """Build unified summary section."""
123
+ summary: dict[str, Any] = {
124
+ "package_name": None,
125
+ "version": None,
126
+ "github_url": None,
127
+ "tarball_url": None,
128
+ "license": None,
129
+ "health_grade": None,
130
+ "burnout_risk": None,
131
+ "has_binaries": False,
132
+ "key_metrics": {},
133
+ }
134
+
135
+ # From discovery. Re-parse the PURL defensively: `to_dict()` runs
136
+ # inside cache writes and from `analyze_batch`, so a malformed PURL
137
+ # raising here would have killed the whole batch before fixes landed.
138
+ if self.discovery:
139
+ try:
140
+ parsed = PURLParser.parse(self.purl)
141
+ summary["package_name"] = parsed.full_name
142
+ summary["version"] = parsed.version
143
+ except Exception:
144
+ summary["package_name"] = self.discovery.name
145
+ summary["version"] = self.discovery.version
146
+ summary["github_url"] = self.discovery.github_url
147
+ summary["tarball_url"] = self.discovery.tarball_url
148
+ if self.discovery.metadata:
149
+ summary["license"] = self.discovery.metadata.get("license")
150
+
151
+ # From scores
152
+ if self.health_score:
153
+ summary["health_grade"] = self.health_score.grade
154
+ summary["key_metrics"]["health_score"] = self.health_score.health_score
155
+
156
+ if self.burnout_score:
157
+ summary["burnout_risk"] = self.burnout_score.risk_level
158
+ summary["key_metrics"]["burnout_score"] = self.burnout_score.burnout_score
159
+
160
+ # From git metrics
161
+ if self.git_metrics:
162
+ window = self.git_metrics.time_windows.get("90_days")
163
+ if window:
164
+ summary["key_metrics"]["bus_factor"] = window.bus_factor
165
+ summary["key_metrics"]["pony_factor"] = window.pony_factor
166
+ summary["key_metrics"]["unique_contributors_90d"] = window.unique_contributors
167
+
168
+ # From GitHub metrics
169
+ if self.github_metrics:
170
+ summary["key_metrics"]["stars"] = self.github_metrics.repository.stars
171
+ summary["key_metrics"]["open_issues"] = self.github_metrics.issues.open_count
172
+ summary["key_metrics"]["open_prs"] = self.github_metrics.pull_requests.open_count
173
+
174
+ # From tarball scan
175
+ if self.tarball_scan:
176
+ summary["has_binaries"] = bool(self.tarball_scan.binaries.get("files"))
177
+ if self.tarball_scan.license_files:
178
+ # Use detected license if not in metadata
179
+ if not summary["license"]:
180
+ for lf in self.tarball_scan.license_files:
181
+ if lf.spdx_id:
182
+ summary["license"] = lf.spdx_id
183
+ break
184
+
185
+ return summary
186
+
187
+
188
+ class AnalysisPipeline:
189
+ """Orchestrates the full analysis pipeline."""
190
+
191
+ def __init__(
192
+ self,
193
+ cache_dir: Path | None = None,
194
+ progress_callback: Callable[[str, str], None] | None = None,
195
+ ):
196
+ """Initialize pipeline.
197
+
198
+ Args:
199
+ cache_dir: Custom cache directory
200
+ progress_callback: Callback for progress updates (step_name, status)
201
+ """
202
+ self.config = get_config()
203
+ self.cache = CacheManager(base_dir=cache_dir) if cache_dir else CacheManager()
204
+ self.git_manager = GitManager(cache_dir=cache_dir)
205
+ self.progress_callback = progress_callback or (lambda *_: None)
206
+
207
+ async def analyze(
208
+ self,
209
+ purl: str,
210
+ skip_clone: bool = False,
211
+ skip_tarball: bool = False,
212
+ skip_github: bool = False,
213
+ force_refresh: bool = False,
214
+ ) -> AnalysisResult:
215
+ """Run full analysis on a package.
216
+
217
+ Args:
218
+ purl: Package URL to analyze
219
+ skip_clone: Skip git clone step
220
+ skip_tarball: Skip tarball scanning
221
+ skip_github: Skip GitHub API metrics
222
+ force_refresh: Force refresh all cached data
223
+
224
+ Returns:
225
+ AnalysisResult with all collected data
226
+ """
227
+ start_time = datetime.now()
228
+ result = AnalysisResult(purl=purl, analyzed_at=start_time)
229
+
230
+ # Step 1: Discovery
231
+ discovery_step = AnalysisStep(name="discovery")
232
+ result.steps.append(discovery_step)
233
+ await self._run_step(
234
+ discovery_step,
235
+ self._discovery_step,
236
+ result,
237
+ purl,
238
+ force_refresh,
239
+ )
240
+
241
+ # Step 2: Clone (if GitHub URL found)
242
+ if not skip_clone and result.discovery and result.discovery.github_url:
243
+ clone_step = AnalysisStep(name="clone")
244
+ result.steps.append(clone_step)
245
+ await self._run_step(
246
+ clone_step,
247
+ self._clone_step,
248
+ result,
249
+ result.discovery.github_url,
250
+ )
251
+
252
+ # Step 3: Git Metrics (if cloned)
253
+ if result.clone and result.clone.success:
254
+ git_step = AnalysisStep(name="git_metrics")
255
+ result.steps.append(git_step)
256
+ await self._run_step(
257
+ git_step,
258
+ self._git_metrics_step,
259
+ result,
260
+ result.clone.local_path,
261
+ result.clone.repo_url,
262
+ )
263
+
264
+ # Step 4: GitHub Metrics
265
+ if not skip_github and result.discovery and result.discovery.github_url:
266
+ github_step = AnalysisStep(name="github_metrics")
267
+ result.steps.append(github_step)
268
+ await self._run_step(
269
+ github_step,
270
+ self._github_metrics_step,
271
+ result,
272
+ result.discovery.github_url,
273
+ )
274
+
275
+ # Step 5: Tarball Scan
276
+ if not skip_tarball:
277
+ tarball_step = AnalysisStep(name="tarball_scan")
278
+ result.steps.append(tarball_step)
279
+ await self._run_step(
280
+ tarball_step,
281
+ self._tarball_step,
282
+ result,
283
+ purl,
284
+ )
285
+
286
+ # Step 6: Health Score
287
+ health_step = AnalysisStep(name="health_score")
288
+ result.steps.append(health_step)
289
+ await self._run_step(
290
+ health_step,
291
+ self._health_score_step,
292
+ result,
293
+ )
294
+
295
+ # Step 7: Burnout Score
296
+ burnout_step = AnalysisStep(name="burnout_score")
297
+ result.steps.append(burnout_step)
298
+ await self._run_step(
299
+ burnout_step,
300
+ self._burnout_score_step,
301
+ result,
302
+ )
303
+
304
+ # Calculate total duration
305
+ result.total_duration_seconds = (datetime.now() - start_time).total_seconds()
306
+
307
+ # Save to cache
308
+ self.cache.save_package_data(purl, "unified.json", result.to_dict())
309
+
310
+ return result
311
+
312
+ async def _run_step(
313
+ self,
314
+ step: AnalysisStep,
315
+ func: Callable,
316
+ result: AnalysisResult,
317
+ *args,
318
+ ) -> None:
319
+ """Run a step with timing and error handling."""
320
+ step.status = "running"
321
+ self.progress_callback(step.name, "running")
322
+ start = datetime.now()
323
+
324
+ try:
325
+ await func(result, *args)
326
+ step.status = "completed"
327
+ except Exception as e:
328
+ step.status = "failed"
329
+ step.error = str(e)
330
+ result.errors.append(f"{step.name}: {e}")
331
+
332
+ step.duration_seconds = (datetime.now() - start).total_seconds()
333
+ self.progress_callback(step.name, step.status)
334
+
335
+ async def _discovery_step(
336
+ self,
337
+ result: AnalysisResult,
338
+ purl: str,
339
+ force_refresh: bool,
340
+ ) -> None:
341
+ """Run discovery step.
342
+
343
+ For pkg:github/* the GitHub URL is encoded in the PURL itself, so we
344
+ skip the registry round-trip and synthesise a DiscoveryResult directly.
345
+ Every other PURL type goes through PackageResolver, which queries
346
+ deps.dev / ecosyste.ms / the package registry.
347
+ """
348
+ if not force_refresh:
349
+ cached = self.cache.get_package_data(purl, "discovery.json")
350
+ if cached:
351
+ result.discovery = DiscoveryResult.from_dict(cached.data)
352
+ return
353
+
354
+ try:
355
+ parsed = PURLParser.parse(purl)
356
+ except Exception as e:
357
+ log.debug("PURL parse failed for %s: %s", purl, e)
358
+ parsed = None
359
+
360
+ if parsed is not None and parsed.type == "github":
361
+ github_url = f"https://github.com/{parsed.namespace}/{parsed.name}"
362
+ result.discovery = DiscoveryResult(
363
+ purl=purl,
364
+ name=f"{parsed.namespace}/{parsed.name}",
365
+ version=parsed.version,
366
+ repository_url=github_url,
367
+ registry_data={"source": "purl"},
368
+ sources=["github"],
369
+ )
370
+ else:
371
+ resolver = PackageResolver()
372
+ result.discovery = await resolver.discover(purl)
373
+
374
+ self.cache.save_package_data(purl, "discovery.json", result.discovery.to_dict())
375
+
376
+ async def _clone_step(
377
+ self,
378
+ result: AnalysisResult,
379
+ github_url: str,
380
+ ) -> None:
381
+ """Run clone step."""
382
+ result.clone = await self.git_manager.clone(github_url)
383
+
384
+ async def _git_metrics_step(
385
+ self,
386
+ result: AnalysisResult,
387
+ local_path: Path,
388
+ repo_url: str,
389
+ ) -> None:
390
+ """Run git metrics step."""
391
+ # Run in executor since it's CPU-bound
392
+ loop = asyncio.get_event_loop()
393
+ analyzer = GitMetricsAnalyzer(local_path)
394
+ result.git_metrics = await loop.run_in_executor(
395
+ None, analyzer.analyze, repo_url
396
+ )
397
+
398
+ async def _github_metrics_step(
399
+ self,
400
+ result: AnalysisResult,
401
+ github_url: str,
402
+ ) -> None:
403
+ """Run GitHub metrics step."""
404
+ collector = GitHubMetricsCollector()
405
+ result.github_metrics = await collector.collect(github_url)
406
+
407
+ async def _tarball_step(
408
+ self,
409
+ result: AnalysisResult,
410
+ purl: str,
411
+ ) -> None:
412
+ """Run tarball scan step."""
413
+ scanner = TarballScanner()
414
+ result.tarball_scan = await scanner.scan_purl(purl)
415
+
416
+ async def _health_score_step(
417
+ self,
418
+ result: AnalysisResult,
419
+ ) -> None:
420
+ """Calculate health score."""
421
+ calculator = HealthScoreCalculator()
422
+ result.health_score = calculator.calculate(
423
+ result.purl,
424
+ result.git_metrics,
425
+ result.github_metrics,
426
+ )
427
+
428
+ async def _burnout_score_step(
429
+ self,
430
+ result: AnalysisResult,
431
+ ) -> None:
432
+ """Calculate burnout score."""
433
+ calculator = BurnoutScoreCalculator()
434
+ result.burnout_score = calculator.calculate(
435
+ result.purl,
436
+ result.git_metrics,
437
+ result.github_metrics,
438
+ )
439
+
440
+ async def analyze_batch(
441
+ self,
442
+ purls: list[str],
443
+ concurrency: int = 3,
444
+ **kwargs,
445
+ ) -> list[AnalysisResult]:
446
+ """Analyze multiple packages concurrently.
447
+
448
+ One bad PURL no longer kills the whole batch: any exception that
449
+ escapes analyze() is captured into a synthetic AnalysisResult with the
450
+ error recorded in its `errors` list, so the caller can inspect failures
451
+ per-package.
452
+ """
453
+ semaphore = asyncio.Semaphore(concurrency)
454
+
455
+ async def analyze_with_semaphore(purl: str) -> AnalysisResult:
456
+ async with semaphore:
457
+ return await self.analyze(purl, **kwargs)
458
+
459
+ tasks = [analyze_with_semaphore(purl) for purl in purls]
460
+ raw = await asyncio.gather(*tasks, return_exceptions=True)
461
+
462
+ results: list[AnalysisResult] = []
463
+ for purl, item in zip(purls, raw, strict=True):
464
+ if isinstance(item, BaseException):
465
+ placeholder = AnalysisResult(purl=purl, analyzed_at=datetime.now())
466
+ placeholder.errors.append(f"analyze raised: {item!r}")
467
+ results.append(placeholder)
468
+ else:
469
+ results.append(item)
470
+ return results
@@ -0,0 +1,5 @@
1
+ """Cache management for suphm."""
2
+
3
+ from suphm.cache.manager import CacheEntry, CacheManager
4
+
5
+ __all__ = ["CacheManager", "CacheEntry"]