src2purl 1.3.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- src2id/__init__.py +15 -0
- src2id/cli/__init__.py +1 -0
- src2id/cli/main.py +318 -0
- src2id/cli/validate.py +93 -0
- src2id/core/__init__.py +1 -0
- src2id/core/cache.py +180 -0
- src2id/core/client.py +370 -0
- src2id/core/config.py +45 -0
- src2id/core/extractor.py +302 -0
- src2id/core/models.py +93 -0
- src2id/core/orchestrator.py +796 -0
- src2id/core/package_identifier.py +123 -0
- src2id/core/purl.py +238 -0
- src2id/core/scanner.py +369 -0
- src2id/core/scorer.py +217 -0
- src2id/core/subcomponent_detector.py +353 -0
- src2id/core/swhid.py +324 -0
- src2id/integrations/__init__.py +1 -0
- src2id/integrations/manifest_parser.py +652 -0
- src2id/integrations/oslili.py +228 -0
- src2id/integrations/upmex.py +305 -0
- src2id/search/__init__.py +34 -0
- src2id/search/hash_search.py +206 -0
- src2id/search/providers.py +310 -0
- src2id/search/strategies.py +391 -0
- src2id/utils/__init__.py +1 -0
- src2id/utils/datetime_utils.py +49 -0
- src2purl-1.3.2.dist-info/METADATA +279 -0
- src2purl-1.3.2.dist-info/RECORD +33 -0
- src2purl-1.3.2.dist-info/WHEEL +5 -0
- src2purl-1.3.2.dist-info/entry_points.txt +3 -0
- src2purl-1.3.2.dist-info/licenses/LICENSE +661 -0
- src2purl-1.3.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,391 @@
|
|
|
1
|
+
"""Unified source identification strategies."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
import os
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Dict, List, Optional, Tuple, Any
|
|
7
|
+
from collections import Counter
|
|
8
|
+
|
|
9
|
+
from rich.console import Console
|
|
10
|
+
from rich.table import Table
|
|
11
|
+
|
|
12
|
+
from ..core.models import DirectoryCandidate, ContentCandidate
|
|
13
|
+
from ..core.scanner import DirectoryScanner
|
|
14
|
+
from ..core.client import SoftwareHeritageClient
|
|
15
|
+
from .hash_search import HashSearcher
|
|
16
|
+
from .providers import (
|
|
17
|
+
SearchProviderRegistry,
|
|
18
|
+
create_default_registry,
|
|
19
|
+
SCANOSSProvider
|
|
20
|
+
)
|
|
21
|
+
|
|
22
|
+
console = Console()
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
class SourceIdentifier:
|
|
26
|
+
"""Unified source identification using multiple strategies."""
|
|
27
|
+
|
|
28
|
+
def __init__(
|
|
29
|
+
self,
|
|
30
|
+
swh_client: Optional[SoftwareHeritageClient] = None,
|
|
31
|
+
search_registry: Optional[SearchProviderRegistry] = None,
|
|
32
|
+
verbose: bool = False
|
|
33
|
+
):
|
|
34
|
+
"""Initialize the source identifier.
|
|
35
|
+
|
|
36
|
+
Args:
|
|
37
|
+
swh_client: Software Heritage client instance
|
|
38
|
+
search_registry: Registry of search providers
|
|
39
|
+
verbose: Enable verbose output
|
|
40
|
+
"""
|
|
41
|
+
# Only store the client if provided, don't create it automatically
|
|
42
|
+
self.swh_client = swh_client
|
|
43
|
+
self.search_registry = search_registry or create_default_registry(verbose=verbose)
|
|
44
|
+
self.verbose = verbose
|
|
45
|
+
self.hash_searcher = HashSearcher(verbose=verbose)
|
|
46
|
+
self._swh_config = None # Store config for lazy initialization
|
|
47
|
+
|
|
48
|
+
async def identify(
|
|
49
|
+
self,
|
|
50
|
+
path: Path,
|
|
51
|
+
max_depth: int = 3,
|
|
52
|
+
confidence_threshold: float = 0.5,
|
|
53
|
+
strategies: Optional[List[str]] = None,
|
|
54
|
+
use_swh: bool = False
|
|
55
|
+
) -> Dict[str, Any]:
|
|
56
|
+
"""Identify the source of a directory using multiple strategies.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
path: Path to analyze
|
|
60
|
+
max_depth: Maximum depth for recursive scanning
|
|
61
|
+
confidence_threshold: Minimum confidence for identification
|
|
62
|
+
strategies: List of strategies to use (default: optimized order)
|
|
63
|
+
use_swh: Whether to include Software Heritage checking
|
|
64
|
+
|
|
65
|
+
Returns:
|
|
66
|
+
Dictionary with identification results
|
|
67
|
+
"""
|
|
68
|
+
results = {
|
|
69
|
+
"path": str(path),
|
|
70
|
+
"identified": False,
|
|
71
|
+
"confidence": 0.0,
|
|
72
|
+
"strategies_used": [],
|
|
73
|
+
"candidates": [],
|
|
74
|
+
"final_origin": None
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
available_strategies = {
|
|
78
|
+
"hash_search": self._identify_via_hash_search,
|
|
79
|
+
"web_search": self._identify_via_web_search,
|
|
80
|
+
"scanoss": self._identify_via_scanoss,
|
|
81
|
+
"swh": self._identify_via_swh
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
# Default optimized order (local methods first, then external APIs)
|
|
85
|
+
if strategies is None:
|
|
86
|
+
strategies_to_use = ["hash_search", "web_search", "scanoss"]
|
|
87
|
+
if use_swh:
|
|
88
|
+
strategies_to_use.append("swh")
|
|
89
|
+
else:
|
|
90
|
+
strategies_to_use = strategies
|
|
91
|
+
|
|
92
|
+
all_candidates = []
|
|
93
|
+
|
|
94
|
+
for strategy_name in strategies_to_use:
|
|
95
|
+
if strategy_name not in available_strategies:
|
|
96
|
+
continue
|
|
97
|
+
|
|
98
|
+
strategy_func = available_strategies[strategy_name]
|
|
99
|
+
|
|
100
|
+
try:
|
|
101
|
+
if self.verbose:
|
|
102
|
+
console.print(f"[cyan]Running {strategy_name} strategy...[/cyan]")
|
|
103
|
+
|
|
104
|
+
candidates = await strategy_func(path, max_depth)
|
|
105
|
+
|
|
106
|
+
if candidates:
|
|
107
|
+
all_candidates.extend(candidates)
|
|
108
|
+
results["strategies_used"].append(strategy_name)
|
|
109
|
+
|
|
110
|
+
if self.verbose:
|
|
111
|
+
console.print(f"[green]✓ {strategy_name} found {len(candidates)} candidates[/green]")
|
|
112
|
+
|
|
113
|
+
except Exception as e:
|
|
114
|
+
if self.verbose:
|
|
115
|
+
console.print(f"[yellow]⚠ {strategy_name} failed: {e}[/yellow]")
|
|
116
|
+
|
|
117
|
+
# Clean up any open sessions
|
|
118
|
+
if hasattr(self, 'search_registry') and self.search_registry:
|
|
119
|
+
await self.search_registry.close_all()
|
|
120
|
+
|
|
121
|
+
# Aggregate and score candidates
|
|
122
|
+
if all_candidates:
|
|
123
|
+
origin_scores = Counter()
|
|
124
|
+
for candidate in all_candidates:
|
|
125
|
+
if "origin" in candidate:
|
|
126
|
+
origin_scores[candidate["origin"]] += candidate.get("confidence", 1.0)
|
|
127
|
+
|
|
128
|
+
if origin_scores:
|
|
129
|
+
best_origin, best_score = origin_scores.most_common(1)[0]
|
|
130
|
+
total_strategies = len(results["strategies_used"])
|
|
131
|
+
confidence = best_score / total_strategies if total_strategies > 0 else 0
|
|
132
|
+
|
|
133
|
+
if confidence >= confidence_threshold:
|
|
134
|
+
results["identified"] = True
|
|
135
|
+
results["confidence"] = confidence
|
|
136
|
+
results["final_origin"] = best_origin
|
|
137
|
+
results["candidates"] = all_candidates[:10] # Top 10 candidates
|
|
138
|
+
|
|
139
|
+
return results
|
|
140
|
+
|
|
141
|
+
async def _identify_via_swh(
|
|
142
|
+
self,
|
|
143
|
+
path: Path,
|
|
144
|
+
max_depth: int
|
|
145
|
+
) -> List[Dict[str, Any]]:
|
|
146
|
+
"""Identify using Software Heritage archive."""
|
|
147
|
+
candidates = []
|
|
148
|
+
|
|
149
|
+
# Lazily create SWH client only when needed
|
|
150
|
+
if self.swh_client is None:
|
|
151
|
+
from ..core.config import SWHPIConfig
|
|
152
|
+
config = SWHPIConfig(verbose=self.verbose)
|
|
153
|
+
self.swh_client = SoftwareHeritageClient(config)
|
|
154
|
+
|
|
155
|
+
# Scan directory
|
|
156
|
+
from ..core.config import SWHPIConfig
|
|
157
|
+
from ..core.swhid import SWHIDGenerator
|
|
158
|
+
config = SWHPIConfig(verbose=self.verbose, max_depth=max_depth)
|
|
159
|
+
scanner = DirectoryScanner(config, SWHIDGenerator())
|
|
160
|
+
dir_candidates, file_candidates = scanner.scan_recursive(path)
|
|
161
|
+
|
|
162
|
+
# Check SWHIDs
|
|
163
|
+
all_swhids = [c.swhid for c in dir_candidates] + [c.swhid for c in file_candidates]
|
|
164
|
+
|
|
165
|
+
if all_swhids:
|
|
166
|
+
known_swhids = await self.swh_client.check_swhids_known(all_swhids[:100])
|
|
167
|
+
|
|
168
|
+
for swhid, is_known in known_swhids.items():
|
|
169
|
+
if is_known:
|
|
170
|
+
# Try to get origin information
|
|
171
|
+
origin = await self.swh_client.get_origin_for_swhid(swhid)
|
|
172
|
+
if origin:
|
|
173
|
+
candidates.append({
|
|
174
|
+
"source": "swh",
|
|
175
|
+
"swhid": swhid,
|
|
176
|
+
"origin": origin,
|
|
177
|
+
"confidence": 1.0
|
|
178
|
+
})
|
|
179
|
+
|
|
180
|
+
return candidates
|
|
181
|
+
|
|
182
|
+
async def _identify_via_hash_search(
|
|
183
|
+
self,
|
|
184
|
+
path: Path,
|
|
185
|
+
max_depth: int
|
|
186
|
+
) -> List[Dict[str, Any]]:
|
|
187
|
+
"""Identify using hash-based web search."""
|
|
188
|
+
candidates = []
|
|
189
|
+
|
|
190
|
+
# Scan for hashes
|
|
191
|
+
from ..core.config import SWHPIConfig
|
|
192
|
+
from ..core.swhid import SWHIDGenerator
|
|
193
|
+
config = SWHPIConfig(verbose=self.verbose, max_depth=1)
|
|
194
|
+
scanner = DirectoryScanner(config, SWHIDGenerator())
|
|
195
|
+
dir_candidates, _ = scanner.scan_recursive(path)
|
|
196
|
+
|
|
197
|
+
if dir_candidates:
|
|
198
|
+
# Extract hash from SWHID
|
|
199
|
+
swhid = dir_candidates[0].swhid
|
|
200
|
+
hash_value = swhid.split(":")[-1] if ":" in swhid else swhid
|
|
201
|
+
|
|
202
|
+
# Search for hash
|
|
203
|
+
urls = await self.hash_searcher.search_hash(hash_value)
|
|
204
|
+
|
|
205
|
+
for url in urls:
|
|
206
|
+
candidates.append({
|
|
207
|
+
"source": "hash_search",
|
|
208
|
+
"hash": hash_value,
|
|
209
|
+
"origin": url,
|
|
210
|
+
"confidence": 0.8
|
|
211
|
+
})
|
|
212
|
+
|
|
213
|
+
return candidates
|
|
214
|
+
|
|
215
|
+
async def _identify_via_scanoss(
|
|
216
|
+
self,
|
|
217
|
+
path: Path,
|
|
218
|
+
max_depth: int
|
|
219
|
+
) -> List[Dict[str, Any]]:
|
|
220
|
+
"""Identify using SCANOSS fingerprinting."""
|
|
221
|
+
candidates = []
|
|
222
|
+
|
|
223
|
+
# Initialize SCANOSS provider
|
|
224
|
+
scanoss = SCANOSSProvider(verbose=self.verbose)
|
|
225
|
+
|
|
226
|
+
try:
|
|
227
|
+
await scanoss.ensure_session()
|
|
228
|
+
|
|
229
|
+
# Use scan_file for individual files as scan_directory is not available
|
|
230
|
+
# in the consolidated provider
|
|
231
|
+
test_files = list(path.glob("**/*.c"))[:2] + list(path.glob("**/*.md"))[:1]
|
|
232
|
+
|
|
233
|
+
if not test_files:
|
|
234
|
+
test_files = list(path.iterdir())[:3]
|
|
235
|
+
|
|
236
|
+
for test_file in test_files:
|
|
237
|
+
if test_file.is_file():
|
|
238
|
+
result = await scanoss.scan_file(test_file)
|
|
239
|
+
# Parse SCANOSS results if available
|
|
240
|
+
if result and isinstance(result, dict):
|
|
241
|
+
for file_name, file_data in result.items():
|
|
242
|
+
if isinstance(file_data, list):
|
|
243
|
+
for match in file_data:
|
|
244
|
+
if isinstance(match, dict) and match.get("component"):
|
|
245
|
+
url = match.get("url", "")
|
|
246
|
+
if self._is_trusted_git_host(url):
|
|
247
|
+
# Ensure matched is numeric
|
|
248
|
+
matched_val = match.get("matched", 0)
|
|
249
|
+
if isinstance(matched_val, str):
|
|
250
|
+
try:
|
|
251
|
+
matched_val = float(matched_val)
|
|
252
|
+
except (ValueError, TypeError):
|
|
253
|
+
matched_val = 0
|
|
254
|
+
candidates.append({
|
|
255
|
+
"source": "scanoss",
|
|
256
|
+
"component": match.get("component", ""),
|
|
257
|
+
"origin": url,
|
|
258
|
+
"confidence": matched_val / 100.0 if matched_val else 0.5
|
|
259
|
+
})
|
|
260
|
+
finally:
|
|
261
|
+
await scanoss.close()
|
|
262
|
+
|
|
263
|
+
return candidates
|
|
264
|
+
|
|
265
|
+
def _is_trusted_git_host(self, url: str) -> bool:
|
|
266
|
+
"""
|
|
267
|
+
Check if URL belongs to a trusted Git hosting service.
|
|
268
|
+
|
|
269
|
+
Uses proper URL parsing to avoid substring attacks.
|
|
270
|
+
"""
|
|
271
|
+
if not url:
|
|
272
|
+
return False
|
|
273
|
+
|
|
274
|
+
try:
|
|
275
|
+
from urllib.parse import urlparse
|
|
276
|
+
parsed = urlparse(url)
|
|
277
|
+
|
|
278
|
+
# Check if hostname exactly matches or is a subdomain of trusted hosts
|
|
279
|
+
trusted_hosts = {
|
|
280
|
+
'github.com',
|
|
281
|
+
'gitlab.com',
|
|
282
|
+
'gitorious.org', # Found in our test results
|
|
283
|
+
'src.fedoraproject.org' # Found in our test results
|
|
284
|
+
}
|
|
285
|
+
|
|
286
|
+
hostname = parsed.hostname
|
|
287
|
+
if not hostname:
|
|
288
|
+
return False
|
|
289
|
+
|
|
290
|
+
# Exact match or subdomain of trusted host
|
|
291
|
+
hostname = hostname.lower()
|
|
292
|
+
for trusted in trusted_hosts:
|
|
293
|
+
if hostname == trusted or hostname.endswith('.' + trusted):
|
|
294
|
+
return True
|
|
295
|
+
|
|
296
|
+
return False
|
|
297
|
+
|
|
298
|
+
except Exception:
|
|
299
|
+
# If URL parsing fails, be conservative and reject
|
|
300
|
+
return False
|
|
301
|
+
|
|
302
|
+
async def _identify_via_web_search(
|
|
303
|
+
self,
|
|
304
|
+
path: Path,
|
|
305
|
+
max_depth: int
|
|
306
|
+
) -> List[Dict[str, Any]]:
|
|
307
|
+
"""Identify using web search providers."""
|
|
308
|
+
candidates = []
|
|
309
|
+
|
|
310
|
+
# Get project name from path
|
|
311
|
+
project_name = path.name
|
|
312
|
+
|
|
313
|
+
# Search using all available providers
|
|
314
|
+
search_results = await self.search_registry.search_all(
|
|
315
|
+
f'"{project_name}" repository site:github.com OR site:gitlab.com'
|
|
316
|
+
)
|
|
317
|
+
|
|
318
|
+
for provider, urls in search_results.items():
|
|
319
|
+
for url in urls:
|
|
320
|
+
candidates.append({
|
|
321
|
+
"source": f"web_search_{provider}",
|
|
322
|
+
"query": project_name,
|
|
323
|
+
"origin": url,
|
|
324
|
+
"confidence": 0.6
|
|
325
|
+
})
|
|
326
|
+
|
|
327
|
+
return candidates
|
|
328
|
+
|
|
329
|
+
def print_results(self, results: Dict[str, Any]):
|
|
330
|
+
"""Print identification results in a formatted table."""
|
|
331
|
+
table = Table(title="Source Identification Results")
|
|
332
|
+
table.add_column("Property", style="cyan")
|
|
333
|
+
table.add_column("Value", style="white")
|
|
334
|
+
|
|
335
|
+
table.add_row("Path", results["path"])
|
|
336
|
+
table.add_row("Identified", "✅ Yes" if results["identified"] else "❌ No")
|
|
337
|
+
table.add_row("Confidence", f"{results['confidence']:.1%}")
|
|
338
|
+
table.add_row("Strategies Used", ", ".join(results["strategies_used"]))
|
|
339
|
+
|
|
340
|
+
if results["final_origin"]:
|
|
341
|
+
table.add_row("Repository", results["final_origin"])
|
|
342
|
+
|
|
343
|
+
console.print(table)
|
|
344
|
+
|
|
345
|
+
if results["candidates"] and self.verbose:
|
|
346
|
+
console.print("\n[bold]Top Candidates:[/bold]")
|
|
347
|
+
for i, candidate in enumerate(results["candidates"][:5], 1):
|
|
348
|
+
console.print(f"{i}. {candidate.get('origin', 'Unknown')} "
|
|
349
|
+
f"(via {candidate.get('source', 'unknown')}, "
|
|
350
|
+
f"confidence: {candidate.get('confidence', 0):.1%})")
|
|
351
|
+
|
|
352
|
+
|
|
353
|
+
async def identify_source(
|
|
354
|
+
path: Path,
|
|
355
|
+
max_depth: int = 3,
|
|
356
|
+
confidence_threshold: float = 0.5,
|
|
357
|
+
verbose: bool = False,
|
|
358
|
+
strategies: Optional[List[str]] = None,
|
|
359
|
+
use_swh: bool = False
|
|
360
|
+
) -> Dict[str, Any]:
|
|
361
|
+
"""Convenience function for source identification.
|
|
362
|
+
|
|
363
|
+
Args:
|
|
364
|
+
path: Path to analyze
|
|
365
|
+
max_depth: Maximum depth for recursive scanning
|
|
366
|
+
confidence_threshold: Minimum confidence for identification
|
|
367
|
+
verbose: Enable verbose output
|
|
368
|
+
strategies: List of strategies to use
|
|
369
|
+
use_swh: Whether to include Software Heritage checking
|
|
370
|
+
|
|
371
|
+
Returns:
|
|
372
|
+
Identification results
|
|
373
|
+
"""
|
|
374
|
+
identifier = SourceIdentifier(verbose=verbose)
|
|
375
|
+
try:
|
|
376
|
+
results = await identifier.identify(
|
|
377
|
+
path=path,
|
|
378
|
+
max_depth=max_depth,
|
|
379
|
+
confidence_threshold=confidence_threshold,
|
|
380
|
+
strategies=strategies,
|
|
381
|
+
use_swh=use_swh
|
|
382
|
+
)
|
|
383
|
+
|
|
384
|
+
if verbose:
|
|
385
|
+
identifier.print_results(results)
|
|
386
|
+
|
|
387
|
+
return results
|
|
388
|
+
finally:
|
|
389
|
+
# Ensure cleanup of any open sessions
|
|
390
|
+
if hasattr(identifier, 'search_registry') and identifier.search_registry:
|
|
391
|
+
await identifier.search_registry.close_all()
|
src2id/utils/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""Utility functions for SWHPI."""
|
|
@@ -0,0 +1,49 @@
|
|
|
1
|
+
"""DateTime utility functions."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime
|
|
4
|
+
from typing import Any, Optional
|
|
5
|
+
|
|
6
|
+
|
|
7
|
+
def parse_datetime(date_str: Any) -> Optional[datetime]:
|
|
8
|
+
"""
|
|
9
|
+
Parse datetime from various formats.
|
|
10
|
+
|
|
11
|
+
Args:
|
|
12
|
+
date_str: Date string, timestamp, or datetime object
|
|
13
|
+
|
|
14
|
+
Returns:
|
|
15
|
+
Datetime object or None if parsing fails
|
|
16
|
+
"""
|
|
17
|
+
if date_str is None:
|
|
18
|
+
return None
|
|
19
|
+
|
|
20
|
+
if isinstance(date_str, datetime):
|
|
21
|
+
return date_str
|
|
22
|
+
|
|
23
|
+
if isinstance(date_str, (int, float)):
|
|
24
|
+
try:
|
|
25
|
+
return datetime.fromtimestamp(date_str)
|
|
26
|
+
except (ValueError, OSError):
|
|
27
|
+
return None
|
|
28
|
+
|
|
29
|
+
if isinstance(date_str, str):
|
|
30
|
+
# Try ISO format first (most common)
|
|
31
|
+
try:
|
|
32
|
+
return datetime.fromisoformat(date_str.replace('Z', '+00:00'))
|
|
33
|
+
except ValueError:
|
|
34
|
+
pass
|
|
35
|
+
|
|
36
|
+
# Try other common formats
|
|
37
|
+
formats = [
|
|
38
|
+
"%Y-%m-%d %H:%M:%S",
|
|
39
|
+
"%Y-%m-%dT%H:%M:%S",
|
|
40
|
+
"%Y-%m-%d",
|
|
41
|
+
]
|
|
42
|
+
|
|
43
|
+
for fmt in formats:
|
|
44
|
+
try:
|
|
45
|
+
return datetime.strptime(date_str, fmt)
|
|
46
|
+
except ValueError:
|
|
47
|
+
continue
|
|
48
|
+
|
|
49
|
+
return None
|