src2purl 1.3.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,391 @@
1
+ """Unified source identification strategies."""
2
+
3
+ import asyncio
4
+ import os
5
+ from pathlib import Path
6
+ from typing import Dict, List, Optional, Tuple, Any
7
+ from collections import Counter
8
+
9
+ from rich.console import Console
10
+ from rich.table import Table
11
+
12
+ from ..core.models import DirectoryCandidate, ContentCandidate
13
+ from ..core.scanner import DirectoryScanner
14
+ from ..core.client import SoftwareHeritageClient
15
+ from .hash_search import HashSearcher
16
+ from .providers import (
17
+ SearchProviderRegistry,
18
+ create_default_registry,
19
+ SCANOSSProvider
20
+ )
21
+
22
+ console = Console()
23
+
24
+
25
+ class SourceIdentifier:
26
+ """Unified source identification using multiple strategies."""
27
+
28
+ def __init__(
29
+ self,
30
+ swh_client: Optional[SoftwareHeritageClient] = None,
31
+ search_registry: Optional[SearchProviderRegistry] = None,
32
+ verbose: bool = False
33
+ ):
34
+ """Initialize the source identifier.
35
+
36
+ Args:
37
+ swh_client: Software Heritage client instance
38
+ search_registry: Registry of search providers
39
+ verbose: Enable verbose output
40
+ """
41
+ # Only store the client if provided, don't create it automatically
42
+ self.swh_client = swh_client
43
+ self.search_registry = search_registry or create_default_registry(verbose=verbose)
44
+ self.verbose = verbose
45
+ self.hash_searcher = HashSearcher(verbose=verbose)
46
+ self._swh_config = None # Store config for lazy initialization
47
+
48
+ async def identify(
49
+ self,
50
+ path: Path,
51
+ max_depth: int = 3,
52
+ confidence_threshold: float = 0.5,
53
+ strategies: Optional[List[str]] = None,
54
+ use_swh: bool = False
55
+ ) -> Dict[str, Any]:
56
+ """Identify the source of a directory using multiple strategies.
57
+
58
+ Args:
59
+ path: Path to analyze
60
+ max_depth: Maximum depth for recursive scanning
61
+ confidence_threshold: Minimum confidence for identification
62
+ strategies: List of strategies to use (default: optimized order)
63
+ use_swh: Whether to include Software Heritage checking
64
+
65
+ Returns:
66
+ Dictionary with identification results
67
+ """
68
+ results = {
69
+ "path": str(path),
70
+ "identified": False,
71
+ "confidence": 0.0,
72
+ "strategies_used": [],
73
+ "candidates": [],
74
+ "final_origin": None
75
+ }
76
+
77
+ available_strategies = {
78
+ "hash_search": self._identify_via_hash_search,
79
+ "web_search": self._identify_via_web_search,
80
+ "scanoss": self._identify_via_scanoss,
81
+ "swh": self._identify_via_swh
82
+ }
83
+
84
+ # Default optimized order (local methods first, then external APIs)
85
+ if strategies is None:
86
+ strategies_to_use = ["hash_search", "web_search", "scanoss"]
87
+ if use_swh:
88
+ strategies_to_use.append("swh")
89
+ else:
90
+ strategies_to_use = strategies
91
+
92
+ all_candidates = []
93
+
94
+ for strategy_name in strategies_to_use:
95
+ if strategy_name not in available_strategies:
96
+ continue
97
+
98
+ strategy_func = available_strategies[strategy_name]
99
+
100
+ try:
101
+ if self.verbose:
102
+ console.print(f"[cyan]Running {strategy_name} strategy...[/cyan]")
103
+
104
+ candidates = await strategy_func(path, max_depth)
105
+
106
+ if candidates:
107
+ all_candidates.extend(candidates)
108
+ results["strategies_used"].append(strategy_name)
109
+
110
+ if self.verbose:
111
+ console.print(f"[green]✓ {strategy_name} found {len(candidates)} candidates[/green]")
112
+
113
+ except Exception as e:
114
+ if self.verbose:
115
+ console.print(f"[yellow]⚠ {strategy_name} failed: {e}[/yellow]")
116
+
117
+ # Clean up any open sessions
118
+ if hasattr(self, 'search_registry') and self.search_registry:
119
+ await self.search_registry.close_all()
120
+
121
+ # Aggregate and score candidates
122
+ if all_candidates:
123
+ origin_scores = Counter()
124
+ for candidate in all_candidates:
125
+ if "origin" in candidate:
126
+ origin_scores[candidate["origin"]] += candidate.get("confidence", 1.0)
127
+
128
+ if origin_scores:
129
+ best_origin, best_score = origin_scores.most_common(1)[0]
130
+ total_strategies = len(results["strategies_used"])
131
+ confidence = best_score / total_strategies if total_strategies > 0 else 0
132
+
133
+ if confidence >= confidence_threshold:
134
+ results["identified"] = True
135
+ results["confidence"] = confidence
136
+ results["final_origin"] = best_origin
137
+ results["candidates"] = all_candidates[:10] # Top 10 candidates
138
+
139
+ return results
140
+
141
+ async def _identify_via_swh(
142
+ self,
143
+ path: Path,
144
+ max_depth: int
145
+ ) -> List[Dict[str, Any]]:
146
+ """Identify using Software Heritage archive."""
147
+ candidates = []
148
+
149
+ # Lazily create SWH client only when needed
150
+ if self.swh_client is None:
151
+ from ..core.config import SWHPIConfig
152
+ config = SWHPIConfig(verbose=self.verbose)
153
+ self.swh_client = SoftwareHeritageClient(config)
154
+
155
+ # Scan directory
156
+ from ..core.config import SWHPIConfig
157
+ from ..core.swhid import SWHIDGenerator
158
+ config = SWHPIConfig(verbose=self.verbose, max_depth=max_depth)
159
+ scanner = DirectoryScanner(config, SWHIDGenerator())
160
+ dir_candidates, file_candidates = scanner.scan_recursive(path)
161
+
162
+ # Check SWHIDs
163
+ all_swhids = [c.swhid for c in dir_candidates] + [c.swhid for c in file_candidates]
164
+
165
+ if all_swhids:
166
+ known_swhids = await self.swh_client.check_swhids_known(all_swhids[:100])
167
+
168
+ for swhid, is_known in known_swhids.items():
169
+ if is_known:
170
+ # Try to get origin information
171
+ origin = await self.swh_client.get_origin_for_swhid(swhid)
172
+ if origin:
173
+ candidates.append({
174
+ "source": "swh",
175
+ "swhid": swhid,
176
+ "origin": origin,
177
+ "confidence": 1.0
178
+ })
179
+
180
+ return candidates
181
+
182
+ async def _identify_via_hash_search(
183
+ self,
184
+ path: Path,
185
+ max_depth: int
186
+ ) -> List[Dict[str, Any]]:
187
+ """Identify using hash-based web search."""
188
+ candidates = []
189
+
190
+ # Scan for hashes
191
+ from ..core.config import SWHPIConfig
192
+ from ..core.swhid import SWHIDGenerator
193
+ config = SWHPIConfig(verbose=self.verbose, max_depth=1)
194
+ scanner = DirectoryScanner(config, SWHIDGenerator())
195
+ dir_candidates, _ = scanner.scan_recursive(path)
196
+
197
+ if dir_candidates:
198
+ # Extract hash from SWHID
199
+ swhid = dir_candidates[0].swhid
200
+ hash_value = swhid.split(":")[-1] if ":" in swhid else swhid
201
+
202
+ # Search for hash
203
+ urls = await self.hash_searcher.search_hash(hash_value)
204
+
205
+ for url in urls:
206
+ candidates.append({
207
+ "source": "hash_search",
208
+ "hash": hash_value,
209
+ "origin": url,
210
+ "confidence": 0.8
211
+ })
212
+
213
+ return candidates
214
+
215
+ async def _identify_via_scanoss(
216
+ self,
217
+ path: Path,
218
+ max_depth: int
219
+ ) -> List[Dict[str, Any]]:
220
+ """Identify using SCANOSS fingerprinting."""
221
+ candidates = []
222
+
223
+ # Initialize SCANOSS provider
224
+ scanoss = SCANOSSProvider(verbose=self.verbose)
225
+
226
+ try:
227
+ await scanoss.ensure_session()
228
+
229
+ # Use scan_file for individual files as scan_directory is not available
230
+ # in the consolidated provider
231
+ test_files = list(path.glob("**/*.c"))[:2] + list(path.glob("**/*.md"))[:1]
232
+
233
+ if not test_files:
234
+ test_files = list(path.iterdir())[:3]
235
+
236
+ for test_file in test_files:
237
+ if test_file.is_file():
238
+ result = await scanoss.scan_file(test_file)
239
+ # Parse SCANOSS results if available
240
+ if result and isinstance(result, dict):
241
+ for file_name, file_data in result.items():
242
+ if isinstance(file_data, list):
243
+ for match in file_data:
244
+ if isinstance(match, dict) and match.get("component"):
245
+ url = match.get("url", "")
246
+ if self._is_trusted_git_host(url):
247
+ # Ensure matched is numeric
248
+ matched_val = match.get("matched", 0)
249
+ if isinstance(matched_val, str):
250
+ try:
251
+ matched_val = float(matched_val)
252
+ except (ValueError, TypeError):
253
+ matched_val = 0
254
+ candidates.append({
255
+ "source": "scanoss",
256
+ "component": match.get("component", ""),
257
+ "origin": url,
258
+ "confidence": matched_val / 100.0 if matched_val else 0.5
259
+ })
260
+ finally:
261
+ await scanoss.close()
262
+
263
+ return candidates
264
+
265
+ def _is_trusted_git_host(self, url: str) -> bool:
266
+ """
267
+ Check if URL belongs to a trusted Git hosting service.
268
+
269
+ Uses proper URL parsing to avoid substring attacks.
270
+ """
271
+ if not url:
272
+ return False
273
+
274
+ try:
275
+ from urllib.parse import urlparse
276
+ parsed = urlparse(url)
277
+
278
+ # Check if hostname exactly matches or is a subdomain of trusted hosts
279
+ trusted_hosts = {
280
+ 'github.com',
281
+ 'gitlab.com',
282
+ 'gitorious.org', # Found in our test results
283
+ 'src.fedoraproject.org' # Found in our test results
284
+ }
285
+
286
+ hostname = parsed.hostname
287
+ if not hostname:
288
+ return False
289
+
290
+ # Exact match or subdomain of trusted host
291
+ hostname = hostname.lower()
292
+ for trusted in trusted_hosts:
293
+ if hostname == trusted or hostname.endswith('.' + trusted):
294
+ return True
295
+
296
+ return False
297
+
298
+ except Exception:
299
+ # If URL parsing fails, be conservative and reject
300
+ return False
301
+
302
+ async def _identify_via_web_search(
303
+ self,
304
+ path: Path,
305
+ max_depth: int
306
+ ) -> List[Dict[str, Any]]:
307
+ """Identify using web search providers."""
308
+ candidates = []
309
+
310
+ # Get project name from path
311
+ project_name = path.name
312
+
313
+ # Search using all available providers
314
+ search_results = await self.search_registry.search_all(
315
+ f'"{project_name}" repository site:github.com OR site:gitlab.com'
316
+ )
317
+
318
+ for provider, urls in search_results.items():
319
+ for url in urls:
320
+ candidates.append({
321
+ "source": f"web_search_{provider}",
322
+ "query": project_name,
323
+ "origin": url,
324
+ "confidence": 0.6
325
+ })
326
+
327
+ return candidates
328
+
329
+ def print_results(self, results: Dict[str, Any]):
330
+ """Print identification results in a formatted table."""
331
+ table = Table(title="Source Identification Results")
332
+ table.add_column("Property", style="cyan")
333
+ table.add_column("Value", style="white")
334
+
335
+ table.add_row("Path", results["path"])
336
+ table.add_row("Identified", "✅ Yes" if results["identified"] else "❌ No")
337
+ table.add_row("Confidence", f"{results['confidence']:.1%}")
338
+ table.add_row("Strategies Used", ", ".join(results["strategies_used"]))
339
+
340
+ if results["final_origin"]:
341
+ table.add_row("Repository", results["final_origin"])
342
+
343
+ console.print(table)
344
+
345
+ if results["candidates"] and self.verbose:
346
+ console.print("\n[bold]Top Candidates:[/bold]")
347
+ for i, candidate in enumerate(results["candidates"][:5], 1):
348
+ console.print(f"{i}. {candidate.get('origin', 'Unknown')} "
349
+ f"(via {candidate.get('source', 'unknown')}, "
350
+ f"confidence: {candidate.get('confidence', 0):.1%})")
351
+
352
+
353
+ async def identify_source(
354
+ path: Path,
355
+ max_depth: int = 3,
356
+ confidence_threshold: float = 0.5,
357
+ verbose: bool = False,
358
+ strategies: Optional[List[str]] = None,
359
+ use_swh: bool = False
360
+ ) -> Dict[str, Any]:
361
+ """Convenience function for source identification.
362
+
363
+ Args:
364
+ path: Path to analyze
365
+ max_depth: Maximum depth for recursive scanning
366
+ confidence_threshold: Minimum confidence for identification
367
+ verbose: Enable verbose output
368
+ strategies: List of strategies to use
369
+ use_swh: Whether to include Software Heritage checking
370
+
371
+ Returns:
372
+ Identification results
373
+ """
374
+ identifier = SourceIdentifier(verbose=verbose)
375
+ try:
376
+ results = await identifier.identify(
377
+ path=path,
378
+ max_depth=max_depth,
379
+ confidence_threshold=confidence_threshold,
380
+ strategies=strategies,
381
+ use_swh=use_swh
382
+ )
383
+
384
+ if verbose:
385
+ identifier.print_results(results)
386
+
387
+ return results
388
+ finally:
389
+ # Ensure cleanup of any open sessions
390
+ if hasattr(identifier, 'search_registry') and identifier.search_registry:
391
+ await identifier.search_registry.close_all()
@@ -0,0 +1 @@
1
+ """Utility functions for SWHPI."""
@@ -0,0 +1,49 @@
1
+ """DateTime utility functions."""
2
+
3
+ from datetime import datetime
4
+ from typing import Any, Optional
5
+
6
+
7
+ def parse_datetime(date_str: Any) -> Optional[datetime]:
8
+ """
9
+ Parse datetime from various formats.
10
+
11
+ Args:
12
+ date_str: Date string, timestamp, or datetime object
13
+
14
+ Returns:
15
+ Datetime object or None if parsing fails
16
+ """
17
+ if date_str is None:
18
+ return None
19
+
20
+ if isinstance(date_str, datetime):
21
+ return date_str
22
+
23
+ if isinstance(date_str, (int, float)):
24
+ try:
25
+ return datetime.fromtimestamp(date_str)
26
+ except (ValueError, OSError):
27
+ return None
28
+
29
+ if isinstance(date_str, str):
30
+ # Try ISO format first (most common)
31
+ try:
32
+ return datetime.fromisoformat(date_str.replace('Z', '+00:00'))
33
+ except ValueError:
34
+ pass
35
+
36
+ # Try other common formats
37
+ formats = [
38
+ "%Y-%m-%d %H:%M:%S",
39
+ "%Y-%m-%dT%H:%M:%S",
40
+ "%Y-%m-%d",
41
+ ]
42
+
43
+ for fmt in formats:
44
+ try:
45
+ return datetime.strptime(date_str, fmt)
46
+ except ValueError:
47
+ continue
48
+
49
+ return None