src2purl 1.3.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,206 @@
1
+ """Optimized hash-based search functionality."""
2
+
3
+ import hashlib
4
+ from pathlib import Path
5
+ from typing import Dict, List, Optional
6
+ import asyncio
7
+
8
+ from rich.console import Console
9
+
10
+ from .providers import SearchProviderRegistry
11
+
12
+ console = Console()
13
+
14
+
15
+ class HashSearcher:
16
+ """Optimized search for files by their content hashes."""
17
+
18
+ def __init__(
19
+ self,
20
+ search_registry: Optional[SearchProviderRegistry] = None,
21
+ verbose: bool = False
22
+ ):
23
+ """Initialize the hash searcher.
24
+
25
+ Args:
26
+ search_registry: Registry of search providers to use
27
+ verbose: Whether to show verbose output
28
+ """
29
+ self.verbose = verbose
30
+ self.search_registry = search_registry
31
+
32
+ def compute_file_hashes(self, file_path: Path) -> Dict[str, str]:
33
+ """Compute various hashes for a file.
34
+
35
+ Args:
36
+ file_path: Path to the file
37
+
38
+ Returns:
39
+ Dictionary of hash types to hash values
40
+ """
41
+ hashes = {}
42
+
43
+ try:
44
+ content = file_path.read_bytes()
45
+
46
+ # SHA1 (used by Git and SWH)
47
+ sha1 = hashlib.sha1(content).hexdigest()
48
+ hashes['sha1'] = sha1
49
+
50
+ # Git blob hash (includes header)
51
+ git_header = f"blob {len(content)}\0".encode()
52
+ git_content = git_header + content
53
+ git_sha1 = hashlib.sha1(git_content).hexdigest()
54
+ hashes['sha1_git'] = git_sha1
55
+
56
+ # SHA256 (modern standard)
57
+ sha256 = hashlib.sha256(content).hexdigest()
58
+ hashes['sha256'] = sha256
59
+
60
+ # MD5 (legacy, but still used)
61
+ md5 = hashlib.md5(content).hexdigest()
62
+ hashes['md5'] = md5
63
+
64
+ except Exception as e:
65
+ if self.verbose:
66
+ console.print(f"[yellow]Error computing hashes for {file_path}: {e}[/yellow]")
67
+
68
+ return hashes
69
+
70
+ def compute_directory_hash(self, directory: Path) -> str:
71
+ """Compute directory hash similar to SWH directory identifier.
72
+
73
+ Args:
74
+ directory: Path to directory
75
+
76
+ Returns:
77
+ Directory hash
78
+ """
79
+ entries = []
80
+
81
+ try:
82
+ for item in sorted(directory.iterdir()):
83
+ if item.name.startswith('.'):
84
+ continue
85
+
86
+ if item.is_file():
87
+ try:
88
+ content = item.read_bytes()
89
+ sha1 = hashlib.sha1(content).hexdigest()
90
+ entries.append(f"100644 {item.name}\0{bytes.fromhex(sha1)}")
91
+ except:
92
+ pass
93
+ elif item.is_dir() and not item.is_symlink():
94
+ # Recursive directory hash
95
+ dir_hash = self.compute_directory_hash(item)
96
+ if dir_hash:
97
+ entries.append(f"40000 {item.name}\0{bytes.fromhex(dir_hash)}")
98
+
99
+ if entries:
100
+ tree_content = b''.join(e.encode() if isinstance(e, str) else e for e in entries)
101
+ return hashlib.sha1(tree_content).hexdigest()
102
+
103
+ except Exception as e:
104
+ if self.verbose:
105
+ console.print(f"[yellow]Error computing directory hash: {e}[/yellow]")
106
+
107
+ return ""
108
+
109
+ async def search_hash(
110
+ self,
111
+ hash_value: str,
112
+ hash_type: str = "auto"
113
+ ) -> List[str]:
114
+ """Search for a hash value across all providers.
115
+
116
+ Args:
117
+ hash_value: The hash value to search for
118
+ hash_type: Type of hash (sha1, sha256, md5, auto)
119
+
120
+ Returns:
121
+ List of unique repository URLs found
122
+ """
123
+ if not self.search_registry:
124
+ return []
125
+
126
+ all_urls = set()
127
+
128
+ # Determine hash type if auto
129
+ if hash_type == "auto":
130
+ if len(hash_value) == 32:
131
+ hash_type = "md5"
132
+ elif len(hash_value) == 40:
133
+ hash_type = "sha1"
134
+ elif len(hash_value) == 64:
135
+ hash_type = "sha256"
136
+
137
+ # Build optimized search queries
138
+ queries = [
139
+ f'"{hash_value}" site:github.com',
140
+ f'"{hash_value}" site:gitlab.com',
141
+ f'"sha1_git:{hash_value}" site:archive.softwareheritage.org'
142
+ ]
143
+
144
+ # Execute searches in parallel
145
+ tasks = []
146
+ for query in queries:
147
+ if self.verbose:
148
+ console.print(f"[dim]Searching: {query}[/dim]")
149
+ tasks.append(self.search_registry.search_all(query))
150
+
151
+ try:
152
+ results = await asyncio.gather(*tasks, return_exceptions=True)
153
+
154
+ for result in results:
155
+ if isinstance(result, dict):
156
+ for provider, urls in result.items():
157
+ all_urls.update(urls)
158
+
159
+ except Exception as e:
160
+ if self.verbose:
161
+ console.print(f"[yellow]Search error: {e}[/yellow]")
162
+
163
+ return list(all_urls)
164
+
165
+ async def search_file(
166
+ self,
167
+ file_path: Path
168
+ ) -> Dict[str, List[str]]:
169
+ """Search for a file by computing and searching its hashes.
170
+
171
+ Args:
172
+ file_path: Path to the file
173
+
174
+ Returns:
175
+ Dictionary mapping hash types to lists of repository URLs
176
+ """
177
+ if not file_path.exists():
178
+ if self.verbose:
179
+ console.print(f"[red]File not found: {file_path}[/red]")
180
+ return {}
181
+
182
+ hashes = self.compute_file_hashes(file_path)
183
+
184
+ if self.verbose:
185
+ console.print(f"\n[bold]Computed hashes for {file_path.name}:[/bold]")
186
+ for hash_type, hash_value in hashes.items():
187
+ console.print(f" {hash_type}: {hash_value}")
188
+
189
+ results = {}
190
+
191
+ # Search for most relevant hashes in parallel
192
+ search_tasks = []
193
+ for hash_type in ['sha1_git', 'sha1']: # Prioritize git hashes
194
+ if hash_type in hashes:
195
+ search_tasks.append((hash_type, self.search_hash(hashes[hash_type], hash_type)))
196
+
197
+ if search_tasks:
198
+ task_results = await asyncio.gather(*[task for _, task in search_tasks])
199
+
200
+ for (hash_type, _), urls in zip(search_tasks, task_results):
201
+ if urls:
202
+ results[hash_type] = urls
203
+ if self.verbose:
204
+ console.print(f"[green]Found {len(urls)} results for {hash_type}[/green]")
205
+
206
+ return results
@@ -0,0 +1,310 @@
1
+ """Search providers for source identification."""
2
+
3
+ import os
4
+ import json
5
+ import hashlib
6
+ from abc import ABC, abstractmethod
7
+ from pathlib import Path
8
+ from typing import Dict, List, Optional, Any
9
+ import asyncio
10
+
11
+ import aiohttp
12
+ from rich.console import Console
13
+
14
+ console = Console()
15
+
16
+
17
+ class SearchProvider(ABC):
18
+ """Abstract base class for search providers."""
19
+
20
+ def __init__(self, api_key: Optional[str] = None, verbose: bool = False):
21
+ """Initialize the search provider."""
22
+ self.api_key = api_key
23
+ self.verbose = verbose
24
+ self.session = None
25
+
26
+ @property
27
+ @abstractmethod
28
+ def name(self) -> str:
29
+ """Return the name of the search provider."""
30
+ pass
31
+
32
+ @property
33
+ @abstractmethod
34
+ def requires_api_key(self) -> bool:
35
+ """Return whether this provider requires an API key."""
36
+ pass
37
+
38
+ @abstractmethod
39
+ async def search(self, query: str, **kwargs) -> List[str]:
40
+ """Perform a search and return repository URLs."""
41
+ pass
42
+
43
+ async def ensure_session(self):
44
+ """Ensure we have an active session."""
45
+ if not self.session:
46
+ self.session = aiohttp.ClientSession()
47
+
48
+ async def close(self):
49
+ """Close the session."""
50
+ if self.session:
51
+ await self.session.close()
52
+ self.session = None
53
+
54
+ def extract_repo_urls(self, urls: List[str]) -> List[str]:
55
+ """Extract and filter repository URLs from a list of URLs."""
56
+ from urllib.parse import urlparse
57
+
58
+ repo_urls = []
59
+ for url in urls:
60
+ try:
61
+ parsed = urlparse(url)
62
+ hostname = parsed.hostname.lower() if parsed.hostname else ''
63
+
64
+ if hostname in ('github.com', 'gitlab.com', 'bitbucket.org'):
65
+ # Clean up the URL - remove file-specific paths
66
+ if "/blob/" in url or "/tree/" in url:
67
+ if hostname in ('github.com', 'gitlab.com'):
68
+ path_parts = parsed.path.strip('/').split('/')
69
+ if len(path_parts) >= 2:
70
+ # Keep only owner/repo part
71
+ clean_path = f"/{path_parts[0]}/{path_parts[1]}"
72
+ clean_url = f"{parsed.scheme}://{hostname}{clean_path}"
73
+ repo_urls.append(clean_url)
74
+ else:
75
+ repo_urls.append(url)
76
+ except Exception:
77
+ # If URL parsing fails, skip this URL
78
+ continue
79
+ return list(set(repo_urls))
80
+
81
+
82
+ class SerpAPIProvider(SearchProvider):
83
+ """Search provider using SerpAPI (Google Search)."""
84
+
85
+ @property
86
+ def name(self) -> str:
87
+ return "SerpAPI"
88
+
89
+ @property
90
+ def requires_api_key(self) -> bool:
91
+ return True
92
+
93
+ async def search(self, query: str, **kwargs) -> List[str]:
94
+ """Search using SerpAPI."""
95
+ if not self.api_key:
96
+ return []
97
+
98
+ await self.ensure_session()
99
+
100
+ params = {
101
+ "q": query,
102
+ "api_key": self.api_key,
103
+ "engine": kwargs.get("engine", "google"),
104
+ "num": kwargs.get("num", 10)
105
+ }
106
+
107
+ try:
108
+ async with self.session.get("https://serpapi.com/search", params=params) as response:
109
+ if response.status == 200:
110
+ data = await response.json()
111
+ urls = [
112
+ result.get("link", "")
113
+ for result in data.get("organic_results", [])
114
+ if result.get("link")
115
+ ]
116
+ return self.extract_repo_urls(urls)
117
+ return []
118
+ except Exception as e:
119
+ if self.verbose:
120
+ console.print(f"[red]{self.name} error: {e}[/red]")
121
+ return []
122
+
123
+
124
+ class GitHubSearchProvider(SearchProvider):
125
+ """Search provider using GitHub's search API."""
126
+
127
+ @property
128
+ def name(self) -> str:
129
+ return "GitHub"
130
+
131
+ @property
132
+ def requires_api_key(self) -> bool:
133
+ return False # Optional
134
+
135
+ async def search(self, query: str, **kwargs) -> List[str]:
136
+ """Search using GitHub API."""
137
+ await self.ensure_session()
138
+
139
+ headers = {"Accept": "application/vnd.github.v3+json"}
140
+ if self.api_key:
141
+ headers["Authorization"] = f"token {self.api_key}"
142
+
143
+ # Extract search terms
144
+ import re
145
+ search_terms = re.findall(r'"([^"]+)"', query)
146
+ if not search_terms:
147
+ search_terms = [query]
148
+
149
+ urls = []
150
+ for term in search_terms[:1]: # Limit to avoid rate limits
151
+ try:
152
+ params = {"q": term[:100], "per_page": 5, "sort": "stars"}
153
+ async with self.session.get(
154
+ "https://api.github.com/search/repositories",
155
+ params=params,
156
+ headers=headers
157
+ ) as response:
158
+ if response.status == 200:
159
+ data = await response.json()
160
+ urls.extend([
161
+ item.get("html_url", "")
162
+ for item in data.get("items", [])
163
+ if item.get("html_url")
164
+ ])
165
+ except Exception:
166
+ pass
167
+
168
+ return self.extract_repo_urls(urls)
169
+
170
+
171
+ class SourcegraphProvider(SearchProvider):
172
+ """Search provider using Sourcegraph code search."""
173
+
174
+ @property
175
+ def name(self) -> str:
176
+ return "Sourcegraph"
177
+
178
+ @property
179
+ def requires_api_key(self) -> bool:
180
+ return False
181
+
182
+ async def search(self, query: str, **kwargs) -> List[str]:
183
+ """Search using Sourcegraph (simplified implementation)."""
184
+ # Note: Full implementation would use Sourcegraph API
185
+ # This is a placeholder that returns empty results
186
+ return []
187
+
188
+
189
+ class SCANOSSProvider(SearchProvider):
190
+ """Provider for SCANOSS code identification service."""
191
+
192
+ DEFAULT_URL = 'https://api.osskb.org/scan/direct'
193
+ PREMIUM_URL = 'https://api.scanoss.com/scan/direct'
194
+
195
+ def __init__(self, api_key: Optional[str] = None, verbose: bool = False):
196
+ """Initialize SCANOSS provider."""
197
+ super().__init__(api_key, verbose)
198
+ self.url = self.PREMIUM_URL if api_key else self.DEFAULT_URL
199
+
200
+ @property
201
+ def name(self) -> str:
202
+ return "SCANOSS"
203
+
204
+ @property
205
+ def requires_api_key(self) -> bool:
206
+ return False # Works without key
207
+
208
+ async def search(self, query: str, **kwargs) -> List[str]:
209
+ """SCANOSS doesn't do text search - use scan_file instead."""
210
+ return []
211
+
212
+ async def scan_file(self, file_path: Path) -> Dict:
213
+ """Scan a file using SCANOSS winnowing algorithm."""
214
+ await self.ensure_session()
215
+
216
+ try:
217
+ content = file_path.read_bytes()
218
+ wfp = self._create_wfp(file_path, content)
219
+
220
+ headers = {'User-Agent': 'swhpi-scanner/1.0'}
221
+ if self.api_key:
222
+ headers['X-Session'] = self.api_key
223
+
224
+ form_data = aiohttp.FormData()
225
+ form_data.add_field('file', wfp, filename='scan.wfp')
226
+ form_data.add_field('format', 'plain')
227
+
228
+ async with self.session.post(
229
+ self.url,
230
+ data=form_data,
231
+ headers=headers,
232
+ timeout=aiohttp.ClientTimeout(total=30)
233
+ ) as response:
234
+ if response.status == 200:
235
+ result_text = await response.text()
236
+ return json.loads(result_text)
237
+ return {}
238
+ except Exception as e:
239
+ if self.verbose:
240
+ console.print(f"[red]SCANOSS error: {e}[/red]")
241
+ return {}
242
+
243
+ def _create_wfp(self, file_path: Path, content: bytes) -> str:
244
+ """Create simplified WFP for SCANOSS."""
245
+ md5_hash = hashlib.md5(content).hexdigest()
246
+ return f"file={md5_hash},{len(content)},{file_path.name}\n1=00000000"
247
+
248
+
249
+ class SearchProviderRegistry:
250
+ """Registry for managing search providers."""
251
+
252
+ def __init__(self, verbose: bool = False):
253
+ """Initialize the registry."""
254
+ self.providers: Dict[str, SearchProvider] = {}
255
+ self.verbose = verbose
256
+
257
+ def register_provider(self, name: str, provider: SearchProvider):
258
+ """Register a search provider."""
259
+ self.providers[name] = provider
260
+
261
+ def get_provider(self, name: str) -> Optional[SearchProvider]:
262
+ """Get a provider by name."""
263
+ return self.providers.get(name)
264
+
265
+ async def search_all(self, query: str, **kwargs) -> Dict[str, List[str]]:
266
+ """Search using all registered providers."""
267
+ results = {}
268
+ for name, provider in self.providers.items():
269
+ if provider.requires_api_key and not provider.api_key:
270
+ continue
271
+ try:
272
+ urls = await provider.search(query, **kwargs)
273
+ if urls:
274
+ results[name] = urls
275
+ except Exception:
276
+ pass
277
+ return results
278
+
279
+ async def close_all(self):
280
+ """Close all provider sessions."""
281
+ for provider in self.providers.values():
282
+ await provider.close()
283
+
284
+
285
+ def create_default_registry(verbose: bool = False) -> SearchProviderRegistry:
286
+ """Create a registry with default providers configured from environment."""
287
+ registry = SearchProviderRegistry(verbose=verbose)
288
+
289
+ # SerpAPI
290
+ if serpapi_key := os.environ.get("SERPAPI_KEY"):
291
+ registry.register_provider(
292
+ "serpapi",
293
+ SerpAPIProvider(api_key=serpapi_key, verbose=verbose)
294
+ )
295
+
296
+ # GitHub
297
+ github_token = os.environ.get("GITHUB_TOKEN")
298
+ registry.register_provider(
299
+ "github",
300
+ GitHubSearchProvider(api_key=github_token, verbose=verbose)
301
+ )
302
+
303
+ # SCANOSS
304
+ scanoss_key = os.environ.get("SCANOSS_API_KEY")
305
+ registry.register_provider(
306
+ "scanoss",
307
+ SCANOSSProvider(api_key=scanoss_key, verbose=verbose)
308
+ )
309
+
310
+ return registry