src2purl 1.3.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- src2id/__init__.py +15 -0
- src2id/cli/__init__.py +1 -0
- src2id/cli/main.py +318 -0
- src2id/cli/validate.py +93 -0
- src2id/core/__init__.py +1 -0
- src2id/core/cache.py +180 -0
- src2id/core/client.py +370 -0
- src2id/core/config.py +45 -0
- src2id/core/extractor.py +302 -0
- src2id/core/models.py +93 -0
- src2id/core/orchestrator.py +796 -0
- src2id/core/package_identifier.py +123 -0
- src2id/core/purl.py +238 -0
- src2id/core/scanner.py +369 -0
- src2id/core/scorer.py +217 -0
- src2id/core/subcomponent_detector.py +353 -0
- src2id/core/swhid.py +324 -0
- src2id/integrations/__init__.py +1 -0
- src2id/integrations/manifest_parser.py +652 -0
- src2id/integrations/oslili.py +228 -0
- src2id/integrations/upmex.py +305 -0
- src2id/search/__init__.py +34 -0
- src2id/search/hash_search.py +206 -0
- src2id/search/providers.py +310 -0
- src2id/search/strategies.py +391 -0
- src2id/utils/__init__.py +1 -0
- src2id/utils/datetime_utils.py +49 -0
- src2purl-1.3.2.dist-info/METADATA +279 -0
- src2purl-1.3.2.dist-info/RECORD +33 -0
- src2purl-1.3.2.dist-info/WHEEL +5 -0
- src2purl-1.3.2.dist-info/entry_points.txt +3 -0
- src2purl-1.3.2.dist-info/licenses/LICENSE +661 -0
- src2purl-1.3.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
"""Optimized hash-based search functionality."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Dict, List, Optional
|
|
6
|
+
import asyncio
|
|
7
|
+
|
|
8
|
+
from rich.console import Console
|
|
9
|
+
|
|
10
|
+
from .providers import SearchProviderRegistry
|
|
11
|
+
|
|
12
|
+
console = Console()
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
class HashSearcher:
|
|
16
|
+
"""Optimized search for files by their content hashes."""
|
|
17
|
+
|
|
18
|
+
def __init__(
|
|
19
|
+
self,
|
|
20
|
+
search_registry: Optional[SearchProviderRegistry] = None,
|
|
21
|
+
verbose: bool = False
|
|
22
|
+
):
|
|
23
|
+
"""Initialize the hash searcher.
|
|
24
|
+
|
|
25
|
+
Args:
|
|
26
|
+
search_registry: Registry of search providers to use
|
|
27
|
+
verbose: Whether to show verbose output
|
|
28
|
+
"""
|
|
29
|
+
self.verbose = verbose
|
|
30
|
+
self.search_registry = search_registry
|
|
31
|
+
|
|
32
|
+
def compute_file_hashes(self, file_path: Path) -> Dict[str, str]:
|
|
33
|
+
"""Compute various hashes for a file.
|
|
34
|
+
|
|
35
|
+
Args:
|
|
36
|
+
file_path: Path to the file
|
|
37
|
+
|
|
38
|
+
Returns:
|
|
39
|
+
Dictionary of hash types to hash values
|
|
40
|
+
"""
|
|
41
|
+
hashes = {}
|
|
42
|
+
|
|
43
|
+
try:
|
|
44
|
+
content = file_path.read_bytes()
|
|
45
|
+
|
|
46
|
+
# SHA1 (used by Git and SWH)
|
|
47
|
+
sha1 = hashlib.sha1(content).hexdigest()
|
|
48
|
+
hashes['sha1'] = sha1
|
|
49
|
+
|
|
50
|
+
# Git blob hash (includes header)
|
|
51
|
+
git_header = f"blob {len(content)}\0".encode()
|
|
52
|
+
git_content = git_header + content
|
|
53
|
+
git_sha1 = hashlib.sha1(git_content).hexdigest()
|
|
54
|
+
hashes['sha1_git'] = git_sha1
|
|
55
|
+
|
|
56
|
+
# SHA256 (modern standard)
|
|
57
|
+
sha256 = hashlib.sha256(content).hexdigest()
|
|
58
|
+
hashes['sha256'] = sha256
|
|
59
|
+
|
|
60
|
+
# MD5 (legacy, but still used)
|
|
61
|
+
md5 = hashlib.md5(content).hexdigest()
|
|
62
|
+
hashes['md5'] = md5
|
|
63
|
+
|
|
64
|
+
except Exception as e:
|
|
65
|
+
if self.verbose:
|
|
66
|
+
console.print(f"[yellow]Error computing hashes for {file_path}: {e}[/yellow]")
|
|
67
|
+
|
|
68
|
+
return hashes
|
|
69
|
+
|
|
70
|
+
def compute_directory_hash(self, directory: Path) -> str:
|
|
71
|
+
"""Compute directory hash similar to SWH directory identifier.
|
|
72
|
+
|
|
73
|
+
Args:
|
|
74
|
+
directory: Path to directory
|
|
75
|
+
|
|
76
|
+
Returns:
|
|
77
|
+
Directory hash
|
|
78
|
+
"""
|
|
79
|
+
entries = []
|
|
80
|
+
|
|
81
|
+
try:
|
|
82
|
+
for item in sorted(directory.iterdir()):
|
|
83
|
+
if item.name.startswith('.'):
|
|
84
|
+
continue
|
|
85
|
+
|
|
86
|
+
if item.is_file():
|
|
87
|
+
try:
|
|
88
|
+
content = item.read_bytes()
|
|
89
|
+
sha1 = hashlib.sha1(content).hexdigest()
|
|
90
|
+
entries.append(f"100644 {item.name}\0{bytes.fromhex(sha1)}")
|
|
91
|
+
except:
|
|
92
|
+
pass
|
|
93
|
+
elif item.is_dir() and not item.is_symlink():
|
|
94
|
+
# Recursive directory hash
|
|
95
|
+
dir_hash = self.compute_directory_hash(item)
|
|
96
|
+
if dir_hash:
|
|
97
|
+
entries.append(f"40000 {item.name}\0{bytes.fromhex(dir_hash)}")
|
|
98
|
+
|
|
99
|
+
if entries:
|
|
100
|
+
tree_content = b''.join(e.encode() if isinstance(e, str) else e for e in entries)
|
|
101
|
+
return hashlib.sha1(tree_content).hexdigest()
|
|
102
|
+
|
|
103
|
+
except Exception as e:
|
|
104
|
+
if self.verbose:
|
|
105
|
+
console.print(f"[yellow]Error computing directory hash: {e}[/yellow]")
|
|
106
|
+
|
|
107
|
+
return ""
|
|
108
|
+
|
|
109
|
+
async def search_hash(
|
|
110
|
+
self,
|
|
111
|
+
hash_value: str,
|
|
112
|
+
hash_type: str = "auto"
|
|
113
|
+
) -> List[str]:
|
|
114
|
+
"""Search for a hash value across all providers.
|
|
115
|
+
|
|
116
|
+
Args:
|
|
117
|
+
hash_value: The hash value to search for
|
|
118
|
+
hash_type: Type of hash (sha1, sha256, md5, auto)
|
|
119
|
+
|
|
120
|
+
Returns:
|
|
121
|
+
List of unique repository URLs found
|
|
122
|
+
"""
|
|
123
|
+
if not self.search_registry:
|
|
124
|
+
return []
|
|
125
|
+
|
|
126
|
+
all_urls = set()
|
|
127
|
+
|
|
128
|
+
# Determine hash type if auto
|
|
129
|
+
if hash_type == "auto":
|
|
130
|
+
if len(hash_value) == 32:
|
|
131
|
+
hash_type = "md5"
|
|
132
|
+
elif len(hash_value) == 40:
|
|
133
|
+
hash_type = "sha1"
|
|
134
|
+
elif len(hash_value) == 64:
|
|
135
|
+
hash_type = "sha256"
|
|
136
|
+
|
|
137
|
+
# Build optimized search queries
|
|
138
|
+
queries = [
|
|
139
|
+
f'"{hash_value}" site:github.com',
|
|
140
|
+
f'"{hash_value}" site:gitlab.com',
|
|
141
|
+
f'"sha1_git:{hash_value}" site:archive.softwareheritage.org'
|
|
142
|
+
]
|
|
143
|
+
|
|
144
|
+
# Execute searches in parallel
|
|
145
|
+
tasks = []
|
|
146
|
+
for query in queries:
|
|
147
|
+
if self.verbose:
|
|
148
|
+
console.print(f"[dim]Searching: {query}[/dim]")
|
|
149
|
+
tasks.append(self.search_registry.search_all(query))
|
|
150
|
+
|
|
151
|
+
try:
|
|
152
|
+
results = await asyncio.gather(*tasks, return_exceptions=True)
|
|
153
|
+
|
|
154
|
+
for result in results:
|
|
155
|
+
if isinstance(result, dict):
|
|
156
|
+
for provider, urls in result.items():
|
|
157
|
+
all_urls.update(urls)
|
|
158
|
+
|
|
159
|
+
except Exception as e:
|
|
160
|
+
if self.verbose:
|
|
161
|
+
console.print(f"[yellow]Search error: {e}[/yellow]")
|
|
162
|
+
|
|
163
|
+
return list(all_urls)
|
|
164
|
+
|
|
165
|
+
async def search_file(
|
|
166
|
+
self,
|
|
167
|
+
file_path: Path
|
|
168
|
+
) -> Dict[str, List[str]]:
|
|
169
|
+
"""Search for a file by computing and searching its hashes.
|
|
170
|
+
|
|
171
|
+
Args:
|
|
172
|
+
file_path: Path to the file
|
|
173
|
+
|
|
174
|
+
Returns:
|
|
175
|
+
Dictionary mapping hash types to lists of repository URLs
|
|
176
|
+
"""
|
|
177
|
+
if not file_path.exists():
|
|
178
|
+
if self.verbose:
|
|
179
|
+
console.print(f"[red]File not found: {file_path}[/red]")
|
|
180
|
+
return {}
|
|
181
|
+
|
|
182
|
+
hashes = self.compute_file_hashes(file_path)
|
|
183
|
+
|
|
184
|
+
if self.verbose:
|
|
185
|
+
console.print(f"\n[bold]Computed hashes for {file_path.name}:[/bold]")
|
|
186
|
+
for hash_type, hash_value in hashes.items():
|
|
187
|
+
console.print(f" {hash_type}: {hash_value}")
|
|
188
|
+
|
|
189
|
+
results = {}
|
|
190
|
+
|
|
191
|
+
# Search for most relevant hashes in parallel
|
|
192
|
+
search_tasks = []
|
|
193
|
+
for hash_type in ['sha1_git', 'sha1']: # Prioritize git hashes
|
|
194
|
+
if hash_type in hashes:
|
|
195
|
+
search_tasks.append((hash_type, self.search_hash(hashes[hash_type], hash_type)))
|
|
196
|
+
|
|
197
|
+
if search_tasks:
|
|
198
|
+
task_results = await asyncio.gather(*[task for _, task in search_tasks])
|
|
199
|
+
|
|
200
|
+
for (hash_type, _), urls in zip(search_tasks, task_results):
|
|
201
|
+
if urls:
|
|
202
|
+
results[hash_type] = urls
|
|
203
|
+
if self.verbose:
|
|
204
|
+
console.print(f"[green]Found {len(urls)} results for {hash_type}[/green]")
|
|
205
|
+
|
|
206
|
+
return results
|
|
@@ -0,0 +1,310 @@
|
|
|
1
|
+
"""Search providers for source identification."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
import json
|
|
5
|
+
import hashlib
|
|
6
|
+
from abc import ABC, abstractmethod
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Dict, List, Optional, Any
|
|
9
|
+
import asyncio
|
|
10
|
+
|
|
11
|
+
import aiohttp
|
|
12
|
+
from rich.console import Console
|
|
13
|
+
|
|
14
|
+
console = Console()
|
|
15
|
+
|
|
16
|
+
|
|
17
|
+
class SearchProvider(ABC):
|
|
18
|
+
"""Abstract base class for search providers."""
|
|
19
|
+
|
|
20
|
+
def __init__(self, api_key: Optional[str] = None, verbose: bool = False):
|
|
21
|
+
"""Initialize the search provider."""
|
|
22
|
+
self.api_key = api_key
|
|
23
|
+
self.verbose = verbose
|
|
24
|
+
self.session = None
|
|
25
|
+
|
|
26
|
+
@property
|
|
27
|
+
@abstractmethod
|
|
28
|
+
def name(self) -> str:
|
|
29
|
+
"""Return the name of the search provider."""
|
|
30
|
+
pass
|
|
31
|
+
|
|
32
|
+
@property
|
|
33
|
+
@abstractmethod
|
|
34
|
+
def requires_api_key(self) -> bool:
|
|
35
|
+
"""Return whether this provider requires an API key."""
|
|
36
|
+
pass
|
|
37
|
+
|
|
38
|
+
@abstractmethod
|
|
39
|
+
async def search(self, query: str, **kwargs) -> List[str]:
|
|
40
|
+
"""Perform a search and return repository URLs."""
|
|
41
|
+
pass
|
|
42
|
+
|
|
43
|
+
async def ensure_session(self):
|
|
44
|
+
"""Ensure we have an active session."""
|
|
45
|
+
if not self.session:
|
|
46
|
+
self.session = aiohttp.ClientSession()
|
|
47
|
+
|
|
48
|
+
async def close(self):
|
|
49
|
+
"""Close the session."""
|
|
50
|
+
if self.session:
|
|
51
|
+
await self.session.close()
|
|
52
|
+
self.session = None
|
|
53
|
+
|
|
54
|
+
def extract_repo_urls(self, urls: List[str]) -> List[str]:
|
|
55
|
+
"""Extract and filter repository URLs from a list of URLs."""
|
|
56
|
+
from urllib.parse import urlparse
|
|
57
|
+
|
|
58
|
+
repo_urls = []
|
|
59
|
+
for url in urls:
|
|
60
|
+
try:
|
|
61
|
+
parsed = urlparse(url)
|
|
62
|
+
hostname = parsed.hostname.lower() if parsed.hostname else ''
|
|
63
|
+
|
|
64
|
+
if hostname in ('github.com', 'gitlab.com', 'bitbucket.org'):
|
|
65
|
+
# Clean up the URL - remove file-specific paths
|
|
66
|
+
if "/blob/" in url or "/tree/" in url:
|
|
67
|
+
if hostname in ('github.com', 'gitlab.com'):
|
|
68
|
+
path_parts = parsed.path.strip('/').split('/')
|
|
69
|
+
if len(path_parts) >= 2:
|
|
70
|
+
# Keep only owner/repo part
|
|
71
|
+
clean_path = f"/{path_parts[0]}/{path_parts[1]}"
|
|
72
|
+
clean_url = f"{parsed.scheme}://{hostname}{clean_path}"
|
|
73
|
+
repo_urls.append(clean_url)
|
|
74
|
+
else:
|
|
75
|
+
repo_urls.append(url)
|
|
76
|
+
except Exception:
|
|
77
|
+
# If URL parsing fails, skip this URL
|
|
78
|
+
continue
|
|
79
|
+
return list(set(repo_urls))
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
class SerpAPIProvider(SearchProvider):
|
|
83
|
+
"""Search provider using SerpAPI (Google Search)."""
|
|
84
|
+
|
|
85
|
+
@property
|
|
86
|
+
def name(self) -> str:
|
|
87
|
+
return "SerpAPI"
|
|
88
|
+
|
|
89
|
+
@property
|
|
90
|
+
def requires_api_key(self) -> bool:
|
|
91
|
+
return True
|
|
92
|
+
|
|
93
|
+
async def search(self, query: str, **kwargs) -> List[str]:
|
|
94
|
+
"""Search using SerpAPI."""
|
|
95
|
+
if not self.api_key:
|
|
96
|
+
return []
|
|
97
|
+
|
|
98
|
+
await self.ensure_session()
|
|
99
|
+
|
|
100
|
+
params = {
|
|
101
|
+
"q": query,
|
|
102
|
+
"api_key": self.api_key,
|
|
103
|
+
"engine": kwargs.get("engine", "google"),
|
|
104
|
+
"num": kwargs.get("num", 10)
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
try:
|
|
108
|
+
async with self.session.get("https://serpapi.com/search", params=params) as response:
|
|
109
|
+
if response.status == 200:
|
|
110
|
+
data = await response.json()
|
|
111
|
+
urls = [
|
|
112
|
+
result.get("link", "")
|
|
113
|
+
for result in data.get("organic_results", [])
|
|
114
|
+
if result.get("link")
|
|
115
|
+
]
|
|
116
|
+
return self.extract_repo_urls(urls)
|
|
117
|
+
return []
|
|
118
|
+
except Exception as e:
|
|
119
|
+
if self.verbose:
|
|
120
|
+
console.print(f"[red]{self.name} error: {e}[/red]")
|
|
121
|
+
return []
|
|
122
|
+
|
|
123
|
+
|
|
124
|
+
class GitHubSearchProvider(SearchProvider):
|
|
125
|
+
"""Search provider using GitHub's search API."""
|
|
126
|
+
|
|
127
|
+
@property
|
|
128
|
+
def name(self) -> str:
|
|
129
|
+
return "GitHub"
|
|
130
|
+
|
|
131
|
+
@property
|
|
132
|
+
def requires_api_key(self) -> bool:
|
|
133
|
+
return False # Optional
|
|
134
|
+
|
|
135
|
+
async def search(self, query: str, **kwargs) -> List[str]:
|
|
136
|
+
"""Search using GitHub API."""
|
|
137
|
+
await self.ensure_session()
|
|
138
|
+
|
|
139
|
+
headers = {"Accept": "application/vnd.github.v3+json"}
|
|
140
|
+
if self.api_key:
|
|
141
|
+
headers["Authorization"] = f"token {self.api_key}"
|
|
142
|
+
|
|
143
|
+
# Extract search terms
|
|
144
|
+
import re
|
|
145
|
+
search_terms = re.findall(r'"([^"]+)"', query)
|
|
146
|
+
if not search_terms:
|
|
147
|
+
search_terms = [query]
|
|
148
|
+
|
|
149
|
+
urls = []
|
|
150
|
+
for term in search_terms[:1]: # Limit to avoid rate limits
|
|
151
|
+
try:
|
|
152
|
+
params = {"q": term[:100], "per_page": 5, "sort": "stars"}
|
|
153
|
+
async with self.session.get(
|
|
154
|
+
"https://api.github.com/search/repositories",
|
|
155
|
+
params=params,
|
|
156
|
+
headers=headers
|
|
157
|
+
) as response:
|
|
158
|
+
if response.status == 200:
|
|
159
|
+
data = await response.json()
|
|
160
|
+
urls.extend([
|
|
161
|
+
item.get("html_url", "")
|
|
162
|
+
for item in data.get("items", [])
|
|
163
|
+
if item.get("html_url")
|
|
164
|
+
])
|
|
165
|
+
except Exception:
|
|
166
|
+
pass
|
|
167
|
+
|
|
168
|
+
return self.extract_repo_urls(urls)
|
|
169
|
+
|
|
170
|
+
|
|
171
|
+
class SourcegraphProvider(SearchProvider):
|
|
172
|
+
"""Search provider using Sourcegraph code search."""
|
|
173
|
+
|
|
174
|
+
@property
|
|
175
|
+
def name(self) -> str:
|
|
176
|
+
return "Sourcegraph"
|
|
177
|
+
|
|
178
|
+
@property
|
|
179
|
+
def requires_api_key(self) -> bool:
|
|
180
|
+
return False
|
|
181
|
+
|
|
182
|
+
async def search(self, query: str, **kwargs) -> List[str]:
|
|
183
|
+
"""Search using Sourcegraph (simplified implementation)."""
|
|
184
|
+
# Note: Full implementation would use Sourcegraph API
|
|
185
|
+
# This is a placeholder that returns empty results
|
|
186
|
+
return []
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
class SCANOSSProvider(SearchProvider):
|
|
190
|
+
"""Provider for SCANOSS code identification service."""
|
|
191
|
+
|
|
192
|
+
DEFAULT_URL = 'https://api.osskb.org/scan/direct'
|
|
193
|
+
PREMIUM_URL = 'https://api.scanoss.com/scan/direct'
|
|
194
|
+
|
|
195
|
+
def __init__(self, api_key: Optional[str] = None, verbose: bool = False):
|
|
196
|
+
"""Initialize SCANOSS provider."""
|
|
197
|
+
super().__init__(api_key, verbose)
|
|
198
|
+
self.url = self.PREMIUM_URL if api_key else self.DEFAULT_URL
|
|
199
|
+
|
|
200
|
+
@property
|
|
201
|
+
def name(self) -> str:
|
|
202
|
+
return "SCANOSS"
|
|
203
|
+
|
|
204
|
+
@property
|
|
205
|
+
def requires_api_key(self) -> bool:
|
|
206
|
+
return False # Works without key
|
|
207
|
+
|
|
208
|
+
async def search(self, query: str, **kwargs) -> List[str]:
|
|
209
|
+
"""SCANOSS doesn't do text search - use scan_file instead."""
|
|
210
|
+
return []
|
|
211
|
+
|
|
212
|
+
async def scan_file(self, file_path: Path) -> Dict:
|
|
213
|
+
"""Scan a file using SCANOSS winnowing algorithm."""
|
|
214
|
+
await self.ensure_session()
|
|
215
|
+
|
|
216
|
+
try:
|
|
217
|
+
content = file_path.read_bytes()
|
|
218
|
+
wfp = self._create_wfp(file_path, content)
|
|
219
|
+
|
|
220
|
+
headers = {'User-Agent': 'swhpi-scanner/1.0'}
|
|
221
|
+
if self.api_key:
|
|
222
|
+
headers['X-Session'] = self.api_key
|
|
223
|
+
|
|
224
|
+
form_data = aiohttp.FormData()
|
|
225
|
+
form_data.add_field('file', wfp, filename='scan.wfp')
|
|
226
|
+
form_data.add_field('format', 'plain')
|
|
227
|
+
|
|
228
|
+
async with self.session.post(
|
|
229
|
+
self.url,
|
|
230
|
+
data=form_data,
|
|
231
|
+
headers=headers,
|
|
232
|
+
timeout=aiohttp.ClientTimeout(total=30)
|
|
233
|
+
) as response:
|
|
234
|
+
if response.status == 200:
|
|
235
|
+
result_text = await response.text()
|
|
236
|
+
return json.loads(result_text)
|
|
237
|
+
return {}
|
|
238
|
+
except Exception as e:
|
|
239
|
+
if self.verbose:
|
|
240
|
+
console.print(f"[red]SCANOSS error: {e}[/red]")
|
|
241
|
+
return {}
|
|
242
|
+
|
|
243
|
+
def _create_wfp(self, file_path: Path, content: bytes) -> str:
|
|
244
|
+
"""Create simplified WFP for SCANOSS."""
|
|
245
|
+
md5_hash = hashlib.md5(content).hexdigest()
|
|
246
|
+
return f"file={md5_hash},{len(content)},{file_path.name}\n1=00000000"
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
class SearchProviderRegistry:
|
|
250
|
+
"""Registry for managing search providers."""
|
|
251
|
+
|
|
252
|
+
def __init__(self, verbose: bool = False):
|
|
253
|
+
"""Initialize the registry."""
|
|
254
|
+
self.providers: Dict[str, SearchProvider] = {}
|
|
255
|
+
self.verbose = verbose
|
|
256
|
+
|
|
257
|
+
def register_provider(self, name: str, provider: SearchProvider):
|
|
258
|
+
"""Register a search provider."""
|
|
259
|
+
self.providers[name] = provider
|
|
260
|
+
|
|
261
|
+
def get_provider(self, name: str) -> Optional[SearchProvider]:
|
|
262
|
+
"""Get a provider by name."""
|
|
263
|
+
return self.providers.get(name)
|
|
264
|
+
|
|
265
|
+
async def search_all(self, query: str, **kwargs) -> Dict[str, List[str]]:
|
|
266
|
+
"""Search using all registered providers."""
|
|
267
|
+
results = {}
|
|
268
|
+
for name, provider in self.providers.items():
|
|
269
|
+
if provider.requires_api_key and not provider.api_key:
|
|
270
|
+
continue
|
|
271
|
+
try:
|
|
272
|
+
urls = await provider.search(query, **kwargs)
|
|
273
|
+
if urls:
|
|
274
|
+
results[name] = urls
|
|
275
|
+
except Exception:
|
|
276
|
+
pass
|
|
277
|
+
return results
|
|
278
|
+
|
|
279
|
+
async def close_all(self):
|
|
280
|
+
"""Close all provider sessions."""
|
|
281
|
+
for provider in self.providers.values():
|
|
282
|
+
await provider.close()
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
def create_default_registry(verbose: bool = False) -> SearchProviderRegistry:
|
|
286
|
+
"""Create a registry with default providers configured from environment."""
|
|
287
|
+
registry = SearchProviderRegistry(verbose=verbose)
|
|
288
|
+
|
|
289
|
+
# SerpAPI
|
|
290
|
+
if serpapi_key := os.environ.get("SERPAPI_KEY"):
|
|
291
|
+
registry.register_provider(
|
|
292
|
+
"serpapi",
|
|
293
|
+
SerpAPIProvider(api_key=serpapi_key, verbose=verbose)
|
|
294
|
+
)
|
|
295
|
+
|
|
296
|
+
# GitHub
|
|
297
|
+
github_token = os.environ.get("GITHUB_TOKEN")
|
|
298
|
+
registry.register_provider(
|
|
299
|
+
"github",
|
|
300
|
+
GitHubSearchProvider(api_key=github_token, verbose=verbose)
|
|
301
|
+
)
|
|
302
|
+
|
|
303
|
+
# SCANOSS
|
|
304
|
+
scanoss_key = os.environ.get("SCANOSS_API_KEY")
|
|
305
|
+
registry.register_provider(
|
|
306
|
+
"scanoss",
|
|
307
|
+
SCANOSSProvider(api_key=scanoss_key, verbose=verbose)
|
|
308
|
+
)
|
|
309
|
+
|
|
310
|
+
return registry
|