src2purl 1.3.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- src2id/__init__.py +15 -0
- src2id/cli/__init__.py +1 -0
- src2id/cli/main.py +318 -0
- src2id/cli/validate.py +93 -0
- src2id/core/__init__.py +1 -0
- src2id/core/cache.py +180 -0
- src2id/core/client.py +370 -0
- src2id/core/config.py +45 -0
- src2id/core/extractor.py +302 -0
- src2id/core/models.py +93 -0
- src2id/core/orchestrator.py +796 -0
- src2id/core/package_identifier.py +123 -0
- src2id/core/purl.py +238 -0
- src2id/core/scanner.py +369 -0
- src2id/core/scorer.py +217 -0
- src2id/core/subcomponent_detector.py +353 -0
- src2id/core/swhid.py +324 -0
- src2id/integrations/__init__.py +1 -0
- src2id/integrations/manifest_parser.py +652 -0
- src2id/integrations/oslili.py +228 -0
- src2id/integrations/upmex.py +305 -0
- src2id/search/__init__.py +34 -0
- src2id/search/hash_search.py +206 -0
- src2id/search/providers.py +310 -0
- src2id/search/strategies.py +391 -0
- src2id/utils/__init__.py +1 -0
- src2id/utils/datetime_utils.py +49 -0
- src2purl-1.3.2.dist-info/METADATA +279 -0
- src2purl-1.3.2.dist-info/RECORD +33 -0
- src2purl-1.3.2.dist-info/WHEEL +5 -0
- src2purl-1.3.2.dist-info/entry_points.txt +3 -0
- src2purl-1.3.2.dist-info/licenses/LICENSE +661 -0
- src2purl-1.3.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,652 @@
|
|
|
1
|
+
"""Direct package manifest parser for common package formats."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import re
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Dict, List, Optional, Set
|
|
7
|
+
|
|
8
|
+
# Handle TOML parsing with fallback for older Python versions
|
|
9
|
+
try:
|
|
10
|
+
import tomllib # Python 3.11+
|
|
11
|
+
except ImportError:
|
|
12
|
+
try:
|
|
13
|
+
import tomli as tomllib # Fallback for Python < 3.11
|
|
14
|
+
except ImportError:
|
|
15
|
+
tomllib = None # No TOML support available
|
|
16
|
+
|
|
17
|
+
from src2id.core.models import PackageMatch, MatchType
|
|
18
|
+
|
|
19
|
+
class DirectManifestParser:
|
|
20
|
+
"""Parser for common package manifest files."""
|
|
21
|
+
|
|
22
|
+
# Supported manifest file patterns
|
|
23
|
+
SUPPORTED_FILES = {
|
|
24
|
+
# Python
|
|
25
|
+
'setup.py', 'pyproject.toml', 'setup.cfg', 'PKG-INFO',
|
|
26
|
+
# Node.js/NPM
|
|
27
|
+
'package.json',
|
|
28
|
+
# Java/Maven
|
|
29
|
+
'pom.xml',
|
|
30
|
+
# Go
|
|
31
|
+
'go.mod', 'go.sum',
|
|
32
|
+
# Rust
|
|
33
|
+
'Cargo.toml', 'Cargo.lock',
|
|
34
|
+
# Ruby
|
|
35
|
+
'Gemfile', 'Gemfile.lock',
|
|
36
|
+
# CocoaPods
|
|
37
|
+
'Podfile',
|
|
38
|
+
# Conda
|
|
39
|
+
'meta.yaml', 'conda.yaml', 'environment.yml',
|
|
40
|
+
# Perl
|
|
41
|
+
'META.json', 'META.yml', 'Makefile.PL',
|
|
42
|
+
# Conan
|
|
43
|
+
'conanfile.py', 'conanfile.txt',
|
|
44
|
+
# NuGet
|
|
45
|
+
'packages.config',
|
|
46
|
+
# Debian
|
|
47
|
+
'control',
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
def __init__(self):
|
|
51
|
+
"""Initialize the manifest parser."""
|
|
52
|
+
pass
|
|
53
|
+
|
|
54
|
+
def scan_directory_for_manifests(self, directory: Path) -> List[Path]:
|
|
55
|
+
"""
|
|
56
|
+
Scan directory for supported manifest files.
|
|
57
|
+
|
|
58
|
+
Args:
|
|
59
|
+
directory: Directory to scan
|
|
60
|
+
|
|
61
|
+
Returns:
|
|
62
|
+
List of manifest file paths found
|
|
63
|
+
"""
|
|
64
|
+
manifest_files = []
|
|
65
|
+
|
|
66
|
+
try:
|
|
67
|
+
# Scan current directory
|
|
68
|
+
manifest_files.extend(self._scan_single_directory(directory))
|
|
69
|
+
|
|
70
|
+
# Scan deeper subdirectories for manifest files (up to 3 levels deep)
|
|
71
|
+
self._scan_recursive_manifests(directory, manifest_files, max_depth=3, current_depth=0)
|
|
72
|
+
|
|
73
|
+
except (PermissionError, OSError):
|
|
74
|
+
pass
|
|
75
|
+
|
|
76
|
+
return manifest_files
|
|
77
|
+
|
|
78
|
+
def _scan_single_directory(self, directory: Path) -> List[Path]:
|
|
79
|
+
"""Scan a single directory for manifest files."""
|
|
80
|
+
manifest_files = []
|
|
81
|
+
|
|
82
|
+
try:
|
|
83
|
+
for file_path in directory.iterdir():
|
|
84
|
+
if file_path.is_file():
|
|
85
|
+
# Check exact filename matches
|
|
86
|
+
if file_path.name in self.SUPPORTED_FILES:
|
|
87
|
+
manifest_files.append(file_path)
|
|
88
|
+
# Check pattern matches (e.g., *.gemspec, *.csproj)
|
|
89
|
+
elif self._matches_pattern(file_path.name):
|
|
90
|
+
manifest_files.append(file_path)
|
|
91
|
+
except (PermissionError, OSError):
|
|
92
|
+
pass
|
|
93
|
+
|
|
94
|
+
return manifest_files
|
|
95
|
+
|
|
96
|
+
def _matches_pattern(self, filename: str) -> bool:
|
|
97
|
+
"""Check if filename matches supported patterns."""
|
|
98
|
+
patterns = [
|
|
99
|
+
r'.*\.gemspec$', # Ruby gemspec files
|
|
100
|
+
r'.*\.csproj$', # .NET project files
|
|
101
|
+
r'.*\.nuspec$', # NuGet package spec files
|
|
102
|
+
r'.*\.podspec$', # CocoaPods spec files
|
|
103
|
+
r'build\.gradle.*$', # Gradle build files
|
|
104
|
+
]
|
|
105
|
+
|
|
106
|
+
return any(re.match(pattern, filename) for pattern in patterns)
|
|
107
|
+
|
|
108
|
+
def extract_metadata_from_file(self, file_path: Path) -> Optional[PackageMatch]:
|
|
109
|
+
"""
|
|
110
|
+
Extract package metadata from a manifest file.
|
|
111
|
+
|
|
112
|
+
Args:
|
|
113
|
+
file_path: Path to manifest file
|
|
114
|
+
|
|
115
|
+
Returns:
|
|
116
|
+
PackageMatch if extraction successful, None otherwise
|
|
117
|
+
"""
|
|
118
|
+
try:
|
|
119
|
+
if file_path.name == 'package.json':
|
|
120
|
+
return self._parse_package_json(file_path)
|
|
121
|
+
elif file_path.name == 'pyproject.toml':
|
|
122
|
+
return self._parse_pyproject_toml(file_path)
|
|
123
|
+
elif file_path.name == 'setup.py':
|
|
124
|
+
return self._parse_setup_py(file_path)
|
|
125
|
+
elif file_path.name == 'setup.cfg':
|
|
126
|
+
return self._parse_setup_cfg(file_path)
|
|
127
|
+
elif file_path.name == 'pom.xml':
|
|
128
|
+
return self._parse_pom_xml(file_path)
|
|
129
|
+
elif file_path.name == 'go.mod':
|
|
130
|
+
return self._parse_go_mod(file_path)
|
|
131
|
+
elif file_path.name == 'Cargo.toml':
|
|
132
|
+
return self._parse_cargo_toml(file_path)
|
|
133
|
+
elif file_path.name.endswith('.gemspec'):
|
|
134
|
+
return self._parse_gemspec(file_path)
|
|
135
|
+
elif file_path.name.endswith(('.podspec', '.podspec.json')):
|
|
136
|
+
return self._parse_podspec(file_path)
|
|
137
|
+
# Add more parsers as needed
|
|
138
|
+
|
|
139
|
+
except Exception as e:
|
|
140
|
+
# Silently ignore parsing errors
|
|
141
|
+
pass
|
|
142
|
+
|
|
143
|
+
return None
|
|
144
|
+
|
|
145
|
+
def extract_metadata_from_directory(self, directory: Path) -> List[PackageMatch]:
|
|
146
|
+
"""
|
|
147
|
+
Extract metadata from all manifest files in a directory.
|
|
148
|
+
|
|
149
|
+
Args:
|
|
150
|
+
directory: Directory to scan
|
|
151
|
+
|
|
152
|
+
Returns:
|
|
153
|
+
List of PackageMatch objects (deduplicated by package name)
|
|
154
|
+
"""
|
|
155
|
+
matches = []
|
|
156
|
+
manifest_files = self.scan_directory_for_manifests(directory)
|
|
157
|
+
|
|
158
|
+
for file_path in manifest_files:
|
|
159
|
+
match = self.extract_metadata_from_file(file_path)
|
|
160
|
+
if match:
|
|
161
|
+
matches.append(match)
|
|
162
|
+
|
|
163
|
+
# Deduplicate by package name, keeping the best quality match
|
|
164
|
+
return self._deduplicate_matches(matches)
|
|
165
|
+
|
|
166
|
+
def _parse_package_json(self, file_path: Path) -> Optional[PackageMatch]:
|
|
167
|
+
"""Parse NPM package.json file."""
|
|
168
|
+
try:
|
|
169
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
170
|
+
data = json.load(f)
|
|
171
|
+
|
|
172
|
+
name = data.get('name')
|
|
173
|
+
version = data.get('version')
|
|
174
|
+
license_info = data.get('license')
|
|
175
|
+
description = data.get('description')
|
|
176
|
+
homepage = data.get('homepage')
|
|
177
|
+
repository = data.get('repository', {})
|
|
178
|
+
|
|
179
|
+
# Extract repository URL
|
|
180
|
+
repo_url = None
|
|
181
|
+
if isinstance(repository, dict):
|
|
182
|
+
repo_url = repository.get('url')
|
|
183
|
+
elif isinstance(repository, str):
|
|
184
|
+
repo_url = repository
|
|
185
|
+
|
|
186
|
+
# Generate PURL
|
|
187
|
+
purl = None
|
|
188
|
+
if name:
|
|
189
|
+
if name.startswith('@'):
|
|
190
|
+
# Scoped package
|
|
191
|
+
parts = name[1:].split('/', 1)
|
|
192
|
+
if len(parts) == 2:
|
|
193
|
+
namespace, pkg_name = parts
|
|
194
|
+
purl = f"pkg:npm/{namespace}/{pkg_name}"
|
|
195
|
+
if version:
|
|
196
|
+
purl += f"@{version}"
|
|
197
|
+
else:
|
|
198
|
+
purl = f"pkg:npm/{name}"
|
|
199
|
+
if version:
|
|
200
|
+
purl += f"@{version}"
|
|
201
|
+
|
|
202
|
+
return PackageMatch(
|
|
203
|
+
download_url=repo_url or homepage or f"https://www.npmjs.com/package/{name}",
|
|
204
|
+
match_type=MatchType.EXACT,
|
|
205
|
+
confidence_score=0.9,
|
|
206
|
+
name=name,
|
|
207
|
+
version=version,
|
|
208
|
+
license=license_info,
|
|
209
|
+
purl=purl,
|
|
210
|
+
is_official_org=self._is_official_npm_org(repo_url or homepage or ''),
|
|
211
|
+
)
|
|
212
|
+
|
|
213
|
+
except (json.JSONDecodeError, KeyError, FileNotFoundError):
|
|
214
|
+
return None
|
|
215
|
+
|
|
216
|
+
def _parse_pyproject_toml(self, file_path: Path) -> Optional[PackageMatch]:
|
|
217
|
+
"""Parse Python pyproject.toml file."""
|
|
218
|
+
if tomllib is None:
|
|
219
|
+
return None # TOML support not available
|
|
220
|
+
|
|
221
|
+
try:
|
|
222
|
+
with open(file_path, 'rb') as f:
|
|
223
|
+
data = tomllib.load(f)
|
|
224
|
+
|
|
225
|
+
project = data.get('project', {})
|
|
226
|
+
name = project.get('name')
|
|
227
|
+
version = project.get('version')
|
|
228
|
+
|
|
229
|
+
# Handle license
|
|
230
|
+
license_info = None
|
|
231
|
+
license_data = project.get('license')
|
|
232
|
+
if isinstance(license_data, dict):
|
|
233
|
+
license_info = license_data.get('text')
|
|
234
|
+
elif isinstance(license_data, str):
|
|
235
|
+
license_info = license_data
|
|
236
|
+
|
|
237
|
+
# Get URLs
|
|
238
|
+
urls = project.get('urls', {})
|
|
239
|
+
homepage = urls.get('Homepage') or urls.get('home')
|
|
240
|
+
repository = urls.get('Repository') or urls.get('repository')
|
|
241
|
+
|
|
242
|
+
# Generate PURL
|
|
243
|
+
purl = None
|
|
244
|
+
if name:
|
|
245
|
+
purl = f"pkg:pypi/{name.lower().replace('_', '-')}"
|
|
246
|
+
if version:
|
|
247
|
+
purl += f"@{version}"
|
|
248
|
+
|
|
249
|
+
return PackageMatch(
|
|
250
|
+
download_url=repository or homepage or f"https://pypi.org/project/{name}/",
|
|
251
|
+
match_type=MatchType.EXACT,
|
|
252
|
+
confidence_score=0.95,
|
|
253
|
+
name=name,
|
|
254
|
+
version=version,
|
|
255
|
+
license=license_info,
|
|
256
|
+
purl=purl,
|
|
257
|
+
is_official_org=self._is_official_python_org(repository or homepage or ''),
|
|
258
|
+
)
|
|
259
|
+
|
|
260
|
+
except (tomllib.TOMLDecodeError, KeyError, FileNotFoundError):
|
|
261
|
+
return None
|
|
262
|
+
|
|
263
|
+
def _parse_setup_py(self, file_path: Path) -> Optional[PackageMatch]:
|
|
264
|
+
"""Parse Python setup.py file (basic regex-based parsing)."""
|
|
265
|
+
try:
|
|
266
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
267
|
+
content = f.read()
|
|
268
|
+
|
|
269
|
+
# Use regex to extract basic information
|
|
270
|
+
name_match = re.search(r'name\s*=\s*["\']([^"\']+)["\']', content)
|
|
271
|
+
version_match = re.search(r'version\s*=\s*["\']([^"\']+)["\']', content)
|
|
272
|
+
license_match = re.search(r'license\s*=\s*["\']([^"\']+)["\']', content)
|
|
273
|
+
url_match = re.search(r'url\s*=\s*["\']([^"\']+)["\']', content)
|
|
274
|
+
|
|
275
|
+
name = name_match.group(1) if name_match else None
|
|
276
|
+
version = version_match.group(1) if version_match else None
|
|
277
|
+
license_info = license_match.group(1) if license_match else None
|
|
278
|
+
url = url_match.group(1) if url_match else None
|
|
279
|
+
|
|
280
|
+
if not name:
|
|
281
|
+
return None
|
|
282
|
+
|
|
283
|
+
# Generate PURL
|
|
284
|
+
purl = f"pkg:pypi/{name.lower().replace('_', '-')}"
|
|
285
|
+
if version:
|
|
286
|
+
purl += f"@{version}"
|
|
287
|
+
|
|
288
|
+
return PackageMatch(
|
|
289
|
+
download_url=url or f"https://pypi.org/project/{name}/",
|
|
290
|
+
match_type=MatchType.EXACT,
|
|
291
|
+
confidence_score=0.85, # Lower confidence for regex parsing
|
|
292
|
+
name=name,
|
|
293
|
+
version=version,
|
|
294
|
+
license=license_info,
|
|
295
|
+
purl=purl,
|
|
296
|
+
is_official_org=self._is_official_python_org(url or ''),
|
|
297
|
+
)
|
|
298
|
+
|
|
299
|
+
except (FileNotFoundError, UnicodeDecodeError):
|
|
300
|
+
return None
|
|
301
|
+
|
|
302
|
+
def _parse_pom_xml(self, file_path: Path) -> Optional[PackageMatch]:
|
|
303
|
+
"""Parse Maven pom.xml file (basic regex-based parsing)."""
|
|
304
|
+
try:
|
|
305
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
306
|
+
content = f.read()
|
|
307
|
+
|
|
308
|
+
# Extract basic Maven coordinates
|
|
309
|
+
group_match = re.search(r'<groupId>([^<]+)</groupId>', content)
|
|
310
|
+
artifact_match = re.search(r'<artifactId>([^<]+)</artifactId>', content)
|
|
311
|
+
version_match = re.search(r'<version>([^<]+)</version>', content)
|
|
312
|
+
url_match = re.search(r'<url>([^<]+)</url>', content)
|
|
313
|
+
|
|
314
|
+
group_id = group_match.group(1) if group_match else None
|
|
315
|
+
artifact_id = artifact_match.group(1) if artifact_match else None
|
|
316
|
+
version = version_match.group(1) if version_match else None
|
|
317
|
+
url = url_match.group(1) if url_match else None
|
|
318
|
+
|
|
319
|
+
if not artifact_id:
|
|
320
|
+
return None
|
|
321
|
+
|
|
322
|
+
# Generate PURL
|
|
323
|
+
purl = f"pkg:maven/{group_id or 'unknown'}/{artifact_id}"
|
|
324
|
+
if version and not version.startswith('${'): # Skip property placeholders
|
|
325
|
+
purl += f"@{version}"
|
|
326
|
+
|
|
327
|
+
return PackageMatch(
|
|
328
|
+
download_url=url or f"https://mvnrepository.com/artifact/{group_id}/{artifact_id}",
|
|
329
|
+
match_type=MatchType.EXACT,
|
|
330
|
+
confidence_score=0.9,
|
|
331
|
+
name=f"{group_id}:{artifact_id}" if group_id else artifact_id,
|
|
332
|
+
version=version if not version.startswith('${') else None,
|
|
333
|
+
purl=purl,
|
|
334
|
+
is_official_org=self._is_official_java_org(url or group_id or ''),
|
|
335
|
+
)
|
|
336
|
+
|
|
337
|
+
except (FileNotFoundError, UnicodeDecodeError):
|
|
338
|
+
return None
|
|
339
|
+
|
|
340
|
+
def _parse_go_mod(self, file_path: Path) -> Optional[PackageMatch]:
|
|
341
|
+
"""Parse Go go.mod file."""
|
|
342
|
+
try:
|
|
343
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
344
|
+
content = f.read()
|
|
345
|
+
|
|
346
|
+
# Extract module name
|
|
347
|
+
module_match = re.search(r'^module\s+([^\s]+)', content, re.MULTILINE)
|
|
348
|
+
go_version_match = re.search(r'^go\s+([^\s]+)', content, re.MULTILINE)
|
|
349
|
+
|
|
350
|
+
module_name = module_match.group(1) if module_match else None
|
|
351
|
+
go_version = go_version_match.group(1) if go_version_match else None
|
|
352
|
+
|
|
353
|
+
if not module_name:
|
|
354
|
+
return None
|
|
355
|
+
|
|
356
|
+
# Generate PURL
|
|
357
|
+
purl = f"pkg:golang/{module_name}"
|
|
358
|
+
|
|
359
|
+
# Try to construct repository URL
|
|
360
|
+
repo_url = None
|
|
361
|
+
if module_name.startswith(('github.com/', 'gitlab.com/', 'bitbucket.org/')):
|
|
362
|
+
repo_url = f"https://{module_name}"
|
|
363
|
+
|
|
364
|
+
return PackageMatch(
|
|
365
|
+
download_url=repo_url or f"https://pkg.go.dev/{module_name}",
|
|
366
|
+
match_type=MatchType.EXACT,
|
|
367
|
+
confidence_score=0.9,
|
|
368
|
+
name=module_name,
|
|
369
|
+
version=go_version,
|
|
370
|
+
purl=purl,
|
|
371
|
+
is_official_org=self._is_official_go_org(module_name),
|
|
372
|
+
)
|
|
373
|
+
|
|
374
|
+
except (FileNotFoundError, UnicodeDecodeError):
|
|
375
|
+
return None
|
|
376
|
+
|
|
377
|
+
def _parse_cargo_toml(self, file_path: Path) -> Optional[PackageMatch]:
|
|
378
|
+
"""Parse Rust Cargo.toml file."""
|
|
379
|
+
if tomllib is None:
|
|
380
|
+
return None # TOML support not available
|
|
381
|
+
|
|
382
|
+
try:
|
|
383
|
+
with open(file_path, 'rb') as f:
|
|
384
|
+
data = tomllib.load(f)
|
|
385
|
+
|
|
386
|
+
package = data.get('package', {})
|
|
387
|
+
name = package.get('name')
|
|
388
|
+
version = package.get('version')
|
|
389
|
+
license_info = package.get('license')
|
|
390
|
+
repository = package.get('repository')
|
|
391
|
+
homepage = package.get('homepage')
|
|
392
|
+
|
|
393
|
+
if not name:
|
|
394
|
+
return None
|
|
395
|
+
|
|
396
|
+
# Generate PURL
|
|
397
|
+
purl = f"pkg:cargo/{name}"
|
|
398
|
+
if version:
|
|
399
|
+
purl += f"@{version}"
|
|
400
|
+
|
|
401
|
+
return PackageMatch(
|
|
402
|
+
download_url=repository or homepage or f"https://crates.io/crates/{name}",
|
|
403
|
+
match_type=MatchType.EXACT,
|
|
404
|
+
confidence_score=0.9,
|
|
405
|
+
name=name,
|
|
406
|
+
version=version,
|
|
407
|
+
license=license_info,
|
|
408
|
+
purl=purl,
|
|
409
|
+
is_official_org=self._is_official_rust_org(repository or homepage or ''),
|
|
410
|
+
)
|
|
411
|
+
|
|
412
|
+
except (tomllib.TOMLDecodeError, KeyError, FileNotFoundError):
|
|
413
|
+
return None
|
|
414
|
+
|
|
415
|
+
def _parse_gemspec(self, file_path: Path) -> Optional[PackageMatch]:
|
|
416
|
+
"""Parse Ruby .gemspec file (basic regex-based parsing)."""
|
|
417
|
+
try:
|
|
418
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
419
|
+
content = f.read()
|
|
420
|
+
|
|
421
|
+
# Extract basic gem information
|
|
422
|
+
name_match = re.search(r'spec\.name\s*=\s*["\']([^"\']+)["\']', content)
|
|
423
|
+
version_match = re.search(r'spec\.version\s*=\s*["\']([^"\']+)["\']', content)
|
|
424
|
+
license_match = re.search(r'spec\.license\s*=\s*["\']([^"\']+)["\']', content)
|
|
425
|
+
homepage_match = re.search(r'spec\.homepage\s*=\s*["\']([^"\']+)["\']', content)
|
|
426
|
+
|
|
427
|
+
name = name_match.group(1) if name_match else None
|
|
428
|
+
version = version_match.group(1) if version_match else None
|
|
429
|
+
license_info = license_match.group(1) if license_match else None
|
|
430
|
+
homepage = homepage_match.group(1) if homepage_match else None
|
|
431
|
+
|
|
432
|
+
if not name:
|
|
433
|
+
return None
|
|
434
|
+
|
|
435
|
+
# Generate PURL
|
|
436
|
+
purl = f"pkg:gem/{name}"
|
|
437
|
+
if version:
|
|
438
|
+
purl += f"@{version}"
|
|
439
|
+
|
|
440
|
+
return PackageMatch(
|
|
441
|
+
download_url=homepage or f"https://rubygems.org/gems/{name}",
|
|
442
|
+
match_type=MatchType.EXACT,
|
|
443
|
+
confidence_score=0.85, # Lower confidence for regex parsing
|
|
444
|
+
name=name,
|
|
445
|
+
version=version,
|
|
446
|
+
license=license_info,
|
|
447
|
+
purl=purl,
|
|
448
|
+
is_official_org=False, # Most gems are not from official orgs
|
|
449
|
+
)
|
|
450
|
+
|
|
451
|
+
except (FileNotFoundError, UnicodeDecodeError):
|
|
452
|
+
return None
|
|
453
|
+
|
|
454
|
+
def _parse_podspec(self, file_path: Path) -> Optional[PackageMatch]:
|
|
455
|
+
"""Parse CocoaPods .podspec file."""
|
|
456
|
+
try:
|
|
457
|
+
if file_path.name.endswith('.json'):
|
|
458
|
+
# JSON podspec
|
|
459
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
460
|
+
data = json.load(f)
|
|
461
|
+
|
|
462
|
+
name = data.get('name')
|
|
463
|
+
version = data.get('version')
|
|
464
|
+
license_info = data.get('license')
|
|
465
|
+
homepage = data.get('homepage')
|
|
466
|
+
source = data.get('source', {})
|
|
467
|
+
|
|
468
|
+
# Extract git URL from source
|
|
469
|
+
git_url = None
|
|
470
|
+
if isinstance(source, dict):
|
|
471
|
+
git_url = source.get('git')
|
|
472
|
+
|
|
473
|
+
else:
|
|
474
|
+
# Ruby podspec (basic regex parsing)
|
|
475
|
+
with open(file_path, 'r', encoding='utf-8') as f:
|
|
476
|
+
content = f.read()
|
|
477
|
+
|
|
478
|
+
name_match = re.search(r's\.name\s*=\s*["\']([^"\']+)["\']', content)
|
|
479
|
+
version_match = re.search(r's\.version\s*=\s*["\']([^"\']+)["\']', content)
|
|
480
|
+
homepage_match = re.search(r's\.homepage\s*=\s*["\']([^"\']+)["\']', content)
|
|
481
|
+
|
|
482
|
+
name = name_match.group(1) if name_match else None
|
|
483
|
+
version = version_match.group(1) if version_match else None
|
|
484
|
+
homepage = homepage_match.group(1) if homepage_match else None
|
|
485
|
+
license_info = None
|
|
486
|
+
git_url = None
|
|
487
|
+
|
|
488
|
+
if not name:
|
|
489
|
+
return None
|
|
490
|
+
|
|
491
|
+
return PackageMatch(
|
|
492
|
+
download_url=git_url or homepage or f"https://cocoapods.org/pods/{name}",
|
|
493
|
+
match_type=MatchType.EXACT,
|
|
494
|
+
confidence_score=0.9,
|
|
495
|
+
name=name,
|
|
496
|
+
version=version,
|
|
497
|
+
license=license_info if isinstance(license_info, str) else None,
|
|
498
|
+
is_official_org=False,
|
|
499
|
+
)
|
|
500
|
+
|
|
501
|
+
except (json.JSONDecodeError, FileNotFoundError, UnicodeDecodeError):
|
|
502
|
+
return None
|
|
503
|
+
|
|
504
|
+
def _deduplicate_matches(self, matches: List[PackageMatch]) -> List[PackageMatch]:
|
|
505
|
+
"""Deduplicate matches by package name, keeping the best quality match."""
|
|
506
|
+
if not matches:
|
|
507
|
+
return matches
|
|
508
|
+
|
|
509
|
+
# Group by package name
|
|
510
|
+
grouped = {}
|
|
511
|
+
for match in matches:
|
|
512
|
+
name = match.name or "unknown"
|
|
513
|
+
if name not in grouped or match.confidence_score > grouped[name].confidence_score:
|
|
514
|
+
grouped[name] = match
|
|
515
|
+
|
|
516
|
+
return list(grouped.values())
|
|
517
|
+
|
|
518
|
+
def _scan_recursive_manifests(self, directory: Path, manifest_files: List[Path], max_depth: int, current_depth: int):
|
|
519
|
+
"""Recursively scan subdirectories for manifest files."""
|
|
520
|
+
if current_depth >= max_depth:
|
|
521
|
+
return
|
|
522
|
+
|
|
523
|
+
try:
|
|
524
|
+
for subdir in directory.iterdir():
|
|
525
|
+
if (subdir.is_dir() and
|
|
526
|
+
not subdir.name.startswith('.') and
|
|
527
|
+
subdir.name not in {'__pycache__', 'node_modules', 'target', 'build', 'dist', '.git'}):
|
|
528
|
+
|
|
529
|
+
# Scan this subdirectory
|
|
530
|
+
manifest_files.extend(self._scan_single_directory(subdir))
|
|
531
|
+
|
|
532
|
+
# Recurse deeper
|
|
533
|
+
self._scan_recursive_manifests(subdir, manifest_files, max_depth, current_depth + 1)
|
|
534
|
+
|
|
535
|
+
except (PermissionError, OSError):
|
|
536
|
+
pass
|
|
537
|
+
|
|
538
|
+
def _parse_setup_cfg(self, file_path: Path) -> Optional[PackageMatch]:
|
|
539
|
+
"""Parse Python setup.cfg file."""
|
|
540
|
+
try:
|
|
541
|
+
import configparser
|
|
542
|
+
|
|
543
|
+
config = configparser.ConfigParser()
|
|
544
|
+
config.read(file_path)
|
|
545
|
+
|
|
546
|
+
# Extract metadata from [metadata] section
|
|
547
|
+
if not config.has_section('metadata'):
|
|
548
|
+
return None
|
|
549
|
+
|
|
550
|
+
metadata_section = config['metadata']
|
|
551
|
+
name = metadata_section.get('name')
|
|
552
|
+
version = metadata_section.get('version')
|
|
553
|
+
license_info = metadata_section.get('license')
|
|
554
|
+
url = metadata_section.get('url')
|
|
555
|
+
|
|
556
|
+
# Try to get repository URL from project_urls
|
|
557
|
+
repo_url = url
|
|
558
|
+
if config.has_option('metadata', 'project_urls'):
|
|
559
|
+
project_urls = metadata_section.get('project_urls', '')
|
|
560
|
+
for line in project_urls.split('\n'):
|
|
561
|
+
if '=' in line:
|
|
562
|
+
key, value = line.split('=', 1)
|
|
563
|
+
key = key.strip().lower()
|
|
564
|
+
if key in ['source', 'repository', 'homepage']:
|
|
565
|
+
repo_url = value.strip()
|
|
566
|
+
break
|
|
567
|
+
|
|
568
|
+
if not name:
|
|
569
|
+
return None
|
|
570
|
+
|
|
571
|
+
# Generate PURL
|
|
572
|
+
purl = f"pkg:pypi/{name.lower().replace('_', '-')}"
|
|
573
|
+
if version and not version.startswith('attr:'):
|
|
574
|
+
purl += f"@{version}"
|
|
575
|
+
|
|
576
|
+
return PackageMatch(
|
|
577
|
+
download_url=repo_url or f"https://pypi.org/project/{name}/",
|
|
578
|
+
match_type=MatchType.EXACT,
|
|
579
|
+
confidence_score=0.9,
|
|
580
|
+
name=name,
|
|
581
|
+
version=version if not version.startswith('attr:') else None,
|
|
582
|
+
license=license_info,
|
|
583
|
+
purl=purl,
|
|
584
|
+
is_official_org=self._is_official_python_org(repo_url or ''),
|
|
585
|
+
)
|
|
586
|
+
|
|
587
|
+
except (configparser.Error, FileNotFoundError, UnicodeDecodeError):
|
|
588
|
+
return None
|
|
589
|
+
|
|
590
|
+
def _is_official_npm_org(self, url: str) -> bool:
|
|
591
|
+
"""Check if NPM package is from an official organization."""
|
|
592
|
+
official_patterns = [
|
|
593
|
+
'github.com/nodejs/',
|
|
594
|
+
'github.com/npm/',
|
|
595
|
+
'github.com/microsoft/',
|
|
596
|
+
'github.com/google/',
|
|
597
|
+
'github.com/facebook/',
|
|
598
|
+
'github.com/angular/',
|
|
599
|
+
'github.com/reactjs/',
|
|
600
|
+
]
|
|
601
|
+
return any(pattern in url.lower() for pattern in official_patterns)
|
|
602
|
+
|
|
603
|
+
def _is_official_python_org(self, url: str) -> bool:
|
|
604
|
+
"""Check if Python package is from an official organization."""
|
|
605
|
+
official_patterns = [
|
|
606
|
+
'github.com/python/',
|
|
607
|
+
'github.com/psf/',
|
|
608
|
+
'github.com/microsoft/',
|
|
609
|
+
'github.com/google/',
|
|
610
|
+
'github.com/numpy/',
|
|
611
|
+
'github.com/scipy/',
|
|
612
|
+
'github.com/pandas-dev/',
|
|
613
|
+
]
|
|
614
|
+
return any(pattern in url.lower() for pattern in official_patterns)
|
|
615
|
+
|
|
616
|
+
def _is_official_java_org(self, identifier: str) -> bool:
|
|
617
|
+
"""Check if Java package is from an official organization."""
|
|
618
|
+
official_patterns = [
|
|
619
|
+
'apache',
|
|
620
|
+
'eclipse',
|
|
621
|
+
'springframework',
|
|
622
|
+
'com.google',
|
|
623
|
+
'com.microsoft',
|
|
624
|
+
'org.apache',
|
|
625
|
+
'org.eclipse',
|
|
626
|
+
'org.springframework',
|
|
627
|
+
]
|
|
628
|
+
return any(pattern in identifier.lower() for pattern in official_patterns)
|
|
629
|
+
|
|
630
|
+
def _is_official_go_org(self, module: str) -> bool:
|
|
631
|
+
"""Check if Go module is from an official organization."""
|
|
632
|
+
official_patterns = [
|
|
633
|
+
'github.com/golang/',
|
|
634
|
+
'github.com/google/',
|
|
635
|
+
'github.com/microsoft/',
|
|
636
|
+
'go.uber.org/',
|
|
637
|
+
'google.golang.org/',
|
|
638
|
+
]
|
|
639
|
+
return any(pattern in module.lower() for pattern in official_patterns)
|
|
640
|
+
|
|
641
|
+
def _is_official_rust_org(self, url: str) -> bool:
|
|
642
|
+
"""Check if Rust crate is from an official organization."""
|
|
643
|
+
official_patterns = [
|
|
644
|
+
'github.com/rust-lang/',
|
|
645
|
+
'github.com/tokio-rs/',
|
|
646
|
+
'github.com/serde-rs/',
|
|
647
|
+
]
|
|
648
|
+
return any(pattern in url.lower() for pattern in official_patterns)
|
|
649
|
+
|
|
650
|
+
def get_supported_file_types(self) -> Set[str]:
|
|
651
|
+
"""Get set of file types supported by the manifest parser."""
|
|
652
|
+
return self.SUPPORTED_FILES.copy()
|