src2purl 1.3.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- src2id/__init__.py +15 -0
- src2id/cli/__init__.py +1 -0
- src2id/cli/main.py +318 -0
- src2id/cli/validate.py +93 -0
- src2id/core/__init__.py +1 -0
- src2id/core/cache.py +180 -0
- src2id/core/client.py +370 -0
- src2id/core/config.py +45 -0
- src2id/core/extractor.py +302 -0
- src2id/core/models.py +93 -0
- src2id/core/orchestrator.py +796 -0
- src2id/core/package_identifier.py +123 -0
- src2id/core/purl.py +238 -0
- src2id/core/scanner.py +369 -0
- src2id/core/scorer.py +217 -0
- src2id/core/subcomponent_detector.py +353 -0
- src2id/core/swhid.py +324 -0
- src2id/integrations/__init__.py +1 -0
- src2id/integrations/manifest_parser.py +652 -0
- src2id/integrations/oslili.py +228 -0
- src2id/integrations/upmex.py +305 -0
- src2id/search/__init__.py +34 -0
- src2id/search/hash_search.py +206 -0
- src2id/search/providers.py +310 -0
- src2id/search/strategies.py +391 -0
- src2id/utils/__init__.py +1 -0
- src2id/utils/datetime_utils.py +49 -0
- src2purl-1.3.2.dist-info/METADATA +279 -0
- src2purl-1.3.2.dist-info/RECORD +33 -0
- src2purl-1.3.2.dist-info/WHEEL +5 -0
- src2purl-1.3.2.dist-info/entry_points.txt +3 -0
- src2purl-1.3.2.dist-info/licenses/LICENSE +661 -0
- src2purl-1.3.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
"""Integration with oslili for enhanced license detection.
|
|
2
|
+
|
|
3
|
+
This module provides integration with the oslili (Open Source License
|
|
4
|
+
Identification Library) tool for more accurate license detection in packages.
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
from pathlib import Path
|
|
8
|
+
from typing import Any, Dict, List, Optional
|
|
9
|
+
|
|
10
|
+
try:
|
|
11
|
+
from semantic_copycat_oslili import (
|
|
12
|
+
LicenseCopyrightDetector,
|
|
13
|
+
DetectionResult,
|
|
14
|
+
DetectedLicense,
|
|
15
|
+
CopyrightInfo,
|
|
16
|
+
Config
|
|
17
|
+
)
|
|
18
|
+
HAS_OSLILI = True
|
|
19
|
+
except ImportError:
|
|
20
|
+
HAS_OSLILI = False
|
|
21
|
+
Config = None # Define Config as None for type hints when oslili not available
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
class OsliliIntegration:
|
|
25
|
+
"""Integration with oslili for license detection."""
|
|
26
|
+
|
|
27
|
+
def __init__(self, config: Optional[Any] = None):
|
|
28
|
+
"""Initialize oslili integration.
|
|
29
|
+
|
|
30
|
+
Args:
|
|
31
|
+
config: Optional Config object for oslili configuration
|
|
32
|
+
"""
|
|
33
|
+
self.available = HAS_OSLILI
|
|
34
|
+
if self.available:
|
|
35
|
+
try:
|
|
36
|
+
# Create config if not provided
|
|
37
|
+
if config is None and HAS_OSLILI:
|
|
38
|
+
config = Config(
|
|
39
|
+
verbose=False,
|
|
40
|
+
debug=False,
|
|
41
|
+
thread_count=4,
|
|
42
|
+
similarity_threshold=0.97,
|
|
43
|
+
max_recursion_depth=5
|
|
44
|
+
)
|
|
45
|
+
self.detector = LicenseCopyrightDetector(config)
|
|
46
|
+
except Exception:
|
|
47
|
+
self.detector = None
|
|
48
|
+
self.available = False
|
|
49
|
+
else:
|
|
50
|
+
self.detector = None
|
|
51
|
+
|
|
52
|
+
def detect_licenses(self, path: Path) -> Dict[str, Any]:
|
|
53
|
+
"""
|
|
54
|
+
Detect licenses in a directory or file.
|
|
55
|
+
|
|
56
|
+
Args:
|
|
57
|
+
path: Path to analyze
|
|
58
|
+
|
|
59
|
+
Returns:
|
|
60
|
+
Dictionary with license information:
|
|
61
|
+
- licenses: List of detected license identifiers
|
|
62
|
+
- confidence: Confidence score (0-1)
|
|
63
|
+
- files: Dict mapping files to their licenses
|
|
64
|
+
- summary: Human-readable summary
|
|
65
|
+
"""
|
|
66
|
+
if not self.available or not self.detector:
|
|
67
|
+
return {
|
|
68
|
+
"licenses": [],
|
|
69
|
+
"confidence": 0.0,
|
|
70
|
+
"files": {},
|
|
71
|
+
"summary": "oslili not available",
|
|
72
|
+
"error": "oslili integration not available"
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
try:
|
|
76
|
+
# Process the path using oslili v1.3.2
|
|
77
|
+
result: DetectionResult = self.detector.process_local_path(str(path))
|
|
78
|
+
|
|
79
|
+
if result and result.licenses:
|
|
80
|
+
# Extract license information
|
|
81
|
+
licenses = []
|
|
82
|
+
confidence_scores = []
|
|
83
|
+
files = {}
|
|
84
|
+
|
|
85
|
+
# Group licenses by category for better understanding
|
|
86
|
+
declared = [l for l in result.licenses if l.category == "declared"]
|
|
87
|
+
detected = [l for l in result.licenses if l.category == "detected"]
|
|
88
|
+
referenced = [l for l in result.licenses if l.category == "referenced"]
|
|
89
|
+
|
|
90
|
+
# Prioritize declared licenses, then detected, then referenced
|
|
91
|
+
all_licenses = declared + detected + referenced
|
|
92
|
+
|
|
93
|
+
for license_info in all_licenses:
|
|
94
|
+
licenses.append(license_info.spdx_id)
|
|
95
|
+
confidence_scores.append(license_info.confidence)
|
|
96
|
+
|
|
97
|
+
# Map files to licenses
|
|
98
|
+
if license_info.source_file:
|
|
99
|
+
if license_info.source_file not in files:
|
|
100
|
+
files[license_info.source_file] = []
|
|
101
|
+
files[license_info.source_file].append({
|
|
102
|
+
"spdx_id": license_info.spdx_id,
|
|
103
|
+
"confidence": license_info.confidence,
|
|
104
|
+
"category": license_info.category,
|
|
105
|
+
"method": license_info.detection_method
|
|
106
|
+
})
|
|
107
|
+
|
|
108
|
+
# Calculate average confidence
|
|
109
|
+
avg_confidence = sum(confidence_scores) / len(confidence_scores) if confidence_scores else 0.8
|
|
110
|
+
|
|
111
|
+
# Generate summary
|
|
112
|
+
if licenses:
|
|
113
|
+
# Remove duplicates while preserving order
|
|
114
|
+
unique_licenses = self._deduplicate_licenses(licenses)
|
|
115
|
+
summary = f"Found {len(unique_licenses)} unique license(s): {', '.join(unique_licenses[:3])}"
|
|
116
|
+
if len(unique_licenses) > 3:
|
|
117
|
+
summary += f" and {len(unique_licenses) - 3} more"
|
|
118
|
+
|
|
119
|
+
# Add category info to summary
|
|
120
|
+
if declared:
|
|
121
|
+
summary += f" (Declared: {len(declared)})"
|
|
122
|
+
if detected:
|
|
123
|
+
summary += f" (Detected: {len(detected)})"
|
|
124
|
+
else:
|
|
125
|
+
summary = "No licenses detected"
|
|
126
|
+
|
|
127
|
+
return {
|
|
128
|
+
"licenses": licenses,
|
|
129
|
+
"confidence": avg_confidence,
|
|
130
|
+
"files": files,
|
|
131
|
+
"summary": summary,
|
|
132
|
+
"copyrights": [c.to_dict() for c in result.copyrights] if result.copyrights else []
|
|
133
|
+
}
|
|
134
|
+
else:
|
|
135
|
+
return {
|
|
136
|
+
"licenses": [],
|
|
137
|
+
"confidence": 0.0,
|
|
138
|
+
"files": {},
|
|
139
|
+
"summary": "No licenses detected",
|
|
140
|
+
"copyrights": []
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
except Exception as e:
|
|
144
|
+
return {
|
|
145
|
+
"licenses": [],
|
|
146
|
+
"confidence": 0.0,
|
|
147
|
+
"files": {},
|
|
148
|
+
"summary": f"Error during detection: {e}",
|
|
149
|
+
"error": str(e)
|
|
150
|
+
}
|
|
151
|
+
|
|
152
|
+
def _deduplicate_licenses(self, licenses: List[str]) -> List[str]:
|
|
153
|
+
"""Remove duplicates while preserving order."""
|
|
154
|
+
return list(dict.fromkeys(licenses))
|
|
155
|
+
|
|
156
|
+
def enhance_package_match(self, match: "PackageMatch", path: Path) -> "PackageMatch":
|
|
157
|
+
"""
|
|
158
|
+
Enhance a package match with oslili license detection.
|
|
159
|
+
|
|
160
|
+
Args:
|
|
161
|
+
match: Package match to enhance
|
|
162
|
+
path: Path to the package directory
|
|
163
|
+
|
|
164
|
+
Returns:
|
|
165
|
+
Enhanced package match with better license information
|
|
166
|
+
"""
|
|
167
|
+
if not self.available:
|
|
168
|
+
return match
|
|
169
|
+
|
|
170
|
+
# Detect licenses
|
|
171
|
+
license_info = self.detect_licenses(path)
|
|
172
|
+
|
|
173
|
+
# Update match if we found licenses with good confidence
|
|
174
|
+
if license_info["licenses"] and license_info["confidence"] > 0.7:
|
|
175
|
+
# Remove duplicates while preserving order
|
|
176
|
+
unique_licenses = self._deduplicate_licenses(license_info["licenses"])
|
|
177
|
+
|
|
178
|
+
# Use the most confident license as primary
|
|
179
|
+
primary_license = unique_licenses[0]
|
|
180
|
+
|
|
181
|
+
# Update match license if not already set or if ours is more confident
|
|
182
|
+
if not match.license or license_info["confidence"] > 0.85:
|
|
183
|
+
match.license = primary_license
|
|
184
|
+
|
|
185
|
+
# Add metadata about additional licenses and copyrights
|
|
186
|
+
if not hasattr(match, "metadata"):
|
|
187
|
+
match.metadata = {}
|
|
188
|
+
|
|
189
|
+
if len(unique_licenses) > 1:
|
|
190
|
+
match.metadata["additional_licenses"] = unique_licenses[1:]
|
|
191
|
+
|
|
192
|
+
match.metadata["license_confidence"] = license_info["confidence"]
|
|
193
|
+
|
|
194
|
+
# Add copyright information if available
|
|
195
|
+
if license_info.get("copyrights"):
|
|
196
|
+
match.metadata["copyrights"] = license_info["copyrights"]
|
|
197
|
+
|
|
198
|
+
# Add file mapping if available
|
|
199
|
+
if license_info.get("files"):
|
|
200
|
+
match.metadata["license_files"] = license_info["files"]
|
|
201
|
+
|
|
202
|
+
return match
|
|
203
|
+
|
|
204
|
+
|
|
205
|
+
|
|
206
|
+
def enhance_with_oslili(package_matches: List["PackageMatch"], base_path: Path) -> List["PackageMatch"]:
|
|
207
|
+
"""
|
|
208
|
+
Enhance package matches with oslili license detection.
|
|
209
|
+
|
|
210
|
+
Args:
|
|
211
|
+
package_matches: List of package matches to enhance
|
|
212
|
+
base_path: Base path where the code is located
|
|
213
|
+
|
|
214
|
+
Returns:
|
|
215
|
+
Enhanced package matches
|
|
216
|
+
"""
|
|
217
|
+
integration = OsliliIntegration()
|
|
218
|
+
|
|
219
|
+
if not integration.available:
|
|
220
|
+
print("oslili not available - skipping license enhancement")
|
|
221
|
+
return package_matches
|
|
222
|
+
|
|
223
|
+
enhanced_matches = []
|
|
224
|
+
for match in package_matches:
|
|
225
|
+
enhanced = integration.enhance_package_match(match, base_path)
|
|
226
|
+
enhanced_matches.append(enhanced)
|
|
227
|
+
|
|
228
|
+
return enhanced_matches
|
|
@@ -0,0 +1,305 @@
|
|
|
1
|
+
"""Integration with UPMEX for package metadata extraction."""
|
|
2
|
+
|
|
3
|
+
import json
|
|
4
|
+
import logging
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Dict, List, Optional, Set
|
|
7
|
+
|
|
8
|
+
from src2id.core.models import PackageMatch, MatchType
|
|
9
|
+
from src2id.integrations.manifest_parser import DirectManifestParser
|
|
10
|
+
|
|
11
|
+
logger = logging.getLogger(__name__)
|
|
12
|
+
|
|
13
|
+
class UpmexIntegration:
|
|
14
|
+
"""Integration with UPMEX Universal Package Metadata Extractor."""
|
|
15
|
+
|
|
16
|
+
# Package file patterns that UPMEX can handle
|
|
17
|
+
UPMEX_SUPPORTED_FILES = {
|
|
18
|
+
# Python
|
|
19
|
+
'setup.py', 'pyproject.toml', 'setup.cfg',
|
|
20
|
+
# Node.js/NPM
|
|
21
|
+
'package.json', 'package-lock.json',
|
|
22
|
+
# Java/Maven
|
|
23
|
+
'pom.xml',
|
|
24
|
+
# Gradle
|
|
25
|
+
'build.gradle', 'build.gradle.kts',
|
|
26
|
+
# Go
|
|
27
|
+
'go.mod', 'go.sum',
|
|
28
|
+
# Rust
|
|
29
|
+
'Cargo.toml', 'Cargo.lock',
|
|
30
|
+
# Ruby
|
|
31
|
+
'*.gemspec', 'Gemfile', 'Gemfile.lock',
|
|
32
|
+
# CocoaPods
|
|
33
|
+
'*.podspec', '*.podspec.json',
|
|
34
|
+
# Conda
|
|
35
|
+
'meta.yaml', 'conda.yaml',
|
|
36
|
+
# Perl
|
|
37
|
+
'META.json', 'META.yml', 'Makefile.PL',
|
|
38
|
+
# Conan
|
|
39
|
+
'conanfile.py', 'conanfile.txt',
|
|
40
|
+
# NuGet
|
|
41
|
+
'*.csproj', '*.nuspec', 'packages.config',
|
|
42
|
+
# Debian
|
|
43
|
+
'control', 'debian/control',
|
|
44
|
+
}
|
|
45
|
+
|
|
46
|
+
def __init__(self, enabled: bool = True):
|
|
47
|
+
"""
|
|
48
|
+
Initialize UPMEX integration.
|
|
49
|
+
|
|
50
|
+
Args:
|
|
51
|
+
enabled: Whether UPMEX integration is enabled
|
|
52
|
+
"""
|
|
53
|
+
self.enabled = enabled
|
|
54
|
+
self._manifest_parser = DirectManifestParser()
|
|
55
|
+
|
|
56
|
+
if enabled:
|
|
57
|
+
from upmex import PackageExtractor
|
|
58
|
+
self._extractor = PackageExtractor()
|
|
59
|
+
logger.info("UPMEX integration initialized successfully")
|
|
60
|
+
|
|
61
|
+
def scan_directory_for_packages(self, directory: Path) -> List[Path]:
|
|
62
|
+
"""
|
|
63
|
+
Scan directory for package files using manifest parser.
|
|
64
|
+
|
|
65
|
+
Args:
|
|
66
|
+
directory: Directory to scan
|
|
67
|
+
|
|
68
|
+
Returns:
|
|
69
|
+
List of package file paths found
|
|
70
|
+
"""
|
|
71
|
+
if not self.enabled:
|
|
72
|
+
return []
|
|
73
|
+
|
|
74
|
+
return self._manifest_parser.scan_directory_for_manifests(directory)
|
|
75
|
+
|
|
76
|
+
# Note: Directory scanning is now handled by DirectManifestParser
|
|
77
|
+
|
|
78
|
+
def extract_metadata_from_file(self, file_path: Path) -> Optional[PackageMatch]:
|
|
79
|
+
"""
|
|
80
|
+
Extract package metadata from a single package file.
|
|
81
|
+
|
|
82
|
+
Args:
|
|
83
|
+
file_path: Path to package file
|
|
84
|
+
|
|
85
|
+
Returns:
|
|
86
|
+
PackageMatch if extraction successful, None otherwise
|
|
87
|
+
"""
|
|
88
|
+
if not self.enabled:
|
|
89
|
+
return None
|
|
90
|
+
|
|
91
|
+
try:
|
|
92
|
+
# Try UPMEX extractor first
|
|
93
|
+
upmex_metadata = self._extractor.extract_from_file(str(file_path))
|
|
94
|
+
if upmex_metadata:
|
|
95
|
+
return self._convert_upmex_metadata(upmex_metadata, file_path)
|
|
96
|
+
except Exception as e:
|
|
97
|
+
logger.debug(f"UPMEX extraction failed for {file_path}: {e}")
|
|
98
|
+
|
|
99
|
+
# Fallback to direct manifest parser
|
|
100
|
+
return self._manifest_parser.extract_metadata_from_file(file_path)
|
|
101
|
+
|
|
102
|
+
def extract_metadata_from_directory(self, directory: Path) -> List[PackageMatch]:
|
|
103
|
+
"""
|
|
104
|
+
Extract metadata from all package files found in a directory.
|
|
105
|
+
|
|
106
|
+
Args:
|
|
107
|
+
directory: Directory to scan and extract from
|
|
108
|
+
|
|
109
|
+
Returns:
|
|
110
|
+
List of PackageMatch objects (deduplicated by package name)
|
|
111
|
+
"""
|
|
112
|
+
if not self.enabled:
|
|
113
|
+
return []
|
|
114
|
+
|
|
115
|
+
matches = []
|
|
116
|
+
|
|
117
|
+
try:
|
|
118
|
+
# Try UPMEX extractor for directory
|
|
119
|
+
upmex_results = self._extractor.extract_from_directory(str(directory))
|
|
120
|
+
for upmex_metadata in upmex_results:
|
|
121
|
+
if upmex_metadata:
|
|
122
|
+
match = self._convert_upmex_metadata(upmex_metadata, directory)
|
|
123
|
+
if match:
|
|
124
|
+
matches.append(match)
|
|
125
|
+
except Exception as e:
|
|
126
|
+
logger.debug(f"UPMEX directory extraction failed for {directory}: {e}")
|
|
127
|
+
|
|
128
|
+
# Also use direct manifest parser for additional coverage
|
|
129
|
+
manifest_matches = self._manifest_parser.extract_metadata_from_directory(directory)
|
|
130
|
+
matches.extend(manifest_matches)
|
|
131
|
+
|
|
132
|
+
# Deduplicate by package name
|
|
133
|
+
seen_packages = set()
|
|
134
|
+
unique_matches = []
|
|
135
|
+
for match in matches:
|
|
136
|
+
if match.name and match.name not in seen_packages:
|
|
137
|
+
seen_packages.add(match.name)
|
|
138
|
+
unique_matches.append(match)
|
|
139
|
+
|
|
140
|
+
return unique_matches
|
|
141
|
+
|
|
142
|
+
def _convert_upmex_metadata(self, upmex_metadata, source_file: Path) -> PackageMatch:
|
|
143
|
+
"""
|
|
144
|
+
Convert UPMEX metadata to PackageMatch format.
|
|
145
|
+
|
|
146
|
+
Args:
|
|
147
|
+
upmex_metadata: UPMEX PackageMetadata object
|
|
148
|
+
source_file: Source file that was processed
|
|
149
|
+
|
|
150
|
+
Returns:
|
|
151
|
+
PackageMatch object
|
|
152
|
+
"""
|
|
153
|
+
# Extract basic package information
|
|
154
|
+
name = getattr(upmex_metadata, 'name', None)
|
|
155
|
+
version = getattr(upmex_metadata, 'version', None)
|
|
156
|
+
|
|
157
|
+
# Debug: Print the metadata structure
|
|
158
|
+
if logger.isEnabledFor(logging.DEBUG):
|
|
159
|
+
logger.debug(f"UPMEX metadata for {source_file.name}: {vars(upmex_metadata)}")
|
|
160
|
+
|
|
161
|
+
# Get download URL from repository information
|
|
162
|
+
download_url = self._extract_download_url(upmex_metadata)
|
|
163
|
+
|
|
164
|
+
# Extract license information
|
|
165
|
+
license_info = self._extract_license_info(upmex_metadata)
|
|
166
|
+
|
|
167
|
+
# Generate PURL if available
|
|
168
|
+
purl = getattr(upmex_metadata, 'purl', None)
|
|
169
|
+
|
|
170
|
+
# Determine confidence based on data completeness
|
|
171
|
+
confidence = self._calculate_confidence(upmex_metadata)
|
|
172
|
+
|
|
173
|
+
# Create PackageMatch
|
|
174
|
+
package_match = PackageMatch(
|
|
175
|
+
download_url=download_url or f"file://{source_file.parent}",
|
|
176
|
+
match_type=MatchType.EXACT, # Direct metadata extraction is exact
|
|
177
|
+
confidence_score=confidence,
|
|
178
|
+
name=name,
|
|
179
|
+
version=version,
|
|
180
|
+
license=license_info,
|
|
181
|
+
sh_url=None, # No Software Heritage URL for direct extraction
|
|
182
|
+
frequency_count=0,
|
|
183
|
+
is_official_org=self._is_official_organization(download_url),
|
|
184
|
+
purl=purl,
|
|
185
|
+
)
|
|
186
|
+
|
|
187
|
+
return package_match
|
|
188
|
+
|
|
189
|
+
def _extract_download_url(self, metadata) -> Optional[str]:
|
|
190
|
+
"""Extract download URL from UPMEX metadata."""
|
|
191
|
+
# Try to get repository URL
|
|
192
|
+
if hasattr(metadata, 'repository') and metadata.repository:
|
|
193
|
+
if isinstance(metadata.repository, dict):
|
|
194
|
+
return metadata.repository.get('url')
|
|
195
|
+
elif isinstance(metadata.repository, str):
|
|
196
|
+
return metadata.repository
|
|
197
|
+
|
|
198
|
+
# Try homepage
|
|
199
|
+
if hasattr(metadata, 'homepage') and metadata.homepage:
|
|
200
|
+
return metadata.homepage
|
|
201
|
+
|
|
202
|
+
# Try to construct from package type and name
|
|
203
|
+
if hasattr(metadata, 'package_type') and hasattr(metadata, 'name'):
|
|
204
|
+
package_type = getattr(metadata.package_type, 'value', str(metadata.package_type))
|
|
205
|
+
name = metadata.name
|
|
206
|
+
|
|
207
|
+
# Generate conventional URLs for major ecosystems
|
|
208
|
+
url_patterns = {
|
|
209
|
+
'python': f"https://pypi.org/project/{name}/",
|
|
210
|
+
'npm': f"https://www.npmjs.com/package/{name}",
|
|
211
|
+
'maven': f"https://mvnrepository.com/artifact/{name}",
|
|
212
|
+
'gem': f"https://rubygems.org/gems/{name}",
|
|
213
|
+
'cargo': f"https://crates.io/crates/{name}",
|
|
214
|
+
'go': f"https://pkg.go.dev/{name}",
|
|
215
|
+
}
|
|
216
|
+
|
|
217
|
+
return url_patterns.get(package_type.lower())
|
|
218
|
+
|
|
219
|
+
return None
|
|
220
|
+
|
|
221
|
+
def _extract_license_info(self, metadata) -> Optional[str]:
|
|
222
|
+
"""Extract license information from UPMEX metadata."""
|
|
223
|
+
# Try licenses list first
|
|
224
|
+
if hasattr(metadata, 'licenses') and metadata.licenses:
|
|
225
|
+
license_ids = []
|
|
226
|
+
for license_obj in metadata.licenses:
|
|
227
|
+
if hasattr(license_obj, 'spdx_id') and license_obj.spdx_id:
|
|
228
|
+
license_ids.append(license_obj.spdx_id)
|
|
229
|
+
elif hasattr(license_obj, 'name') and license_obj.name:
|
|
230
|
+
license_ids.append(license_obj.name)
|
|
231
|
+
|
|
232
|
+
if license_ids:
|
|
233
|
+
return ', '.join(license_ids)
|
|
234
|
+
|
|
235
|
+
# Try direct license field
|
|
236
|
+
if hasattr(metadata, 'license') and metadata.license:
|
|
237
|
+
return metadata.license
|
|
238
|
+
|
|
239
|
+
return None
|
|
240
|
+
|
|
241
|
+
def _calculate_confidence(self, metadata) -> float:
|
|
242
|
+
"""Calculate confidence score based on metadata completeness."""
|
|
243
|
+
confidence = 0.7 # Base confidence for direct metadata extraction
|
|
244
|
+
|
|
245
|
+
# Boost confidence for complete metadata
|
|
246
|
+
if hasattr(metadata, 'name') and metadata.name:
|
|
247
|
+
confidence += 0.1
|
|
248
|
+
|
|
249
|
+
if hasattr(metadata, 'version') and metadata.version:
|
|
250
|
+
confidence += 0.1
|
|
251
|
+
|
|
252
|
+
if hasattr(metadata, 'licenses') and metadata.licenses:
|
|
253
|
+
confidence += 0.05
|
|
254
|
+
|
|
255
|
+
if (hasattr(metadata, 'repository') and metadata.repository) or \
|
|
256
|
+
(hasattr(metadata, 'homepage') and metadata.homepage):
|
|
257
|
+
confidence += 0.05
|
|
258
|
+
|
|
259
|
+
return min(1.0, confidence)
|
|
260
|
+
|
|
261
|
+
def _is_official_organization(self, url: Optional[str]) -> bool:
|
|
262
|
+
"""Check if URL belongs to an official organization."""
|
|
263
|
+
if not url:
|
|
264
|
+
return False
|
|
265
|
+
|
|
266
|
+
# Official organization patterns
|
|
267
|
+
official_patterns = [
|
|
268
|
+
'github.com/python/',
|
|
269
|
+
'github.com/nodejs/',
|
|
270
|
+
'github.com/golang/',
|
|
271
|
+
'github.com/rust-lang/',
|
|
272
|
+
'github.com/microsoft/',
|
|
273
|
+
'github.com/google/',
|
|
274
|
+
'github.com/apache/',
|
|
275
|
+
'github.com/eclipse/',
|
|
276
|
+
'gitlab.com/gitlab-org/',
|
|
277
|
+
]
|
|
278
|
+
|
|
279
|
+
return any(pattern in url.lower() for pattern in official_patterns)
|
|
280
|
+
|
|
281
|
+
def _deduplicate_matches(self, matches: List[PackageMatch]) -> List[PackageMatch]:
|
|
282
|
+
"""
|
|
283
|
+
Deduplicate matches by package name, keeping the best quality match.
|
|
284
|
+
|
|
285
|
+
Args:
|
|
286
|
+
matches: List of package matches
|
|
287
|
+
|
|
288
|
+
Returns:
|
|
289
|
+
Deduplicated list of matches
|
|
290
|
+
"""
|
|
291
|
+
if not matches:
|
|
292
|
+
return matches
|
|
293
|
+
|
|
294
|
+
# Group by package name
|
|
295
|
+
grouped = {}
|
|
296
|
+
for match in matches:
|
|
297
|
+
name = match.name or "unknown"
|
|
298
|
+
if name not in grouped or match.confidence_score > grouped[name].confidence_score:
|
|
299
|
+
grouped[name] = match
|
|
300
|
+
|
|
301
|
+
return list(grouped.values())
|
|
302
|
+
|
|
303
|
+
def get_supported_file_types(self) -> Set[str]:
|
|
304
|
+
"""Get set of file types supported by UPMEX integration."""
|
|
305
|
+
return self.UPMEX_SUPPORTED_FILES.copy()
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""Unified search module for source identification."""
|
|
2
|
+
|
|
3
|
+
from .providers import (
|
|
4
|
+
SearchProvider,
|
|
5
|
+
SerpAPIProvider,
|
|
6
|
+
GitHubSearchProvider,
|
|
7
|
+
SourcegraphProvider,
|
|
8
|
+
SCANOSSProvider,
|
|
9
|
+
SearchProviderRegistry,
|
|
10
|
+
create_default_registry
|
|
11
|
+
)
|
|
12
|
+
|
|
13
|
+
from .strategies import (
|
|
14
|
+
SourceIdentifier,
|
|
15
|
+
identify_source
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
from .hash_search import HashSearcher
|
|
19
|
+
|
|
20
|
+
__all__ = [
|
|
21
|
+
# Providers
|
|
22
|
+
'SearchProvider',
|
|
23
|
+
'SerpAPIProvider',
|
|
24
|
+
'GitHubSearchProvider',
|
|
25
|
+
'SourcegraphProvider',
|
|
26
|
+
'SCANOSSProvider',
|
|
27
|
+
'SearchProviderRegistry',
|
|
28
|
+
'create_default_registry',
|
|
29
|
+
# Strategies
|
|
30
|
+
'SourceIdentifier',
|
|
31
|
+
'identify_source',
|
|
32
|
+
# Hash search
|
|
33
|
+
'HashSearcher',
|
|
34
|
+
]
|