src2purl 1.3.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,228 @@
1
+ """Integration with oslili for enhanced license detection.
2
+
3
+ This module provides integration with the oslili (Open Source License
4
+ Identification Library) tool for more accurate license detection in packages.
5
+ """
6
+
7
+ from pathlib import Path
8
+ from typing import Any, Dict, List, Optional
9
+
10
+ try:
11
+ from semantic_copycat_oslili import (
12
+ LicenseCopyrightDetector,
13
+ DetectionResult,
14
+ DetectedLicense,
15
+ CopyrightInfo,
16
+ Config
17
+ )
18
+ HAS_OSLILI = True
19
+ except ImportError:
20
+ HAS_OSLILI = False
21
+ Config = None # Define Config as None for type hints when oslili not available
22
+
23
+
24
+ class OsliliIntegration:
25
+ """Integration with oslili for license detection."""
26
+
27
+ def __init__(self, config: Optional[Any] = None):
28
+ """Initialize oslili integration.
29
+
30
+ Args:
31
+ config: Optional Config object for oslili configuration
32
+ """
33
+ self.available = HAS_OSLILI
34
+ if self.available:
35
+ try:
36
+ # Create config if not provided
37
+ if config is None and HAS_OSLILI:
38
+ config = Config(
39
+ verbose=False,
40
+ debug=False,
41
+ thread_count=4,
42
+ similarity_threshold=0.97,
43
+ max_recursion_depth=5
44
+ )
45
+ self.detector = LicenseCopyrightDetector(config)
46
+ except Exception:
47
+ self.detector = None
48
+ self.available = False
49
+ else:
50
+ self.detector = None
51
+
52
+ def detect_licenses(self, path: Path) -> Dict[str, Any]:
53
+ """
54
+ Detect licenses in a directory or file.
55
+
56
+ Args:
57
+ path: Path to analyze
58
+
59
+ Returns:
60
+ Dictionary with license information:
61
+ - licenses: List of detected license identifiers
62
+ - confidence: Confidence score (0-1)
63
+ - files: Dict mapping files to their licenses
64
+ - summary: Human-readable summary
65
+ """
66
+ if not self.available or not self.detector:
67
+ return {
68
+ "licenses": [],
69
+ "confidence": 0.0,
70
+ "files": {},
71
+ "summary": "oslili not available",
72
+ "error": "oslili integration not available"
73
+ }
74
+
75
+ try:
76
+ # Process the path using oslili v1.3.2
77
+ result: DetectionResult = self.detector.process_local_path(str(path))
78
+
79
+ if result and result.licenses:
80
+ # Extract license information
81
+ licenses = []
82
+ confidence_scores = []
83
+ files = {}
84
+
85
+ # Group licenses by category for better understanding
86
+ declared = [l for l in result.licenses if l.category == "declared"]
87
+ detected = [l for l in result.licenses if l.category == "detected"]
88
+ referenced = [l for l in result.licenses if l.category == "referenced"]
89
+
90
+ # Prioritize declared licenses, then detected, then referenced
91
+ all_licenses = declared + detected + referenced
92
+
93
+ for license_info in all_licenses:
94
+ licenses.append(license_info.spdx_id)
95
+ confidence_scores.append(license_info.confidence)
96
+
97
+ # Map files to licenses
98
+ if license_info.source_file:
99
+ if license_info.source_file not in files:
100
+ files[license_info.source_file] = []
101
+ files[license_info.source_file].append({
102
+ "spdx_id": license_info.spdx_id,
103
+ "confidence": license_info.confidence,
104
+ "category": license_info.category,
105
+ "method": license_info.detection_method
106
+ })
107
+
108
+ # Calculate average confidence
109
+ avg_confidence = sum(confidence_scores) / len(confidence_scores) if confidence_scores else 0.8
110
+
111
+ # Generate summary
112
+ if licenses:
113
+ # Remove duplicates while preserving order
114
+ unique_licenses = self._deduplicate_licenses(licenses)
115
+ summary = f"Found {len(unique_licenses)} unique license(s): {', '.join(unique_licenses[:3])}"
116
+ if len(unique_licenses) > 3:
117
+ summary += f" and {len(unique_licenses) - 3} more"
118
+
119
+ # Add category info to summary
120
+ if declared:
121
+ summary += f" (Declared: {len(declared)})"
122
+ if detected:
123
+ summary += f" (Detected: {len(detected)})"
124
+ else:
125
+ summary = "No licenses detected"
126
+
127
+ return {
128
+ "licenses": licenses,
129
+ "confidence": avg_confidence,
130
+ "files": files,
131
+ "summary": summary,
132
+ "copyrights": [c.to_dict() for c in result.copyrights] if result.copyrights else []
133
+ }
134
+ else:
135
+ return {
136
+ "licenses": [],
137
+ "confidence": 0.0,
138
+ "files": {},
139
+ "summary": "No licenses detected",
140
+ "copyrights": []
141
+ }
142
+
143
+ except Exception as e:
144
+ return {
145
+ "licenses": [],
146
+ "confidence": 0.0,
147
+ "files": {},
148
+ "summary": f"Error during detection: {e}",
149
+ "error": str(e)
150
+ }
151
+
152
+ def _deduplicate_licenses(self, licenses: List[str]) -> List[str]:
153
+ """Remove duplicates while preserving order."""
154
+ return list(dict.fromkeys(licenses))
155
+
156
+ def enhance_package_match(self, match: "PackageMatch", path: Path) -> "PackageMatch":
157
+ """
158
+ Enhance a package match with oslili license detection.
159
+
160
+ Args:
161
+ match: Package match to enhance
162
+ path: Path to the package directory
163
+
164
+ Returns:
165
+ Enhanced package match with better license information
166
+ """
167
+ if not self.available:
168
+ return match
169
+
170
+ # Detect licenses
171
+ license_info = self.detect_licenses(path)
172
+
173
+ # Update match if we found licenses with good confidence
174
+ if license_info["licenses"] and license_info["confidence"] > 0.7:
175
+ # Remove duplicates while preserving order
176
+ unique_licenses = self._deduplicate_licenses(license_info["licenses"])
177
+
178
+ # Use the most confident license as primary
179
+ primary_license = unique_licenses[0]
180
+
181
+ # Update match license if not already set or if ours is more confident
182
+ if not match.license or license_info["confidence"] > 0.85:
183
+ match.license = primary_license
184
+
185
+ # Add metadata about additional licenses and copyrights
186
+ if not hasattr(match, "metadata"):
187
+ match.metadata = {}
188
+
189
+ if len(unique_licenses) > 1:
190
+ match.metadata["additional_licenses"] = unique_licenses[1:]
191
+
192
+ match.metadata["license_confidence"] = license_info["confidence"]
193
+
194
+ # Add copyright information if available
195
+ if license_info.get("copyrights"):
196
+ match.metadata["copyrights"] = license_info["copyrights"]
197
+
198
+ # Add file mapping if available
199
+ if license_info.get("files"):
200
+ match.metadata["license_files"] = license_info["files"]
201
+
202
+ return match
203
+
204
+
205
+
206
+ def enhance_with_oslili(package_matches: List["PackageMatch"], base_path: Path) -> List["PackageMatch"]:
207
+ """
208
+ Enhance package matches with oslili license detection.
209
+
210
+ Args:
211
+ package_matches: List of package matches to enhance
212
+ base_path: Base path where the code is located
213
+
214
+ Returns:
215
+ Enhanced package matches
216
+ """
217
+ integration = OsliliIntegration()
218
+
219
+ if not integration.available:
220
+ print("oslili not available - skipping license enhancement")
221
+ return package_matches
222
+
223
+ enhanced_matches = []
224
+ for match in package_matches:
225
+ enhanced = integration.enhance_package_match(match, base_path)
226
+ enhanced_matches.append(enhanced)
227
+
228
+ return enhanced_matches
@@ -0,0 +1,305 @@
1
+ """Integration with UPMEX for package metadata extraction."""
2
+
3
+ import json
4
+ import logging
5
+ from pathlib import Path
6
+ from typing import Dict, List, Optional, Set
7
+
8
+ from src2id.core.models import PackageMatch, MatchType
9
+ from src2id.integrations.manifest_parser import DirectManifestParser
10
+
11
+ logger = logging.getLogger(__name__)
12
+
13
+ class UpmexIntegration:
14
+ """Integration with UPMEX Universal Package Metadata Extractor."""
15
+
16
+ # Package file patterns that UPMEX can handle
17
+ UPMEX_SUPPORTED_FILES = {
18
+ # Python
19
+ 'setup.py', 'pyproject.toml', 'setup.cfg',
20
+ # Node.js/NPM
21
+ 'package.json', 'package-lock.json',
22
+ # Java/Maven
23
+ 'pom.xml',
24
+ # Gradle
25
+ 'build.gradle', 'build.gradle.kts',
26
+ # Go
27
+ 'go.mod', 'go.sum',
28
+ # Rust
29
+ 'Cargo.toml', 'Cargo.lock',
30
+ # Ruby
31
+ '*.gemspec', 'Gemfile', 'Gemfile.lock',
32
+ # CocoaPods
33
+ '*.podspec', '*.podspec.json',
34
+ # Conda
35
+ 'meta.yaml', 'conda.yaml',
36
+ # Perl
37
+ 'META.json', 'META.yml', 'Makefile.PL',
38
+ # Conan
39
+ 'conanfile.py', 'conanfile.txt',
40
+ # NuGet
41
+ '*.csproj', '*.nuspec', 'packages.config',
42
+ # Debian
43
+ 'control', 'debian/control',
44
+ }
45
+
46
+ def __init__(self, enabled: bool = True):
47
+ """
48
+ Initialize UPMEX integration.
49
+
50
+ Args:
51
+ enabled: Whether UPMEX integration is enabled
52
+ """
53
+ self.enabled = enabled
54
+ self._manifest_parser = DirectManifestParser()
55
+
56
+ if enabled:
57
+ from upmex import PackageExtractor
58
+ self._extractor = PackageExtractor()
59
+ logger.info("UPMEX integration initialized successfully")
60
+
61
+ def scan_directory_for_packages(self, directory: Path) -> List[Path]:
62
+ """
63
+ Scan directory for package files using manifest parser.
64
+
65
+ Args:
66
+ directory: Directory to scan
67
+
68
+ Returns:
69
+ List of package file paths found
70
+ """
71
+ if not self.enabled:
72
+ return []
73
+
74
+ return self._manifest_parser.scan_directory_for_manifests(directory)
75
+
76
+ # Note: Directory scanning is now handled by DirectManifestParser
77
+
78
+ def extract_metadata_from_file(self, file_path: Path) -> Optional[PackageMatch]:
79
+ """
80
+ Extract package metadata from a single package file.
81
+
82
+ Args:
83
+ file_path: Path to package file
84
+
85
+ Returns:
86
+ PackageMatch if extraction successful, None otherwise
87
+ """
88
+ if not self.enabled:
89
+ return None
90
+
91
+ try:
92
+ # Try UPMEX extractor first
93
+ upmex_metadata = self._extractor.extract_from_file(str(file_path))
94
+ if upmex_metadata:
95
+ return self._convert_upmex_metadata(upmex_metadata, file_path)
96
+ except Exception as e:
97
+ logger.debug(f"UPMEX extraction failed for {file_path}: {e}")
98
+
99
+ # Fallback to direct manifest parser
100
+ return self._manifest_parser.extract_metadata_from_file(file_path)
101
+
102
+ def extract_metadata_from_directory(self, directory: Path) -> List[PackageMatch]:
103
+ """
104
+ Extract metadata from all package files found in a directory.
105
+
106
+ Args:
107
+ directory: Directory to scan and extract from
108
+
109
+ Returns:
110
+ List of PackageMatch objects (deduplicated by package name)
111
+ """
112
+ if not self.enabled:
113
+ return []
114
+
115
+ matches = []
116
+
117
+ try:
118
+ # Try UPMEX extractor for directory
119
+ upmex_results = self._extractor.extract_from_directory(str(directory))
120
+ for upmex_metadata in upmex_results:
121
+ if upmex_metadata:
122
+ match = self._convert_upmex_metadata(upmex_metadata, directory)
123
+ if match:
124
+ matches.append(match)
125
+ except Exception as e:
126
+ logger.debug(f"UPMEX directory extraction failed for {directory}: {e}")
127
+
128
+ # Also use direct manifest parser for additional coverage
129
+ manifest_matches = self._manifest_parser.extract_metadata_from_directory(directory)
130
+ matches.extend(manifest_matches)
131
+
132
+ # Deduplicate by package name
133
+ seen_packages = set()
134
+ unique_matches = []
135
+ for match in matches:
136
+ if match.name and match.name not in seen_packages:
137
+ seen_packages.add(match.name)
138
+ unique_matches.append(match)
139
+
140
+ return unique_matches
141
+
142
+ def _convert_upmex_metadata(self, upmex_metadata, source_file: Path) -> PackageMatch:
143
+ """
144
+ Convert UPMEX metadata to PackageMatch format.
145
+
146
+ Args:
147
+ upmex_metadata: UPMEX PackageMetadata object
148
+ source_file: Source file that was processed
149
+
150
+ Returns:
151
+ PackageMatch object
152
+ """
153
+ # Extract basic package information
154
+ name = getattr(upmex_metadata, 'name', None)
155
+ version = getattr(upmex_metadata, 'version', None)
156
+
157
+ # Debug: Print the metadata structure
158
+ if logger.isEnabledFor(logging.DEBUG):
159
+ logger.debug(f"UPMEX metadata for {source_file.name}: {vars(upmex_metadata)}")
160
+
161
+ # Get download URL from repository information
162
+ download_url = self._extract_download_url(upmex_metadata)
163
+
164
+ # Extract license information
165
+ license_info = self._extract_license_info(upmex_metadata)
166
+
167
+ # Generate PURL if available
168
+ purl = getattr(upmex_metadata, 'purl', None)
169
+
170
+ # Determine confidence based on data completeness
171
+ confidence = self._calculate_confidence(upmex_metadata)
172
+
173
+ # Create PackageMatch
174
+ package_match = PackageMatch(
175
+ download_url=download_url or f"file://{source_file.parent}",
176
+ match_type=MatchType.EXACT, # Direct metadata extraction is exact
177
+ confidence_score=confidence,
178
+ name=name,
179
+ version=version,
180
+ license=license_info,
181
+ sh_url=None, # No Software Heritage URL for direct extraction
182
+ frequency_count=0,
183
+ is_official_org=self._is_official_organization(download_url),
184
+ purl=purl,
185
+ )
186
+
187
+ return package_match
188
+
189
+ def _extract_download_url(self, metadata) -> Optional[str]:
190
+ """Extract download URL from UPMEX metadata."""
191
+ # Try to get repository URL
192
+ if hasattr(metadata, 'repository') and metadata.repository:
193
+ if isinstance(metadata.repository, dict):
194
+ return metadata.repository.get('url')
195
+ elif isinstance(metadata.repository, str):
196
+ return metadata.repository
197
+
198
+ # Try homepage
199
+ if hasattr(metadata, 'homepage') and metadata.homepage:
200
+ return metadata.homepage
201
+
202
+ # Try to construct from package type and name
203
+ if hasattr(metadata, 'package_type') and hasattr(metadata, 'name'):
204
+ package_type = getattr(metadata.package_type, 'value', str(metadata.package_type))
205
+ name = metadata.name
206
+
207
+ # Generate conventional URLs for major ecosystems
208
+ url_patterns = {
209
+ 'python': f"https://pypi.org/project/{name}/",
210
+ 'npm': f"https://www.npmjs.com/package/{name}",
211
+ 'maven': f"https://mvnrepository.com/artifact/{name}",
212
+ 'gem': f"https://rubygems.org/gems/{name}",
213
+ 'cargo': f"https://crates.io/crates/{name}",
214
+ 'go': f"https://pkg.go.dev/{name}",
215
+ }
216
+
217
+ return url_patterns.get(package_type.lower())
218
+
219
+ return None
220
+
221
+ def _extract_license_info(self, metadata) -> Optional[str]:
222
+ """Extract license information from UPMEX metadata."""
223
+ # Try licenses list first
224
+ if hasattr(metadata, 'licenses') and metadata.licenses:
225
+ license_ids = []
226
+ for license_obj in metadata.licenses:
227
+ if hasattr(license_obj, 'spdx_id') and license_obj.spdx_id:
228
+ license_ids.append(license_obj.spdx_id)
229
+ elif hasattr(license_obj, 'name') and license_obj.name:
230
+ license_ids.append(license_obj.name)
231
+
232
+ if license_ids:
233
+ return ', '.join(license_ids)
234
+
235
+ # Try direct license field
236
+ if hasattr(metadata, 'license') and metadata.license:
237
+ return metadata.license
238
+
239
+ return None
240
+
241
+ def _calculate_confidence(self, metadata) -> float:
242
+ """Calculate confidence score based on metadata completeness."""
243
+ confidence = 0.7 # Base confidence for direct metadata extraction
244
+
245
+ # Boost confidence for complete metadata
246
+ if hasattr(metadata, 'name') and metadata.name:
247
+ confidence += 0.1
248
+
249
+ if hasattr(metadata, 'version') and metadata.version:
250
+ confidence += 0.1
251
+
252
+ if hasattr(metadata, 'licenses') and metadata.licenses:
253
+ confidence += 0.05
254
+
255
+ if (hasattr(metadata, 'repository') and metadata.repository) or \
256
+ (hasattr(metadata, 'homepage') and metadata.homepage):
257
+ confidence += 0.05
258
+
259
+ return min(1.0, confidence)
260
+
261
+ def _is_official_organization(self, url: Optional[str]) -> bool:
262
+ """Check if URL belongs to an official organization."""
263
+ if not url:
264
+ return False
265
+
266
+ # Official organization patterns
267
+ official_patterns = [
268
+ 'github.com/python/',
269
+ 'github.com/nodejs/',
270
+ 'github.com/golang/',
271
+ 'github.com/rust-lang/',
272
+ 'github.com/microsoft/',
273
+ 'github.com/google/',
274
+ 'github.com/apache/',
275
+ 'github.com/eclipse/',
276
+ 'gitlab.com/gitlab-org/',
277
+ ]
278
+
279
+ return any(pattern in url.lower() for pattern in official_patterns)
280
+
281
+ def _deduplicate_matches(self, matches: List[PackageMatch]) -> List[PackageMatch]:
282
+ """
283
+ Deduplicate matches by package name, keeping the best quality match.
284
+
285
+ Args:
286
+ matches: List of package matches
287
+
288
+ Returns:
289
+ Deduplicated list of matches
290
+ """
291
+ if not matches:
292
+ return matches
293
+
294
+ # Group by package name
295
+ grouped = {}
296
+ for match in matches:
297
+ name = match.name or "unknown"
298
+ if name not in grouped or match.confidence_score > grouped[name].confidence_score:
299
+ grouped[name] = match
300
+
301
+ return list(grouped.values())
302
+
303
+ def get_supported_file_types(self) -> Set[str]:
304
+ """Get set of file types supported by UPMEX integration."""
305
+ return self.UPMEX_SUPPORTED_FILES.copy()
@@ -0,0 +1,34 @@
1
+ """Unified search module for source identification."""
2
+
3
+ from .providers import (
4
+ SearchProvider,
5
+ SerpAPIProvider,
6
+ GitHubSearchProvider,
7
+ SourcegraphProvider,
8
+ SCANOSSProvider,
9
+ SearchProviderRegistry,
10
+ create_default_registry
11
+ )
12
+
13
+ from .strategies import (
14
+ SourceIdentifier,
15
+ identify_source
16
+ )
17
+
18
+ from .hash_search import HashSearcher
19
+
20
+ __all__ = [
21
+ # Providers
22
+ 'SearchProvider',
23
+ 'SerpAPIProvider',
24
+ 'GitHubSearchProvider',
25
+ 'SourcegraphProvider',
26
+ 'SCANOSSProvider',
27
+ 'SearchProviderRegistry',
28
+ 'create_default_registry',
29
+ # Strategies
30
+ 'SourceIdentifier',
31
+ 'identify_source',
32
+ # Hash search
33
+ 'HashSearcher',
34
+ ]