src2purl 1.3.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- src2id/__init__.py +15 -0
- src2id/cli/__init__.py +1 -0
- src2id/cli/main.py +318 -0
- src2id/cli/validate.py +93 -0
- src2id/core/__init__.py +1 -0
- src2id/core/cache.py +180 -0
- src2id/core/client.py +370 -0
- src2id/core/config.py +45 -0
- src2id/core/extractor.py +302 -0
- src2id/core/models.py +93 -0
- src2id/core/orchestrator.py +796 -0
- src2id/core/package_identifier.py +123 -0
- src2id/core/purl.py +238 -0
- src2id/core/scanner.py +369 -0
- src2id/core/scorer.py +217 -0
- src2id/core/subcomponent_detector.py +353 -0
- src2id/core/swhid.py +324 -0
- src2id/integrations/__init__.py +1 -0
- src2id/integrations/manifest_parser.py +652 -0
- src2id/integrations/oslili.py +228 -0
- src2id/integrations/upmex.py +305 -0
- src2id/search/__init__.py +34 -0
- src2id/search/hash_search.py +206 -0
- src2id/search/providers.py +310 -0
- src2id/search/strategies.py +391 -0
- src2id/utils/__init__.py +1 -0
- src2id/utils/datetime_utils.py +49 -0
- src2purl-1.3.2.dist-info/METADATA +279 -0
- src2purl-1.3.2.dist-info/RECORD +33 -0
- src2purl-1.3.2.dist-info/WHEEL +5 -0
- src2purl-1.3.2.dist-info/entry_points.txt +3 -0
- src2purl-1.3.2.dist-info/licenses/LICENSE +661 -0
- src2purl-1.3.2.dist-info/top_level.txt +1 -0
src2id/core/scanner.py
ADDED
|
@@ -0,0 +1,369 @@
|
|
|
1
|
+
"""Directory scanner for generating SWHID candidates."""
|
|
2
|
+
|
|
3
|
+
import os
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import List, Optional, Set, Tuple
|
|
6
|
+
|
|
7
|
+
from src2id.core.config import SWHPIConfig
|
|
8
|
+
from src2id.core.models import DirectoryCandidate, ContentCandidate
|
|
9
|
+
|
|
10
|
+
|
|
11
|
+
class DirectoryScanner:
|
|
12
|
+
"""Scans filesystem and generates directory candidates for SH matching."""
|
|
13
|
+
|
|
14
|
+
# Directories to skip during scanning
|
|
15
|
+
SKIP_DIRS = {
|
|
16
|
+
'.git', '.svn', '.hg', '.bzr', # Version control
|
|
17
|
+
'__pycache__', '.pytest_cache', '.mypy_cache', # Python
|
|
18
|
+
'node_modules', 'bower_components', # JavaScript
|
|
19
|
+
'target', 'build', 'dist', 'out', # Build directories
|
|
20
|
+
'.idea', '.vscode', '.vs', # IDE directories
|
|
21
|
+
'venv', 'env', '.env', # Virtual environments
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
# File extensions to consider as source code
|
|
25
|
+
SOURCE_EXTENSIONS = {
|
|
26
|
+
# C/C++
|
|
27
|
+
'.c', '.h', '.cpp', '.hpp', '.cc', '.cxx', '.hxx', '.C', '.H',
|
|
28
|
+
# Python
|
|
29
|
+
'.py', '.pyx', '.pxd', '.pyi',
|
|
30
|
+
# JavaScript/TypeScript
|
|
31
|
+
'.js', '.jsx', '.ts', '.tsx', '.mjs', '.cjs',
|
|
32
|
+
# Java/Kotlin
|
|
33
|
+
'.java', '.kt', '.kts',
|
|
34
|
+
# Rust
|
|
35
|
+
'.rs',
|
|
36
|
+
# Go
|
|
37
|
+
'.go',
|
|
38
|
+
# Ruby
|
|
39
|
+
'.rb',
|
|
40
|
+
# Swift
|
|
41
|
+
'.swift',
|
|
42
|
+
# Other
|
|
43
|
+
'.sh', '.bash', '.zsh', '.fish',
|
|
44
|
+
'.yaml', '.yml', '.json', '.xml', '.toml',
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
def __init__(self, config: SWHPIConfig, swhid_generator):
|
|
48
|
+
"""
|
|
49
|
+
Initialize the directory scanner.
|
|
50
|
+
|
|
51
|
+
Args:
|
|
52
|
+
config: Configuration settings
|
|
53
|
+
swhid_generator: SWHID generator instance
|
|
54
|
+
"""
|
|
55
|
+
self.config = config
|
|
56
|
+
self.swhid_generator = swhid_generator
|
|
57
|
+
|
|
58
|
+
def scan_recursive(self, start_path: Path) -> Tuple[List[DirectoryCandidate], List[ContentCandidate]]:
|
|
59
|
+
"""
|
|
60
|
+
Generate directory and file candidates using depth-first approach.
|
|
61
|
+
|
|
62
|
+
Scans the starting directory and files first, then subdirectories up to max_depth.
|
|
63
|
+
|
|
64
|
+
Args:
|
|
65
|
+
start_path: Starting directory path
|
|
66
|
+
|
|
67
|
+
Returns:
|
|
68
|
+
Tuple of (directory candidates, file candidates)
|
|
69
|
+
"""
|
|
70
|
+
dir_candidates = []
|
|
71
|
+
file_candidates = []
|
|
72
|
+
current = start_path.resolve()
|
|
73
|
+
|
|
74
|
+
# Track total files scanned (for limiting)
|
|
75
|
+
max_files = 100 # Reasonable limit to avoid overwhelming the API
|
|
76
|
+
files_scanned = 0
|
|
77
|
+
|
|
78
|
+
# Scan subdirectories recursively
|
|
79
|
+
def scan_directory(path: Path, depth: int):
|
|
80
|
+
nonlocal files_scanned
|
|
81
|
+
|
|
82
|
+
if depth > self.config.max_depth:
|
|
83
|
+
return
|
|
84
|
+
|
|
85
|
+
# Process current directory
|
|
86
|
+
if self._is_meaningful_directory(path):
|
|
87
|
+
self._add_candidate(dir_candidates, path, depth, current)
|
|
88
|
+
|
|
89
|
+
# Process files in current directory (only at depths we're scanning)
|
|
90
|
+
if files_scanned < max_files:
|
|
91
|
+
try:
|
|
92
|
+
for item in path.iterdir():
|
|
93
|
+
if item.is_file() and not item.name.startswith('.'):
|
|
94
|
+
# Skip very large files and common non-source files
|
|
95
|
+
if item.suffix not in {'.pyc', '.pyo', '.so', '.dll', '.exe', '.jpg', '.png', '.gif'}:
|
|
96
|
+
if item.stat().st_size < 10_000_000: # Skip files > 10MB
|
|
97
|
+
self._add_file_candidate(file_candidates, item, depth, current)
|
|
98
|
+
files_scanned += 1
|
|
99
|
+
if files_scanned >= max_files:
|
|
100
|
+
break
|
|
101
|
+
except (PermissionError, OSError):
|
|
102
|
+
pass
|
|
103
|
+
|
|
104
|
+
# Scan subdirectories
|
|
105
|
+
if depth < self.config.max_depth:
|
|
106
|
+
try:
|
|
107
|
+
for subdir in path.iterdir():
|
|
108
|
+
if (subdir.is_dir() and
|
|
109
|
+
subdir.name not in self.SKIP_DIRS and
|
|
110
|
+
not subdir.name.startswith('.')):
|
|
111
|
+
scan_directory(subdir, depth + 1)
|
|
112
|
+
except (PermissionError, OSError):
|
|
113
|
+
if self.config.verbose:
|
|
114
|
+
print(f"Permission denied scanning: {path}")
|
|
115
|
+
|
|
116
|
+
# Start scanning from the target directory
|
|
117
|
+
scan_directory(current, 0)
|
|
118
|
+
|
|
119
|
+
# Sort directory candidates by specificity score (highest first)
|
|
120
|
+
dir_candidates.sort(key=lambda c: c.specificity_score, reverse=True)
|
|
121
|
+
|
|
122
|
+
if self.config.verbose and file_candidates:
|
|
123
|
+
print(f"Collected {len(file_candidates)} files for checking")
|
|
124
|
+
|
|
125
|
+
return dir_candidates, file_candidates
|
|
126
|
+
|
|
127
|
+
def _add_candidate(self, candidates: List[DirectoryCandidate], path: Path, depth: int, start_path: Path):
|
|
128
|
+
"""Add a directory candidate to the list."""
|
|
129
|
+
try:
|
|
130
|
+
# Generate SWHID for the directory
|
|
131
|
+
swhid = self.swhid_generator.generate_directory_swhid(path)
|
|
132
|
+
|
|
133
|
+
# Count relevant files
|
|
134
|
+
file_count = self._count_relevant_files(path)
|
|
135
|
+
|
|
136
|
+
# Calculate specificity score (higher for more specific directories)
|
|
137
|
+
specificity_score = self._calculate_specificity_score(
|
|
138
|
+
path, start_path, depth, file_count
|
|
139
|
+
)
|
|
140
|
+
|
|
141
|
+
candidate = DirectoryCandidate(
|
|
142
|
+
path=path,
|
|
143
|
+
swhid=swhid,
|
|
144
|
+
depth=depth,
|
|
145
|
+
specificity_score=specificity_score,
|
|
146
|
+
file_count=file_count
|
|
147
|
+
)
|
|
148
|
+
candidates.append(candidate)
|
|
149
|
+
|
|
150
|
+
if self.config.verbose:
|
|
151
|
+
print(f"Scanned: {path.relative_to(start_path.parent) if path != start_path else path.name} (depth={depth}, files={file_count})")
|
|
152
|
+
print(f" SWHID: {swhid}")
|
|
153
|
+
|
|
154
|
+
except Exception as e:
|
|
155
|
+
if self.config.verbose:
|
|
156
|
+
print(f"Error scanning {path}: {e}")
|
|
157
|
+
|
|
158
|
+
def _add_file_candidate(self, candidates: List[ContentCandidate], file_path: Path, depth: int, start_path: Path):
|
|
159
|
+
"""Add a file candidate to the list."""
|
|
160
|
+
try:
|
|
161
|
+
# Generate SWHID for the file
|
|
162
|
+
swhid = self.swhid_generator.generate_content_swhid(file_path)
|
|
163
|
+
|
|
164
|
+
# Get file size
|
|
165
|
+
size = file_path.stat().st_size
|
|
166
|
+
|
|
167
|
+
candidate = ContentCandidate(
|
|
168
|
+
path=file_path,
|
|
169
|
+
swhid=swhid,
|
|
170
|
+
depth=depth,
|
|
171
|
+
size=size
|
|
172
|
+
)
|
|
173
|
+
candidates.append(candidate)
|
|
174
|
+
|
|
175
|
+
except Exception as e:
|
|
176
|
+
if self.config.verbose:
|
|
177
|
+
print(f"Error scanning file {file_path}: {e}")
|
|
178
|
+
|
|
179
|
+
def _is_meaningful_directory(self, path: Path) -> bool:
|
|
180
|
+
"""
|
|
181
|
+
Check if directory likely contains package content.
|
|
182
|
+
|
|
183
|
+
Args:
|
|
184
|
+
path: Directory path to check
|
|
185
|
+
|
|
186
|
+
Returns:
|
|
187
|
+
True if directory should be scanned
|
|
188
|
+
"""
|
|
189
|
+
if not path.is_dir():
|
|
190
|
+
return False
|
|
191
|
+
|
|
192
|
+
# Skip if directory name is in skip list
|
|
193
|
+
if path.name in self.SKIP_DIRS:
|
|
194
|
+
return False
|
|
195
|
+
|
|
196
|
+
# Skip hidden directories (except current directory)
|
|
197
|
+
if path.name.startswith('.') and path.name != '.':
|
|
198
|
+
return False
|
|
199
|
+
|
|
200
|
+
# Count source files (including in immediate subdirectories)
|
|
201
|
+
source_file_count = 0
|
|
202
|
+
try:
|
|
203
|
+
# Check files in the directory itself
|
|
204
|
+
for item in path.iterdir():
|
|
205
|
+
if item.is_file() and item.suffix in self.SOURCE_EXTENSIONS:
|
|
206
|
+
source_file_count += 1
|
|
207
|
+
if source_file_count >= self.config.min_files:
|
|
208
|
+
return True
|
|
209
|
+
|
|
210
|
+
# If not enough files in root, check immediate subdirectories
|
|
211
|
+
if source_file_count < self.config.min_files:
|
|
212
|
+
for subdir in path.iterdir():
|
|
213
|
+
if subdir.is_dir() and subdir.name not in self.SKIP_DIRS:
|
|
214
|
+
for item in subdir.iterdir():
|
|
215
|
+
if item.is_file() and item.suffix in self.SOURCE_EXTENSIONS:
|
|
216
|
+
source_file_count += 1
|
|
217
|
+
if source_file_count >= self.config.min_files:
|
|
218
|
+
return True
|
|
219
|
+
if source_file_count > 10: # Early exit for performance
|
|
220
|
+
return True
|
|
221
|
+
except (PermissionError, OSError):
|
|
222
|
+
return False
|
|
223
|
+
|
|
224
|
+
# Also check for package indicators even with fewer source files
|
|
225
|
+
if source_file_count > 0 and self._check_package_indicators(path) > 0:
|
|
226
|
+
return True
|
|
227
|
+
|
|
228
|
+
return source_file_count >= self.config.min_files
|
|
229
|
+
|
|
230
|
+
def _count_relevant_files(self, path: Path) -> int:
|
|
231
|
+
"""
|
|
232
|
+
Count source files, ignoring build artifacts.
|
|
233
|
+
|
|
234
|
+
Args:
|
|
235
|
+
path: Directory path
|
|
236
|
+
|
|
237
|
+
Returns:
|
|
238
|
+
Number of relevant files
|
|
239
|
+
"""
|
|
240
|
+
count = 0
|
|
241
|
+
try:
|
|
242
|
+
for root, dirs, files in os.walk(path):
|
|
243
|
+
# Skip directories in the skip list
|
|
244
|
+
dirs[:] = [d for d in dirs if d not in self.SKIP_DIRS]
|
|
245
|
+
|
|
246
|
+
# Count source files
|
|
247
|
+
for file in files:
|
|
248
|
+
if any(file.endswith(ext) for ext in self.SOURCE_EXTENSIONS):
|
|
249
|
+
count += 1
|
|
250
|
+
|
|
251
|
+
# Limit traversal depth for performance
|
|
252
|
+
if count > 1000: # Arbitrary limit
|
|
253
|
+
break
|
|
254
|
+
except (PermissionError, OSError):
|
|
255
|
+
pass
|
|
256
|
+
|
|
257
|
+
return count
|
|
258
|
+
|
|
259
|
+
def _calculate_specificity_score(
|
|
260
|
+
self,
|
|
261
|
+
path: Path,
|
|
262
|
+
start_path: Path,
|
|
263
|
+
depth: int,
|
|
264
|
+
file_count: int
|
|
265
|
+
) -> float:
|
|
266
|
+
"""
|
|
267
|
+
Calculate how specific/relevant a directory is.
|
|
268
|
+
|
|
269
|
+
Args:
|
|
270
|
+
path: Directory path being scored
|
|
271
|
+
start_path: Original starting path
|
|
272
|
+
depth: Depth from starting path
|
|
273
|
+
file_count: Number of relevant files
|
|
274
|
+
|
|
275
|
+
Returns:
|
|
276
|
+
Specificity score between 0 and 1
|
|
277
|
+
"""
|
|
278
|
+
# Base score from depth (closer to start = higher score)
|
|
279
|
+
depth_score = 1.0 / (depth + 1)
|
|
280
|
+
|
|
281
|
+
# Bonus for being the exact start path
|
|
282
|
+
if path == start_path:
|
|
283
|
+
depth_score *= 1.5
|
|
284
|
+
|
|
285
|
+
# File count factor (more files = more likely to be root)
|
|
286
|
+
file_factor = min(1.0, file_count / 100.0)
|
|
287
|
+
|
|
288
|
+
# Check for package indicators
|
|
289
|
+
package_indicators = self._check_package_indicators(path)
|
|
290
|
+
|
|
291
|
+
# Combine factors
|
|
292
|
+
score = (
|
|
293
|
+
depth_score * self.config.score_weights['specificity'] +
|
|
294
|
+
file_factor * 0.3 +
|
|
295
|
+
package_indicators * 0.3
|
|
296
|
+
)
|
|
297
|
+
|
|
298
|
+
return min(1.0, score)
|
|
299
|
+
|
|
300
|
+
def _check_package_indicators(self, path: Path) -> float:
|
|
301
|
+
"""
|
|
302
|
+
Check for indicators that this is a package root.
|
|
303
|
+
|
|
304
|
+
Args:
|
|
305
|
+
path: Directory path to check
|
|
306
|
+
|
|
307
|
+
Returns:
|
|
308
|
+
Score between 0 and 1 based on package indicators
|
|
309
|
+
"""
|
|
310
|
+
score = 0.0
|
|
311
|
+
indicators = {
|
|
312
|
+
# Build files
|
|
313
|
+
'CMakeLists.txt': 0.3,
|
|
314
|
+
'Makefile': 0.2,
|
|
315
|
+
'configure': 0.2,
|
|
316
|
+
'setup.py': 0.3,
|
|
317
|
+
'package.json': 0.3,
|
|
318
|
+
'Cargo.toml': 0.3,
|
|
319
|
+
'go.mod': 0.3,
|
|
320
|
+
'pom.xml': 0.3,
|
|
321
|
+
'build.gradle': 0.3,
|
|
322
|
+
# Documentation
|
|
323
|
+
'README.md': 0.1,
|
|
324
|
+
'README.rst': 0.1,
|
|
325
|
+
'README.txt': 0.1,
|
|
326
|
+
'LICENSE': 0.1,
|
|
327
|
+
'LICENSE.txt': 0.1,
|
|
328
|
+
'COPYING': 0.1,
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
for filename, weight in indicators.items():
|
|
332
|
+
if (path / filename).exists():
|
|
333
|
+
score += weight
|
|
334
|
+
|
|
335
|
+
return min(1.0, score)
|
|
336
|
+
|
|
337
|
+
def detect_git_submodules(self, repo_path: Path) -> List[Path]:
|
|
338
|
+
"""
|
|
339
|
+
Parse .gitmodules and return submodule paths.
|
|
340
|
+
|
|
341
|
+
Args:
|
|
342
|
+
repo_path: Repository root path
|
|
343
|
+
|
|
344
|
+
Returns:
|
|
345
|
+
List of submodule paths
|
|
346
|
+
"""
|
|
347
|
+
submodules = []
|
|
348
|
+
gitmodules_path = repo_path / '.gitmodules'
|
|
349
|
+
|
|
350
|
+
if not gitmodules_path.exists():
|
|
351
|
+
return submodules
|
|
352
|
+
|
|
353
|
+
try:
|
|
354
|
+
with open(gitmodules_path, 'r') as f:
|
|
355
|
+
lines = f.readlines()
|
|
356
|
+
|
|
357
|
+
current_path = None
|
|
358
|
+
for line in lines:
|
|
359
|
+
line = line.strip()
|
|
360
|
+
if line.startswith('path ='):
|
|
361
|
+
current_path = line.split('=', 1)[1].strip()
|
|
362
|
+
submodule_path = repo_path / current_path
|
|
363
|
+
if submodule_path.exists():
|
|
364
|
+
submodules.append(submodule_path)
|
|
365
|
+
except Exception as e:
|
|
366
|
+
if self.config.verbose:
|
|
367
|
+
print(f"Error parsing .gitmodules: {e}")
|
|
368
|
+
|
|
369
|
+
return submodules
|
src2id/core/scorer.py
ADDED
|
@@ -0,0 +1,217 @@
|
|
|
1
|
+
"""Confidence scoring for package matches."""
|
|
2
|
+
|
|
3
|
+
from datetime import datetime, timedelta
|
|
4
|
+
from typing import Any, Dict
|
|
5
|
+
|
|
6
|
+
from src2id.core.config import SWHPIConfig
|
|
7
|
+
from src2id.utils.datetime_utils import parse_datetime
|
|
8
|
+
|
|
9
|
+
|
|
10
|
+
class ConfidenceScorer:
|
|
11
|
+
"""Calculates confidence scores for package matches."""
|
|
12
|
+
|
|
13
|
+
def __init__(self, config: SWHPIConfig):
|
|
14
|
+
"""
|
|
15
|
+
Initialize the confidence scorer.
|
|
16
|
+
|
|
17
|
+
Args:
|
|
18
|
+
config: Configuration settings
|
|
19
|
+
"""
|
|
20
|
+
self.config = config
|
|
21
|
+
|
|
22
|
+
def calculate_confidence(self, match_data: Dict[str, Any]) -> float:
|
|
23
|
+
"""
|
|
24
|
+
Multi-factor confidence scoring.
|
|
25
|
+
|
|
26
|
+
Factors considered:
|
|
27
|
+
- Match type (exact vs fuzzy)
|
|
28
|
+
- Frequency/popularity
|
|
29
|
+
- Official organization authority
|
|
30
|
+
- Recency of activity
|
|
31
|
+
|
|
32
|
+
Args:
|
|
33
|
+
match_data: Dictionary with match information
|
|
34
|
+
|
|
35
|
+
Returns:
|
|
36
|
+
Confidence score between 0 and 1
|
|
37
|
+
"""
|
|
38
|
+
# Get base score from match type
|
|
39
|
+
base_score = self._get_base_score(match_data)
|
|
40
|
+
|
|
41
|
+
# Calculate individual factor scores
|
|
42
|
+
frequency_score = self._frequency_score(match_data.get('frequency_rank', 1))
|
|
43
|
+
authority_score = self._authority_score(match_data.get('is_official_org', False))
|
|
44
|
+
recency_score = self._recency_score(match_data.get('last_activity'))
|
|
45
|
+
|
|
46
|
+
# Combine scores using configured weights
|
|
47
|
+
weights = self.config.score_weights
|
|
48
|
+
|
|
49
|
+
# Weighted combination
|
|
50
|
+
final_score = (
|
|
51
|
+
base_score * 0.4 + # Base match quality
|
|
52
|
+
frequency_score * weights.get('popularity', 0.2) +
|
|
53
|
+
authority_score * weights.get('authority', 0.3) +
|
|
54
|
+
recency_score * weights.get('recency', 0.3)
|
|
55
|
+
)
|
|
56
|
+
|
|
57
|
+
# Apply multipliers
|
|
58
|
+
multipliers = [
|
|
59
|
+
self._frequency_multiplier(match_data.get('frequency_rank', 1)),
|
|
60
|
+
self._authority_multiplier(match_data.get('is_official_org', False)),
|
|
61
|
+
self._recency_multiplier(match_data.get('last_activity')),
|
|
62
|
+
]
|
|
63
|
+
|
|
64
|
+
for multiplier in multipliers:
|
|
65
|
+
final_score *= multiplier
|
|
66
|
+
|
|
67
|
+
# Ensure score is within bounds
|
|
68
|
+
return min(1.0, max(0.0, final_score))
|
|
69
|
+
|
|
70
|
+
def _get_base_score(self, match_data: Dict[str, Any]) -> float:
|
|
71
|
+
"""
|
|
72
|
+
Get base confidence score from match type.
|
|
73
|
+
|
|
74
|
+
Args:
|
|
75
|
+
match_data: Match information
|
|
76
|
+
|
|
77
|
+
Returns:
|
|
78
|
+
Base score
|
|
79
|
+
"""
|
|
80
|
+
match_type = match_data.get('match_type')
|
|
81
|
+
|
|
82
|
+
if match_type == 'exact' or match_type.value == 'exact':
|
|
83
|
+
return 0.9
|
|
84
|
+
elif match_type == 'fuzzy' or match_type.value == 'fuzzy':
|
|
85
|
+
# For fuzzy matches, use similarity score
|
|
86
|
+
similarity = match_data.get('similarity_score', 0.5)
|
|
87
|
+
return similarity * 0.8
|
|
88
|
+
else:
|
|
89
|
+
return 0.5
|
|
90
|
+
|
|
91
|
+
def _frequency_score(self, frequency_rank: int) -> float:
|
|
92
|
+
"""
|
|
93
|
+
Calculate score based on frequency/popularity.
|
|
94
|
+
|
|
95
|
+
Args:
|
|
96
|
+
frequency_rank: Number of visits/occurrences
|
|
97
|
+
|
|
98
|
+
Returns:
|
|
99
|
+
Frequency score between 0 and 1
|
|
100
|
+
"""
|
|
101
|
+
if frequency_rank <= 0:
|
|
102
|
+
return 0.0
|
|
103
|
+
elif frequency_rank == 1:
|
|
104
|
+
return 0.3
|
|
105
|
+
elif frequency_rank < 5:
|
|
106
|
+
return 0.5
|
|
107
|
+
elif frequency_rank < 10:
|
|
108
|
+
return 0.7
|
|
109
|
+
elif frequency_rank < 50:
|
|
110
|
+
return 0.85
|
|
111
|
+
else:
|
|
112
|
+
return 1.0
|
|
113
|
+
|
|
114
|
+
def _authority_score(self, is_official: bool) -> float:
|
|
115
|
+
"""
|
|
116
|
+
Calculate score based on official organization status.
|
|
117
|
+
|
|
118
|
+
Args:
|
|
119
|
+
is_official: Whether from official organization
|
|
120
|
+
|
|
121
|
+
Returns:
|
|
122
|
+
Authority score
|
|
123
|
+
"""
|
|
124
|
+
return 1.0 if is_official else 0.7
|
|
125
|
+
|
|
126
|
+
def _recency_score(self, last_activity: Any) -> float:
|
|
127
|
+
"""
|
|
128
|
+
Calculate score based on recency of activity.
|
|
129
|
+
|
|
130
|
+
Args:
|
|
131
|
+
last_activity: Last activity datetime
|
|
132
|
+
|
|
133
|
+
Returns:
|
|
134
|
+
Recency score between 0 and 1
|
|
135
|
+
"""
|
|
136
|
+
parsed = parse_datetime(last_activity)
|
|
137
|
+
if not parsed:
|
|
138
|
+
return 0.5
|
|
139
|
+
last_activity = parsed
|
|
140
|
+
|
|
141
|
+
# Calculate days since last activity
|
|
142
|
+
now = datetime.now(last_activity.tzinfo) if last_activity.tzinfo else datetime.now()
|
|
143
|
+
days_ago = (now - last_activity).days
|
|
144
|
+
|
|
145
|
+
if days_ago < 30:
|
|
146
|
+
return 1.0
|
|
147
|
+
elif days_ago < 90:
|
|
148
|
+
return 0.9
|
|
149
|
+
elif days_ago < 180:
|
|
150
|
+
return 0.8
|
|
151
|
+
elif days_ago < 365:
|
|
152
|
+
return 0.7
|
|
153
|
+
elif days_ago < 730: # 2 years
|
|
154
|
+
return 0.6
|
|
155
|
+
else:
|
|
156
|
+
return 0.5
|
|
157
|
+
|
|
158
|
+
def _frequency_multiplier(self, frequency_rank: int) -> float:
|
|
159
|
+
"""
|
|
160
|
+
Boost confidence for frequently appearing packages.
|
|
161
|
+
|
|
162
|
+
Args:
|
|
163
|
+
frequency_rank: Number of visits/occurrences
|
|
164
|
+
|
|
165
|
+
Returns:
|
|
166
|
+
Multiplier value
|
|
167
|
+
"""
|
|
168
|
+
if frequency_rank < 2:
|
|
169
|
+
return 0.95
|
|
170
|
+
elif frequency_rank < 5:
|
|
171
|
+
return 1.0
|
|
172
|
+
elif frequency_rank < 20:
|
|
173
|
+
return 1.05
|
|
174
|
+
elif frequency_rank < 100:
|
|
175
|
+
return 1.1
|
|
176
|
+
else:
|
|
177
|
+
return 1.15
|
|
178
|
+
|
|
179
|
+
def _authority_multiplier(self, is_official: bool) -> float:
|
|
180
|
+
"""
|
|
181
|
+
Boost confidence for official organizations.
|
|
182
|
+
|
|
183
|
+
Args:
|
|
184
|
+
is_official: Whether from official organization
|
|
185
|
+
|
|
186
|
+
Returns:
|
|
187
|
+
Multiplier value
|
|
188
|
+
"""
|
|
189
|
+
return 1.15 if is_official else 1.0
|
|
190
|
+
|
|
191
|
+
def _recency_multiplier(self, last_activity: Any) -> float:
|
|
192
|
+
"""
|
|
193
|
+
Boost confidence for recently active repositories.
|
|
194
|
+
|
|
195
|
+
Args:
|
|
196
|
+
last_activity: Last activity datetime
|
|
197
|
+
|
|
198
|
+
Returns:
|
|
199
|
+
Multiplier value
|
|
200
|
+
"""
|
|
201
|
+
parsed = parse_datetime(last_activity)
|
|
202
|
+
if not parsed:
|
|
203
|
+
return 1.0
|
|
204
|
+
last_activity = parsed
|
|
205
|
+
|
|
206
|
+
# Calculate days since last activity
|
|
207
|
+
now = datetime.now(last_activity.tzinfo) if last_activity.tzinfo else datetime.now()
|
|
208
|
+
days_ago = (now - last_activity).days
|
|
209
|
+
|
|
210
|
+
if days_ago < 30:
|
|
211
|
+
return 1.1
|
|
212
|
+
elif days_ago < 365:
|
|
213
|
+
return 1.05
|
|
214
|
+
elif days_ago > 1095: # 3 years
|
|
215
|
+
return 0.95
|
|
216
|
+
else:
|
|
217
|
+
return 1.0
|