src2purl 1.3.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
src2id/core/scanner.py ADDED
@@ -0,0 +1,369 @@
1
+ """Directory scanner for generating SWHID candidates."""
2
+
3
+ import os
4
+ from pathlib import Path
5
+ from typing import List, Optional, Set, Tuple
6
+
7
+ from src2id.core.config import SWHPIConfig
8
+ from src2id.core.models import DirectoryCandidate, ContentCandidate
9
+
10
+
11
+ class DirectoryScanner:
12
+ """Scans filesystem and generates directory candidates for SH matching."""
13
+
14
+ # Directories to skip during scanning
15
+ SKIP_DIRS = {
16
+ '.git', '.svn', '.hg', '.bzr', # Version control
17
+ '__pycache__', '.pytest_cache', '.mypy_cache', # Python
18
+ 'node_modules', 'bower_components', # JavaScript
19
+ 'target', 'build', 'dist', 'out', # Build directories
20
+ '.idea', '.vscode', '.vs', # IDE directories
21
+ 'venv', 'env', '.env', # Virtual environments
22
+ }
23
+
24
+ # File extensions to consider as source code
25
+ SOURCE_EXTENSIONS = {
26
+ # C/C++
27
+ '.c', '.h', '.cpp', '.hpp', '.cc', '.cxx', '.hxx', '.C', '.H',
28
+ # Python
29
+ '.py', '.pyx', '.pxd', '.pyi',
30
+ # JavaScript/TypeScript
31
+ '.js', '.jsx', '.ts', '.tsx', '.mjs', '.cjs',
32
+ # Java/Kotlin
33
+ '.java', '.kt', '.kts',
34
+ # Rust
35
+ '.rs',
36
+ # Go
37
+ '.go',
38
+ # Ruby
39
+ '.rb',
40
+ # Swift
41
+ '.swift',
42
+ # Other
43
+ '.sh', '.bash', '.zsh', '.fish',
44
+ '.yaml', '.yml', '.json', '.xml', '.toml',
45
+ }
46
+
47
+ def __init__(self, config: SWHPIConfig, swhid_generator):
48
+ """
49
+ Initialize the directory scanner.
50
+
51
+ Args:
52
+ config: Configuration settings
53
+ swhid_generator: SWHID generator instance
54
+ """
55
+ self.config = config
56
+ self.swhid_generator = swhid_generator
57
+
58
+ def scan_recursive(self, start_path: Path) -> Tuple[List[DirectoryCandidate], List[ContentCandidate]]:
59
+ """
60
+ Generate directory and file candidates using depth-first approach.
61
+
62
+ Scans the starting directory and files first, then subdirectories up to max_depth.
63
+
64
+ Args:
65
+ start_path: Starting directory path
66
+
67
+ Returns:
68
+ Tuple of (directory candidates, file candidates)
69
+ """
70
+ dir_candidates = []
71
+ file_candidates = []
72
+ current = start_path.resolve()
73
+
74
+ # Track total files scanned (for limiting)
75
+ max_files = 100 # Reasonable limit to avoid overwhelming the API
76
+ files_scanned = 0
77
+
78
+ # Scan subdirectories recursively
79
+ def scan_directory(path: Path, depth: int):
80
+ nonlocal files_scanned
81
+
82
+ if depth > self.config.max_depth:
83
+ return
84
+
85
+ # Process current directory
86
+ if self._is_meaningful_directory(path):
87
+ self._add_candidate(dir_candidates, path, depth, current)
88
+
89
+ # Process files in current directory (only at depths we're scanning)
90
+ if files_scanned < max_files:
91
+ try:
92
+ for item in path.iterdir():
93
+ if item.is_file() and not item.name.startswith('.'):
94
+ # Skip very large files and common non-source files
95
+ if item.suffix not in {'.pyc', '.pyo', '.so', '.dll', '.exe', '.jpg', '.png', '.gif'}:
96
+ if item.stat().st_size < 10_000_000: # Skip files > 10MB
97
+ self._add_file_candidate(file_candidates, item, depth, current)
98
+ files_scanned += 1
99
+ if files_scanned >= max_files:
100
+ break
101
+ except (PermissionError, OSError):
102
+ pass
103
+
104
+ # Scan subdirectories
105
+ if depth < self.config.max_depth:
106
+ try:
107
+ for subdir in path.iterdir():
108
+ if (subdir.is_dir() and
109
+ subdir.name not in self.SKIP_DIRS and
110
+ not subdir.name.startswith('.')):
111
+ scan_directory(subdir, depth + 1)
112
+ except (PermissionError, OSError):
113
+ if self.config.verbose:
114
+ print(f"Permission denied scanning: {path}")
115
+
116
+ # Start scanning from the target directory
117
+ scan_directory(current, 0)
118
+
119
+ # Sort directory candidates by specificity score (highest first)
120
+ dir_candidates.sort(key=lambda c: c.specificity_score, reverse=True)
121
+
122
+ if self.config.verbose and file_candidates:
123
+ print(f"Collected {len(file_candidates)} files for checking")
124
+
125
+ return dir_candidates, file_candidates
126
+
127
+ def _add_candidate(self, candidates: List[DirectoryCandidate], path: Path, depth: int, start_path: Path):
128
+ """Add a directory candidate to the list."""
129
+ try:
130
+ # Generate SWHID for the directory
131
+ swhid = self.swhid_generator.generate_directory_swhid(path)
132
+
133
+ # Count relevant files
134
+ file_count = self._count_relevant_files(path)
135
+
136
+ # Calculate specificity score (higher for more specific directories)
137
+ specificity_score = self._calculate_specificity_score(
138
+ path, start_path, depth, file_count
139
+ )
140
+
141
+ candidate = DirectoryCandidate(
142
+ path=path,
143
+ swhid=swhid,
144
+ depth=depth,
145
+ specificity_score=specificity_score,
146
+ file_count=file_count
147
+ )
148
+ candidates.append(candidate)
149
+
150
+ if self.config.verbose:
151
+ print(f"Scanned: {path.relative_to(start_path.parent) if path != start_path else path.name} (depth={depth}, files={file_count})")
152
+ print(f" SWHID: {swhid}")
153
+
154
+ except Exception as e:
155
+ if self.config.verbose:
156
+ print(f"Error scanning {path}: {e}")
157
+
158
+ def _add_file_candidate(self, candidates: List[ContentCandidate], file_path: Path, depth: int, start_path: Path):
159
+ """Add a file candidate to the list."""
160
+ try:
161
+ # Generate SWHID for the file
162
+ swhid = self.swhid_generator.generate_content_swhid(file_path)
163
+
164
+ # Get file size
165
+ size = file_path.stat().st_size
166
+
167
+ candidate = ContentCandidate(
168
+ path=file_path,
169
+ swhid=swhid,
170
+ depth=depth,
171
+ size=size
172
+ )
173
+ candidates.append(candidate)
174
+
175
+ except Exception as e:
176
+ if self.config.verbose:
177
+ print(f"Error scanning file {file_path}: {e}")
178
+
179
+ def _is_meaningful_directory(self, path: Path) -> bool:
180
+ """
181
+ Check if directory likely contains package content.
182
+
183
+ Args:
184
+ path: Directory path to check
185
+
186
+ Returns:
187
+ True if directory should be scanned
188
+ """
189
+ if not path.is_dir():
190
+ return False
191
+
192
+ # Skip if directory name is in skip list
193
+ if path.name in self.SKIP_DIRS:
194
+ return False
195
+
196
+ # Skip hidden directories (except current directory)
197
+ if path.name.startswith('.') and path.name != '.':
198
+ return False
199
+
200
+ # Count source files (including in immediate subdirectories)
201
+ source_file_count = 0
202
+ try:
203
+ # Check files in the directory itself
204
+ for item in path.iterdir():
205
+ if item.is_file() and item.suffix in self.SOURCE_EXTENSIONS:
206
+ source_file_count += 1
207
+ if source_file_count >= self.config.min_files:
208
+ return True
209
+
210
+ # If not enough files in root, check immediate subdirectories
211
+ if source_file_count < self.config.min_files:
212
+ for subdir in path.iterdir():
213
+ if subdir.is_dir() and subdir.name not in self.SKIP_DIRS:
214
+ for item in subdir.iterdir():
215
+ if item.is_file() and item.suffix in self.SOURCE_EXTENSIONS:
216
+ source_file_count += 1
217
+ if source_file_count >= self.config.min_files:
218
+ return True
219
+ if source_file_count > 10: # Early exit for performance
220
+ return True
221
+ except (PermissionError, OSError):
222
+ return False
223
+
224
+ # Also check for package indicators even with fewer source files
225
+ if source_file_count > 0 and self._check_package_indicators(path) > 0:
226
+ return True
227
+
228
+ return source_file_count >= self.config.min_files
229
+
230
+ def _count_relevant_files(self, path: Path) -> int:
231
+ """
232
+ Count source files, ignoring build artifacts.
233
+
234
+ Args:
235
+ path: Directory path
236
+
237
+ Returns:
238
+ Number of relevant files
239
+ """
240
+ count = 0
241
+ try:
242
+ for root, dirs, files in os.walk(path):
243
+ # Skip directories in the skip list
244
+ dirs[:] = [d for d in dirs if d not in self.SKIP_DIRS]
245
+
246
+ # Count source files
247
+ for file in files:
248
+ if any(file.endswith(ext) for ext in self.SOURCE_EXTENSIONS):
249
+ count += 1
250
+
251
+ # Limit traversal depth for performance
252
+ if count > 1000: # Arbitrary limit
253
+ break
254
+ except (PermissionError, OSError):
255
+ pass
256
+
257
+ return count
258
+
259
+ def _calculate_specificity_score(
260
+ self,
261
+ path: Path,
262
+ start_path: Path,
263
+ depth: int,
264
+ file_count: int
265
+ ) -> float:
266
+ """
267
+ Calculate how specific/relevant a directory is.
268
+
269
+ Args:
270
+ path: Directory path being scored
271
+ start_path: Original starting path
272
+ depth: Depth from starting path
273
+ file_count: Number of relevant files
274
+
275
+ Returns:
276
+ Specificity score between 0 and 1
277
+ """
278
+ # Base score from depth (closer to start = higher score)
279
+ depth_score = 1.0 / (depth + 1)
280
+
281
+ # Bonus for being the exact start path
282
+ if path == start_path:
283
+ depth_score *= 1.5
284
+
285
+ # File count factor (more files = more likely to be root)
286
+ file_factor = min(1.0, file_count / 100.0)
287
+
288
+ # Check for package indicators
289
+ package_indicators = self._check_package_indicators(path)
290
+
291
+ # Combine factors
292
+ score = (
293
+ depth_score * self.config.score_weights['specificity'] +
294
+ file_factor * 0.3 +
295
+ package_indicators * 0.3
296
+ )
297
+
298
+ return min(1.0, score)
299
+
300
+ def _check_package_indicators(self, path: Path) -> float:
301
+ """
302
+ Check for indicators that this is a package root.
303
+
304
+ Args:
305
+ path: Directory path to check
306
+
307
+ Returns:
308
+ Score between 0 and 1 based on package indicators
309
+ """
310
+ score = 0.0
311
+ indicators = {
312
+ # Build files
313
+ 'CMakeLists.txt': 0.3,
314
+ 'Makefile': 0.2,
315
+ 'configure': 0.2,
316
+ 'setup.py': 0.3,
317
+ 'package.json': 0.3,
318
+ 'Cargo.toml': 0.3,
319
+ 'go.mod': 0.3,
320
+ 'pom.xml': 0.3,
321
+ 'build.gradle': 0.3,
322
+ # Documentation
323
+ 'README.md': 0.1,
324
+ 'README.rst': 0.1,
325
+ 'README.txt': 0.1,
326
+ 'LICENSE': 0.1,
327
+ 'LICENSE.txt': 0.1,
328
+ 'COPYING': 0.1,
329
+ }
330
+
331
+ for filename, weight in indicators.items():
332
+ if (path / filename).exists():
333
+ score += weight
334
+
335
+ return min(1.0, score)
336
+
337
+ def detect_git_submodules(self, repo_path: Path) -> List[Path]:
338
+ """
339
+ Parse .gitmodules and return submodule paths.
340
+
341
+ Args:
342
+ repo_path: Repository root path
343
+
344
+ Returns:
345
+ List of submodule paths
346
+ """
347
+ submodules = []
348
+ gitmodules_path = repo_path / '.gitmodules'
349
+
350
+ if not gitmodules_path.exists():
351
+ return submodules
352
+
353
+ try:
354
+ with open(gitmodules_path, 'r') as f:
355
+ lines = f.readlines()
356
+
357
+ current_path = None
358
+ for line in lines:
359
+ line = line.strip()
360
+ if line.startswith('path ='):
361
+ current_path = line.split('=', 1)[1].strip()
362
+ submodule_path = repo_path / current_path
363
+ if submodule_path.exists():
364
+ submodules.append(submodule_path)
365
+ except Exception as e:
366
+ if self.config.verbose:
367
+ print(f"Error parsing .gitmodules: {e}")
368
+
369
+ return submodules
src2id/core/scorer.py ADDED
@@ -0,0 +1,217 @@
1
+ """Confidence scoring for package matches."""
2
+
3
+ from datetime import datetime, timedelta
4
+ from typing import Any, Dict
5
+
6
+ from src2id.core.config import SWHPIConfig
7
+ from src2id.utils.datetime_utils import parse_datetime
8
+
9
+
10
+ class ConfidenceScorer:
11
+ """Calculates confidence scores for package matches."""
12
+
13
+ def __init__(self, config: SWHPIConfig):
14
+ """
15
+ Initialize the confidence scorer.
16
+
17
+ Args:
18
+ config: Configuration settings
19
+ """
20
+ self.config = config
21
+
22
+ def calculate_confidence(self, match_data: Dict[str, Any]) -> float:
23
+ """
24
+ Multi-factor confidence scoring.
25
+
26
+ Factors considered:
27
+ - Match type (exact vs fuzzy)
28
+ - Frequency/popularity
29
+ - Official organization authority
30
+ - Recency of activity
31
+
32
+ Args:
33
+ match_data: Dictionary with match information
34
+
35
+ Returns:
36
+ Confidence score between 0 and 1
37
+ """
38
+ # Get base score from match type
39
+ base_score = self._get_base_score(match_data)
40
+
41
+ # Calculate individual factor scores
42
+ frequency_score = self._frequency_score(match_data.get('frequency_rank', 1))
43
+ authority_score = self._authority_score(match_data.get('is_official_org', False))
44
+ recency_score = self._recency_score(match_data.get('last_activity'))
45
+
46
+ # Combine scores using configured weights
47
+ weights = self.config.score_weights
48
+
49
+ # Weighted combination
50
+ final_score = (
51
+ base_score * 0.4 + # Base match quality
52
+ frequency_score * weights.get('popularity', 0.2) +
53
+ authority_score * weights.get('authority', 0.3) +
54
+ recency_score * weights.get('recency', 0.3)
55
+ )
56
+
57
+ # Apply multipliers
58
+ multipliers = [
59
+ self._frequency_multiplier(match_data.get('frequency_rank', 1)),
60
+ self._authority_multiplier(match_data.get('is_official_org', False)),
61
+ self._recency_multiplier(match_data.get('last_activity')),
62
+ ]
63
+
64
+ for multiplier in multipliers:
65
+ final_score *= multiplier
66
+
67
+ # Ensure score is within bounds
68
+ return min(1.0, max(0.0, final_score))
69
+
70
+ def _get_base_score(self, match_data: Dict[str, Any]) -> float:
71
+ """
72
+ Get base confidence score from match type.
73
+
74
+ Args:
75
+ match_data: Match information
76
+
77
+ Returns:
78
+ Base score
79
+ """
80
+ match_type = match_data.get('match_type')
81
+
82
+ if match_type == 'exact' or match_type.value == 'exact':
83
+ return 0.9
84
+ elif match_type == 'fuzzy' or match_type.value == 'fuzzy':
85
+ # For fuzzy matches, use similarity score
86
+ similarity = match_data.get('similarity_score', 0.5)
87
+ return similarity * 0.8
88
+ else:
89
+ return 0.5
90
+
91
+ def _frequency_score(self, frequency_rank: int) -> float:
92
+ """
93
+ Calculate score based on frequency/popularity.
94
+
95
+ Args:
96
+ frequency_rank: Number of visits/occurrences
97
+
98
+ Returns:
99
+ Frequency score between 0 and 1
100
+ """
101
+ if frequency_rank <= 0:
102
+ return 0.0
103
+ elif frequency_rank == 1:
104
+ return 0.3
105
+ elif frequency_rank < 5:
106
+ return 0.5
107
+ elif frequency_rank < 10:
108
+ return 0.7
109
+ elif frequency_rank < 50:
110
+ return 0.85
111
+ else:
112
+ return 1.0
113
+
114
+ def _authority_score(self, is_official: bool) -> float:
115
+ """
116
+ Calculate score based on official organization status.
117
+
118
+ Args:
119
+ is_official: Whether from official organization
120
+
121
+ Returns:
122
+ Authority score
123
+ """
124
+ return 1.0 if is_official else 0.7
125
+
126
+ def _recency_score(self, last_activity: Any) -> float:
127
+ """
128
+ Calculate score based on recency of activity.
129
+
130
+ Args:
131
+ last_activity: Last activity datetime
132
+
133
+ Returns:
134
+ Recency score between 0 and 1
135
+ """
136
+ parsed = parse_datetime(last_activity)
137
+ if not parsed:
138
+ return 0.5
139
+ last_activity = parsed
140
+
141
+ # Calculate days since last activity
142
+ now = datetime.now(last_activity.tzinfo) if last_activity.tzinfo else datetime.now()
143
+ days_ago = (now - last_activity).days
144
+
145
+ if days_ago < 30:
146
+ return 1.0
147
+ elif days_ago < 90:
148
+ return 0.9
149
+ elif days_ago < 180:
150
+ return 0.8
151
+ elif days_ago < 365:
152
+ return 0.7
153
+ elif days_ago < 730: # 2 years
154
+ return 0.6
155
+ else:
156
+ return 0.5
157
+
158
+ def _frequency_multiplier(self, frequency_rank: int) -> float:
159
+ """
160
+ Boost confidence for frequently appearing packages.
161
+
162
+ Args:
163
+ frequency_rank: Number of visits/occurrences
164
+
165
+ Returns:
166
+ Multiplier value
167
+ """
168
+ if frequency_rank < 2:
169
+ return 0.95
170
+ elif frequency_rank < 5:
171
+ return 1.0
172
+ elif frequency_rank < 20:
173
+ return 1.05
174
+ elif frequency_rank < 100:
175
+ return 1.1
176
+ else:
177
+ return 1.15
178
+
179
+ def _authority_multiplier(self, is_official: bool) -> float:
180
+ """
181
+ Boost confidence for official organizations.
182
+
183
+ Args:
184
+ is_official: Whether from official organization
185
+
186
+ Returns:
187
+ Multiplier value
188
+ """
189
+ return 1.15 if is_official else 1.0
190
+
191
+ def _recency_multiplier(self, last_activity: Any) -> float:
192
+ """
193
+ Boost confidence for recently active repositories.
194
+
195
+ Args:
196
+ last_activity: Last activity datetime
197
+
198
+ Returns:
199
+ Multiplier value
200
+ """
201
+ parsed = parse_datetime(last_activity)
202
+ if not parsed:
203
+ return 1.0
204
+ last_activity = parsed
205
+
206
+ # Calculate days since last activity
207
+ now = datetime.now(last_activity.tzinfo) if last_activity.tzinfo else datetime.now()
208
+ days_ago = (now - last_activity).days
209
+
210
+ if days_ago < 30:
211
+ return 1.1
212
+ elif days_ago < 365:
213
+ return 1.05
214
+ elif days_ago > 1095: # 3 years
215
+ return 0.95
216
+ else:
217
+ return 1.0