src2purl 1.3.2__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- src2id/__init__.py +15 -0
- src2id/cli/__init__.py +1 -0
- src2id/cli/main.py +318 -0
- src2id/cli/validate.py +93 -0
- src2id/core/__init__.py +1 -0
- src2id/core/cache.py +180 -0
- src2id/core/client.py +370 -0
- src2id/core/config.py +45 -0
- src2id/core/extractor.py +302 -0
- src2id/core/models.py +93 -0
- src2id/core/orchestrator.py +796 -0
- src2id/core/package_identifier.py +123 -0
- src2id/core/purl.py +238 -0
- src2id/core/scanner.py +369 -0
- src2id/core/scorer.py +217 -0
- src2id/core/subcomponent_detector.py +353 -0
- src2id/core/swhid.py +324 -0
- src2id/integrations/__init__.py +1 -0
- src2id/integrations/manifest_parser.py +652 -0
- src2id/integrations/oslili.py +228 -0
- src2id/integrations/upmex.py +305 -0
- src2id/search/__init__.py +34 -0
- src2id/search/hash_search.py +206 -0
- src2id/search/providers.py +310 -0
- src2id/search/strategies.py +391 -0
- src2id/utils/__init__.py +1 -0
- src2id/utils/datetime_utils.py +49 -0
- src2purl-1.3.2.dist-info/METADATA +279 -0
- src2purl-1.3.2.dist-info/RECORD +33 -0
- src2purl-1.3.2.dist-info/WHEEL +5 -0
- src2purl-1.3.2.dist-info/entry_points.txt +3 -0
- src2purl-1.3.2.dist-info/licenses/LICENSE +661 -0
- src2purl-1.3.2.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,353 @@
|
|
|
1
|
+
"""Detect and identify multiple subcomponents in a project."""
|
|
2
|
+
|
|
3
|
+
import asyncio
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import List, Dict, Any, Optional
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
|
|
8
|
+
from rich.console import Console
|
|
9
|
+
from rich.table import Table
|
|
10
|
+
|
|
11
|
+
console = Console()
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass
|
|
15
|
+
class Subcomponent:
|
|
16
|
+
"""Represents a detected subcomponent in a project."""
|
|
17
|
+
path: Path
|
|
18
|
+
type: str # 'npm', 'python', 'rust', 'go', 'java', 'mono', 'multi'
|
|
19
|
+
name: Optional[str] = None
|
|
20
|
+
markers: List[str] = None
|
|
21
|
+
|
|
22
|
+
def __post_init__(self):
|
|
23
|
+
if self.markers is None:
|
|
24
|
+
self.markers = []
|
|
25
|
+
|
|
26
|
+
|
|
27
|
+
class SubcomponentDetector:
|
|
28
|
+
"""Detects multiple subcomponents in a project structure."""
|
|
29
|
+
|
|
30
|
+
# Package markers that indicate a component boundary
|
|
31
|
+
PACKAGE_MARKERS = {
|
|
32
|
+
'package.json': 'npm',
|
|
33
|
+
'pyproject.toml': 'python',
|
|
34
|
+
'setup.py': 'python',
|
|
35
|
+
'requirements.txt': 'python',
|
|
36
|
+
'Cargo.toml': 'rust',
|
|
37
|
+
'go.mod': 'go',
|
|
38
|
+
'pom.xml': 'java',
|
|
39
|
+
'build.gradle': 'java',
|
|
40
|
+
'build.gradle.kts': 'java',
|
|
41
|
+
'Gemfile': 'ruby',
|
|
42
|
+
'composer.json': 'php',
|
|
43
|
+
'CMakeLists.txt': 'cmake',
|
|
44
|
+
'Makefile': 'make',
|
|
45
|
+
'.csproj': 'dotnet',
|
|
46
|
+
'mix.exs': 'elixir',
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
# Directories that typically indicate subcomponents
|
|
50
|
+
SUBCOMPONENT_PATTERNS = {
|
|
51
|
+
'packages/*': 'monorepo', # Lerna/Yarn workspaces
|
|
52
|
+
'apps/*': 'monorepo', # Nx/Turborepo
|
|
53
|
+
'services/*': 'microservices',
|
|
54
|
+
'libs/*': 'libraries',
|
|
55
|
+
'modules/*': 'modules',
|
|
56
|
+
'components/*': 'components',
|
|
57
|
+
'plugins/*': 'plugins',
|
|
58
|
+
'extensions/*': 'extensions',
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
def __init__(self, verbose: bool = False):
|
|
62
|
+
"""Initialize the detector."""
|
|
63
|
+
self.verbose = verbose
|
|
64
|
+
|
|
65
|
+
def detect_subcomponents(
|
|
66
|
+
self,
|
|
67
|
+
root_path: Path,
|
|
68
|
+
max_depth: int = 3
|
|
69
|
+
) -> List[Subcomponent]:
|
|
70
|
+
"""
|
|
71
|
+
Detect all subcomponents in a project.
|
|
72
|
+
|
|
73
|
+
Args:
|
|
74
|
+
root_path: Root directory to scan
|
|
75
|
+
max_depth: Maximum depth to search for subcomponents
|
|
76
|
+
|
|
77
|
+
Returns:
|
|
78
|
+
List of detected subcomponents
|
|
79
|
+
"""
|
|
80
|
+
subcomponents = []
|
|
81
|
+
visited = set()
|
|
82
|
+
|
|
83
|
+
# Check if root itself is a component
|
|
84
|
+
root_markers = self._check_markers(root_path)
|
|
85
|
+
if root_markers:
|
|
86
|
+
root_component = Subcomponent(
|
|
87
|
+
path=root_path,
|
|
88
|
+
type=self._determine_type(root_markers),
|
|
89
|
+
name=root_path.name,
|
|
90
|
+
markers=root_markers
|
|
91
|
+
)
|
|
92
|
+
subcomponents.append(root_component)
|
|
93
|
+
visited.add(root_path)
|
|
94
|
+
|
|
95
|
+
# Search for nested subcomponents
|
|
96
|
+
self._scan_for_subcomponents(
|
|
97
|
+
root_path,
|
|
98
|
+
subcomponents,
|
|
99
|
+
visited,
|
|
100
|
+
current_depth=0,
|
|
101
|
+
max_depth=max_depth
|
|
102
|
+
)
|
|
103
|
+
|
|
104
|
+
# Deduplicate and organize
|
|
105
|
+
subcomponents = self._organize_subcomponents(subcomponents)
|
|
106
|
+
|
|
107
|
+
if self.verbose:
|
|
108
|
+
self._print_detection_results(subcomponents)
|
|
109
|
+
|
|
110
|
+
return subcomponents
|
|
111
|
+
|
|
112
|
+
def _scan_for_subcomponents(
|
|
113
|
+
self,
|
|
114
|
+
path: Path,
|
|
115
|
+
subcomponents: List[Subcomponent],
|
|
116
|
+
visited: set,
|
|
117
|
+
current_depth: int,
|
|
118
|
+
max_depth: int
|
|
119
|
+
):
|
|
120
|
+
"""Recursively scan for subcomponents."""
|
|
121
|
+
if current_depth >= max_depth:
|
|
122
|
+
return
|
|
123
|
+
|
|
124
|
+
try:
|
|
125
|
+
for item in path.iterdir():
|
|
126
|
+
if item.is_dir() and item not in visited:
|
|
127
|
+
# Skip common non-component directories
|
|
128
|
+
if item.name in {'.git', '__pycache__', 'node_modules',
|
|
129
|
+
'venv', 'build', 'dist', '.idea', '.vscode'}:
|
|
130
|
+
continue
|
|
131
|
+
|
|
132
|
+
# Check for package markers
|
|
133
|
+
markers = self._check_markers(item)
|
|
134
|
+
|
|
135
|
+
if markers:
|
|
136
|
+
# Found a subcomponent
|
|
137
|
+
component = Subcomponent(
|
|
138
|
+
path=item,
|
|
139
|
+
type=self._determine_type(markers),
|
|
140
|
+
name=item.name,
|
|
141
|
+
markers=markers
|
|
142
|
+
)
|
|
143
|
+
subcomponents.append(component)
|
|
144
|
+
visited.add(item)
|
|
145
|
+
|
|
146
|
+
# Don't scan inside detected components by default
|
|
147
|
+
# unless it's a monorepo pattern
|
|
148
|
+
if self._is_monorepo_pattern(item):
|
|
149
|
+
self._scan_for_subcomponents(
|
|
150
|
+
item, subcomponents, visited,
|
|
151
|
+
current_depth + 1, max_depth
|
|
152
|
+
)
|
|
153
|
+
else:
|
|
154
|
+
# Continue scanning subdirectories
|
|
155
|
+
self._scan_for_subcomponents(
|
|
156
|
+
item, subcomponents, visited,
|
|
157
|
+
current_depth + 1, max_depth
|
|
158
|
+
)
|
|
159
|
+
except PermissionError:
|
|
160
|
+
pass
|
|
161
|
+
|
|
162
|
+
def _check_markers(self, path: Path) -> List[str]:
|
|
163
|
+
"""Check for package markers in a directory."""
|
|
164
|
+
markers = []
|
|
165
|
+
|
|
166
|
+
for marker_file, _ in self.PACKAGE_MARKERS.items():
|
|
167
|
+
marker_path = path / marker_file
|
|
168
|
+
if marker_path.exists():
|
|
169
|
+
markers.append(marker_file)
|
|
170
|
+
|
|
171
|
+
return markers
|
|
172
|
+
|
|
173
|
+
def _determine_type(self, markers: List[str]) -> str:
|
|
174
|
+
"""Determine component type from markers."""
|
|
175
|
+
types = set()
|
|
176
|
+
for marker in markers:
|
|
177
|
+
if marker in self.PACKAGE_MARKERS:
|
|
178
|
+
types.add(self.PACKAGE_MARKERS[marker])
|
|
179
|
+
|
|
180
|
+
if len(types) == 1:
|
|
181
|
+
return list(types)[0]
|
|
182
|
+
elif len(types) > 1:
|
|
183
|
+
return 'multi'
|
|
184
|
+
else:
|
|
185
|
+
return 'unknown'
|
|
186
|
+
|
|
187
|
+
def _is_monorepo_pattern(self, path: Path) -> bool:
|
|
188
|
+
"""Check if this looks like a monorepo."""
|
|
189
|
+
# Check for workspace configuration files
|
|
190
|
+
workspace_files = [
|
|
191
|
+
'lerna.json',
|
|
192
|
+
'nx.json',
|
|
193
|
+
'pnpm-workspace.yaml',
|
|
194
|
+
'rush.json',
|
|
195
|
+
'turbo.json'
|
|
196
|
+
]
|
|
197
|
+
|
|
198
|
+
for ws_file in workspace_files:
|
|
199
|
+
if (path / ws_file).exists():
|
|
200
|
+
return True
|
|
201
|
+
|
|
202
|
+
# Check for packages/apps directories
|
|
203
|
+
monorepo_dirs = {'packages', 'apps', 'services', 'libs'}
|
|
204
|
+
subdirs = {item.name for item in path.iterdir() if item.is_dir()}
|
|
205
|
+
|
|
206
|
+
return bool(monorepo_dirs & subdirs)
|
|
207
|
+
|
|
208
|
+
def _organize_subcomponents(
|
|
209
|
+
self,
|
|
210
|
+
subcomponents: List[Subcomponent]
|
|
211
|
+
) -> List[Subcomponent]:
|
|
212
|
+
"""Organize and deduplicate subcomponents."""
|
|
213
|
+
# Remove duplicates based on path
|
|
214
|
+
seen_paths = set()
|
|
215
|
+
unique_components = []
|
|
216
|
+
|
|
217
|
+
for comp in subcomponents:
|
|
218
|
+
if comp.path not in seen_paths:
|
|
219
|
+
unique_components.append(comp)
|
|
220
|
+
seen_paths.add(comp.path)
|
|
221
|
+
|
|
222
|
+
# Sort by path depth (root first, then nested)
|
|
223
|
+
unique_components.sort(key=lambda c: len(c.path.parts))
|
|
224
|
+
|
|
225
|
+
return unique_components
|
|
226
|
+
|
|
227
|
+
def _print_detection_results(self, subcomponents: List[Subcomponent]):
|
|
228
|
+
"""Print detected subcomponents in a nice table."""
|
|
229
|
+
if not subcomponents:
|
|
230
|
+
console.print("[yellow]No subcomponents detected[/yellow]")
|
|
231
|
+
return
|
|
232
|
+
|
|
233
|
+
table = Table(title=f"Detected {len(subcomponents)} Subcomponents")
|
|
234
|
+
table.add_column("Path", style="cyan")
|
|
235
|
+
table.add_column("Type", style="green")
|
|
236
|
+
table.add_column("Markers", style="yellow")
|
|
237
|
+
|
|
238
|
+
for comp in subcomponents:
|
|
239
|
+
# Try to get relative path, fallback to absolute
|
|
240
|
+
try:
|
|
241
|
+
rel_path = str(comp.path.relative_to(Path.cwd()))
|
|
242
|
+
except ValueError:
|
|
243
|
+
# If not relative to cwd, just use the path as-is
|
|
244
|
+
rel_path = str(comp.path)
|
|
245
|
+
|
|
246
|
+
markers_str = ", ".join(comp.markers[:3])
|
|
247
|
+
if len(comp.markers) > 3:
|
|
248
|
+
markers_str += f" +{len(comp.markers)-3}"
|
|
249
|
+
|
|
250
|
+
table.add_row(rel_path, comp.type, markers_str)
|
|
251
|
+
|
|
252
|
+
console.print(table)
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
async def identify_subcomponents(
|
|
256
|
+
root_path: Path,
|
|
257
|
+
max_depth: int = 3,
|
|
258
|
+
confidence_threshold: float = 0.5,
|
|
259
|
+
verbose: bool = False,
|
|
260
|
+
use_swh: bool = False
|
|
261
|
+
) -> Dict[str, Any]:
|
|
262
|
+
"""
|
|
263
|
+
Identify all subcomponents in a project.
|
|
264
|
+
|
|
265
|
+
Args:
|
|
266
|
+
root_path: Root directory to analyze
|
|
267
|
+
max_depth: Maximum depth for scanning
|
|
268
|
+
confidence_threshold: Minimum confidence for identification
|
|
269
|
+
verbose: Enable verbose output
|
|
270
|
+
use_swh: Include Software Heritage checking
|
|
271
|
+
|
|
272
|
+
Returns:
|
|
273
|
+
Dictionary with identification results for all subcomponents
|
|
274
|
+
"""
|
|
275
|
+
from src2id.search import identify_source
|
|
276
|
+
|
|
277
|
+
# Detect subcomponents
|
|
278
|
+
detector = SubcomponentDetector(verbose=verbose)
|
|
279
|
+
subcomponents = detector.detect_subcomponents(root_path, max_depth)
|
|
280
|
+
|
|
281
|
+
if not subcomponents:
|
|
282
|
+
# No subcomponents detected, identify as single project
|
|
283
|
+
if verbose:
|
|
284
|
+
console.print("[dim]No subcomponents detected, analyzing as single project[/dim]")
|
|
285
|
+
|
|
286
|
+
result = await identify_source(
|
|
287
|
+
path=root_path,
|
|
288
|
+
max_depth=max_depth,
|
|
289
|
+
confidence_threshold=confidence_threshold,
|
|
290
|
+
verbose=verbose,
|
|
291
|
+
use_swh=use_swh
|
|
292
|
+
)
|
|
293
|
+
|
|
294
|
+
return {
|
|
295
|
+
"root": root_path,
|
|
296
|
+
"subcomponents": [],
|
|
297
|
+
"single_result": result
|
|
298
|
+
}
|
|
299
|
+
|
|
300
|
+
# Identify each subcomponent
|
|
301
|
+
results = {
|
|
302
|
+
"root": root_path,
|
|
303
|
+
"subcomponents": [],
|
|
304
|
+
"total_identified": 0,
|
|
305
|
+
"total_components": len(subcomponents)
|
|
306
|
+
}
|
|
307
|
+
|
|
308
|
+
if verbose:
|
|
309
|
+
console.print(f"\n[bold]Identifying {len(subcomponents)} subcomponents...[/bold]\n")
|
|
310
|
+
|
|
311
|
+
for i, comp in enumerate(subcomponents, 1):
|
|
312
|
+
if verbose:
|
|
313
|
+
console.print(f"[cyan]Component {i}/{len(subcomponents)}: {comp.path.name}[/cyan]")
|
|
314
|
+
|
|
315
|
+
# Identify this component
|
|
316
|
+
comp_result = await identify_source(
|
|
317
|
+
path=comp.path,
|
|
318
|
+
max_depth=1, # Don't go too deep for subcomponents
|
|
319
|
+
confidence_threshold=confidence_threshold,
|
|
320
|
+
verbose=False, # Less verbose for individual components
|
|
321
|
+
use_swh=use_swh
|
|
322
|
+
)
|
|
323
|
+
|
|
324
|
+
# Add component info to result
|
|
325
|
+
comp_info = {
|
|
326
|
+
"path": str(comp.path),
|
|
327
|
+
"type": comp.type,
|
|
328
|
+
"markers": comp.markers,
|
|
329
|
+
"identified": comp_result["identified"],
|
|
330
|
+
"confidence": comp_result["confidence"],
|
|
331
|
+
"repository": comp_result.get("final_origin"),
|
|
332
|
+
"strategies_used": comp_result.get("strategies_used", [])
|
|
333
|
+
}
|
|
334
|
+
|
|
335
|
+
results["subcomponents"].append(comp_info)
|
|
336
|
+
|
|
337
|
+
if comp_result["identified"]:
|
|
338
|
+
results["total_identified"] += 1
|
|
339
|
+
|
|
340
|
+
if verbose:
|
|
341
|
+
if comp_result["identified"]:
|
|
342
|
+
console.print(f" ✓ Identified: {comp_result['final_origin']}")
|
|
343
|
+
else:
|
|
344
|
+
console.print(f" ✗ Not identified")
|
|
345
|
+
|
|
346
|
+
# Print summary
|
|
347
|
+
if verbose:
|
|
348
|
+
console.print(f"\n[bold green]Summary:[/bold green]")
|
|
349
|
+
console.print(f"Total components: {results['total_components']}")
|
|
350
|
+
console.print(f"Identified: {results['total_identified']}")
|
|
351
|
+
console.print(f"Success rate: {results['total_identified']/results['total_components']:.1%}")
|
|
352
|
+
|
|
353
|
+
return results
|
src2id/core/swhid.py
ADDED
|
@@ -0,0 +1,324 @@
|
|
|
1
|
+
"""SWHID generation using Software Heritage tools."""
|
|
2
|
+
|
|
3
|
+
import hashlib
|
|
4
|
+
import os
|
|
5
|
+
from pathlib import Path
|
|
6
|
+
from typing import Optional
|
|
7
|
+
|
|
8
|
+
# Try different SWHID generation methods in order of preference
|
|
9
|
+
HAS_SWH_MODEL = False
|
|
10
|
+
HAS_MINISWHID = False
|
|
11
|
+
|
|
12
|
+
try:
|
|
13
|
+
from swh.model.cli import model_of_dir
|
|
14
|
+
from swh.model.from_disk import Content
|
|
15
|
+
HAS_SWH_MODEL = True
|
|
16
|
+
except ImportError:
|
|
17
|
+
try:
|
|
18
|
+
import miniswhid
|
|
19
|
+
HAS_MINISWHID = True
|
|
20
|
+
except ImportError:
|
|
21
|
+
import warnings
|
|
22
|
+
warnings.warn(
|
|
23
|
+
"No accurate SWHID generation available. "
|
|
24
|
+
"Install swh.model or miniswhid for accurate SWHID generation: "
|
|
25
|
+
"pip install swh.model"
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class SWHIDGenerator:
|
|
30
|
+
"""
|
|
31
|
+
Generates Software Heritage Identifiers using miniswhid or custom implementation.
|
|
32
|
+
"""
|
|
33
|
+
|
|
34
|
+
def __init__(self, use_swh_model: bool = True):
|
|
35
|
+
"""
|
|
36
|
+
Initialize the SWHID generator.
|
|
37
|
+
|
|
38
|
+
Args:
|
|
39
|
+
use_swh_model: Whether to use swh.model if available
|
|
40
|
+
"""
|
|
41
|
+
self.use_swh_model = use_swh_model and HAS_SWH_MODEL
|
|
42
|
+
self.use_miniswhid = (not self.use_swh_model) and HAS_MINISWHID
|
|
43
|
+
|
|
44
|
+
def generate_directory_swhid(self, path: Path) -> str:
|
|
45
|
+
"""
|
|
46
|
+
Generate SWHID for directory content.
|
|
47
|
+
|
|
48
|
+
Args:
|
|
49
|
+
path: Directory path
|
|
50
|
+
|
|
51
|
+
Returns:
|
|
52
|
+
SWHID string in format swh:1:dir:HASH
|
|
53
|
+
"""
|
|
54
|
+
if not path.is_dir():
|
|
55
|
+
raise ValueError(f"Path {path} is not a directory")
|
|
56
|
+
|
|
57
|
+
if self.use_swh_model:
|
|
58
|
+
return self._generate_with_swh_model(path)
|
|
59
|
+
elif self.use_miniswhid:
|
|
60
|
+
return self._generate_with_miniswhid(path)
|
|
61
|
+
else:
|
|
62
|
+
return self._generate_fallback(path)
|
|
63
|
+
|
|
64
|
+
def generate_content_swhid(self, file_path: Path) -> str:
|
|
65
|
+
"""
|
|
66
|
+
Generate SWHID for individual file content.
|
|
67
|
+
|
|
68
|
+
Args:
|
|
69
|
+
file_path: File path
|
|
70
|
+
|
|
71
|
+
Returns:
|
|
72
|
+
SWHID string in format swh:1:cnt:HASH
|
|
73
|
+
"""
|
|
74
|
+
if not file_path.is_file():
|
|
75
|
+
raise ValueError(f"Path {file_path} is not a file")
|
|
76
|
+
|
|
77
|
+
if self.use_swh_model:
|
|
78
|
+
# Use swh.model for file content
|
|
79
|
+
try:
|
|
80
|
+
from swh.model.from_disk import Content
|
|
81
|
+
content = Content.from_file(path=bytes(file_path))
|
|
82
|
+
return f"swh:1:cnt:{content.hash}"
|
|
83
|
+
except Exception as e:
|
|
84
|
+
print(f"Warning: swh.model failed for {file_path}: {e}")
|
|
85
|
+
return self._hash_file_content(file_path)
|
|
86
|
+
elif self.use_miniswhid:
|
|
87
|
+
# Use miniswhid for file content
|
|
88
|
+
try:
|
|
89
|
+
result = miniswhid.compute_swhid(str(file_path))
|
|
90
|
+
if isinstance(result, dict) and 'swhid' in result:
|
|
91
|
+
return result['swhid']
|
|
92
|
+
return str(result)
|
|
93
|
+
except Exception as e:
|
|
94
|
+
if hasattr(miniswhid, 'hash_file'):
|
|
95
|
+
# Alternative API
|
|
96
|
+
return f"swh:1:cnt:{miniswhid.hash_file(str(file_path))}"
|
|
97
|
+
raise e
|
|
98
|
+
else:
|
|
99
|
+
# Fallback implementation
|
|
100
|
+
return self._hash_file_content(file_path)
|
|
101
|
+
|
|
102
|
+
def _generate_with_swh_model(self, path: Path) -> str:
|
|
103
|
+
"""
|
|
104
|
+
Generate SWHID using swh.model library.
|
|
105
|
+
|
|
106
|
+
Args:
|
|
107
|
+
path: Directory path
|
|
108
|
+
|
|
109
|
+
Returns:
|
|
110
|
+
SWHID string
|
|
111
|
+
"""
|
|
112
|
+
try:
|
|
113
|
+
# Use official exclusion patterns like SWH scanner
|
|
114
|
+
exclusion_patterns = [
|
|
115
|
+
b'.git', b'.hg', b'.svn', b'__pycache__',
|
|
116
|
+
b'.mypy_cache', b'.tox', b'*.egg-info',
|
|
117
|
+
b'.bzr', b'.coverage', b'.eggs'
|
|
118
|
+
]
|
|
119
|
+
|
|
120
|
+
# Use official model_of_dir method (same as SWH scanner)
|
|
121
|
+
source_tree = model_of_dir(
|
|
122
|
+
str(path).encode(),
|
|
123
|
+
exclusion_patterns
|
|
124
|
+
)
|
|
125
|
+
|
|
126
|
+
# Generate SWHID using official method
|
|
127
|
+
swhid = source_tree.swhid()
|
|
128
|
+
return str(swhid)
|
|
129
|
+
|
|
130
|
+
except Exception as e:
|
|
131
|
+
print(f"Warning: swh.model failed for {path}: {e}")
|
|
132
|
+
print("Falling back to custom implementation")
|
|
133
|
+
return self._generate_fallback(path)
|
|
134
|
+
|
|
135
|
+
def _generate_with_miniswhid(self, path: Path) -> str:
|
|
136
|
+
"""
|
|
137
|
+
Generate SWHID using miniswhid library.
|
|
138
|
+
|
|
139
|
+
Args:
|
|
140
|
+
path: Directory path
|
|
141
|
+
|
|
142
|
+
Returns:
|
|
143
|
+
SWHID string
|
|
144
|
+
"""
|
|
145
|
+
try:
|
|
146
|
+
# Try the main API
|
|
147
|
+
result = miniswhid.compute_swhid(str(path))
|
|
148
|
+
|
|
149
|
+
# Handle different return types from miniswhid
|
|
150
|
+
if isinstance(result, dict):
|
|
151
|
+
if 'swhid' in result:
|
|
152
|
+
return result['swhid']
|
|
153
|
+
elif 'directory' in result:
|
|
154
|
+
return f"swh:1:dir:{result['directory']}"
|
|
155
|
+
elif isinstance(result, str):
|
|
156
|
+
if result.startswith('swh:'):
|
|
157
|
+
return result
|
|
158
|
+
else:
|
|
159
|
+
return f"swh:1:dir:{result}"
|
|
160
|
+
|
|
161
|
+
# If we get here, try alternative API if available
|
|
162
|
+
if hasattr(miniswhid, 'hash_directory'):
|
|
163
|
+
dir_hash = miniswhid.hash_directory(str(path))
|
|
164
|
+
return f"swh:1:dir:{dir_hash}"
|
|
165
|
+
|
|
166
|
+
# Last resort: use the result as-is
|
|
167
|
+
return str(result)
|
|
168
|
+
|
|
169
|
+
except Exception as e:
|
|
170
|
+
print(f"Warning: miniswhid failed for {path}: {e}")
|
|
171
|
+
print("Falling back to custom implementation")
|
|
172
|
+
return self._generate_fallback(path)
|
|
173
|
+
|
|
174
|
+
def _generate_fallback(self, path: Path) -> str:
|
|
175
|
+
"""
|
|
176
|
+
Custom fallback implementation for SWHID generation.
|
|
177
|
+
|
|
178
|
+
This is a simplified version that may not match SH exactly
|
|
179
|
+
but provides a consistent hash for the directory content.
|
|
180
|
+
|
|
181
|
+
Args:
|
|
182
|
+
path: Directory path
|
|
183
|
+
|
|
184
|
+
Returns:
|
|
185
|
+
SWHID string
|
|
186
|
+
"""
|
|
187
|
+
# Create a hash of the directory structure and content
|
|
188
|
+
hasher = hashlib.sha1()
|
|
189
|
+
|
|
190
|
+
# Get all files in sorted order for consistency
|
|
191
|
+
all_files = []
|
|
192
|
+
for root, dirs, files in os.walk(path):
|
|
193
|
+
# Skip hidden and build directories
|
|
194
|
+
dirs[:] = [d for d in dirs if not d.startswith('.') and d not in {
|
|
195
|
+
'__pycache__', 'node_modules', 'build', 'dist', 'target'
|
|
196
|
+
}]
|
|
197
|
+
dirs.sort() # Ensure consistent ordering
|
|
198
|
+
|
|
199
|
+
for file in sorted(files):
|
|
200
|
+
if not file.startswith('.'):
|
|
201
|
+
file_path = Path(root) / file
|
|
202
|
+
relative_path = file_path.relative_to(path)
|
|
203
|
+
all_files.append(relative_path)
|
|
204
|
+
|
|
205
|
+
# Sort all files for consistent ordering
|
|
206
|
+
all_files.sort()
|
|
207
|
+
|
|
208
|
+
# Hash the directory structure
|
|
209
|
+
for relative_path in all_files:
|
|
210
|
+
file_path = path / relative_path
|
|
211
|
+
|
|
212
|
+
# Add file path to hash
|
|
213
|
+
hasher.update(str(relative_path).encode('utf-8'))
|
|
214
|
+
|
|
215
|
+
# Add file content hash
|
|
216
|
+
try:
|
|
217
|
+
with open(file_path, 'rb') as f:
|
|
218
|
+
file_hasher = hashlib.sha1()
|
|
219
|
+
while chunk := f.read(8192):
|
|
220
|
+
file_hasher.update(chunk)
|
|
221
|
+
hasher.update(file_hasher.digest())
|
|
222
|
+
except (PermissionError, OSError):
|
|
223
|
+
# Skip files we can't read
|
|
224
|
+
continue
|
|
225
|
+
|
|
226
|
+
# Add file mode (simplified)
|
|
227
|
+
try:
|
|
228
|
+
mode = oct(file_path.stat().st_mode)[-3:]
|
|
229
|
+
hasher.update(mode.encode('utf-8'))
|
|
230
|
+
except (PermissionError, OSError):
|
|
231
|
+
hasher.update(b'644') # Default mode
|
|
232
|
+
|
|
233
|
+
# Generate the final hash
|
|
234
|
+
dir_hash = hasher.hexdigest()
|
|
235
|
+
|
|
236
|
+
# Return in SWHID format
|
|
237
|
+
# Note: This is a simplified format and may not match SH exactly
|
|
238
|
+
return f"swh:1:dir:{dir_hash}"
|
|
239
|
+
|
|
240
|
+
def generate_content_swhid(self, file_path: Path) -> str:
|
|
241
|
+
"""
|
|
242
|
+
Generate SWHID for file content.
|
|
243
|
+
|
|
244
|
+
Args:
|
|
245
|
+
file_path: File path
|
|
246
|
+
|
|
247
|
+
Returns:
|
|
248
|
+
SWHID string for the file content
|
|
249
|
+
"""
|
|
250
|
+
if not file_path.is_file():
|
|
251
|
+
raise ValueError(f"Not a file: {file_path}")
|
|
252
|
+
|
|
253
|
+
if self.use_swh_model:
|
|
254
|
+
try:
|
|
255
|
+
# Use official SWH model for accurate content hashing
|
|
256
|
+
content = Content.from_file(path=file_path, max_content_length=10_000_000)
|
|
257
|
+
return str(content.swhid())
|
|
258
|
+
except Exception:
|
|
259
|
+
# Fall back to basic implementation
|
|
260
|
+
pass
|
|
261
|
+
|
|
262
|
+
# Fallback: Generate git-compatible SHA1
|
|
263
|
+
return self._hash_file_content(file_path)
|
|
264
|
+
|
|
265
|
+
def _hash_file_content(self, file_path: Path) -> str:
|
|
266
|
+
"""
|
|
267
|
+
Generate content hash for a single file.
|
|
268
|
+
|
|
269
|
+
Args:
|
|
270
|
+
file_path: File path
|
|
271
|
+
|
|
272
|
+
Returns:
|
|
273
|
+
SWHID string for content
|
|
274
|
+
"""
|
|
275
|
+
hasher = hashlib.sha1()
|
|
276
|
+
|
|
277
|
+
try:
|
|
278
|
+
with open(file_path, 'rb') as f:
|
|
279
|
+
# Add git-style header
|
|
280
|
+
content = f.read()
|
|
281
|
+
header = f"blob {len(content)}\0".encode('utf-8')
|
|
282
|
+
hasher.update(header)
|
|
283
|
+
hasher.update(content)
|
|
284
|
+
except (PermissionError, OSError) as e:
|
|
285
|
+
raise ValueError(f"Cannot read file {file_path}: {e}")
|
|
286
|
+
|
|
287
|
+
content_hash = hasher.hexdigest()
|
|
288
|
+
return f"swh:1:cnt:{content_hash}"
|
|
289
|
+
|
|
290
|
+
def validate_swhid(self, swhid: str) -> bool:
|
|
291
|
+
"""
|
|
292
|
+
Validate SWHID format.
|
|
293
|
+
|
|
294
|
+
Args:
|
|
295
|
+
swhid: SWHID string to validate
|
|
296
|
+
|
|
297
|
+
Returns:
|
|
298
|
+
True if valid SWHID format
|
|
299
|
+
"""
|
|
300
|
+
if not isinstance(swhid, str):
|
|
301
|
+
return False
|
|
302
|
+
|
|
303
|
+
parts = swhid.split(':')
|
|
304
|
+
if len(parts) != 4:
|
|
305
|
+
return False
|
|
306
|
+
|
|
307
|
+
if parts[0] != 'swh':
|
|
308
|
+
return False
|
|
309
|
+
|
|
310
|
+
if parts[1] != '1': # Version
|
|
311
|
+
return False
|
|
312
|
+
|
|
313
|
+
if parts[2] not in {'cnt', 'dir', 'rev', 'rel', 'snp', 'ori'}:
|
|
314
|
+
return False
|
|
315
|
+
|
|
316
|
+
# Check if hash is valid hex
|
|
317
|
+
try:
|
|
318
|
+
int(parts[3], 16)
|
|
319
|
+
if len(parts[3]) != 40: # SHA1 length
|
|
320
|
+
return False
|
|
321
|
+
except ValueError:
|
|
322
|
+
return False
|
|
323
|
+
|
|
324
|
+
return True
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
"""External integrations for SWHPI."""
|