src2purl 1.3.2__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,353 @@
1
+ """Detect and identify multiple subcomponents in a project."""
2
+
3
+ import asyncio
4
+ from pathlib import Path
5
+ from typing import List, Dict, Any, Optional
6
+ from dataclasses import dataclass
7
+
8
+ from rich.console import Console
9
+ from rich.table import Table
10
+
11
+ console = Console()
12
+
13
+
14
+ @dataclass
15
+ class Subcomponent:
16
+ """Represents a detected subcomponent in a project."""
17
+ path: Path
18
+ type: str # 'npm', 'python', 'rust', 'go', 'java', 'mono', 'multi'
19
+ name: Optional[str] = None
20
+ markers: List[str] = None
21
+
22
+ def __post_init__(self):
23
+ if self.markers is None:
24
+ self.markers = []
25
+
26
+
27
+ class SubcomponentDetector:
28
+ """Detects multiple subcomponents in a project structure."""
29
+
30
+ # Package markers that indicate a component boundary
31
+ PACKAGE_MARKERS = {
32
+ 'package.json': 'npm',
33
+ 'pyproject.toml': 'python',
34
+ 'setup.py': 'python',
35
+ 'requirements.txt': 'python',
36
+ 'Cargo.toml': 'rust',
37
+ 'go.mod': 'go',
38
+ 'pom.xml': 'java',
39
+ 'build.gradle': 'java',
40
+ 'build.gradle.kts': 'java',
41
+ 'Gemfile': 'ruby',
42
+ 'composer.json': 'php',
43
+ 'CMakeLists.txt': 'cmake',
44
+ 'Makefile': 'make',
45
+ '.csproj': 'dotnet',
46
+ 'mix.exs': 'elixir',
47
+ }
48
+
49
+ # Directories that typically indicate subcomponents
50
+ SUBCOMPONENT_PATTERNS = {
51
+ 'packages/*': 'monorepo', # Lerna/Yarn workspaces
52
+ 'apps/*': 'monorepo', # Nx/Turborepo
53
+ 'services/*': 'microservices',
54
+ 'libs/*': 'libraries',
55
+ 'modules/*': 'modules',
56
+ 'components/*': 'components',
57
+ 'plugins/*': 'plugins',
58
+ 'extensions/*': 'extensions',
59
+ }
60
+
61
+ def __init__(self, verbose: bool = False):
62
+ """Initialize the detector."""
63
+ self.verbose = verbose
64
+
65
+ def detect_subcomponents(
66
+ self,
67
+ root_path: Path,
68
+ max_depth: int = 3
69
+ ) -> List[Subcomponent]:
70
+ """
71
+ Detect all subcomponents in a project.
72
+
73
+ Args:
74
+ root_path: Root directory to scan
75
+ max_depth: Maximum depth to search for subcomponents
76
+
77
+ Returns:
78
+ List of detected subcomponents
79
+ """
80
+ subcomponents = []
81
+ visited = set()
82
+
83
+ # Check if root itself is a component
84
+ root_markers = self._check_markers(root_path)
85
+ if root_markers:
86
+ root_component = Subcomponent(
87
+ path=root_path,
88
+ type=self._determine_type(root_markers),
89
+ name=root_path.name,
90
+ markers=root_markers
91
+ )
92
+ subcomponents.append(root_component)
93
+ visited.add(root_path)
94
+
95
+ # Search for nested subcomponents
96
+ self._scan_for_subcomponents(
97
+ root_path,
98
+ subcomponents,
99
+ visited,
100
+ current_depth=0,
101
+ max_depth=max_depth
102
+ )
103
+
104
+ # Deduplicate and organize
105
+ subcomponents = self._organize_subcomponents(subcomponents)
106
+
107
+ if self.verbose:
108
+ self._print_detection_results(subcomponents)
109
+
110
+ return subcomponents
111
+
112
+ def _scan_for_subcomponents(
113
+ self,
114
+ path: Path,
115
+ subcomponents: List[Subcomponent],
116
+ visited: set,
117
+ current_depth: int,
118
+ max_depth: int
119
+ ):
120
+ """Recursively scan for subcomponents."""
121
+ if current_depth >= max_depth:
122
+ return
123
+
124
+ try:
125
+ for item in path.iterdir():
126
+ if item.is_dir() and item not in visited:
127
+ # Skip common non-component directories
128
+ if item.name in {'.git', '__pycache__', 'node_modules',
129
+ 'venv', 'build', 'dist', '.idea', '.vscode'}:
130
+ continue
131
+
132
+ # Check for package markers
133
+ markers = self._check_markers(item)
134
+
135
+ if markers:
136
+ # Found a subcomponent
137
+ component = Subcomponent(
138
+ path=item,
139
+ type=self._determine_type(markers),
140
+ name=item.name,
141
+ markers=markers
142
+ )
143
+ subcomponents.append(component)
144
+ visited.add(item)
145
+
146
+ # Don't scan inside detected components by default
147
+ # unless it's a monorepo pattern
148
+ if self._is_monorepo_pattern(item):
149
+ self._scan_for_subcomponents(
150
+ item, subcomponents, visited,
151
+ current_depth + 1, max_depth
152
+ )
153
+ else:
154
+ # Continue scanning subdirectories
155
+ self._scan_for_subcomponents(
156
+ item, subcomponents, visited,
157
+ current_depth + 1, max_depth
158
+ )
159
+ except PermissionError:
160
+ pass
161
+
162
+ def _check_markers(self, path: Path) -> List[str]:
163
+ """Check for package markers in a directory."""
164
+ markers = []
165
+
166
+ for marker_file, _ in self.PACKAGE_MARKERS.items():
167
+ marker_path = path / marker_file
168
+ if marker_path.exists():
169
+ markers.append(marker_file)
170
+
171
+ return markers
172
+
173
+ def _determine_type(self, markers: List[str]) -> str:
174
+ """Determine component type from markers."""
175
+ types = set()
176
+ for marker in markers:
177
+ if marker in self.PACKAGE_MARKERS:
178
+ types.add(self.PACKAGE_MARKERS[marker])
179
+
180
+ if len(types) == 1:
181
+ return list(types)[0]
182
+ elif len(types) > 1:
183
+ return 'multi'
184
+ else:
185
+ return 'unknown'
186
+
187
+ def _is_monorepo_pattern(self, path: Path) -> bool:
188
+ """Check if this looks like a monorepo."""
189
+ # Check for workspace configuration files
190
+ workspace_files = [
191
+ 'lerna.json',
192
+ 'nx.json',
193
+ 'pnpm-workspace.yaml',
194
+ 'rush.json',
195
+ 'turbo.json'
196
+ ]
197
+
198
+ for ws_file in workspace_files:
199
+ if (path / ws_file).exists():
200
+ return True
201
+
202
+ # Check for packages/apps directories
203
+ monorepo_dirs = {'packages', 'apps', 'services', 'libs'}
204
+ subdirs = {item.name for item in path.iterdir() if item.is_dir()}
205
+
206
+ return bool(monorepo_dirs & subdirs)
207
+
208
+ def _organize_subcomponents(
209
+ self,
210
+ subcomponents: List[Subcomponent]
211
+ ) -> List[Subcomponent]:
212
+ """Organize and deduplicate subcomponents."""
213
+ # Remove duplicates based on path
214
+ seen_paths = set()
215
+ unique_components = []
216
+
217
+ for comp in subcomponents:
218
+ if comp.path not in seen_paths:
219
+ unique_components.append(comp)
220
+ seen_paths.add(comp.path)
221
+
222
+ # Sort by path depth (root first, then nested)
223
+ unique_components.sort(key=lambda c: len(c.path.parts))
224
+
225
+ return unique_components
226
+
227
+ def _print_detection_results(self, subcomponents: List[Subcomponent]):
228
+ """Print detected subcomponents in a nice table."""
229
+ if not subcomponents:
230
+ console.print("[yellow]No subcomponents detected[/yellow]")
231
+ return
232
+
233
+ table = Table(title=f"Detected {len(subcomponents)} Subcomponents")
234
+ table.add_column("Path", style="cyan")
235
+ table.add_column("Type", style="green")
236
+ table.add_column("Markers", style="yellow")
237
+
238
+ for comp in subcomponents:
239
+ # Try to get relative path, fallback to absolute
240
+ try:
241
+ rel_path = str(comp.path.relative_to(Path.cwd()))
242
+ except ValueError:
243
+ # If not relative to cwd, just use the path as-is
244
+ rel_path = str(comp.path)
245
+
246
+ markers_str = ", ".join(comp.markers[:3])
247
+ if len(comp.markers) > 3:
248
+ markers_str += f" +{len(comp.markers)-3}"
249
+
250
+ table.add_row(rel_path, comp.type, markers_str)
251
+
252
+ console.print(table)
253
+
254
+
255
+ async def identify_subcomponents(
256
+ root_path: Path,
257
+ max_depth: int = 3,
258
+ confidence_threshold: float = 0.5,
259
+ verbose: bool = False,
260
+ use_swh: bool = False
261
+ ) -> Dict[str, Any]:
262
+ """
263
+ Identify all subcomponents in a project.
264
+
265
+ Args:
266
+ root_path: Root directory to analyze
267
+ max_depth: Maximum depth for scanning
268
+ confidence_threshold: Minimum confidence for identification
269
+ verbose: Enable verbose output
270
+ use_swh: Include Software Heritage checking
271
+
272
+ Returns:
273
+ Dictionary with identification results for all subcomponents
274
+ """
275
+ from src2id.search import identify_source
276
+
277
+ # Detect subcomponents
278
+ detector = SubcomponentDetector(verbose=verbose)
279
+ subcomponents = detector.detect_subcomponents(root_path, max_depth)
280
+
281
+ if not subcomponents:
282
+ # No subcomponents detected, identify as single project
283
+ if verbose:
284
+ console.print("[dim]No subcomponents detected, analyzing as single project[/dim]")
285
+
286
+ result = await identify_source(
287
+ path=root_path,
288
+ max_depth=max_depth,
289
+ confidence_threshold=confidence_threshold,
290
+ verbose=verbose,
291
+ use_swh=use_swh
292
+ )
293
+
294
+ return {
295
+ "root": root_path,
296
+ "subcomponents": [],
297
+ "single_result": result
298
+ }
299
+
300
+ # Identify each subcomponent
301
+ results = {
302
+ "root": root_path,
303
+ "subcomponents": [],
304
+ "total_identified": 0,
305
+ "total_components": len(subcomponents)
306
+ }
307
+
308
+ if verbose:
309
+ console.print(f"\n[bold]Identifying {len(subcomponents)} subcomponents...[/bold]\n")
310
+
311
+ for i, comp in enumerate(subcomponents, 1):
312
+ if verbose:
313
+ console.print(f"[cyan]Component {i}/{len(subcomponents)}: {comp.path.name}[/cyan]")
314
+
315
+ # Identify this component
316
+ comp_result = await identify_source(
317
+ path=comp.path,
318
+ max_depth=1, # Don't go too deep for subcomponents
319
+ confidence_threshold=confidence_threshold,
320
+ verbose=False, # Less verbose for individual components
321
+ use_swh=use_swh
322
+ )
323
+
324
+ # Add component info to result
325
+ comp_info = {
326
+ "path": str(comp.path),
327
+ "type": comp.type,
328
+ "markers": comp.markers,
329
+ "identified": comp_result["identified"],
330
+ "confidence": comp_result["confidence"],
331
+ "repository": comp_result.get("final_origin"),
332
+ "strategies_used": comp_result.get("strategies_used", [])
333
+ }
334
+
335
+ results["subcomponents"].append(comp_info)
336
+
337
+ if comp_result["identified"]:
338
+ results["total_identified"] += 1
339
+
340
+ if verbose:
341
+ if comp_result["identified"]:
342
+ console.print(f" ✓ Identified: {comp_result['final_origin']}")
343
+ else:
344
+ console.print(f" ✗ Not identified")
345
+
346
+ # Print summary
347
+ if verbose:
348
+ console.print(f"\n[bold green]Summary:[/bold green]")
349
+ console.print(f"Total components: {results['total_components']}")
350
+ console.print(f"Identified: {results['total_identified']}")
351
+ console.print(f"Success rate: {results['total_identified']/results['total_components']:.1%}")
352
+
353
+ return results
src2id/core/swhid.py ADDED
@@ -0,0 +1,324 @@
1
+ """SWHID generation using Software Heritage tools."""
2
+
3
+ import hashlib
4
+ import os
5
+ from pathlib import Path
6
+ from typing import Optional
7
+
8
+ # Try different SWHID generation methods in order of preference
9
+ HAS_SWH_MODEL = False
10
+ HAS_MINISWHID = False
11
+
12
+ try:
13
+ from swh.model.cli import model_of_dir
14
+ from swh.model.from_disk import Content
15
+ HAS_SWH_MODEL = True
16
+ except ImportError:
17
+ try:
18
+ import miniswhid
19
+ HAS_MINISWHID = True
20
+ except ImportError:
21
+ import warnings
22
+ warnings.warn(
23
+ "No accurate SWHID generation available. "
24
+ "Install swh.model or miniswhid for accurate SWHID generation: "
25
+ "pip install swh.model"
26
+ )
27
+
28
+
29
+ class SWHIDGenerator:
30
+ """
31
+ Generates Software Heritage Identifiers using miniswhid or custom implementation.
32
+ """
33
+
34
+ def __init__(self, use_swh_model: bool = True):
35
+ """
36
+ Initialize the SWHID generator.
37
+
38
+ Args:
39
+ use_swh_model: Whether to use swh.model if available
40
+ """
41
+ self.use_swh_model = use_swh_model and HAS_SWH_MODEL
42
+ self.use_miniswhid = (not self.use_swh_model) and HAS_MINISWHID
43
+
44
+ def generate_directory_swhid(self, path: Path) -> str:
45
+ """
46
+ Generate SWHID for directory content.
47
+
48
+ Args:
49
+ path: Directory path
50
+
51
+ Returns:
52
+ SWHID string in format swh:1:dir:HASH
53
+ """
54
+ if not path.is_dir():
55
+ raise ValueError(f"Path {path} is not a directory")
56
+
57
+ if self.use_swh_model:
58
+ return self._generate_with_swh_model(path)
59
+ elif self.use_miniswhid:
60
+ return self._generate_with_miniswhid(path)
61
+ else:
62
+ return self._generate_fallback(path)
63
+
64
+ def generate_content_swhid(self, file_path: Path) -> str:
65
+ """
66
+ Generate SWHID for individual file content.
67
+
68
+ Args:
69
+ file_path: File path
70
+
71
+ Returns:
72
+ SWHID string in format swh:1:cnt:HASH
73
+ """
74
+ if not file_path.is_file():
75
+ raise ValueError(f"Path {file_path} is not a file")
76
+
77
+ if self.use_swh_model:
78
+ # Use swh.model for file content
79
+ try:
80
+ from swh.model.from_disk import Content
81
+ content = Content.from_file(path=bytes(file_path))
82
+ return f"swh:1:cnt:{content.hash}"
83
+ except Exception as e:
84
+ print(f"Warning: swh.model failed for {file_path}: {e}")
85
+ return self._hash_file_content(file_path)
86
+ elif self.use_miniswhid:
87
+ # Use miniswhid for file content
88
+ try:
89
+ result = miniswhid.compute_swhid(str(file_path))
90
+ if isinstance(result, dict) and 'swhid' in result:
91
+ return result['swhid']
92
+ return str(result)
93
+ except Exception as e:
94
+ if hasattr(miniswhid, 'hash_file'):
95
+ # Alternative API
96
+ return f"swh:1:cnt:{miniswhid.hash_file(str(file_path))}"
97
+ raise e
98
+ else:
99
+ # Fallback implementation
100
+ return self._hash_file_content(file_path)
101
+
102
+ def _generate_with_swh_model(self, path: Path) -> str:
103
+ """
104
+ Generate SWHID using swh.model library.
105
+
106
+ Args:
107
+ path: Directory path
108
+
109
+ Returns:
110
+ SWHID string
111
+ """
112
+ try:
113
+ # Use official exclusion patterns like SWH scanner
114
+ exclusion_patterns = [
115
+ b'.git', b'.hg', b'.svn', b'__pycache__',
116
+ b'.mypy_cache', b'.tox', b'*.egg-info',
117
+ b'.bzr', b'.coverage', b'.eggs'
118
+ ]
119
+
120
+ # Use official model_of_dir method (same as SWH scanner)
121
+ source_tree = model_of_dir(
122
+ str(path).encode(),
123
+ exclusion_patterns
124
+ )
125
+
126
+ # Generate SWHID using official method
127
+ swhid = source_tree.swhid()
128
+ return str(swhid)
129
+
130
+ except Exception as e:
131
+ print(f"Warning: swh.model failed for {path}: {e}")
132
+ print("Falling back to custom implementation")
133
+ return self._generate_fallback(path)
134
+
135
+ def _generate_with_miniswhid(self, path: Path) -> str:
136
+ """
137
+ Generate SWHID using miniswhid library.
138
+
139
+ Args:
140
+ path: Directory path
141
+
142
+ Returns:
143
+ SWHID string
144
+ """
145
+ try:
146
+ # Try the main API
147
+ result = miniswhid.compute_swhid(str(path))
148
+
149
+ # Handle different return types from miniswhid
150
+ if isinstance(result, dict):
151
+ if 'swhid' in result:
152
+ return result['swhid']
153
+ elif 'directory' in result:
154
+ return f"swh:1:dir:{result['directory']}"
155
+ elif isinstance(result, str):
156
+ if result.startswith('swh:'):
157
+ return result
158
+ else:
159
+ return f"swh:1:dir:{result}"
160
+
161
+ # If we get here, try alternative API if available
162
+ if hasattr(miniswhid, 'hash_directory'):
163
+ dir_hash = miniswhid.hash_directory(str(path))
164
+ return f"swh:1:dir:{dir_hash}"
165
+
166
+ # Last resort: use the result as-is
167
+ return str(result)
168
+
169
+ except Exception as e:
170
+ print(f"Warning: miniswhid failed for {path}: {e}")
171
+ print("Falling back to custom implementation")
172
+ return self._generate_fallback(path)
173
+
174
+ def _generate_fallback(self, path: Path) -> str:
175
+ """
176
+ Custom fallback implementation for SWHID generation.
177
+
178
+ This is a simplified version that may not match SH exactly
179
+ but provides a consistent hash for the directory content.
180
+
181
+ Args:
182
+ path: Directory path
183
+
184
+ Returns:
185
+ SWHID string
186
+ """
187
+ # Create a hash of the directory structure and content
188
+ hasher = hashlib.sha1()
189
+
190
+ # Get all files in sorted order for consistency
191
+ all_files = []
192
+ for root, dirs, files in os.walk(path):
193
+ # Skip hidden and build directories
194
+ dirs[:] = [d for d in dirs if not d.startswith('.') and d not in {
195
+ '__pycache__', 'node_modules', 'build', 'dist', 'target'
196
+ }]
197
+ dirs.sort() # Ensure consistent ordering
198
+
199
+ for file in sorted(files):
200
+ if not file.startswith('.'):
201
+ file_path = Path(root) / file
202
+ relative_path = file_path.relative_to(path)
203
+ all_files.append(relative_path)
204
+
205
+ # Sort all files for consistent ordering
206
+ all_files.sort()
207
+
208
+ # Hash the directory structure
209
+ for relative_path in all_files:
210
+ file_path = path / relative_path
211
+
212
+ # Add file path to hash
213
+ hasher.update(str(relative_path).encode('utf-8'))
214
+
215
+ # Add file content hash
216
+ try:
217
+ with open(file_path, 'rb') as f:
218
+ file_hasher = hashlib.sha1()
219
+ while chunk := f.read(8192):
220
+ file_hasher.update(chunk)
221
+ hasher.update(file_hasher.digest())
222
+ except (PermissionError, OSError):
223
+ # Skip files we can't read
224
+ continue
225
+
226
+ # Add file mode (simplified)
227
+ try:
228
+ mode = oct(file_path.stat().st_mode)[-3:]
229
+ hasher.update(mode.encode('utf-8'))
230
+ except (PermissionError, OSError):
231
+ hasher.update(b'644') # Default mode
232
+
233
+ # Generate the final hash
234
+ dir_hash = hasher.hexdigest()
235
+
236
+ # Return in SWHID format
237
+ # Note: This is a simplified format and may not match SH exactly
238
+ return f"swh:1:dir:{dir_hash}"
239
+
240
+ def generate_content_swhid(self, file_path: Path) -> str:
241
+ """
242
+ Generate SWHID for file content.
243
+
244
+ Args:
245
+ file_path: File path
246
+
247
+ Returns:
248
+ SWHID string for the file content
249
+ """
250
+ if not file_path.is_file():
251
+ raise ValueError(f"Not a file: {file_path}")
252
+
253
+ if self.use_swh_model:
254
+ try:
255
+ # Use official SWH model for accurate content hashing
256
+ content = Content.from_file(path=file_path, max_content_length=10_000_000)
257
+ return str(content.swhid())
258
+ except Exception:
259
+ # Fall back to basic implementation
260
+ pass
261
+
262
+ # Fallback: Generate git-compatible SHA1
263
+ return self._hash_file_content(file_path)
264
+
265
+ def _hash_file_content(self, file_path: Path) -> str:
266
+ """
267
+ Generate content hash for a single file.
268
+
269
+ Args:
270
+ file_path: File path
271
+
272
+ Returns:
273
+ SWHID string for content
274
+ """
275
+ hasher = hashlib.sha1()
276
+
277
+ try:
278
+ with open(file_path, 'rb') as f:
279
+ # Add git-style header
280
+ content = f.read()
281
+ header = f"blob {len(content)}\0".encode('utf-8')
282
+ hasher.update(header)
283
+ hasher.update(content)
284
+ except (PermissionError, OSError) as e:
285
+ raise ValueError(f"Cannot read file {file_path}: {e}")
286
+
287
+ content_hash = hasher.hexdigest()
288
+ return f"swh:1:cnt:{content_hash}"
289
+
290
+ def validate_swhid(self, swhid: str) -> bool:
291
+ """
292
+ Validate SWHID format.
293
+
294
+ Args:
295
+ swhid: SWHID string to validate
296
+
297
+ Returns:
298
+ True if valid SWHID format
299
+ """
300
+ if not isinstance(swhid, str):
301
+ return False
302
+
303
+ parts = swhid.split(':')
304
+ if len(parts) != 4:
305
+ return False
306
+
307
+ if parts[0] != 'swh':
308
+ return False
309
+
310
+ if parts[1] != '1': # Version
311
+ return False
312
+
313
+ if parts[2] not in {'cnt', 'dir', 'rev', 'rel', 'snp', 'ori'}:
314
+ return False
315
+
316
+ # Check if hash is valid hex
317
+ try:
318
+ int(parts[3], 16)
319
+ if len(parts[3]) != 40: # SHA1 length
320
+ return False
321
+ except ValueError:
322
+ return False
323
+
324
+ return True
@@ -0,0 +1 @@
1
+ """External integrations for SWHPI."""