codeupipe 0.1.0__py3-none-any.whl

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (60) hide show
  1. codeupipe/__init__.py +39 -0
  2. codeupipe/cli.py +1502 -0
  3. codeupipe/converter/__init__.py +9 -0
  4. codeupipe/converter/config.py +119 -0
  5. codeupipe/converter/filters/__init__.py +21 -0
  6. codeupipe/converter/filters/analyze.py +60 -0
  7. codeupipe/converter/filters/classify.py +52 -0
  8. codeupipe/converter/filters/classify_files.py +61 -0
  9. codeupipe/converter/filters/generate_export.py +187 -0
  10. codeupipe/converter/filters/generate_import.py +229 -0
  11. codeupipe/converter/filters/parse_config.py +26 -0
  12. codeupipe/converter/filters/scan_project.py +52 -0
  13. codeupipe/converter/pipelines/__init__.py +8 -0
  14. codeupipe/converter/pipelines/export_pipeline.py +40 -0
  15. codeupipe/converter/pipelines/import_pipeline.py +40 -0
  16. codeupipe/converter/taps/__init__.py +7 -0
  17. codeupipe/converter/taps/conversion_log.py +46 -0
  18. codeupipe/core/__init__.py +20 -0
  19. codeupipe/core/filter.py +27 -0
  20. codeupipe/core/hook.py +34 -0
  21. codeupipe/core/payload.py +94 -0
  22. codeupipe/core/pipeline.py +231 -0
  23. codeupipe/core/state.py +78 -0
  24. codeupipe/core/stream_filter.py +33 -0
  25. codeupipe/core/tap.py +27 -0
  26. codeupipe/core/valve.py +52 -0
  27. codeupipe/linter/__init__.py +57 -0
  28. codeupipe/linter/assemble_doc_report.py +97 -0
  29. codeupipe/linter/assemble_report.py +140 -0
  30. codeupipe/linter/check_bundle.py +47 -0
  31. codeupipe/linter/check_index.py +88 -0
  32. codeupipe/linter/check_naming.py +47 -0
  33. codeupipe/linter/check_protocols.py +73 -0
  34. codeupipe/linter/check_structure.py +41 -0
  35. codeupipe/linter/check_symbols.py +116 -0
  36. codeupipe/linter/check_tests.py +48 -0
  37. codeupipe/linter/coverage_pipeline.py +39 -0
  38. codeupipe/linter/detect_drift.py +44 -0
  39. codeupipe/linter/detect_orphans.py +106 -0
  40. codeupipe/linter/doc_check_pipeline.py +28 -0
  41. codeupipe/linter/git_history.py +130 -0
  42. codeupipe/linter/lint_pipeline.py +51 -0
  43. codeupipe/linter/map_coverage.py +84 -0
  44. codeupipe/linter/report_gaps.py +68 -0
  45. codeupipe/linter/report_pipeline.py +49 -0
  46. codeupipe/linter/resolve_refs.py +48 -0
  47. codeupipe/linter/scan_components.py +95 -0
  48. codeupipe/linter/scan_directory.py +123 -0
  49. codeupipe/linter/scan_docs.py +62 -0
  50. codeupipe/linter/scan_tests.py +104 -0
  51. codeupipe/py.typed +1 -0
  52. codeupipe/testing.py +344 -0
  53. codeupipe/utils/__init__.py +10 -0
  54. codeupipe/utils/error_handling.py +68 -0
  55. codeupipe-0.1.0.dist-info/METADATA +216 -0
  56. codeupipe-0.1.0.dist-info/RECORD +60 -0
  57. codeupipe-0.1.0.dist-info/WHEEL +5 -0
  58. codeupipe-0.1.0.dist-info/entry_points.txt +2 -0
  59. codeupipe-0.1.0.dist-info/licenses/LICENSE +190 -0
  60. codeupipe-0.1.0.dist-info/top_level.txt +1 -0
@@ -0,0 +1,106 @@
1
+ """
2
+ DetectOrphans: Find unreferenced components and stale test files.
3
+ """
4
+
5
+ import ast
6
+ from pathlib import Path
7
+
8
+ from codeupipe import Payload
9
+
10
+
11
+ def _extract_imported_names(tree: ast.Module) -> set:
12
+ """Extract all names imported from relative or absolute imports."""
13
+ names = set()
14
+ for node in ast.walk(tree):
15
+ if isinstance(node, ast.ImportFrom):
16
+ for alias in (node.names or []):
17
+ names.add(alias.name)
18
+ elif isinstance(node, ast.Import):
19
+ for alias in (node.names or []):
20
+ names.add(alias.name.split(".")[-1])
21
+ return names
22
+
23
+
24
+ class DetectOrphans:
25
+ """
26
+ Filter (sync): Detect orphaned components and orphaned test files.
27
+
28
+ Orphaned component = never imported by any other .py file in the directory
29
+ (excluding __init__.py re-exports and test files).
30
+
31
+ Orphaned test = test_*.py file whose stem doesn't match any component.
32
+
33
+ Input keys:
34
+ - components (list[dict]): from ScanComponents
35
+ - directory (str): component directory path
36
+ - tests_dir (str, optional): test directory path
37
+
38
+ Output keys (added):
39
+ - orphaned_components (list[dict]): components never imported
40
+ - orphaned_tests (list[dict]): test files with no matching component
41
+ - import_map (dict[str, list[str]]): component_name → list of importing files
42
+ """
43
+
44
+ def call(self, payload: Payload) -> Payload:
45
+ components = payload.get("components", [])
46
+ directory = payload.get("directory", "")
47
+ tests_dir_str = payload.get("tests_dir", "tests")
48
+
49
+ dir_path = Path(directory)
50
+ component_names = {c["name"] for c in components}
51
+ component_stems = {c["stem"] for c in components}
52
+
53
+ # Build import map: which files import each component name
54
+ import_map = {name: [] for name in component_names}
55
+
56
+ if dir_path.is_dir():
57
+ for py_file in sorted(dir_path.glob("*.py")):
58
+ # Skip __init__.py (re-exports don't count as real usage)
59
+ if py_file.name == "__init__.py":
60
+ continue
61
+ # Skip test files in the component directory
62
+ if py_file.name.startswith("test_"):
63
+ continue
64
+
65
+ try:
66
+ source = py_file.read_text()
67
+ tree = ast.parse(source, filename=str(py_file))
68
+ except (SyntaxError, OSError):
69
+ continue
70
+
71
+ imported = _extract_imported_names(tree)
72
+ for name in component_names:
73
+ if name in imported:
74
+ # Don't count self-imports (file that defines the component)
75
+ comp = next((c for c in components if c["name"] == name), None)
76
+ if comp and Path(comp["file"]).name == py_file.name:
77
+ continue
78
+ import_map[name].append(py_file.name)
79
+
80
+ # Orphaned components: never imported by anyone
81
+ orphaned_components = []
82
+ for comp in components:
83
+ importers = import_map.get(comp["name"], [])
84
+ if not importers:
85
+ orphaned_components.append({
86
+ "name": comp["name"],
87
+ "kind": comp["kind"],
88
+ "file": comp["file"],
89
+ })
90
+
91
+ # Orphaned tests: test_*.py files whose stem doesn't match any component
92
+ orphaned_tests = []
93
+ tests_dir = Path(tests_dir_str)
94
+ if tests_dir.is_dir():
95
+ for test_file in sorted(tests_dir.glob("test_*.py")):
96
+ stem = test_file.stem[5:] # strip "test_"
97
+ if stem not in component_stems:
98
+ orphaned_tests.append({
99
+ "file": str(test_file),
100
+ "stem": stem,
101
+ })
102
+
103
+ payload = payload.insert("orphaned_components", orphaned_components)
104
+ payload = payload.insert("orphaned_tests", orphaned_tests)
105
+ payload = payload.insert("import_map", import_map)
106
+ return payload
@@ -0,0 +1,28 @@
1
+ """
2
+ doc_check_pipeline: Wires the doc-code sync check pipeline.
3
+
4
+ Scans markdown for cup:ref markers, resolves file references,
5
+ checks symbol existence, detects hash drift, validates index
6
+ coverage, and assembles a report.
7
+ """
8
+
9
+ from codeupipe import Pipeline
10
+
11
+ from .scan_docs import ScanDocs
12
+ from .resolve_refs import ResolveRefs
13
+ from .check_symbols import CheckSymbols
14
+ from .detect_drift import DetectDrift
15
+ from .check_index import CheckIndex
16
+ from .assemble_doc_report import AssembleDocReport
17
+
18
+
19
+ def build_doc_check_pipeline() -> Pipeline:
20
+ """Build and return the doc-check pipeline."""
21
+ pipeline = Pipeline()
22
+ pipeline.add_filter(ScanDocs(), "scan_docs")
23
+ pipeline.add_filter(ResolveRefs(), "resolve_refs")
24
+ pipeline.add_filter(CheckSymbols(), "check_symbols")
25
+ pipeline.add_filter(DetectDrift(), "detect_drift")
26
+ pipeline.add_filter(CheckIndex(), "check_index")
27
+ pipeline.add_filter(AssembleDocReport(), "assemble_doc_report")
28
+ return pipeline
@@ -0,0 +1,130 @@
1
+ """
2
+ GitHistory: Retrieve git log data for component and test files.
3
+ """
4
+
5
+ import subprocess
6
+ from datetime import datetime, timezone
7
+ from pathlib import Path
8
+ from typing import Optional
9
+
10
+ from codeupipe import Payload
11
+
12
+
13
+ def _git_file_info(filepath: str, repo_root: str) -> dict:
14
+ """Get git log info for a single file.
15
+
16
+ Returns dict with last_modified, last_author, commit_count, days_since_change.
17
+ """
18
+ try:
19
+ # Get last commit date + author
20
+ result = subprocess.run(
21
+ ["git", "log", "-1", "--format=%aI%n%aN", "--", filepath],
22
+ cwd=repo_root,
23
+ capture_output=True,
24
+ text=True,
25
+ timeout=10,
26
+ )
27
+ lines = result.stdout.strip().split("\n")
28
+ if result.returncode != 0 or not lines or not lines[0]:
29
+ return {
30
+ "last_modified": None,
31
+ "last_author": None,
32
+ "commit_count": 0,
33
+ "days_since_change": None,
34
+ }
35
+
36
+ last_modified_iso = lines[0]
37
+ last_author = lines[1] if len(lines) > 1 else None
38
+
39
+ # Parse date for days_since_change
40
+ # Python 3.9/3.10 fromisoformat doesn't accept 'Z' suffix
41
+ if last_modified_iso.endswith("Z"):
42
+ last_modified_iso = last_modified_iso[:-1] + "+00:00"
43
+ last_dt = datetime.fromisoformat(last_modified_iso)
44
+ now = datetime.now(timezone.utc)
45
+ if last_dt.tzinfo is None:
46
+ last_dt = last_dt.replace(tzinfo=timezone.utc)
47
+ days = (now - last_dt).days
48
+
49
+ # Get commit count
50
+ count_result = subprocess.run(
51
+ ["git", "rev-list", "--count", "HEAD", "--", filepath],
52
+ cwd=repo_root,
53
+ capture_output=True,
54
+ text=True,
55
+ timeout=10,
56
+ )
57
+ commit_count = int(count_result.stdout.strip()) if count_result.returncode == 0 else 0
58
+
59
+ # Normalize date to YYYY-MM-DD
60
+ last_modified = last_dt.strftime("%Y-%m-%d")
61
+
62
+ return {
63
+ "last_modified": last_modified,
64
+ "last_author": last_author,
65
+ "commit_count": commit_count,
66
+ "days_since_change": days,
67
+ }
68
+ except (subprocess.TimeoutExpired, OSError, ValueError):
69
+ return {
70
+ "last_modified": None,
71
+ "last_author": None,
72
+ "commit_count": 0,
73
+ "days_since_change": None,
74
+ }
75
+
76
+
77
+ def _find_repo_root(directory: str) -> Optional[str]:
78
+ """Find the git repository root for a directory."""
79
+ try:
80
+ result = subprocess.run(
81
+ ["git", "rev-parse", "--show-toplevel"],
82
+ cwd=directory,
83
+ capture_output=True,
84
+ text=True,
85
+ timeout=5,
86
+ )
87
+ if result.returncode == 0:
88
+ return result.stdout.strip()
89
+ except (subprocess.TimeoutExpired, OSError):
90
+ pass
91
+ return None
92
+
93
+
94
+ class GitHistory:
95
+ """
96
+ Filter (sync): Retrieve git history for each component and test file.
97
+
98
+ Input keys:
99
+ - components (list[dict]): from ScanComponents (each has 'file')
100
+ - test_map (list[dict], optional): from ScanTests (each has 'test_file')
101
+ - directory (str): component directory path
102
+
103
+ Output keys (added):
104
+ - git_info (dict[str, dict]): filepath → git data
105
+ Each entry has: last_modified, last_author, commit_count, days_since_change
106
+ """
107
+
108
+ def call(self, payload: Payload) -> Payload:
109
+ components = payload.get("components", [])
110
+ test_map = payload.get("test_map", [])
111
+ directory = payload.get("directory", ".")
112
+
113
+ repo_root = _find_repo_root(directory)
114
+ if repo_root is None:
115
+ return payload.insert("git_info", {})
116
+
117
+ # Collect unique file paths
118
+ files = set()
119
+ for comp in components:
120
+ files.add(comp["file"])
121
+ for entry in test_map:
122
+ files.add(entry["test_file"])
123
+
124
+ # Query git for each file
125
+ git_info = {}
126
+ for filepath in sorted(files):
127
+ info = _git_file_info(filepath, repo_root)
128
+ git_info[filepath] = info
129
+
130
+ return payload.insert("git_info", git_info)
@@ -0,0 +1,51 @@
1
+ """
2
+ LintPipeline: CUP standards linter — built with CUP itself (dogfooding).
3
+
4
+ Scans a directory and enforces CUP000–CUP008 rules:
5
+ CUP000: Syntax error in file
6
+ CUP001: Multiple components in one file
7
+ CUP002: Missing test file
8
+ CUP003: Filter missing call()
9
+ CUP004: Tap missing observe()
10
+ CUP005: StreamFilter missing stream()
11
+ CUP006: Hook missing lifecycle methods
12
+ CUP007: File name not snake_case
13
+ CUP008: Stale __init__.py bundle
14
+ """
15
+
16
+ from codeupipe import Pipeline, Payload
17
+
18
+ # TODO: update import paths to match your project layout
19
+ from .scan_directory import ScanDirectory
20
+ from .check_naming import CheckNaming
21
+ from .check_structure import CheckStructure
22
+ from .check_protocols import CheckProtocols
23
+ from .check_tests import CheckTests
24
+ from .check_bundle import CheckBundle
25
+
26
+
27
+ def build_lint_pipeline() -> Pipeline:
28
+ """
29
+ Construct the LintPipeline pipeline.
30
+
31
+ Steps:
32
+ 1. ScanDirectory (Filter)
33
+ 2. CheckNaming (Filter)
34
+ 3. CheckStructure (Filter)
35
+ 4. CheckProtocols (Filter)
36
+ 5. CheckTests (Filter)
37
+ 6. CheckBundle (Filter)
38
+
39
+ Use pipeline.run(payload) for single-payload execution.
40
+ Use pipeline.stream(source) for streaming execution.
41
+ """
42
+ pipeline = Pipeline()
43
+
44
+ pipeline.add_filter(ScanDirectory(), "scan_directory")
45
+ pipeline.add_filter(CheckNaming(), "check_naming")
46
+ pipeline.add_filter(CheckStructure(), "check_structure")
47
+ pipeline.add_filter(CheckProtocols(), "check_protocols")
48
+ pipeline.add_filter(CheckTests(), "check_tests")
49
+ pipeline.add_filter(CheckBundle(), "check_bundle")
50
+
51
+ return pipeline
@@ -0,0 +1,84 @@
1
+ """
2
+ MapCoverage: Cross-reference components against tests to build coverage map.
3
+ """
4
+
5
+ from codeupipe import Payload
6
+
7
+
8
+ class MapCoverage:
9
+ """
10
+ Filter (sync): Join components and test_map to produce per-component coverage.
11
+
12
+ Input keys:
13
+ - components (list[dict]): from ScanComponents
14
+ - test_map (list[dict]): from ScanTests
15
+
16
+ Output keys (added):
17
+ - coverage (list[dict]): each with keys:
18
+ - name (str): component class/function name
19
+ - kind (str): component type
20
+ - file (str): source file path
21
+ - has_test_file (bool): whether a test file exists
22
+ - test_count (int): number of test_* methods
23
+ - methods (list[str]): public method names
24
+ - tested_methods (list[str]): methods referenced in tests
25
+ - untested_methods (list[str]): methods not referenced in tests
26
+ - coverage_pct (float): percentage of methods covered (0-100)
27
+ """
28
+
29
+ def call(self, payload: Payload) -> Payload:
30
+ components = payload.get("components", [])
31
+ test_map = payload.get("test_map", [])
32
+
33
+ # Index test_map by stem for fast lookup
34
+ test_index = {}
35
+ for entry in test_map:
36
+ stem = entry["stem"]
37
+ if stem not in test_index:
38
+ test_index[stem] = {
39
+ "test_methods": [],
40
+ "referenced_methods": set(),
41
+ "imports": set(),
42
+ }
43
+ test_index[stem]["test_methods"].extend(entry["test_methods"])
44
+ test_index[stem]["referenced_methods"] |= entry["referenced_methods"]
45
+ test_index[stem]["imports"] |= entry["imports"]
46
+
47
+ coverage = []
48
+
49
+ for comp in components:
50
+ stem = comp["stem"]
51
+ name = comp["name"]
52
+ kind = comp["kind"]
53
+ methods = comp["methods"]
54
+ test_info = test_index.get(stem)
55
+
56
+ has_test = test_info is not None
57
+ test_count = len(test_info["test_methods"]) if test_info else 0
58
+ referenced = test_info["referenced_methods"] if test_info else set()
59
+
60
+ # For builders (functions), check if the name itself is imported
61
+ if kind == "builder":
62
+ tested = [name] if (test_info and name in test_info["imports"]) else []
63
+ untested = [] if tested else [name]
64
+ total = 1
65
+ else:
66
+ tested = [m for m in methods if m in referenced]
67
+ untested = [m for m in methods if m not in referenced]
68
+ total = len(methods)
69
+
70
+ pct = (len(tested) / total * 100) if total > 0 else 100.0
71
+
72
+ coverage.append({
73
+ "name": name,
74
+ "kind": kind,
75
+ "file": comp["file"],
76
+ "has_test_file": has_test,
77
+ "test_count": test_count,
78
+ "methods": methods if kind != "builder" else [name],
79
+ "tested_methods": tested,
80
+ "untested_methods": untested,
81
+ "coverage_pct": round(pct, 1),
82
+ })
83
+
84
+ return payload.insert("coverage", coverage)
@@ -0,0 +1,68 @@
1
+ """
2
+ ReportGaps: Compute summary statistics and coverage gaps from coverage data.
3
+ """
4
+
5
+ from codeupipe import Payload
6
+
7
+
8
+ class ReportGaps:
9
+ """
10
+ Filter (sync): Produce a human-readable summary from coverage data.
11
+
12
+ Input keys:
13
+ - coverage (list[dict]): from MapCoverage
14
+
15
+ Output keys (added):
16
+ - summary (dict): with keys:
17
+ - total_components (int)
18
+ - tested_components (int): components with at least one test
19
+ - untested_components (int): components with zero tests
20
+ - total_methods (int)
21
+ - tested_methods (int)
22
+ - untested_methods (int)
23
+ - overall_pct (float): aggregate method coverage %
24
+ - gaps (list[dict]): components with coverage < 100%, each with:
25
+ - name (str)
26
+ - kind (str)
27
+ - file (str)
28
+ - coverage_pct (float)
29
+ - missing (list[str]): untested method names
30
+ """
31
+
32
+ def call(self, payload: Payload) -> Payload:
33
+ coverage = payload.get("coverage", [])
34
+
35
+ total_components = len(coverage)
36
+ tested_components = sum(1 for c in coverage if c["has_test_file"])
37
+ untested_components = total_components - tested_components
38
+
39
+ total_methods = sum(len(c["methods"]) for c in coverage)
40
+ tested_methods = sum(len(c["tested_methods"]) for c in coverage)
41
+ untested_methods = total_methods - tested_methods
42
+
43
+ overall_pct = (tested_methods / total_methods * 100) if total_methods > 0 else 100.0
44
+
45
+ summary = {
46
+ "total_components": total_components,
47
+ "tested_components": tested_components,
48
+ "untested_components": untested_components,
49
+ "total_methods": total_methods,
50
+ "tested_methods": tested_methods,
51
+ "untested_methods": untested_methods,
52
+ "overall_pct": round(overall_pct, 1),
53
+ }
54
+
55
+ gaps = []
56
+ for c in coverage:
57
+ if c["coverage_pct"] < 100.0:
58
+ gaps.append({
59
+ "name": c["name"],
60
+ "kind": c["kind"],
61
+ "file": c["file"],
62
+ "coverage_pct": c["coverage_pct"],
63
+ "missing": c["untested_methods"],
64
+ })
65
+
66
+ payload = payload.insert("summary", summary)
67
+ payload = payload.insert("gaps", gaps)
68
+ return payload
@@ -0,0 +1,49 @@
1
+ """
2
+ ReportPipeline: Full codebase health report — built with CUP.
3
+
4
+ Composes coverage analysis, orphan detection, git history, and
5
+ report assembly into a single pipeline.
6
+
7
+ Reuses: ScanComponents, ScanTests, MapCoverage, ReportGaps
8
+ New: DetectOrphans, GitHistory, AssembleReport
9
+ """
10
+
11
+ from codeupipe import Pipeline
12
+
13
+ from .scan_components import ScanComponents
14
+ from .scan_tests import ScanTests
15
+ from .map_coverage import MapCoverage
16
+ from .report_gaps import ReportGaps
17
+ from .detect_orphans import DetectOrphans
18
+ from .git_history import GitHistory
19
+ from .assemble_report import AssembleReport
20
+
21
+
22
+ def build_report_pipeline() -> Pipeline:
23
+ """
24
+ Construct the full report pipeline.
25
+
26
+ Steps:
27
+ 1. ScanComponents — catalog components + public methods
28
+ 2. ScanTests — parse test files, map tested symbols
29
+ 3. MapCoverage — cross-reference coverage
30
+ 4. ReportGaps — compute summary + gaps
31
+ 5. DetectOrphans — find unreferenced components/tests
32
+ 6. GitHistory — retrieve git log per file
33
+ 7. AssembleReport — merge into unified report
34
+
35
+ Use pipeline.run(payload) with:
36
+ - directory (str): component directory path
37
+ - tests_dir (str, optional): test directory (default: "tests")
38
+ """
39
+ pipeline = Pipeline()
40
+
41
+ pipeline.add_filter(ScanComponents(), "scan_components")
42
+ pipeline.add_filter(ScanTests(), "scan_tests")
43
+ pipeline.add_filter(MapCoverage(), "map_coverage")
44
+ pipeline.add_filter(ReportGaps(), "report_gaps")
45
+ pipeline.add_filter(DetectOrphans(), "detect_orphans")
46
+ pipeline.add_filter(GitHistory(), "git_history")
47
+ pipeline.add_filter(AssembleReport(), "assemble_report")
48
+
49
+ return pipeline
@@ -0,0 +1,48 @@
1
+ """
2
+ ResolveRefs: Verify source files exist and compute current content hashes.
3
+
4
+ Takes the doc_refs from ScanDocs and enriches each with existence checks
5
+ and current file content hashes for drift detection.
6
+ """
7
+
8
+ import hashlib
9
+ from pathlib import Path
10
+
11
+ from codeupipe import Payload
12
+
13
+
14
+ class ResolveRefs:
15
+ """
16
+ Filter (sync): Resolve file references and compute current hashes.
17
+
18
+ Input keys:
19
+ - directory (str): root directory
20
+ - doc_refs (list[dict]): from ScanDocs
21
+
22
+ Output keys (added):
23
+ - resolved_refs (list[dict]): enriched refs with:
24
+ exists (bool), current_hash (str|None), abs_path (str)
25
+ """
26
+
27
+ def call(self, payload: Payload) -> Payload:
28
+ directory = Path(payload.get("directory", "."))
29
+ doc_refs = payload.get("doc_refs", [])
30
+ resolved = []
31
+
32
+ for ref in doc_refs:
33
+ abs_path = directory / ref["file"]
34
+ exists = abs_path.is_file()
35
+
36
+ current_hash = None
37
+ if exists:
38
+ content = abs_path.read_bytes()
39
+ current_hash = hashlib.sha256(content).hexdigest()[:7]
40
+
41
+ resolved.append({
42
+ **ref,
43
+ "exists": exists,
44
+ "current_hash": current_hash,
45
+ "abs_path": str(abs_path),
46
+ })
47
+
48
+ return payload.insert("resolved_refs", resolved)
@@ -0,0 +1,95 @@
1
+ """
2
+ ScanComponents: Discover all CUP components and their public methods.
3
+ """
4
+
5
+ import ast
6
+ from pathlib import Path
7
+ from typing import Optional
8
+
9
+ from codeupipe import Payload
10
+
11
+ from .scan_directory import classify_class
12
+
13
+
14
+ def _extract_public_methods(node: ast.ClassDef) -> list:
15
+ """Extract public method names from a class AST node."""
16
+ return [
17
+ n.name
18
+ for n in ast.iter_child_nodes(node)
19
+ if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef))
20
+ and not n.name.startswith("_")
21
+ ]
22
+
23
+
24
+ def _extract_public_functions(tree: ast.Module) -> list:
25
+ """Extract top-level public function names from a module."""
26
+ return [
27
+ n.name
28
+ for n in ast.iter_child_nodes(tree)
29
+ if isinstance(n, (ast.FunctionDef, ast.AsyncFunctionDef))
30
+ and not n.name.startswith("_")
31
+ ]
32
+
33
+
34
+ class ScanComponents:
35
+ """
36
+ Filter (sync): Parse component directory and catalog each
37
+ component class with its public methods.
38
+
39
+ Input keys:
40
+ - directory (str): path to the component directory
41
+
42
+ Output keys (added):
43
+ - components (list[dict]): each with keys:
44
+ - file (str): relative filepath
45
+ - stem (str): filename without .py
46
+ - name (str): class or function name
47
+ - kind (str): component type (filter, tap, hook, stream-filter, builder)
48
+ - methods (list[str]): public method names
49
+ """
50
+
51
+ def call(self, payload: Payload) -> Payload:
52
+ directory = payload.get("directory")
53
+ dir_path = Path(directory)
54
+
55
+ if not dir_path.is_dir():
56
+ raise FileNotFoundError(f"Directory not found: {directory}")
57
+
58
+ components = []
59
+
60
+ for py_file in sorted(dir_path.glob("*.py")):
61
+ if py_file.name == "__init__.py":
62
+ continue
63
+
64
+ try:
65
+ source = py_file.read_text()
66
+ tree = ast.parse(source, filename=str(py_file))
67
+ except (SyntaxError, OSError):
68
+ continue
69
+
70
+ rel = str(py_file)
71
+ stem = py_file.stem
72
+
73
+ for node in ast.iter_child_nodes(tree):
74
+ if isinstance(node, ast.ClassDef) and not node.name.startswith("_"):
75
+ ctype = classify_class(node)
76
+ if ctype is not None:
77
+ methods = _extract_public_methods(node)
78
+ components.append({
79
+ "file": rel,
80
+ "stem": stem,
81
+ "name": node.name,
82
+ "kind": ctype,
83
+ "methods": methods,
84
+ })
85
+ elif isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
86
+ if not node.name.startswith("_") and node.name.startswith("build_"):
87
+ components.append({
88
+ "file": rel,
89
+ "stem": stem,
90
+ "name": node.name,
91
+ "kind": "builder",
92
+ "methods": [],
93
+ })
94
+
95
+ return payload.insert("components", components)