python-surveyor 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- python_surveyor/__init__.py +17 -0
- python_surveyor/__main__.py +6 -0
- python_surveyor/cli.py +102 -0
- python_surveyor/model.py +65 -0
- python_surveyor/report.py +137 -0
- python_surveyor/scanner.py +251 -0
- python_surveyor-0.1.0.dist-info/METADATA +19 -0
- python_surveyor-0.1.0.dist-info/RECORD +12 -0
- python_surveyor-0.1.0.dist-info/WHEEL +5 -0
- python_surveyor-0.1.0.dist-info/entry_points.txt +2 -0
- python_surveyor-0.1.0.dist-info/licenses/LICENSE +21 -0
- python_surveyor-0.1.0.dist-info/top_level.txt +1 -0
|
@@ -0,0 +1,17 @@
|
|
|
1
|
+
"""python-surveyor: AST-based scanner for AI-generated-code smells."""
|
|
2
|
+
|
|
3
|
+
from python_surveyor.model import (
|
|
4
|
+
CallSite,
|
|
5
|
+
Finding,
|
|
6
|
+
Location,
|
|
7
|
+
ParseError,
|
|
8
|
+
SourceExcerpt,
|
|
9
|
+
)
|
|
10
|
+
|
|
11
|
+
__all__ = [
|
|
12
|
+
"CallSite",
|
|
13
|
+
"Finding",
|
|
14
|
+
"Location",
|
|
15
|
+
"ParseError",
|
|
16
|
+
"SourceExcerpt",
|
|
17
|
+
]
|
python_surveyor/cli.py
ADDED
|
@@ -0,0 +1,102 @@
|
|
|
1
|
+
"""``click`` CLI: ``scan`` and ``list-checks`` commands."""
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import TextIO
|
|
6
|
+
|
|
7
|
+
import click
|
|
8
|
+
|
|
9
|
+
from python_surveyor.checks import ALL_CHECKS
|
|
10
|
+
from python_surveyor.report import render_json, render_text
|
|
11
|
+
from python_surveyor.scanner import scan
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@click.group()
|
|
15
|
+
def main() -> None:
|
|
16
|
+
"""python-surveyor: AST-based scanner for AI-generated-code smells."""
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
@main.command("scan")
|
|
20
|
+
@click.argument("paths", nargs=-1, type=click.Path(exists=True))
|
|
21
|
+
@click.option(
|
|
22
|
+
"--check",
|
|
23
|
+
"check_ids",
|
|
24
|
+
multiple=True,
|
|
25
|
+
help="Restrict to this check id (repeatable).",
|
|
26
|
+
)
|
|
27
|
+
@click.option(
|
|
28
|
+
"--exclude",
|
|
29
|
+
"excludes",
|
|
30
|
+
multiple=True,
|
|
31
|
+
help="fnmatch pattern to exclude (repeatable, additive to defaults).",
|
|
32
|
+
)
|
|
33
|
+
@click.option(
|
|
34
|
+
"--format",
|
|
35
|
+
"fmt",
|
|
36
|
+
type=click.Choice(["text", "json"]),
|
|
37
|
+
default="text",
|
|
38
|
+
help="Output format. Default: text.",
|
|
39
|
+
)
|
|
40
|
+
@click.option(
|
|
41
|
+
"--output",
|
|
42
|
+
"output",
|
|
43
|
+
type=click.Path(path_type=Path),
|
|
44
|
+
default=None,
|
|
45
|
+
help="Write to PATH instead of stdout.",
|
|
46
|
+
)
|
|
47
|
+
@click.option(
|
|
48
|
+
"--max-excerpt-lines",
|
|
49
|
+
"max_excerpt_lines",
|
|
50
|
+
type=int,
|
|
51
|
+
default=20,
|
|
52
|
+
help="Cap excerpt line count (default 20).",
|
|
53
|
+
)
|
|
54
|
+
@click.option(
|
|
55
|
+
"--max-call-sites",
|
|
56
|
+
"max_call_sites",
|
|
57
|
+
type=int,
|
|
58
|
+
default=5,
|
|
59
|
+
help="Cap call-site samples in notes (default 5).",
|
|
60
|
+
)
|
|
61
|
+
def scan_cmd(
|
|
62
|
+
paths: tuple[str, ...],
|
|
63
|
+
check_ids: tuple[str, ...],
|
|
64
|
+
excludes: tuple[str, ...],
|
|
65
|
+
fmt: str,
|
|
66
|
+
output: "Path | None",
|
|
67
|
+
max_excerpt_lines: int,
|
|
68
|
+
max_call_sites: int,
|
|
69
|
+
) -> None:
|
|
70
|
+
"""Scan PATHS for AI-generated-code smells."""
|
|
71
|
+
root_paths: tuple[str, ...] = paths if paths else (".",)
|
|
72
|
+
result = scan(
|
|
73
|
+
root_paths=root_paths,
|
|
74
|
+
check_ids=check_ids if check_ids else None,
|
|
75
|
+
excludes=excludes,
|
|
76
|
+
max_call_sites=max_call_sites,
|
|
77
|
+
)
|
|
78
|
+
stream: TextIO
|
|
79
|
+
if output is not None:
|
|
80
|
+
stream = output.open("w", encoding="utf-8")
|
|
81
|
+
else:
|
|
82
|
+
stream = sys.stdout
|
|
83
|
+
try:
|
|
84
|
+
if fmt == "json":
|
|
85
|
+
render_json(result, max_excerpt_lines, stream)
|
|
86
|
+
else:
|
|
87
|
+
render_text(result, max_excerpt_lines, stream)
|
|
88
|
+
finally:
|
|
89
|
+
if output is not None:
|
|
90
|
+
stream.close()
|
|
91
|
+
|
|
92
|
+
|
|
93
|
+
@main.command("list-checks")
|
|
94
|
+
def list_checks_cmd() -> None:
|
|
95
|
+
"""List available check ids and descriptions."""
|
|
96
|
+
click.echo("id\tdescription")
|
|
97
|
+
for check_spec in ALL_CHECKS:
|
|
98
|
+
click.echo(f"{check_spec.check_id}\t{check_spec.description}")
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
if __name__ == "__main__":
|
|
102
|
+
main()
|
python_surveyor/model.py
ADDED
|
@@ -0,0 +1,65 @@
|
|
|
1
|
+
"""Frozen dataclasses shared by scanner, checks, and report renderers.
|
|
2
|
+
|
|
3
|
+
All construction uses required positional fields; there are no
|
|
4
|
+
default-valued fields on ``Finding``/``SourceExcerpt``, consistent with the
|
|
5
|
+
"no optional params on internal APIs" rule this tool itself checks for.
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
from dataclasses import dataclass
|
|
9
|
+
from pathlib import Path
|
|
10
|
+
|
|
11
|
+
|
|
12
|
+
@dataclass(frozen=True)
|
|
13
|
+
class Location:
|
|
14
|
+
"""A single source position."""
|
|
15
|
+
|
|
16
|
+
path: Path
|
|
17
|
+
line: int
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
@dataclass(frozen=True)
|
|
21
|
+
class CallSite:
|
|
22
|
+
"""A call to a name, with enough info to tell if defaulted params are used.
|
|
23
|
+
|
|
24
|
+
``positional_count`` is the number of positional args (not counting
|
|
25
|
+
``*args`` spreads). ``keywords`` is the set of keyword arg names (not
|
|
26
|
+
counting ``**kwargs`` spreads). Together these let the
|
|
27
|
+
``optional-param-default`` check determine whether a call site uses the
|
|
28
|
+
default value or passes an explicit one.
|
|
29
|
+
"""
|
|
30
|
+
|
|
31
|
+
location: Location
|
|
32
|
+
positional_count: int
|
|
33
|
+
keywords: frozenset[str]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
@dataclass(frozen=True)
|
|
37
|
+
class SourceExcerpt:
|
|
38
|
+
"""A literal slice of source worth showing verbatim."""
|
|
39
|
+
|
|
40
|
+
label: str
|
|
41
|
+
path: Path
|
|
42
|
+
start_line: int
|
|
43
|
+
end_line: int
|
|
44
|
+
|
|
45
|
+
|
|
46
|
+
@dataclass(frozen=True)
|
|
47
|
+
class Finding:
|
|
48
|
+
"""A candidate smell detected by a check."""
|
|
49
|
+
|
|
50
|
+
check_id: str
|
|
51
|
+
path: Path
|
|
52
|
+
line: int
|
|
53
|
+
column: int
|
|
54
|
+
message: str
|
|
55
|
+
excerpts: tuple[SourceExcerpt, ...]
|
|
56
|
+
notes: tuple[str, ...]
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
@dataclass(frozen=True)
|
|
60
|
+
class ParseError:
|
|
61
|
+
"""A file that could not be ``ast.parse``-d."""
|
|
62
|
+
|
|
63
|
+
path: Path
|
|
64
|
+
line: int
|
|
65
|
+
message: str
|
|
@@ -0,0 +1,137 @@
|
|
|
1
|
+
"""Render a ``ScanResult`` as text or JSON.
|
|
2
|
+
|
|
3
|
+
Text output groups findings by ``check_id``; each finding is shown as
|
|
4
|
+
``path:line:col -- message`` followed by labeled excerpts (with the line
|
|
5
|
+
range in the header) and note lines. Parse errors get their own leading
|
|
6
|
+
section. Excerpts are capped to ``--max-excerpt-lines`` with a truncation
|
|
7
|
+
note.
|
|
8
|
+
|
|
9
|
+
JSON output is a flat list of the same fields for programmatic use (excerpt
|
|
10
|
+
text is not included — only the line range, so consumers can lazy-load).
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import json
|
|
14
|
+
from pathlib import Path
|
|
15
|
+
from typing import TextIO
|
|
16
|
+
|
|
17
|
+
from python_surveyor.checks import CHECKS_BY_ID
|
|
18
|
+
from python_surveyor.scanner import ScanResult
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _read_excerpt_lines(
|
|
22
|
+
path: Path, start: int, end: int, max_lines: int
|
|
23
|
+
) -> "tuple[list[str], str | None]":
|
|
24
|
+
"""Read ``[start, end]`` from ``path``, capping to ``max_lines``.
|
|
25
|
+
|
|
26
|
+
Returns ``(lines, truncation_note)``.
|
|
27
|
+
"""
|
|
28
|
+
try:
|
|
29
|
+
all_lines = path.read_text(encoding="utf-8").splitlines()
|
|
30
|
+
except OSError:
|
|
31
|
+
return ([], None)
|
|
32
|
+
available_end = min(end, len(all_lines))
|
|
33
|
+
if available_end < start:
|
|
34
|
+
return ([], None)
|
|
35
|
+
span = available_end - start + 1
|
|
36
|
+
if span <= max_lines:
|
|
37
|
+
return (all_lines[start - 1 : available_end], None)
|
|
38
|
+
capped_end = start + max_lines - 1
|
|
39
|
+
return (
|
|
40
|
+
all_lines[start - 1 : capped_end],
|
|
41
|
+
f"excerpt truncated at {max_lines} lines",
|
|
42
|
+
)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def render_text(
|
|
46
|
+
result: ScanResult,
|
|
47
|
+
max_excerpt_lines: int,
|
|
48
|
+
stream: TextIO,
|
|
49
|
+
) -> None:
|
|
50
|
+
"""Render ``result`` as human-readable text to ``stream``."""
|
|
51
|
+
stream.write(f"# scan: {result.files_scanned} file(s) scanned\n\n")
|
|
52
|
+
if result.parse_errors:
|
|
53
|
+
stream.write("## parse errors\n\n")
|
|
54
|
+
for err in result.parse_errors:
|
|
55
|
+
stream.write(f"{err.path}:{err.line} -- {err.message}\n")
|
|
56
|
+
stream.write("\n")
|
|
57
|
+
current_check: str | None = None
|
|
58
|
+
counter = 0
|
|
59
|
+
for finding in result.findings:
|
|
60
|
+
if finding.check_id != current_check:
|
|
61
|
+
current_check = finding.check_id
|
|
62
|
+
counter = 0
|
|
63
|
+
stream.write(f"## {current_check}\n\n")
|
|
64
|
+
spec = CHECKS_BY_ID.get(current_check)
|
|
65
|
+
if spec is not None:
|
|
66
|
+
stream.write(f"{spec.explanation}\n\n")
|
|
67
|
+
counter += 1
|
|
68
|
+
for excerpt in finding.excerpts:
|
|
69
|
+
lines, truncation = _read_excerpt_lines(
|
|
70
|
+
excerpt.path,
|
|
71
|
+
excerpt.start_line,
|
|
72
|
+
excerpt.end_line,
|
|
73
|
+
max_excerpt_lines,
|
|
74
|
+
)
|
|
75
|
+
stream.write(
|
|
76
|
+
f"({counter}) {excerpt.path}:"
|
|
77
|
+
f"{excerpt.start_line}-{excerpt.end_line}\n"
|
|
78
|
+
)
|
|
79
|
+
for line in lines:
|
|
80
|
+
stream.write(f"{line}\n")
|
|
81
|
+
if truncation:
|
|
82
|
+
stream.write(f"... {truncation}\n")
|
|
83
|
+
for note in finding.notes:
|
|
84
|
+
stream.write(f" note: {note}\n")
|
|
85
|
+
stream.write("\n")
|
|
86
|
+
if not result.findings:
|
|
87
|
+
stream.write("no findings.\n")
|
|
88
|
+
|
|
89
|
+
|
|
90
|
+
def render_json(
|
|
91
|
+
result: ScanResult,
|
|
92
|
+
max_excerpt_lines: int,
|
|
93
|
+
stream: TextIO,
|
|
94
|
+
) -> None:
|
|
95
|
+
"""Render ``result`` as JSON to ``stream``.
|
|
96
|
+
|
|
97
|
+
``max_excerpt_lines`` is accepted for API symmetry with
|
|
98
|
+
:func:`render_text`; JSON emits ranges only, not source text.
|
|
99
|
+
"""
|
|
100
|
+
del max_excerpt_lines
|
|
101
|
+
findings_payload = []
|
|
102
|
+
for finding in result.findings:
|
|
103
|
+
excerpts_payload = [
|
|
104
|
+
{
|
|
105
|
+
"label": excerpt.label,
|
|
106
|
+
"path": str(excerpt.path),
|
|
107
|
+
"start_line": excerpt.start_line,
|
|
108
|
+
"end_line": excerpt.end_line,
|
|
109
|
+
}
|
|
110
|
+
for excerpt in finding.excerpts
|
|
111
|
+
]
|
|
112
|
+
findings_payload.append(
|
|
113
|
+
{
|
|
114
|
+
"check_id": finding.check_id,
|
|
115
|
+
"path": str(finding.path),
|
|
116
|
+
"line": finding.line,
|
|
117
|
+
"column": finding.column,
|
|
118
|
+
"message": finding.message,
|
|
119
|
+
"excerpts": excerpts_payload,
|
|
120
|
+
"notes": list(finding.notes),
|
|
121
|
+
}
|
|
122
|
+
)
|
|
123
|
+
parse_errors_payload = [
|
|
124
|
+
{
|
|
125
|
+
"path": str(err.path),
|
|
126
|
+
"line": err.line,
|
|
127
|
+
"message": err.message,
|
|
128
|
+
}
|
|
129
|
+
for err in result.parse_errors
|
|
130
|
+
]
|
|
131
|
+
payload = {
|
|
132
|
+
"files_scanned": result.files_scanned,
|
|
133
|
+
"findings": findings_payload,
|
|
134
|
+
"parse_errors": parse_errors_payload,
|
|
135
|
+
}
|
|
136
|
+
json.dump(payload, stream, indent=2)
|
|
137
|
+
stream.write("\n")
|
|
@@ -0,0 +1,251 @@
|
|
|
1
|
+
"""Discovery, parsing, ``Corpus`` building, and check orchestration.
|
|
2
|
+
|
|
3
|
+
The scanner runs in two passes:
|
|
4
|
+
|
|
5
|
+
1. **Discover + parse.** Walk the given paths, prune the default
|
|
6
|
+
ignore-directories plus anything matching ``--exclude``, read every
|
|
7
|
+
``*.py`` file, ``ast.parse`` it. Files that fail to parse become a
|
|
8
|
+
``ParseError`` and are excluded from further analysis rather than silently
|
|
9
|
+
dropped.
|
|
10
|
+
2. **Build a ``Corpus``** over the successfully parsed files, giving every
|
|
11
|
+
check a cheap, already-built answer to "does anything in what we scanned
|
|
12
|
+
call/use this identifier" without re-parsing.
|
|
13
|
+
3. **Run checks.** Every check has the same signature so the internal check
|
|
14
|
+
API itself has no optional parameters.
|
|
15
|
+
4. **Sort** findings by ``(check_id, path, line)`` and return a
|
|
16
|
+
``ScanResult``.
|
|
17
|
+
"""
|
|
18
|
+
|
|
19
|
+
import ast
|
|
20
|
+
import fnmatch
|
|
21
|
+
import os
|
|
22
|
+
import tokenize
|
|
23
|
+
from collections.abc import Iterator
|
|
24
|
+
from dataclasses import dataclass
|
|
25
|
+
from pathlib import Path
|
|
26
|
+
|
|
27
|
+
from python_surveyor.checks import ALL_CHECKS, CheckSpec
|
|
28
|
+
from python_surveyor.model import CallSite, Finding, Location, ParseError
|
|
29
|
+
|
|
30
|
+
DEFAULT_PRUNED_DIRS: tuple[str, ...] = (
|
|
31
|
+
".venv",
|
|
32
|
+
".git",
|
|
33
|
+
"__pycache__",
|
|
34
|
+
"build",
|
|
35
|
+
"dist",
|
|
36
|
+
".pytest_cache",
|
|
37
|
+
".mypy_cache",
|
|
38
|
+
".tox",
|
|
39
|
+
".ruff_cache",
|
|
40
|
+
)
|
|
41
|
+
|
|
42
|
+
|
|
43
|
+
def _endswith_egg_info(name: str) -> bool:
|
|
44
|
+
return name.endswith(".egg-info")
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
@dataclass(frozen=True, eq=False)
|
|
48
|
+
class SourceFile:
|
|
49
|
+
"""A parsed file plus its source text and tokens."""
|
|
50
|
+
|
|
51
|
+
path: Path
|
|
52
|
+
source: str
|
|
53
|
+
lines: tuple[str, ...]
|
|
54
|
+
tree: ast.Module
|
|
55
|
+
tokens: tuple[tokenize.TokenInfo, ...]
|
|
56
|
+
|
|
57
|
+
def line_text(self, line_no: int) -> str:
|
|
58
|
+
"""Return the (1-indexed) line's text, or ``""`` if out of range."""
|
|
59
|
+
if 1 <= line_no <= len(self.lines):
|
|
60
|
+
return self.lines[line_no - 1]
|
|
61
|
+
return ""
|
|
62
|
+
|
|
63
|
+
|
|
64
|
+
@dataclass(frozen=True)
|
|
65
|
+
class Corpus:
|
|
66
|
+
"""Cross-file indices built once over every successfully parsed file.
|
|
67
|
+
|
|
68
|
+
``call_sites`` answers "is this name called anywhere we scanned?".
|
|
69
|
+
``param_locations`` answers "is this name used as a parameter anywhere?"
|
|
70
|
+
and pairs each hit with the enclosing function's name so the
|
|
71
|
+
fixture-naming check can spot redefined-outer-name collisions.
|
|
72
|
+
``max_call_sites`` is the cap the ``optional-param-default`` check applies
|
|
73
|
+
when sampling call sites into a note (set by the scanner from the CLI
|
|
74
|
+
flag so checks don't need optional parameters of their own).
|
|
75
|
+
"""
|
|
76
|
+
|
|
77
|
+
call_sites: dict[str, tuple[CallSite, ...]]
|
|
78
|
+
param_locations: dict[str, tuple[tuple[Location, str], ...]]
|
|
79
|
+
max_call_sites: int
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
@dataclass(frozen=True)
|
|
83
|
+
class ScanResult:
|
|
84
|
+
"""The output of a scan: findings plus any parse errors encountered."""
|
|
85
|
+
|
|
86
|
+
findings: tuple[Finding, ...]
|
|
87
|
+
parse_errors: tuple[ParseError, ...]
|
|
88
|
+
files_scanned: int
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def _discover(root_paths: tuple[str, ...], excludes: tuple[str, ...]) -> Iterator[Path]:
|
|
92
|
+
"""Walk ``root_paths`` and yield ``*.py`` file paths, pruning ignored dirs."""
|
|
93
|
+
pruned_dirs = set(DEFAULT_PRUNED_DIRS)
|
|
94
|
+
for root in root_paths:
|
|
95
|
+
root_path = Path(root)
|
|
96
|
+
if root_path.is_file():
|
|
97
|
+
if root_path.suffix == ".py":
|
|
98
|
+
yield root_path
|
|
99
|
+
continue
|
|
100
|
+
for dirpath, dirnames, filenames in os.walk(root_path):
|
|
101
|
+
kept_dirs: list[str] = []
|
|
102
|
+
for dirname in dirnames:
|
|
103
|
+
if dirname in pruned_dirs or _endswith_egg_info(dirname):
|
|
104
|
+
continue
|
|
105
|
+
if any(
|
|
106
|
+
fnmatch.fnmatch(dirname, pattern) for pattern in excludes
|
|
107
|
+
) or any(
|
|
108
|
+
fnmatch.fnmatch(os.path.join(dirpath, dirname), pattern)
|
|
109
|
+
for pattern in excludes
|
|
110
|
+
):
|
|
111
|
+
continue
|
|
112
|
+
kept_dirs.append(dirname)
|
|
113
|
+
dirnames[:] = kept_dirs
|
|
114
|
+
for filename in filenames:
|
|
115
|
+
if not filename.endswith(".py"):
|
|
116
|
+
continue
|
|
117
|
+
if any(fnmatch.fnmatch(filename, pattern) for pattern in excludes):
|
|
118
|
+
continue
|
|
119
|
+
yield Path(dirpath) / filename
|
|
120
|
+
|
|
121
|
+
|
|
122
|
+
def _parse_file(path: Path) -> "tuple[SourceFile | None, ParseError | None]":
|
|
123
|
+
try:
|
|
124
|
+
source = path.read_text(encoding="utf-8")
|
|
125
|
+
except OSError as err:
|
|
126
|
+
return None, ParseError(path=path, line=0, message=str(err))
|
|
127
|
+
try:
|
|
128
|
+
tree = ast.parse(source, filename=str(path))
|
|
129
|
+
except SyntaxError as err:
|
|
130
|
+
line = err.lineno if err.lineno is not None else 0
|
|
131
|
+
return None, ParseError(path=path, line=line, message=err.msg)
|
|
132
|
+
token_list: list[tokenize.TokenInfo] = []
|
|
133
|
+
try:
|
|
134
|
+
with tokenize.open(str(path)) as readline:
|
|
135
|
+
token_list = list(tokenize.generate_tokens(readline.readline))
|
|
136
|
+
except (tokenize.TokenError, IndentationError, SyntaxError):
|
|
137
|
+
token_list = []
|
|
138
|
+
lines = tuple(source.splitlines())
|
|
139
|
+
return (
|
|
140
|
+
SourceFile(
|
|
141
|
+
path=path,
|
|
142
|
+
source=source,
|
|
143
|
+
lines=lines,
|
|
144
|
+
tree=tree,
|
|
145
|
+
tokens=tuple(token_list),
|
|
146
|
+
),
|
|
147
|
+
None,
|
|
148
|
+
)
|
|
149
|
+
|
|
150
|
+
|
|
151
|
+
def _resolve_call_name(node: ast.Call) -> str | None:
|
|
152
|
+
func = node.func
|
|
153
|
+
if isinstance(func, ast.Name):
|
|
154
|
+
return func.id
|
|
155
|
+
if isinstance(func, ast.Attribute):
|
|
156
|
+
return func.attr
|
|
157
|
+
return None
|
|
158
|
+
|
|
159
|
+
|
|
160
|
+
def _call_positional_count(node: ast.Call) -> int:
|
|
161
|
+
"""Count positional args (excluding ``*args`` spreads)."""
|
|
162
|
+
return sum(1 for arg in node.args if not isinstance(arg, ast.Starred))
|
|
163
|
+
|
|
164
|
+
|
|
165
|
+
def _call_keywords(node: ast.Call) -> frozenset[str]:
|
|
166
|
+
"""Return the set of keyword arg names (excluding ``**kwargs`` spreads)."""
|
|
167
|
+
return frozenset(kw.arg for kw in node.keywords if kw.arg is not None)
|
|
168
|
+
|
|
169
|
+
|
|
170
|
+
def _collect_params(
|
|
171
|
+
func_node: "ast.FunctionDef | ast.AsyncFunctionDef",
|
|
172
|
+
path: Path,
|
|
173
|
+
param_locations: dict[str, list[tuple[Location, str]]],
|
|
174
|
+
) -> None:
|
|
175
|
+
args = func_node.args
|
|
176
|
+
skip_names = {"self", "cls"}
|
|
177
|
+
names: list[str] = []
|
|
178
|
+
for arg in args.args:
|
|
179
|
+
if arg.arg not in skip_names:
|
|
180
|
+
names.append(arg.arg)
|
|
181
|
+
for arg in args.kwonlyargs:
|
|
182
|
+
names.append(arg.arg)
|
|
183
|
+
for name in names:
|
|
184
|
+
param_locations.setdefault(name, []).append(
|
|
185
|
+
(Location(path=path, line=func_node.lineno), func_node.name)
|
|
186
|
+
)
|
|
187
|
+
|
|
188
|
+
|
|
189
|
+
def _build_corpus(source_files: list[SourceFile], max_call_sites: int) -> Corpus:
|
|
190
|
+
call_sites: dict[str, list[CallSite]] = {}
|
|
191
|
+
param_locations: dict[str, list[tuple[Location, str]]] = {}
|
|
192
|
+
for source_file in source_files:
|
|
193
|
+
for node in ast.walk(source_file.tree):
|
|
194
|
+
if isinstance(node, ast.Call):
|
|
195
|
+
resolved = _resolve_call_name(node)
|
|
196
|
+
if resolved is None:
|
|
197
|
+
continue
|
|
198
|
+
call_sites.setdefault(resolved, []).append(
|
|
199
|
+
CallSite(
|
|
200
|
+
location=Location(path=source_file.path, line=node.lineno),
|
|
201
|
+
positional_count=_call_positional_count(node),
|
|
202
|
+
keywords=_call_keywords(node),
|
|
203
|
+
)
|
|
204
|
+
)
|
|
205
|
+
continue
|
|
206
|
+
if isinstance(node, (ast.FunctionDef, ast.AsyncFunctionDef)):
|
|
207
|
+
_collect_params(node, source_file.path, param_locations)
|
|
208
|
+
return Corpus(
|
|
209
|
+
call_sites={k: tuple(v) for k, v in call_sites.items()},
|
|
210
|
+
param_locations={k: tuple(v) for k, v in param_locations.items()},
|
|
211
|
+
max_call_sites=max_call_sites,
|
|
212
|
+
)
|
|
213
|
+
|
|
214
|
+
|
|
215
|
+
def scan(
|
|
216
|
+
root_paths: tuple[str, ...],
|
|
217
|
+
check_ids: "tuple[str, ...] | None",
|
|
218
|
+
excludes: tuple[str, ...],
|
|
219
|
+
max_call_sites: int,
|
|
220
|
+
) -> ScanResult:
|
|
221
|
+
"""Run the full discover -> parse -> corpus -> checks pipeline."""
|
|
222
|
+
if check_ids is None:
|
|
223
|
+
enabled_checks: tuple[CheckSpec, ...] = tuple(ALL_CHECKS)
|
|
224
|
+
else:
|
|
225
|
+
wanted = set(check_ids)
|
|
226
|
+
enabled_checks = tuple(
|
|
227
|
+
check_spec for check_spec in ALL_CHECKS if check_spec.check_id in wanted
|
|
228
|
+
)
|
|
229
|
+
missing = wanted - {check_spec.check_id for check_spec in enabled_checks}
|
|
230
|
+
if missing:
|
|
231
|
+
raise ValueError(f"unknown check ids: {sorted(missing)}")
|
|
232
|
+
source_files: list[SourceFile] = []
|
|
233
|
+
parse_errors: list[ParseError] = []
|
|
234
|
+
for path in _discover(root_paths, excludes):
|
|
235
|
+
parsed, parse_err = _parse_file(path)
|
|
236
|
+
if parse_err is not None:
|
|
237
|
+
parse_errors.append(parse_err)
|
|
238
|
+
continue
|
|
239
|
+
assert parsed is not None
|
|
240
|
+
source_files.append(parsed)
|
|
241
|
+
corpus = _build_corpus(source_files, max_call_sites)
|
|
242
|
+
findings: list[Finding] = []
|
|
243
|
+
for source_file in source_files:
|
|
244
|
+
for check_spec in enabled_checks:
|
|
245
|
+
findings.extend(check_spec.run(source_file, corpus))
|
|
246
|
+
findings.sort(key=lambda f: (f.check_id, str(f.path), f.line))
|
|
247
|
+
return ScanResult(
|
|
248
|
+
findings=tuple(findings),
|
|
249
|
+
parse_errors=tuple(parse_errors),
|
|
250
|
+
files_scanned=len(source_files),
|
|
251
|
+
)
|
|
@@ -0,0 +1,19 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: python-surveyor
|
|
3
|
+
Version: 0.1.0
|
|
4
|
+
Summary: Use python introspection to survey source code for final LLM judgement
|
|
5
|
+
Author: Dave Cunningham
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/sparkprime/python-surveyor
|
|
8
|
+
Requires-Python: >=3.12
|
|
9
|
+
License-File: LICENSE
|
|
10
|
+
Requires-Dist: click>=8.1.0
|
|
11
|
+
Provides-Extra: dev
|
|
12
|
+
Requires-Dist: pyright==1.1.411; extra == "dev"
|
|
13
|
+
Requires-Dist: pytest>=8.0.0; extra == "dev"
|
|
14
|
+
Requires-Dist: pytest-cov; extra == "dev"
|
|
15
|
+
Requires-Dist: coverage; extra == "dev"
|
|
16
|
+
Requires-Dist: black==26.5.1; extra == "dev"
|
|
17
|
+
Requires-Dist: isort==8.0.1; extra == "dev"
|
|
18
|
+
Requires-Dist: pylint==4.0.6; extra == "dev"
|
|
19
|
+
Dynamic: license-file
|
|
@@ -0,0 +1,12 @@
|
|
|
1
|
+
python_surveyor/__init__.py,sha256=os_lOpeA_qqAQrIUw-ztrBY0jmrOjSdMF7EuGV2rQ2M,287
|
|
2
|
+
python_surveyor/__main__.py,sha256=qM_59puxdldIVyuFxKcmKt2K9Me4y3opAh1LIEtoOP0,124
|
|
3
|
+
python_surveyor/cli.py,sha256=A5GveO4aWQoQNPx0JhpuEehILrHDk7ZAJYzD43SxOZE,2504
|
|
4
|
+
python_surveyor/model.py,sha256=AcBbwncH1EoHocDzk-umUakxxFx9RwBtX81wSD0BE2I,1548
|
|
5
|
+
python_surveyor/report.py,sha256=AbQuKBSch7kiO7J_MMYGIzunstGqmkCs4RCr9-wDjpo,4426
|
|
6
|
+
python_surveyor/scanner.py,sha256=3fK5wPsOsoFGkcSgmgMlBwBEeSqSOcjSNlEfwbfVi-Y,8856
|
|
7
|
+
python_surveyor-0.1.0.dist-info/licenses/LICENSE,sha256=e0bQqTIJZymR3Os44UpqdNU4g3pl9GLMfZ3YKr-XY3Q,1072
|
|
8
|
+
python_surveyor-0.1.0.dist-info/METADATA,sha256=kiUUo8yj-EmfCs_K-MYq4VmuUhR2xkFOWCSOWcYbuB4,681
|
|
9
|
+
python_surveyor-0.1.0.dist-info/WHEEL,sha256=YVMoNqKzERt-wjUZwJ33xBGAwnFl-4cqbYkTtWa4itE,91
|
|
10
|
+
python_surveyor-0.1.0.dist-info/entry_points.txt,sha256=uDhKSL-x9NOPj8uQOKtqey799vSFh-4xi4ArqHXu8eA,61
|
|
11
|
+
python_surveyor-0.1.0.dist-info/top_level.txt,sha256=8BSg7FuIKg3pHoyhRmrAB52wRr__RuuUj0WJ8CjyfBE,16
|
|
12
|
+
python_surveyor-0.1.0.dist-info/RECORD,,
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Dave Cunningham
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
python_surveyor
|