python-surveyor 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Dave Cunningham
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,19 @@
1
+ Metadata-Version: 2.4
2
+ Name: python-surveyor
3
+ Version: 0.1.0
4
+ Summary: Use python introspection to survey source code for final LLM judgement
5
+ Author: Dave Cunningham
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/sparkprime/python-surveyor
8
+ Requires-Python: >=3.12
9
+ License-File: LICENSE
10
+ Requires-Dist: click>=8.1.0
11
+ Provides-Extra: dev
12
+ Requires-Dist: pyright==1.1.411; extra == "dev"
13
+ Requires-Dist: pytest>=8.0.0; extra == "dev"
14
+ Requires-Dist: pytest-cov; extra == "dev"
15
+ Requires-Dist: coverage; extra == "dev"
16
+ Requires-Dist: black==26.5.1; extra == "dev"
17
+ Requires-Dist: isort==8.0.1; extra == "dev"
18
+ Requires-Dist: pylint==4.0.6; extra == "dev"
19
+ Dynamic: license-file
@@ -0,0 +1,18 @@
1
+ # python-surveyor
2
+
3
+ A scanner for code that finds potential issues.
4
+ It is intentionally **recall-only**: every hit is a *candidate* with just enough
5
+ context attached for a human (or an LLM agent) to make the judgement call. The
6
+ tool never decides whether a hit is a real problem and never fixes anything.
7
+
8
+ ## Install
9
+
10
+ ```bash
11
+ uv pip install -e ".[dev]"
12
+ ```
13
+
14
+ ## Usage
15
+
16
+ ```bash
17
+ python-surveyor --help
18
+ ```
@@ -0,0 +1,84 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68.0"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [tool.setuptools.packages.find]
6
+ where = ["."]
7
+ include = ["python_surveyor"]
8
+
9
+ [project]
10
+ name = "python-surveyor"
11
+ version = "0.1.0"
12
+ description = "Use python introspection to survey source code for final LLM judgement"
13
+ requires-python = ">=3.12"
14
+ license = "MIT"
15
+ license-files = ["LICENSE"]
16
+ authors = [
17
+ { name = "Dave Cunningham" },
18
+ ]
19
+ urls = { Homepage = "https://github.com/sparkprime/python-surveyor" }
20
+
21
+ dependencies = [
22
+ "click>=8.1.0",
23
+ ]
24
+
25
+ [project.optional-dependencies]
26
+ dev = [
27
+ "pyright==1.1.411",
28
+ "pytest>=8.0.0",
29
+ "pytest-cov",
30
+ "coverage",
31
+ "black == 26.5.1",
32
+ "isort == 8.0.1",
33
+ "pylint == 4.0.6",
34
+ ]
35
+
36
+ [project.scripts]
37
+ python-surveyor = "python_surveyor.cli:main"
38
+
39
+ [tool.pytest.ini_options]
40
+ testpaths = ["tests"]
41
+ filterwarnings = [
42
+ "error",
43
+ ]
44
+ xfail_strict = true
45
+ log_level = "INFO"
46
+
47
+ [tool.black]
48
+ line-length = 88
49
+ target-version = ["py312"]
50
+
51
+ [tool.isort]
52
+ profile = "black"
53
+ line_length = 88
54
+ known_first_party = ["python_surveyor"]
55
+
56
+ [tool.pyright]
57
+ include = ["python_surveyor", "tests"]
58
+ venvPath = "."
59
+ venv = ".venv"
60
+ reportMissingTypeStubs = "none"
61
+ reportUnknownMemberType = "none"
62
+ reportUnknownArgumentType = "none"
63
+ reportUnknownVariableType = "none"
64
+ reportImportCycles = "none"
65
+
66
+ [tool.pylint.'MASTER']
67
+ ignore-paths = [
68
+ "^tests/",
69
+ ]
70
+
71
+ [tool.pylint.'MESSAGES CONTROL']
72
+ disable = [
73
+ "too-few-public-methods",
74
+ "too-many-arguments",
75
+ "too-many-positional-arguments",
76
+ "too-many-locals",
77
+ "too-many-branches",
78
+ "too-many-statements",
79
+ "too-many-return-statements",
80
+ "too-many-instance-attributes",
81
+ "too-many-nested-blocks",
82
+ "duplicate-code",
83
+ "fixme",
84
+ ]
@@ -0,0 +1,17 @@
1
+ """python-surveyor: AST-based scanner for AI-generated-code smells."""
2
+
3
+ from python_surveyor.model import (
4
+ CallSite,
5
+ Finding,
6
+ Location,
7
+ ParseError,
8
+ SourceExcerpt,
9
+ )
10
+
11
+ __all__ = [
12
+ "CallSite",
13
+ "Finding",
14
+ "Location",
15
+ "ParseError",
16
+ "SourceExcerpt",
17
+ ]
@@ -0,0 +1,6 @@
1
+ """`python -m python_surveyor` entry point."""
2
+
3
+ from python_surveyor.cli import main
4
+
5
+ if __name__ == "__main__":
6
+ main()
@@ -0,0 +1,102 @@
1
+ """``click`` CLI: ``scan`` and ``list-checks`` commands."""
2
+
3
+ import sys
4
+ from pathlib import Path
5
+ from typing import TextIO
6
+
7
+ import click
8
+
9
+ from python_surveyor.checks import ALL_CHECKS
10
+ from python_surveyor.report import render_json, render_text
11
+ from python_surveyor.scanner import scan
12
+
13
+
14
+ @click.group()
15
+ def main() -> None:
16
+ """python-surveyor: AST-based scanner for AI-generated-code smells."""
17
+
18
+
19
+ @main.command("scan")
20
+ @click.argument("paths", nargs=-1, type=click.Path(exists=True))
21
+ @click.option(
22
+ "--check",
23
+ "check_ids",
24
+ multiple=True,
25
+ help="Restrict to this check id (repeatable).",
26
+ )
27
+ @click.option(
28
+ "--exclude",
29
+ "excludes",
30
+ multiple=True,
31
+ help="fnmatch pattern to exclude (repeatable, additive to defaults).",
32
+ )
33
+ @click.option(
34
+ "--format",
35
+ "fmt",
36
+ type=click.Choice(["text", "json"]),
37
+ default="text",
38
+ help="Output format. Default: text.",
39
+ )
40
+ @click.option(
41
+ "--output",
42
+ "output",
43
+ type=click.Path(path_type=Path),
44
+ default=None,
45
+ help="Write to PATH instead of stdout.",
46
+ )
47
+ @click.option(
48
+ "--max-excerpt-lines",
49
+ "max_excerpt_lines",
50
+ type=int,
51
+ default=20,
52
+ help="Cap excerpt line count (default 20).",
53
+ )
54
+ @click.option(
55
+ "--max-call-sites",
56
+ "max_call_sites",
57
+ type=int,
58
+ default=5,
59
+ help="Cap call-site samples in notes (default 5).",
60
+ )
61
+ def scan_cmd(
62
+ paths: tuple[str, ...],
63
+ check_ids: tuple[str, ...],
64
+ excludes: tuple[str, ...],
65
+ fmt: str,
66
+ output: "Path | None",
67
+ max_excerpt_lines: int,
68
+ max_call_sites: int,
69
+ ) -> None:
70
+ """Scan PATHS for AI-generated-code smells."""
71
+ root_paths: tuple[str, ...] = paths if paths else (".",)
72
+ result = scan(
73
+ root_paths=root_paths,
74
+ check_ids=check_ids if check_ids else None,
75
+ excludes=excludes,
76
+ max_call_sites=max_call_sites,
77
+ )
78
+ stream: TextIO
79
+ if output is not None:
80
+ stream = output.open("w", encoding="utf-8")
81
+ else:
82
+ stream = sys.stdout
83
+ try:
84
+ if fmt == "json":
85
+ render_json(result, max_excerpt_lines, stream)
86
+ else:
87
+ render_text(result, max_excerpt_lines, stream)
88
+ finally:
89
+ if output is not None:
90
+ stream.close()
91
+
92
+
93
+ @main.command("list-checks")
94
+ def list_checks_cmd() -> None:
95
+ """List available check ids and descriptions."""
96
+ click.echo("id\tdescription")
97
+ for check_spec in ALL_CHECKS:
98
+ click.echo(f"{check_spec.check_id}\t{check_spec.description}")
99
+
100
+
101
+ if __name__ == "__main__":
102
+ main()
@@ -0,0 +1,65 @@
1
+ """Frozen dataclasses shared by scanner, checks, and report renderers.
2
+
3
+ All construction uses required positional fields; there are no
4
+ default-valued fields on ``Finding``/``SourceExcerpt``, consistent with the
5
+ "no optional params on internal APIs" rule this tool itself checks for.
6
+ """
7
+
8
+ from dataclasses import dataclass
9
+ from pathlib import Path
10
+
11
+
12
+ @dataclass(frozen=True)
13
+ class Location:
14
+ """A single source position."""
15
+
16
+ path: Path
17
+ line: int
18
+
19
+
20
+ @dataclass(frozen=True)
21
+ class CallSite:
22
+ """A call to a name, with enough info to tell if defaulted params are used.
23
+
24
+ ``positional_count`` is the number of positional args (not counting
25
+ ``*args`` spreads). ``keywords`` is the set of keyword arg names (not
26
+ counting ``**kwargs`` spreads). Together these let the
27
+ ``optional-param-default`` check determine whether a call site uses the
28
+ default value or passes an explicit one.
29
+ """
30
+
31
+ location: Location
32
+ positional_count: int
33
+ keywords: frozenset[str]
34
+
35
+
36
+ @dataclass(frozen=True)
37
+ class SourceExcerpt:
38
+ """A literal slice of source worth showing verbatim."""
39
+
40
+ label: str
41
+ path: Path
42
+ start_line: int
43
+ end_line: int
44
+
45
+
46
+ @dataclass(frozen=True)
47
+ class Finding:
48
+ """A candidate smell detected by a check."""
49
+
50
+ check_id: str
51
+ path: Path
52
+ line: int
53
+ column: int
54
+ message: str
55
+ excerpts: tuple[SourceExcerpt, ...]
56
+ notes: tuple[str, ...]
57
+
58
+
59
+ @dataclass(frozen=True)
60
+ class ParseError:
61
+ """A file that could not be ``ast.parse``-d."""
62
+
63
+ path: Path
64
+ line: int
65
+ message: str
@@ -0,0 +1,137 @@
1
+ """Render a ``ScanResult`` as text or JSON.
2
+
3
+ Text output groups findings by ``check_id``; each finding is shown as
4
+ ``path:line:col -- message`` followed by labeled excerpts (with the line
5
+ range in the header) and note lines. Parse errors get their own leading
6
+ section. Excerpts are capped to ``--max-excerpt-lines`` with a truncation
7
+ note.
8
+
9
+ JSON output is a flat list of the same fields for programmatic use (excerpt
10
+ text is not included — only the line range, so consumers can lazy-load).
11
+ """
12
+
13
+ import json
14
+ from pathlib import Path
15
+ from typing import TextIO
16
+
17
+ from python_surveyor.checks import CHECKS_BY_ID
18
+ from python_surveyor.scanner import ScanResult
19
+
20
+
21
+ def _read_excerpt_lines(
22
+ path: Path, start: int, end: int, max_lines: int
23
+ ) -> "tuple[list[str], str | None]":
24
+ """Read ``[start, end]`` from ``path``, capping to ``max_lines``.
25
+
26
+ Returns ``(lines, truncation_note)``.
27
+ """
28
+ try:
29
+ all_lines = path.read_text(encoding="utf-8").splitlines()
30
+ except OSError:
31
+ return ([], None)
32
+ available_end = min(end, len(all_lines))
33
+ if available_end < start:
34
+ return ([], None)
35
+ span = available_end - start + 1
36
+ if span <= max_lines:
37
+ return (all_lines[start - 1 : available_end], None)
38
+ capped_end = start + max_lines - 1
39
+ return (
40
+ all_lines[start - 1 : capped_end],
41
+ f"excerpt truncated at {max_lines} lines",
42
+ )
43
+
44
+
45
+ def render_text(
46
+ result: ScanResult,
47
+ max_excerpt_lines: int,
48
+ stream: TextIO,
49
+ ) -> None:
50
+ """Render ``result`` as human-readable text to ``stream``."""
51
+ stream.write(f"# scan: {result.files_scanned} file(s) scanned\n\n")
52
+ if result.parse_errors:
53
+ stream.write("## parse errors\n\n")
54
+ for err in result.parse_errors:
55
+ stream.write(f"{err.path}:{err.line} -- {err.message}\n")
56
+ stream.write("\n")
57
+ current_check: str | None = None
58
+ counter = 0
59
+ for finding in result.findings:
60
+ if finding.check_id != current_check:
61
+ current_check = finding.check_id
62
+ counter = 0
63
+ stream.write(f"## {current_check}\n\n")
64
+ spec = CHECKS_BY_ID.get(current_check)
65
+ if spec is not None:
66
+ stream.write(f"{spec.explanation}\n\n")
67
+ counter += 1
68
+ for excerpt in finding.excerpts:
69
+ lines, truncation = _read_excerpt_lines(
70
+ excerpt.path,
71
+ excerpt.start_line,
72
+ excerpt.end_line,
73
+ max_excerpt_lines,
74
+ )
75
+ stream.write(
76
+ f"({counter}) {excerpt.path}:"
77
+ f"{excerpt.start_line}-{excerpt.end_line}\n"
78
+ )
79
+ for line in lines:
80
+ stream.write(f"{line}\n")
81
+ if truncation:
82
+ stream.write(f"... {truncation}\n")
83
+ for note in finding.notes:
84
+ stream.write(f" note: {note}\n")
85
+ stream.write("\n")
86
+ if not result.findings:
87
+ stream.write("no findings.\n")
88
+
89
+
90
+ def render_json(
91
+ result: ScanResult,
92
+ max_excerpt_lines: int,
93
+ stream: TextIO,
94
+ ) -> None:
95
+ """Render ``result`` as JSON to ``stream``.
96
+
97
+ ``max_excerpt_lines`` is accepted for API symmetry with
98
+ :func:`render_text`; JSON emits ranges only, not source text.
99
+ """
100
+ del max_excerpt_lines
101
+ findings_payload = []
102
+ for finding in result.findings:
103
+ excerpts_payload = [
104
+ {
105
+ "label": excerpt.label,
106
+ "path": str(excerpt.path),
107
+ "start_line": excerpt.start_line,
108
+ "end_line": excerpt.end_line,
109
+ }
110
+ for excerpt in finding.excerpts
111
+ ]
112
+ findings_payload.append(
113
+ {
114
+ "check_id": finding.check_id,
115
+ "path": str(finding.path),
116
+ "line": finding.line,
117
+ "column": finding.column,
118
+ "message": finding.message,
119
+ "excerpts": excerpts_payload,
120
+ "notes": list(finding.notes),
121
+ }
122
+ )
123
+ parse_errors_payload = [
124
+ {
125
+ "path": str(err.path),
126
+ "line": err.line,
127
+ "message": err.message,
128
+ }
129
+ for err in result.parse_errors
130
+ ]
131
+ payload = {
132
+ "files_scanned": result.files_scanned,
133
+ "findings": findings_payload,
134
+ "parse_errors": parse_errors_payload,
135
+ }
136
+ json.dump(payload, stream, indent=2)
137
+ stream.write("\n")