slopcount 0.1.0__py3-none-any.whl
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- slopcount/__init__.py +1 -0
- slopcount/app.py +206 -0
- slopcount/cli.py +114 -0
- slopcount/detectors/__init__.py +4 -0
- slopcount/detectors/code_style.py +145 -0
- slopcount/detectors/docs_bloat.py +70 -0
- slopcount/detectors/env_markers.py +70 -0
- slopcount/detectors/git_history.py +133 -0
- slopcount/detectors/perplexity.py +139 -0
- slopcount/detectors/phrase.py +34 -0
- slopcount/download_model.py +19 -0
- slopcount/evidence.py +171 -0
- slopcount/extractors.py +126 -0
- slopcount/i18n.py +61 -0
- slopcount/locale/ru/LC_MESSAGES/slopcount.mo +0 -0
- slopcount/locale/ru/LC_MESSAGES/slopcount.po +518 -0
- slopcount/metrics/__init__.py +0 -0
- slopcount/metrics/costs.py +138 -0
- slopcount/metrics/slocomo.py +107 -0
- slopcount/render/__init__.py +0 -0
- slopcount/render/csv_out.py +27 -0
- slopcount/render/json_out.py +166 -0
- slopcount/render/text.py +311 -0
- slopcount/rules/env_markers.toml +44 -0
- slopcount/rules/languages.toml +414 -0
- slopcount/rules/phrases_en.toml +54 -0
- slopcount/rules/phrases_ru.toml +34 -0
- slopcount/rules.py +110 -0
- slopcount/scales.py +86 -0
- slopcount/scc.py +351 -0
- slopcount-0.1.0.dist-info/METADATA +304 -0
- slopcount-0.1.0.dist-info/RECORD +35 -0
- slopcount-0.1.0.dist-info/WHEEL +4 -0
- slopcount-0.1.0.dist-info/entry_points.txt +2 -0
- slopcount-0.1.0.dist-info/licenses/LICENSE +21 -0
slopcount/__init__.py
ADDED
|
@@ -0,0 +1 @@
|
|
|
1
|
+
__version__ = "0.1.0"
|
slopcount/app.py
ADDED
|
@@ -0,0 +1,206 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
from collections import Counter
|
|
5
|
+
from dataclasses import dataclass, field
|
|
6
|
+
from pathlib import Path
|
|
7
|
+
|
|
8
|
+
from slopcount import scc
|
|
9
|
+
from slopcount.detectors.code_style import CodeStyleDetector
|
|
10
|
+
from slopcount.detectors.docs_bloat import DocsBloatDetector, repo_bloat_evidence
|
|
11
|
+
from slopcount.detectors.env_markers import EnvMarkerDetector
|
|
12
|
+
from slopcount.detectors.phrase import PhraseDetector
|
|
13
|
+
from slopcount.evidence import (
|
|
14
|
+
Evidence,
|
|
15
|
+
LanguageRow,
|
|
16
|
+
Report,
|
|
17
|
+
ScannedFile,
|
|
18
|
+
VolumeStats,
|
|
19
|
+
aggregate_slop,
|
|
20
|
+
read_text,
|
|
21
|
+
safe_ratio,
|
|
22
|
+
)
|
|
23
|
+
from slopcount.i18n import _
|
|
24
|
+
from slopcount.metrics.costs import attribute_locomo, split_cocomo
|
|
25
|
+
from slopcount.metrics.slocomo import compute as slocomo_compute
|
|
26
|
+
from slopcount.rules import load_languages, load_rules
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
@dataclass
|
|
30
|
+
class Options:
|
|
31
|
+
paths: list[str] = field(default_factory=lambda: ["."])
|
|
32
|
+
evidence: bool = False
|
|
33
|
+
json_out: bool = False
|
|
34
|
+
csv_out: bool = False
|
|
35
|
+
history: int | None = None
|
|
36
|
+
perplexity: bool = False
|
|
37
|
+
rules: list[Path] = field(default_factory=list)
|
|
38
|
+
lang: str | None = None
|
|
39
|
+
personcost: float = 4690.50
|
|
40
|
+
overhead: float = 2.4
|
|
41
|
+
coffee_price: float = 4.0
|
|
42
|
+
no_therapy: bool = False
|
|
43
|
+
wide: bool = False
|
|
44
|
+
scc_path: str | None = None
|
|
45
|
+
|
|
46
|
+
|
|
47
|
+
def run(opts: Options) -> Report:
|
|
48
|
+
root = Path(opts.paths[0])
|
|
49
|
+
if len(opts.paths) > 1:
|
|
50
|
+
print(
|
|
51
|
+
_("slopcount: multiple paths given, scanning only the first: %s") % opts.paths[0],
|
|
52
|
+
file=sys.stderr,
|
|
53
|
+
)
|
|
54
|
+
langmap = load_languages(opts.rules)
|
|
55
|
+
manifest = scc.collect(
|
|
56
|
+
root, personcost=opts.personcost, overhead=opts.overhead, scc_path=opts.scc_path
|
|
57
|
+
)
|
|
58
|
+
# kind/language из каталога языков; неизвестный scc-язык → ("data", None)
|
|
59
|
+
files: list[ScannedFile] = []
|
|
60
|
+
for f in manifest.files:
|
|
61
|
+
kind_lang = langmap.get(f.language_name, ("data", None))
|
|
62
|
+
files.append(ScannedFile(f.path, kind_lang[1], kind_lang[0], f.size))
|
|
63
|
+
|
|
64
|
+
# ── Фаза volume: объём проекта из манифеста (спека §4) ──────────────
|
|
65
|
+
# SLOC/комментарии/когнитива — по манифесту целиком: нечитаемые (skip)
|
|
66
|
+
# файлы входят в объём, как и в COCOMO/LOCOMO scc (спека §4);
|
|
67
|
+
# детекторы такие файлы по-прежнему не видят.
|
|
68
|
+
volume = VolumeStats(files_total=len(manifest.files))
|
|
69
|
+
lang_files: Counter[str] = Counter()
|
|
70
|
+
lang_sloc: Counter[str] = Counter()
|
|
71
|
+
bucket_lines: Counter[str] = Counter() # kind → Σ Code (корзины стоимостей)
|
|
72
|
+
for f, sf in zip(manifest.files, files, strict=True):
|
|
73
|
+
bucket_lines[sf.kind] += f.code
|
|
74
|
+
if sf.kind != "code":
|
|
75
|
+
continue
|
|
76
|
+
volume.sloc += f.code
|
|
77
|
+
volume.comment_lines += f.comment
|
|
78
|
+
volume.complexity += f.complexity
|
|
79
|
+
volume.cognitive_total += f.cognitive
|
|
80
|
+
lang_files[f.language_name] += 1
|
|
81
|
+
lang_sloc[f.language_name] += f.code
|
|
82
|
+
volume.languages = [
|
|
83
|
+
LanguageRow(lang, lang_files[lang], lang_sloc[lang])
|
|
84
|
+
for lang in sorted(lang_sloc, key=lambda lg: (-lang_sloc[lg], lg))
|
|
85
|
+
]
|
|
86
|
+
docs_bucket_lines = bucket_lines["markdown"] + bucket_lines["prose"]
|
|
87
|
+
|
|
88
|
+
# ── Фаза detect: один проход по читаемым файлам ─────────────────────
|
|
89
|
+
phrase = PhraseDetector(load_rules(opts.rules))
|
|
90
|
+
docs_bloat = DocsBloatDetector()
|
|
91
|
+
style_detector = CodeStyleDetector()
|
|
92
|
+
pplx = None
|
|
93
|
+
if opts.perplexity:
|
|
94
|
+
from slopcount.detectors.perplexity import PerplexityDetector, available
|
|
95
|
+
|
|
96
|
+
if not available():
|
|
97
|
+
raise RuntimeError(
|
|
98
|
+
_(
|
|
99
|
+
"slopcount: --perplexity requires extras; "
|
|
100
|
+
"pipx install 'slopcount[perplexity]' and "
|
|
101
|
+
"python -m slopcount.download_model"
|
|
102
|
+
)
|
|
103
|
+
)
|
|
104
|
+
pplx = PerplexityDetector()
|
|
105
|
+
by_path = {f.path: f for f in manifest.files}
|
|
106
|
+
evidences: list[Evidence] = []
|
|
107
|
+
infected: list[tuple[str, int]] = []
|
|
108
|
+
md_files = 0
|
|
109
|
+
md_lines = 0
|
|
110
|
+
md_words = 0
|
|
111
|
+
skip = 0
|
|
112
|
+
n = 0
|
|
113
|
+
progressed = False
|
|
114
|
+
is_stderr_tty = sys.stderr.isatty()
|
|
115
|
+
total = sum(1 for f in files if f.kind in ("code", "markdown", "prose"))
|
|
116
|
+
for sf in files:
|
|
117
|
+
if sf.kind not in ("code", "markdown", "prose"):
|
|
118
|
+
continue
|
|
119
|
+
text = read_text(root / sf.path)
|
|
120
|
+
if text is None:
|
|
121
|
+
skip += 1
|
|
122
|
+
continue
|
|
123
|
+
n += 1
|
|
124
|
+
if is_stderr_tty and n % 200 == 0:
|
|
125
|
+
print(f"\rscanned {n}/{total} files...", end="", file=sys.stderr)
|
|
126
|
+
progressed = True
|
|
127
|
+
if sf.kind == "code":
|
|
128
|
+
evidences.extend(style_detector.detect(sf, text))
|
|
129
|
+
else: # markdown | prose — объём доков (спека §5.1)
|
|
130
|
+
md_words += len(text.split())
|
|
131
|
+
if sf.kind == "markdown":
|
|
132
|
+
if sf.path.lower().endswith((".md", ".markdown")):
|
|
133
|
+
md_files += 1
|
|
134
|
+
md_lines += by_path[sf.path].lines
|
|
135
|
+
bloat = docs_bloat.detect(sf, text) # детекция — только markdown (спека §4)
|
|
136
|
+
evidences.extend(bloat.evidences)
|
|
137
|
+
infected.extend(bloat.infected)
|
|
138
|
+
evidences.extend(phrase.detect(sf, text))
|
|
139
|
+
if pplx is not None and sf.kind in ("markdown", "prose"):
|
|
140
|
+
evidences.extend(pplx.detect(sf, text))
|
|
141
|
+
if progressed:
|
|
142
|
+
print(file=sys.stderr) # завершаем строку прогресса
|
|
143
|
+
rb = repo_bloat_evidence(files, volume.sloc)
|
|
144
|
+
if rb:
|
|
145
|
+
evidences.append(rb)
|
|
146
|
+
evidences.extend(EnvMarkerDetector().detect(root, files, read_text))
|
|
147
|
+
history_commits: int | None = None
|
|
148
|
+
if opts.history:
|
|
149
|
+
from slopcount.detectors.git_history import GitUnavailable
|
|
150
|
+
from slopcount.detectors.git_history import detect as git_detect
|
|
151
|
+
|
|
152
|
+
try:
|
|
153
|
+
hist_evs, commits = git_detect(root, opts.history)
|
|
154
|
+
evidences.extend(hist_evs)
|
|
155
|
+
history_commits = commits
|
|
156
|
+
except GitUnavailable:
|
|
157
|
+
print(_("slopcount: git history unavailable; skipping archaeology"), file=sys.stderr)
|
|
158
|
+
volume.md_files = md_files
|
|
159
|
+
volume.md_lines = md_lines
|
|
160
|
+
volume.md_words = md_words
|
|
161
|
+
volume.md_sloc_ratio = safe_ratio(md_lines, volume.sloc)
|
|
162
|
+
volume.comment_sloc_ratio = safe_ratio(volume.comment_lines, volume.sloc)
|
|
163
|
+
|
|
164
|
+
# ── Фаза aggregate: SlopStats ───────────────────────────────────────
|
|
165
|
+
slop = aggregate_slop(evidences, sloc=volume.sloc, infected=infected)
|
|
166
|
+
|
|
167
|
+
# ── Фаза comprehension: SLOCOMO + разбивки стоимостей ───────────────
|
|
168
|
+
# (slocomo_compute импортирован наверху: цикла больше нет — Options в
|
|
169
|
+
# slocomo.py живёт только в TYPE_CHECKING)
|
|
170
|
+
slocomo = slocomo_compute(
|
|
171
|
+
md_words=md_words,
|
|
172
|
+
comment_lines=volume.comment_lines,
|
|
173
|
+
sloc=volume.sloc,
|
|
174
|
+
cognitive_total=volume.cognitive_total,
|
|
175
|
+
slop_ratio=slop.ratio,
|
|
176
|
+
locomo=manifest.locomo,
|
|
177
|
+
opts=opts,
|
|
178
|
+
)
|
|
179
|
+
report = Report(
|
|
180
|
+
root=str(root),
|
|
181
|
+
skip_count=skip,
|
|
182
|
+
history_commits=history_commits,
|
|
183
|
+
volume=volume,
|
|
184
|
+
slop=slop,
|
|
185
|
+
scc_version=manifest.scc_version,
|
|
186
|
+
cocomo=manifest.cocomo,
|
|
187
|
+
locomo=manifest.locomo,
|
|
188
|
+
slocomo=slocomo,
|
|
189
|
+
)
|
|
190
|
+
if manifest.cocomo is not None:
|
|
191
|
+
report.cocomo_breakdown = split_cocomo(
|
|
192
|
+
manifest.cocomo,
|
|
193
|
+
docs_lines=docs_bucket_lines,
|
|
194
|
+
code_lines=bucket_lines["code"],
|
|
195
|
+
data_lines=bucket_lines["data"],
|
|
196
|
+
personcost=opts.personcost,
|
|
197
|
+
overhead=opts.overhead,
|
|
198
|
+
)
|
|
199
|
+
if manifest.locomo is not None:
|
|
200
|
+
report.locomo_breakdown = attribute_locomo(
|
|
201
|
+
manifest.locomo,
|
|
202
|
+
docs_lines=docs_bucket_lines,
|
|
203
|
+
code_lines=bucket_lines["code"],
|
|
204
|
+
data_lines=bucket_lines["data"],
|
|
205
|
+
)
|
|
206
|
+
return report
|
slopcount/cli.py
ADDED
|
@@ -0,0 +1,114 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import sys
|
|
4
|
+
from pathlib import Path
|
|
5
|
+
from typing import Literal
|
|
6
|
+
|
|
7
|
+
import typer
|
|
8
|
+
|
|
9
|
+
from slopcount import __version__, i18n
|
|
10
|
+
from slopcount.app import Options, run
|
|
11
|
+
from slopcount.i18n import _
|
|
12
|
+
from slopcount.render.csv_out import render_csv
|
|
13
|
+
from slopcount.render.json_out import render_json
|
|
14
|
+
from slopcount.render.text import (
|
|
15
|
+
render_comprehension,
|
|
16
|
+
render_evidence,
|
|
17
|
+
render_slop,
|
|
18
|
+
render_volume,
|
|
19
|
+
)
|
|
20
|
+
|
|
21
|
+
app = typer.Typer()
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def _version(value: bool) -> None:
|
|
25
|
+
if value:
|
|
26
|
+
typer.echo(f"slopcount {__version__}")
|
|
27
|
+
raise typer.Exit()
|
|
28
|
+
|
|
29
|
+
|
|
30
|
+
@app.command(context_settings={"help_option_names": ["-h", "--help"]})
|
|
31
|
+
def scan(
|
|
32
|
+
paths: list[str] | None = typer.Argument( # noqa: B008 — идиоматика typer
|
|
33
|
+
None, help="Directory to scan (default: .)"
|
|
34
|
+
),
|
|
35
|
+
evidence: bool = typer.Option(
|
|
36
|
+
False, "--evidence", help="List every finding with its source line"
|
|
37
|
+
),
|
|
38
|
+
json_out: bool = typer.Option(False, "--json", help="Print JSON report"),
|
|
39
|
+
csv_out: bool = typer.Option(False, "--csv", help="Print CSV report"),
|
|
40
|
+
history: int | None = typer.Option(None, "--history", help="Git history depth to scan"),
|
|
41
|
+
perplexity: bool = typer.Option(False, "--perplexity", help="Perplexity detector (extras)"),
|
|
42
|
+
rules: list[Path] | None = typer.Option( # noqa: B008 — идиоматика typer
|
|
43
|
+
None, "--rules", help="Extra rules TOML (repeatable)"
|
|
44
|
+
),
|
|
45
|
+
scc_path: str | None = typer.Option(
|
|
46
|
+
None, "--scc-path", help="Path to the scc binary (else PATH/env/download)"
|
|
47
|
+
),
|
|
48
|
+
lang: Literal["en", "ru"] | None = typer.Option(None, "--lang", help="Interface language"),
|
|
49
|
+
personcost: float = typer.Option(4690.50, "--personcost", help="Monthly person cost, USD"),
|
|
50
|
+
overhead: float = typer.Option(
|
|
51
|
+
2.4, "--overhead", help="Cost overhead multiplier (COCOMO and SLOCOMO)"
|
|
52
|
+
),
|
|
53
|
+
coffee_price: float = typer.Option(4.0, "--coffee-price", help="Coffee cup price, USD"),
|
|
54
|
+
no_therapy: bool = typer.Option(False, "--no-therapy", help="Skip the therapy estimate"),
|
|
55
|
+
wide: bool = typer.Option(False, "--wide", help="Wide table layout"),
|
|
56
|
+
version: bool | None = typer.Option(
|
|
57
|
+
None, "--version", callback=_version, is_eager=True, help="Show version and exit"
|
|
58
|
+
),
|
|
59
|
+
) -> None:
|
|
60
|
+
"""Count the AI slop in a project and the cost of comprehending it."""
|
|
61
|
+
i18n.setup(lang)
|
|
62
|
+
opts = Options(
|
|
63
|
+
paths=paths or ["."],
|
|
64
|
+
evidence=evidence,
|
|
65
|
+
json_out=json_out,
|
|
66
|
+
csv_out=csv_out,
|
|
67
|
+
history=history,
|
|
68
|
+
perplexity=perplexity,
|
|
69
|
+
rules=rules or [],
|
|
70
|
+
scc_path=scc_path,
|
|
71
|
+
lang=lang,
|
|
72
|
+
personcost=personcost,
|
|
73
|
+
overhead=overhead,
|
|
74
|
+
coffee_price=coffee_price,
|
|
75
|
+
no_therapy=no_therapy,
|
|
76
|
+
wide=wide,
|
|
77
|
+
)
|
|
78
|
+
root = Path(opts.paths[0])
|
|
79
|
+
if not root.exists():
|
|
80
|
+
print(_("slopcount: path not found: {path}").format(path=root), file=sys.stderr)
|
|
81
|
+
raise typer.Exit(code=2)
|
|
82
|
+
if root.is_file():
|
|
83
|
+
print(
|
|
84
|
+
_("slopcount: path is a file, directory expected: {path}").format(path=root),
|
|
85
|
+
file=sys.stderr,
|
|
86
|
+
)
|
|
87
|
+
raise typer.Exit(code=2)
|
|
88
|
+
try:
|
|
89
|
+
report = run(opts)
|
|
90
|
+
except RuntimeError as e: # напр. --perplexity без extras: подсказка, exit 2
|
|
91
|
+
print(e, file=sys.stderr)
|
|
92
|
+
raise typer.Exit(code=2) from e
|
|
93
|
+
if json_out:
|
|
94
|
+
print(render_json(report))
|
|
95
|
+
elif csv_out:
|
|
96
|
+
print(render_csv(report), end="")
|
|
97
|
+
else:
|
|
98
|
+
print(render_volume(report))
|
|
99
|
+
print(render_comprehension(report))
|
|
100
|
+
print(render_slop(report))
|
|
101
|
+
if evidence:
|
|
102
|
+
print(render_evidence(report))
|
|
103
|
+
|
|
104
|
+
|
|
105
|
+
def main(argv: list[str] | None = None) -> int:
|
|
106
|
+
try:
|
|
107
|
+
app(argv)
|
|
108
|
+
except SystemExit as e: # typer/click: --version (0) / usage error (2) / Exit(N)
|
|
109
|
+
return int(e.code or 0)
|
|
110
|
+
return 0
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
if __name__ == "__main__":
|
|
114
|
+
raise SystemExit(main())
|
|
@@ -0,0 +1,145 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from collections.abc import Callable
|
|
5
|
+
|
|
6
|
+
from slopcount.detectors import EMOJI_RE
|
|
7
|
+
from slopcount.evidence import Category, Evidence, ScannedFile
|
|
8
|
+
from slopcount.extractors import extract_comments
|
|
9
|
+
from slopcount.i18n import _, ngettext
|
|
10
|
+
|
|
11
|
+
_TRIVIAL_DOC = re.compile(
|
|
12
|
+
r"^(Adds?|Returns?|Gets?|Sets?|Creates?|Initiali[sz]es?|Updates?|Checks?)\b", re.I
|
|
13
|
+
)
|
|
14
|
+
_CATCH_ALL = re.compile(
|
|
15
|
+
r"except\s+(Exception|BaseException)|catch\s*\(\s*(e|err|error|Exception)\b"
|
|
16
|
+
)
|
|
17
|
+
|
|
18
|
+
|
|
19
|
+
class CodeStyleDetector:
|
|
20
|
+
category = Category.STYLE
|
|
21
|
+
|
|
22
|
+
def detect(self, sf: ScannedFile, text: str) -> list[Evidence]:
|
|
23
|
+
checks: list[Callable[[ScannedFile, str], list[Evidence]]] = [
|
|
24
|
+
self._docstrings,
|
|
25
|
+
self._catch_all,
|
|
26
|
+
self._emoji_comments,
|
|
27
|
+
self._docstring_perfection,
|
|
28
|
+
self._monotone_comments,
|
|
29
|
+
]
|
|
30
|
+
out: list[Evidence] = []
|
|
31
|
+
for check in checks:
|
|
32
|
+
out.extend(check(sf, text))
|
|
33
|
+
return out
|
|
34
|
+
|
|
35
|
+
def _docstrings(self, sf, text) -> list[Evidence]:
|
|
36
|
+
evs = []
|
|
37
|
+
blocks = [b for b in extract_comments(text, sf.language or "") if b.is_docstring]
|
|
38
|
+
code_lines = len([line for line in text.split("\n") if line.strip()])
|
|
39
|
+
for b in blocks:
|
|
40
|
+
n = len(b.lines)
|
|
41
|
+
first = b.lines[0] if b.lines else ""
|
|
42
|
+
if n <= 2 and _TRIVIAL_DOC.match(first):
|
|
43
|
+
evs.append(
|
|
44
|
+
Evidence(
|
|
45
|
+
sf.path,
|
|
46
|
+
b.start_line,
|
|
47
|
+
self.category,
|
|
48
|
+
2,
|
|
49
|
+
_("trivial docstring on obvious function"),
|
|
50
|
+
)
|
|
51
|
+
)
|
|
52
|
+
if code_lines and n / code_lines > 0.5 and n >= 5:
|
|
53
|
+
evs.append(
|
|
54
|
+
Evidence(
|
|
55
|
+
sf.path,
|
|
56
|
+
b.start_line,
|
|
57
|
+
self.category,
|
|
58
|
+
2,
|
|
59
|
+
ngettext(
|
|
60
|
+
"docstring longer than body (%d line)",
|
|
61
|
+
"docstring longer than body (%d lines)",
|
|
62
|
+
n,
|
|
63
|
+
)
|
|
64
|
+
% n,
|
|
65
|
+
)
|
|
66
|
+
)
|
|
67
|
+
return evs
|
|
68
|
+
|
|
69
|
+
def _catch_all(self, sf, text) -> list[Evidence]:
|
|
70
|
+
evs = []
|
|
71
|
+
total = 0
|
|
72
|
+
for i, line in enumerate(text.split("\n"), 1):
|
|
73
|
+
if _CATCH_ALL.search(line):
|
|
74
|
+
total += 1
|
|
75
|
+
evs.append(
|
|
76
|
+
Evidence(sf.path, i, self.category, 1, _("catch-all exception swallowing"))
|
|
77
|
+
)
|
|
78
|
+
if total >= 5:
|
|
79
|
+
evs.append(
|
|
80
|
+
Evidence(
|
|
81
|
+
sf.path, 0, self.category, 2, _("defensive catch-all density (%d)") % total
|
|
82
|
+
)
|
|
83
|
+
)
|
|
84
|
+
return evs
|
|
85
|
+
|
|
86
|
+
def _emoji_comments(self, sf, text) -> list[Evidence]:
|
|
87
|
+
evs = []
|
|
88
|
+
for b in extract_comments(text, sf.language or ""):
|
|
89
|
+
for k, line in enumerate(b.lines):
|
|
90
|
+
if EMOJI_RE.search(line):
|
|
91
|
+
evs.append(
|
|
92
|
+
Evidence(
|
|
93
|
+
sf.path, b.start_line + k, self.category, 2, _("emoji in code comment")
|
|
94
|
+
)
|
|
95
|
+
)
|
|
96
|
+
return evs
|
|
97
|
+
|
|
98
|
+
_GOOGLE = re.compile(r"\b(Args|Parameters|Returns|Raises)\s*:", re.I)
|
|
99
|
+
_DEF_LINE = re.compile(r"^\s*(?:async\s+)?def\s+\w+")
|
|
100
|
+
|
|
101
|
+
def _docstring_perfection(self, sf, text) -> list[Evidence]:
|
|
102
|
+
lines = text.split("\n")
|
|
103
|
+
defs = [i for i, line in enumerate(lines, 1) if self._DEF_LINE.match(line)]
|
|
104
|
+
if len(defs) < 5:
|
|
105
|
+
return []
|
|
106
|
+
doc_starts = {
|
|
107
|
+
b.start_line for b in extract_comments(text, sf.language or "") if b.is_docstring
|
|
108
|
+
}
|
|
109
|
+
perfect = sum(
|
|
110
|
+
1
|
|
111
|
+
for d in defs
|
|
112
|
+
if any(ds == d + 1 for ds in doc_starts)
|
|
113
|
+
and any(self._GOOGLE.search(line) for line in lines[d : d + 15])
|
|
114
|
+
)
|
|
115
|
+
if perfect / len(defs) >= 0.8:
|
|
116
|
+
return [
|
|
117
|
+
Evidence(
|
|
118
|
+
sf.path,
|
|
119
|
+
0,
|
|
120
|
+
self.category,
|
|
121
|
+
2,
|
|
122
|
+
_("textbook-perfect docstrings on %d/%d functions") % (perfect, len(defs)),
|
|
123
|
+
)
|
|
124
|
+
]
|
|
125
|
+
return []
|
|
126
|
+
|
|
127
|
+
def _monotone_comments(self, sf, text) -> list[Evidence]:
|
|
128
|
+
lens = [
|
|
129
|
+
len(line) for b in extract_comments(text, sf.language or "") for line in b.lines if line
|
|
130
|
+
]
|
|
131
|
+
if len(lens) < 10:
|
|
132
|
+
return []
|
|
133
|
+
mean = sum(lens) / len(lens)
|
|
134
|
+
var = sum((x - mean) ** 2 for x in lens) / len(lens)
|
|
135
|
+
if var < 25:
|
|
136
|
+
return [
|
|
137
|
+
Evidence(
|
|
138
|
+
sf.path,
|
|
139
|
+
0,
|
|
140
|
+
self.category,
|
|
141
|
+
2,
|
|
142
|
+
_("monotone comment length (var=%.1f) — machine cadence") % var,
|
|
143
|
+
)
|
|
144
|
+
]
|
|
145
|
+
return []
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import re
|
|
4
|
+
from dataclasses import dataclass
|
|
5
|
+
|
|
6
|
+
from slopcount.detectors import EMOJI_RE
|
|
7
|
+
from slopcount.evidence import Category, Evidence, ScannedFile
|
|
8
|
+
from slopcount.i18n import _, ngettext
|
|
9
|
+
|
|
10
|
+
_EMOJI_HEADER = re.compile(r"^#{1,6}\s.*" + EMOJI_RE.pattern)
|
|
11
|
+
_GIANT_LINES = 500 # спека §4.2: «спеки-гиганты»
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class DocsBloatResult:
|
|
16
|
+
evidences: list[Evidence]
|
|
17
|
+
infected: list[tuple[str, int]] # (путь, всего строк файла)
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
class DocsBloatDetector:
|
|
21
|
+
category = Category.DOCS
|
|
22
|
+
|
|
23
|
+
def detect(self, sf: ScannedFile, text: str) -> DocsBloatResult:
|
|
24
|
+
lines = text.split("\n")
|
|
25
|
+
# физические строки: хвостовой "\n" даёт пустой последний элемент
|
|
26
|
+
total = len(lines) - (1 if lines and lines[-1] == "" else 0)
|
|
27
|
+
ev: list[Evidence] = []
|
|
28
|
+
weight = 0
|
|
29
|
+
if total > _GIANT_LINES:
|
|
30
|
+
ev.append(
|
|
31
|
+
Evidence(
|
|
32
|
+
sf.path,
|
|
33
|
+
0,
|
|
34
|
+
self.category,
|
|
35
|
+
5,
|
|
36
|
+
ngettext("spec giant: %d line", "spec giant: %d lines", total) % total,
|
|
37
|
+
)
|
|
38
|
+
)
|
|
39
|
+
weight += 5
|
|
40
|
+
for i, line in enumerate(lines, 1):
|
|
41
|
+
if _EMOJI_HEADER.match(line):
|
|
42
|
+
ev.append(
|
|
43
|
+
Evidence(sf.path, i, self.category, 2, _("emoji-decorated section header"))
|
|
44
|
+
)
|
|
45
|
+
weight += 2
|
|
46
|
+
infected: list[tuple[str, int]] = []
|
|
47
|
+
# гигант заражён целиком: «гигантскость» — утверждение обо всём файле,
|
|
48
|
+
# плотностной порог его никогда не достижим (вес маркера 5 фиксирован);
|
|
49
|
+
# для остальных файлов заражает плотность улико-баллов (спека §4.2)
|
|
50
|
+
if total > _GIANT_LINES or (total > 0 and weight / total > 0.1):
|
|
51
|
+
infected.append((sf.path, total))
|
|
52
|
+
return DocsBloatResult(ev, infected)
|
|
53
|
+
|
|
54
|
+
|
|
55
|
+
def repo_bloat_evidence(files: list[ScannedFile], sloc: int) -> Evidence | None:
|
|
56
|
+
"""Spec-to-Code Ratio из спеки §4.2: >100 КБ markdown на KLOC — тревога."""
|
|
57
|
+
docs_bytes = sum(f.size for f in files if f.kind == "markdown")
|
|
58
|
+
if sloc == 0:
|
|
59
|
+
# docs-only repo: ratio undefined, signal comes from infection instead
|
|
60
|
+
return None
|
|
61
|
+
kb_per_kloc = docs_bytes / 1024 / (sloc / 1000)
|
|
62
|
+
if kb_per_kloc > 100:
|
|
63
|
+
return Evidence(
|
|
64
|
+
"<repo>",
|
|
65
|
+
0,
|
|
66
|
+
Category.DOCS,
|
|
67
|
+
4,
|
|
68
|
+
_("docs bloat: %.0f KB of markdown per KLOC") % kb_per_kloc,
|
|
69
|
+
)
|
|
70
|
+
return None
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
from __future__ import annotations
|
|
2
|
+
|
|
3
|
+
import fnmatch
|
|
4
|
+
import re
|
|
5
|
+
from collections.abc import Callable
|
|
6
|
+
from dataclasses import dataclass
|
|
7
|
+
from importlib import resources
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
|
|
10
|
+
from slopcount.evidence import Category, Evidence, ScannedFile
|
|
11
|
+
from slopcount.i18n import _
|
|
12
|
+
|
|
13
|
+
|
|
14
|
+
@dataclass(frozen=True)
|
|
15
|
+
class _Header:
|
|
16
|
+
pattern: re.Pattern
|
|
17
|
+
weight: int
|
|
18
|
+
description: str
|
|
19
|
+
|
|
20
|
+
|
|
21
|
+
def _load():
|
|
22
|
+
"""Внутренний каталог; битый TOML падает сразу (fail fast, ассет в пакете)."""
|
|
23
|
+
import tomllib
|
|
24
|
+
|
|
25
|
+
base = resources.files("slopcount").joinpath("rules/env_markers.toml")
|
|
26
|
+
data = tomllib.loads(Path(str(base)).read_text(encoding="utf-8"))
|
|
27
|
+
paths = [(m["path"], m["weight"], _(m["description"])) for m in data.get("marker", [])]
|
|
28
|
+
headers = [
|
|
29
|
+
_Header(re.compile(h["pattern"]), h["weight"], _(h["description"]))
|
|
30
|
+
for h in data.get("header", [])
|
|
31
|
+
]
|
|
32
|
+
return paths, headers
|
|
33
|
+
|
|
34
|
+
|
|
35
|
+
class EnvMarkerDetector:
|
|
36
|
+
category = Category.AGENCY
|
|
37
|
+
|
|
38
|
+
def __init__(self):
|
|
39
|
+
self.paths, self.headers = _load()
|
|
40
|
+
|
|
41
|
+
def detect(
|
|
42
|
+
self, root: Path, scanned: list[ScannedFile], reader: Callable[[Path], str | None]
|
|
43
|
+
) -> list[Evidence]:
|
|
44
|
+
evs: list[Evidence] = []
|
|
45
|
+
names = [sf.path for sf in scanned] + _all_entries(root)
|
|
46
|
+
for pat, weight, desc in self.paths:
|
|
47
|
+
for name in names:
|
|
48
|
+
if fnmatch.fnmatch(name, pat) or fnmatch.fnmatch(Path(name).name, pat):
|
|
49
|
+
evs.append(Evidence(name, 0, self.category, weight, desc))
|
|
50
|
+
break
|
|
51
|
+
for sf in scanned:
|
|
52
|
+
# data-файлы не читаем вовсе — ни байта контента до детекторов
|
|
53
|
+
# (матрица §5.3 спеки scc-delegation)
|
|
54
|
+
if sf.kind == "data":
|
|
55
|
+
continue
|
|
56
|
+
text = reader(root / sf.path)
|
|
57
|
+
if not text:
|
|
58
|
+
continue
|
|
59
|
+
for line in text.split("\n")[:5]:
|
|
60
|
+
for h in self.headers:
|
|
61
|
+
if h.pattern.search(line):
|
|
62
|
+
evs.append(Evidence(sf.path, 1, self.category, h.weight, h.description))
|
|
63
|
+
return evs
|
|
64
|
+
|
|
65
|
+
|
|
66
|
+
def _all_entries(root: Path) -> list[str]:
|
|
67
|
+
out = []
|
|
68
|
+
for p in root.iterdir():
|
|
69
|
+
out.append(p.name)
|
|
70
|
+
return out
|