slopcount 0.1.0__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (77) hide show
  1. slopcount-0.1.0/.github/workflows/ci.yml +45 -0
  2. slopcount-0.1.0/.github/workflows/release.yml +43 -0
  3. slopcount-0.1.0/.gitignore +8 -0
  4. slopcount-0.1.0/CLAUDE.md +134 -0
  5. slopcount-0.1.0/LICENSE +21 -0
  6. slopcount-0.1.0/PKG-INFO +304 -0
  7. slopcount-0.1.0/README.md +278 -0
  8. slopcount-0.1.0/README.ru.md +279 -0
  9. slopcount-0.1.0/docs/superpowers/plans/2026-09-25-slopcount-implementation.md +3559 -0
  10. slopcount-0.1.0/docs/superpowers/plans/2026-09-26-md-sloc-ratio.md +919 -0
  11. slopcount-0.1.0/docs/superpowers/plans/2026-09-26-typer-migration.md +358 -0
  12. slopcount-0.1.0/docs/superpowers/plans/2026-09-27-comprehension-redesign.md +2267 -0
  13. slopcount-0.1.0/docs/superpowers/plans/2026-09-27-scc-delegation.md +1786 -0
  14. slopcount-0.1.0/docs/superpowers/plans/2026-09-28-first-release.md +972 -0
  15. slopcount-0.1.0/docs/superpowers/specs/2026-09-25-slopcount-design.md +428 -0
  16. slopcount-0.1.0/docs/superpowers/specs/2026-09-26-md-sloc-ratio-design.md +195 -0
  17. slopcount-0.1.0/docs/superpowers/specs/2026-09-26-typer-migration-design.md +205 -0
  18. slopcount-0.1.0/docs/superpowers/specs/2026-09-27-comprehension-redesign-design.md +571 -0
  19. slopcount-0.1.0/docs/superpowers/specs/2026-09-27-scc-delegation-design.md +450 -0
  20. slopcount-0.1.0/docs/superpowers/specs/2026-09-28-first-release-design.md +208 -0
  21. slopcount-0.1.0/pyproject.toml +58 -0
  22. slopcount-0.1.0/src/slopcount/__init__.py +1 -0
  23. slopcount-0.1.0/src/slopcount/app.py +206 -0
  24. slopcount-0.1.0/src/slopcount/cli.py +114 -0
  25. slopcount-0.1.0/src/slopcount/detectors/__init__.py +4 -0
  26. slopcount-0.1.0/src/slopcount/detectors/code_style.py +145 -0
  27. slopcount-0.1.0/src/slopcount/detectors/docs_bloat.py +70 -0
  28. slopcount-0.1.0/src/slopcount/detectors/env_markers.py +70 -0
  29. slopcount-0.1.0/src/slopcount/detectors/git_history.py +133 -0
  30. slopcount-0.1.0/src/slopcount/detectors/perplexity.py +139 -0
  31. slopcount-0.1.0/src/slopcount/detectors/phrase.py +34 -0
  32. slopcount-0.1.0/src/slopcount/download_model.py +19 -0
  33. slopcount-0.1.0/src/slopcount/evidence.py +171 -0
  34. slopcount-0.1.0/src/slopcount/extractors.py +126 -0
  35. slopcount-0.1.0/src/slopcount/i18n.py +61 -0
  36. slopcount-0.1.0/src/slopcount/locale/ru/LC_MESSAGES/slopcount.mo +0 -0
  37. slopcount-0.1.0/src/slopcount/locale/ru/LC_MESSAGES/slopcount.po +518 -0
  38. slopcount-0.1.0/src/slopcount/metrics/__init__.py +0 -0
  39. slopcount-0.1.0/src/slopcount/metrics/costs.py +138 -0
  40. slopcount-0.1.0/src/slopcount/metrics/slocomo.py +107 -0
  41. slopcount-0.1.0/src/slopcount/render/__init__.py +0 -0
  42. slopcount-0.1.0/src/slopcount/render/csv_out.py +27 -0
  43. slopcount-0.1.0/src/slopcount/render/json_out.py +166 -0
  44. slopcount-0.1.0/src/slopcount/render/text.py +311 -0
  45. slopcount-0.1.0/src/slopcount/rules/env_markers.toml +44 -0
  46. slopcount-0.1.0/src/slopcount/rules/languages.toml +414 -0
  47. slopcount-0.1.0/src/slopcount/rules/phrases_en.toml +54 -0
  48. slopcount-0.1.0/src/slopcount/rules/phrases_ru.toml +34 -0
  49. slopcount-0.1.0/src/slopcount/rules.py +110 -0
  50. slopcount-0.1.0/src/slopcount/scales.py +86 -0
  51. slopcount-0.1.0/src/slopcount/scc.py +351 -0
  52. slopcount-0.1.0/tests/conftest.py +25 -0
  53. slopcount-0.1.0/tests/fixtures/human_project/src/args.c +13 -0
  54. slopcount-0.1.0/tests/fixtures/human_project/src/util.py +6 -0
  55. slopcount-0.1.0/tests/fixtures/scc_slop_project.json2 +145 -0
  56. slopcount-0.1.0/tests/fixtures/slop_project/CLAUDE.md +1 -0
  57. slopcount-0.1.0/tests/fixtures/slop_project/README.md +11 -0
  58. slopcount-0.1.0/tests/fixtures/slop_project/src/defensive.py +14 -0
  59. slopcount-0.1.0/tests/fixtures/slop_project/src/greeter.py +6 -0
  60. slopcount-0.1.0/tests/golden/slop_project_en.txt +77 -0
  61. slopcount-0.1.0/tests/test_cli.py +161 -0
  62. slopcount-0.1.0/tests/test_code_style.py +74 -0
  63. slopcount-0.1.0/tests/test_costs.py +168 -0
  64. slopcount-0.1.0/tests/test_docs_bloat.py +58 -0
  65. slopcount-0.1.0/tests/test_e2e.py +345 -0
  66. slopcount-0.1.0/tests/test_env_markers.py +54 -0
  67. slopcount-0.1.0/tests/test_evidence.py +101 -0
  68. slopcount-0.1.0/tests/test_extractors.py +77 -0
  69. slopcount-0.1.0/tests/test_git_history.py +152 -0
  70. slopcount-0.1.0/tests/test_i18n.py +105 -0
  71. slopcount-0.1.0/tests/test_perplexity.py +140 -0
  72. slopcount-0.1.0/tests/test_phrase.py +36 -0
  73. slopcount-0.1.0/tests/test_render.py +158 -0
  74. slopcount-0.1.0/tests/test_rules.py +108 -0
  75. slopcount-0.1.0/tests/test_scales.py +50 -0
  76. slopcount-0.1.0/tests/test_scc.py +306 -0
  77. slopcount-0.1.0/tests/test_slocomo.py +205 -0
@@ -0,0 +1,45 @@
1
+ name: ci
2
+ on:
3
+ push:
4
+ branches: [main, slopcount-design]
5
+ pull_request:
6
+
7
+ jobs:
8
+ tests:
9
+ runs-on: ubuntu-latest
10
+ steps:
11
+ - uses: actions/checkout@v4
12
+ - uses: actions/setup-python@v5
13
+ with:
14
+ python-version: "3.12"
15
+ - run: python -m pip install -e .[dev]
16
+ - name: Install scc 4.1.0 (pinned, sha256 verified)
17
+ # hash = the pinned table in src/slopcount/scc.py (official checksums.txt v4.1.0)
18
+ run: |
19
+ set -euo pipefail
20
+ curl -fsSL -o /tmp/scc.tar.gz \
21
+ https://github.com/boyter/scc/releases/download/v4.1.0/scc_Linux_x86_64.tar.gz
22
+ echo "c7328436d3027f4357d3d7853f7dc3ac2bbcb4ca08f1adad91a27c593884079b /tmp/scc.tar.gz" \
23
+ | sha256sum -c
24
+ sudo tar xzf /tmp/scc.tar.gz -C /usr/local/bin scc
25
+ - run: ruff check .
26
+ - run: ruff format --check .
27
+ - run: python -m pytest tests/ -q
28
+
29
+ build:
30
+ runs-on: ubuntu-latest
31
+ steps:
32
+ - uses: actions/checkout@v4
33
+ - uses: actions/setup-python@v5
34
+ with:
35
+ python-version: "3.12"
36
+ - run: python -m pip install build twine
37
+ - run: python -m build
38
+ - run: twine check dist/*
39
+ - name: Wheel must ship locale and rule catalogs
40
+ run: |
41
+ unzip -l dist/*.whl | grep -q 'locale/ru/LC_MESSAGES/slopcount\.mo'
42
+ unzip -l dist/*.whl | grep -q 'rules/languages\.toml'
43
+ unzip -l dist/*.whl | grep -q 'rules/phrases_en\.toml'
44
+ unzip -l dist/*.whl | grep -q 'rules/phrases_ru\.toml'
45
+ unzip -l dist/*.whl | grep -q 'rules/env_markers\.toml'
@@ -0,0 +1,43 @@
1
+ name: release
2
+ on:
3
+ push:
4
+ tags: ["v*"]
5
+
6
+ jobs:
7
+ build:
8
+ runs-on: ubuntu-latest
9
+ steps:
10
+ - uses: actions/checkout@v4
11
+ - uses: actions/setup-python@v5
12
+ with:
13
+ python-version: "3.12"
14
+ - run: python -m pip install build twine
15
+ - run: python -m build
16
+ - run: twine check dist/*
17
+ - name: Wheel must ship locale and rule catalogs
18
+ run: |
19
+ unzip -l dist/*.whl | grep -q 'locale/ru/LC_MESSAGES/slopcount\.mo'
20
+ unzip -l dist/*.whl | grep -q 'rules/languages\.toml'
21
+ unzip -l dist/*.whl | grep -q 'rules/phrases_en\.toml'
22
+ unzip -l dist/*.whl | grep -q 'rules/phrases_ru\.toml'
23
+ unzip -l dist/*.whl | grep -q 'rules/env_markers\.toml'
24
+ - name: Tag must match package version
25
+ run: |
26
+ grep -Fq "version = \"${GITHUB_REF_NAME#v}\"" pyproject.toml
27
+ - uses: actions/upload-artifact@v4
28
+ with:
29
+ name: dist
30
+ path: dist/
31
+
32
+ publish:
33
+ needs: build
34
+ runs-on: ubuntu-latest
35
+ environment: pypi
36
+ permissions:
37
+ id-token: write
38
+ steps:
39
+ - uses: actions/download-artifact@v4
40
+ with:
41
+ name: dist
42
+ path: dist/
43
+ - uses: pypa/gh-action-pypi-publish@release/v1
@@ -0,0 +1,8 @@
1
+ __pycache__/
2
+ *.pyc
3
+ .venv/
4
+ .pytest_cache/
5
+ dist/
6
+ *.egg-info/
7
+ .idea/
8
+ .ruff_cache/
@@ -0,0 +1,134 @@
1
+ # CLAUDE.md
2
+
3
+ This file provides guidance to Claude Code (claude.ai/code) when working with code in this repository.
4
+
5
+ ## Что это
6
+
7
+ `slopcount` — сканирует проект, детектирует LLM-слоп и считает стоимость его *понимания* (comprehension,
8
+ SLOCOMO вместо COCOMO). Детекция и формулы честные, единицы — шутливые. Каждая улика
9
+ объяснима (`--evidence`).
10
+
11
+
12
+ ## Команды
13
+
14
+ ```bash
15
+ .venv/bin/python -m pytest tests/ -q # все тесты (~20 с: e2e гоняет настоящий scc)
16
+ .venv/bin/python -m pytest tests/test_slocomo.py -q # один файл
17
+ .venv/bin/python -m pytest tests/test_e2e.py::test_name -q # один тест
18
+ .venv/bin/slopcount . # самоскан
19
+ .venv/bin/slopcount . --scc-path /path/to/scc # явный бинарник scc
20
+ .venv/bin/slopcount tests/fixtures/slop_project --evidence # на эталонном слопе
21
+ .venv/bin/ruff check . # линтер (конфиг в pyproject.toml)
22
+ .venv/bin/ruff format --check . # проверить форматирование
23
+ .venv/bin/ruff format . # отформатировать
24
+ ```
25
+
26
+
27
+
28
+ **После правки `.po` обязательно перекомпилировать `.mo`** — тест
29
+ `test_i18n.py` проверяет побайтовое совпадение (drift guard):
30
+
31
+ ```bash
32
+ msgfmt --check -o src/slopcount/locale/ru/LC_MESSAGES/slopcount.mo \
33
+ src/slopcount/locale/ru/LC_MESSAGES/slopcount.po
34
+ ```
35
+
36
+ ## Архитектура
37
+
38
+ Конвейер: `scc (один subprocess: манифест per-file + COCOMO/LOCOMO) →
39
+ классификация kind/language → volume (VolumeStats: языки/SLOC/комментарии/md) →
40
+ детекторы → Evidence[] → aggregate_slop (SlopStats) → comprehension (SLOCOMO +
41
+ разбивки costs.py) → render (3 секции)`. Оркестратор — `app.py:run()`; `cli.py` —
42
+ тонкая typer-обёртка.
43
+
44
+ - **`scc.py`** — единственная точка работы с бинарником scc. Резолв:
45
+ `--scc-path` > `SLOPCOUNT_SCC_BIN` > PATH (версия ≥ 4.1.0) >
46
+ кеш `~/.cache/slopcount/scc/<ver>/` > скачивание официального релиза
47
+ (sha256 по зашитой таблице, атомарная укладка в кеш). Один subprocess
48
+ `--format json2 --by-file --cognitive --locomo` из нейтрального пустого cwd
49
+ + `--no-config` (гасит авто-детект `.sccconfig`/`SCC_CONFIG_PATH` — чужой
50
+ конфиг искажает COCOMO/LOCOMO на порядки); `--personcost/--overhead`
51
+ транслируются в `--avg-wage/--overhead`; `SKIP_DIRS` → `--exclude-dir`.
52
+ В отчёт идёт фактическая версия бинарника, а не пин.
53
+ - **`rules/languages.toml`** — все 366 display-имён scc 4.1.0 с явной
54
+ классификацией kind: code (251) / markdown (2) / prose (1) / data (112);
55
+ `[extractors]` (31 имя → id для extractors.py). Пользовательский `--rules`
56
+ TOML переопределяет точечно секцией `[languages]` (мердж поверх builtin,
57
+ оверрайд стирает extractor-id, неизвестное имя — предупреждение). Имя,
58
+ которого нет в каталоге (будущая версия scc) → молча `("data", None)`.
59
+ - **Матрица kind → судьба файла** (спека §5.3): `data` не читается вообще —
60
+ только факт существования; SLOC/комментарии/complexity/cognitive — только
61
+ kind=code; числитель MD/SLOC — только `.md`/`.markdown`; COCOMO/LOCOMO scc
62
+ считает по полному дереву (включая data) — «написать/перегенерировать всё
63
+ дерево» против «понять слоп» у SLOCOMO.
64
+ - **`evidence.py`** — ядро модели данных: `Evidence`, `ScannedFile`,
65
+ `read_text` и `Report` (+ `scc_version`/`complexity`/`cognitive_total`/
66
+ `files_total`/`cocomo`/`locomo`) и `aggregate_slop()`. Пять категорий
67
+ (`prose/docs/style/agency/history`) не смешиваются. **AGENCY не входит в
68
+ SLOP и Slop Ratio** — это констатация «в проекте живут агенты», не
69
+ обвинение. Все ratios — доли; SLOC=0 при SLOP>0 даёт ratio=inf — это
70
+ предусмотрено.
71
+ - **Три секции отчёта** (спека 2026-09-27-comprehension-redesign): Project
72
+ Volume / Comprehension Effort & Cost / Detected Slop; три шкалы `scales.py`
73
+ (doc/comment/slop, калибровка §7 спеки: doc 0.05/0.35/0.50/0.75/1.0+RECURSION,
74
+ slop CLEAN<0.005) вместо вердикта; `--evidence` вместо `--details` (все
75
+ улики + сниппеты строк, кеш lru_cache); `--fail-above`/`--verdict-only`
76
+ удалены — CI-гейтинг поверх `--json`.
77
+ - **SLOP** = уникальные (file, line) с уликами по prose/docs/style/history
78
+ + `round(строки × 0.8)` за каждый «заражённый» md-файл (гигант >500
79
+ строк или плотность веса > 0.1 на строку); **MD/SLOC** = md-строки (только `.md`/`.markdown`) / SLOC —
80
+ информационная вместе с comment/SLOC.
81
+ - **Детекторы подключаются вручную в `app.py`** (реестра из спеки §7 нет;
82
+ `detectors/__init__.py` содержит только общий `EMOJI_RE`). Новый детектор =
83
+ новый модуль + явный вызов в `run()` + включение/невключение в SLOP-набор
84
+ категорий в `aggregate_slop()`.
85
+ - **`metrics/`** — `slocomo.py`: ComprehensionStats — чтение доков
86
+ (238 wpm × 2.3) + комментарии (строки × 6 слов) + код (200 SLOC/ч) +
87
+ cognitive×0.5 мин (весь проект); `person_months = reading_hours/152 ×
88
+ (1+slop_ratio)`, всё per person + team_costs (Фибоначчи 1..21);
89
+ токены/окна/GPU — из LOCOMO (in+out). `metrics/costs.py`: разбивки
90
+ стоимостей в единой структуре корзин docs / source_code{total, code,
91
+ comments} / data — COCOMO независимым пересчётом 2.4·K^1.05
92
+ (+drift-guard против estimatedCost scc), LOCOMO атрибуцией по долям
93
+ Code-строк, SLOCOMO точной суммой. Комментарии scc в COCOMO/LOCOMO не
94
+ считает (подстрока «комментарии» — прочерк), data SLOCOMO не читает.
95
+ - **`extractors.py`** — разбор коммент-блоков для phrase- и style-детекторов
96
+ (нумерация улик, `is_docstring`, содержимое блоков). Инвариант:
97
+ `CommentBlock` занимает ровно `len(lines)` физических строк начиная с
98
+ `start_line` (пустые строки внутри блока сохраняются). Известные
99
+ приближения (незакрытый docstring, `/* */` не с начала строки)
100
+ задокументированы в докстринге — не «чинить» молча.
101
+ - **Обход и gitignore** — семантика scc: настоящие gitignore-правила (с
102
+ отрицаниями), детект языка по shebang/имени файла, бинарные файлы в
103
+ манифест не попадают. `read_text` (NUL-байт в первых 1024, строгий UTF-8)
104
+ остаётся страховкой от нечитаемых файлов → счётчик skip.
105
+
106
+ ### i18n (gettext, критичные нюансы)
107
+
108
+ - **Английский — язык исходников**: каждый msgid уже английский текст,
109
+ обёрнут в `_()`; en работает без каталога (NullTranslations-фолбэк).
110
+ - `i18n.setup()` обязан быть вызван до любого `_()` — это глобальное
111
+ состояние; conftest.py сбрасывает в en autouse-фикстурой (детекторы и
112
+ каталоги правил тоже локализуются при загрузке).
113
+ - Плюрализация только через `ngettext` (ru: 1 чашка / 2 чашки / 5 чашек),
114
+ интерполяция только `%`-стилем — msgfmt проверяет плейсхолдеры.
115
+ Числа — `fmt_int`/`fmt_float` (разделители en `,.` / ru ` ,`).
116
+ - **Ключи JSON/CSV не локализуются никогда** — стабильны для CI между языками.
117
+ - Каталоги фраз *детекции* (`rules/phrases_en.toml`, `phrases_ru.toml`)
118
+ ортогональны локали UI: грузятся всегда оба, можно сканировать английский
119
+ слоп с русским интерфейсом. Описания правил проходят `_()` при загрузке
120
+ (локализация улик в `--evidence`/`--csv`).
121
+ - Битый пользовательский `--rules` TOML → `RuntimeError` → exit 2; битый
122
+ встроенный — fail fast.
123
+
124
+ ### Опциональные режимы (lazy import, extras)
125
+
126
+ - **`--perplexity`**: transformers/torch импортируются только при флаге;
127
+ нет extras → `RuntimeError` → exit 2 с подсказкой. Модель — gpt2
128
+ (переопределяется `SLOPCOUNT_PPLX_MODEL`), скачивается
129
+ `python -m slopcount.download_model`. Замер **контекстный** (GPTZero-стиль):
130
+ один проход по файлу, пер-токенный NLL, группировка по предложениям через
131
+ offset mapping; **первое предложение отбрасывается** (нет левого контекста),
132
+ порог — **медиана < 40**. Burstiness убран: на gpt2 классы не разделяет.
133
+ Файлы длиннее окна модели меряются по префиксу; предупреждение токенизатора
134
+ об overlong глушится хирургическим фильтром логгера.
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Egor Chernyshev
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,304 @@
1
+ Metadata-Version: 2.5
2
+ Name: slopcount
3
+ Version: 0.1.0
4
+ Summary: Detect AI slop in a codebase and estimate the cost of comprehending it
5
+ Project-URL: Homepage, https://github.com/echernyshev/slopcount
6
+ Author: Egor Chernyshev
7
+ License-Expression: MIT
8
+ License-File: LICENSE
9
+ Keywords: ai,cocomo,code-quality,cognitive-complexity,llm,sloc,sloccount,slop
10
+ Classifier: Environment :: Console
11
+ Classifier: License :: OSI Approved :: MIT License
12
+ Classifier: Programming Language :: Python :: 3.11
13
+ Classifier: Programming Language :: Python :: 3.12
14
+ Classifier: Programming Language :: Python :: 3.13
15
+ Classifier: Topic :: Software Development
16
+ Classifier: Topic :: Software Development :: Quality Assurance
17
+ Requires-Python: >=3.11
18
+ Requires-Dist: typer<0.28,>=0.27
19
+ Provides-Extra: dev
20
+ Requires-Dist: pytest>=8; extra == 'dev'
21
+ Requires-Dist: ruff>=0.14; extra == 'dev'
22
+ Provides-Extra: perplexity
23
+ Requires-Dist: torch>=2.2; extra == 'perplexity'
24
+ Requires-Dist: transformers>=4.40; extra == 'perplexity'
25
+ Description-Content-Type: text/markdown
26
+
27
+ # slopcount
28
+
29
+ **SLOC is what you paid to write. SLOP is what you must now read.**
30
+
31
+ [![PyPI](https://img.shields.io/pypi/v/slopcount.svg)](https://pypi.org/project/slopcount/)
32
+ [![Python](https://img.shields.io/pypi/pyversions/slopcount.svg)](https://pypi.org/project/slopcount/)
33
+ [![CI](https://github.com/echernyshev/slopcount/actions/workflows/ci.yml/badge.svg)](https://github.com/echernyshev/slopcount/actions/workflows/ci.yml)
34
+ [![License: MIT](https://img.shields.io/pypi/l/slopcount.svg)](LICENSE)
35
+
36
+ [English](README.md) · [Русский](README.ru.md)
37
+
38
+ `slopcount` scans a codebase, detects LLM-generated slop and estimates what the
39
+ project costs to *comprehend* — a spiritual successor to the classic
40
+ [`sloccount`](https://www.dwheeler.com/sloccount/), which estimated what code
41
+ costs to *write*.
42
+
43
+ ## Quick Start
44
+
45
+ ```bash
46
+ pipx install slopcount # or: pip install slopcount
47
+ slopcount .
48
+ ```
49
+
50
+ That's it. On first run slopcount downloads the [`scc`](https://github.com/boyter/scc)
51
+ counter binary (~7 MB) into `~/.cache/slopcount/` and caches it (skipped when
52
+ a suitable scc ≥ 4.1.0 is already on PATH). To use your own build:
53
+ `--scc-path /path/to/scc` or the `SLOPCOUNT_SCC_BIN` environment variable.
54
+
55
+ ```bash
56
+ slopcount . --evidence # every finding: file, line, snippet, matched rule
57
+ slopcount . --json # machine-readable output for CI
58
+ slopcount . --lang ru # Russian interface
59
+ ```
60
+
61
+ ## Why
62
+
63
+ LLM code generation became nearly free. Comprehension did not: a human — or an
64
+ agent burning tokens — still has to read every line, and agents now write a
65
+ growing share of the world's code. Along the way, slop accumulates: fluent,
66
+ confident, low-value text. Comments that restate the code. Docstrings that
67
+ explain the signature and nothing else. Documentation generated on demand that
68
+ nobody asked for and nobody maintains.
69
+
70
+ Classical estimation answered *“what did this cost to write?”* (sloccount,
71
+ COCOMO). Generation being nearly free makes that question obsolete. The question
72
+ that matters now is *“what does this cost to understand?”* — for the engineer
73
+ joining the project, for the reviewer, for the agent about to extend it.
74
+ slopcount answers it:
75
+
76
+ - detects slop markers in comments, docstrings and markdown docs — every
77
+ finding is explainable: file, line, snippet, rule;
78
+ - estimates the effort of reading the whole project — docs, code, comments,
79
+ cognitive complexity — from published, boring formulas;
80
+ - puts three costs side by side: writing the tree from scratch (COCOMO),
81
+ regenerating it with an LLM (LOCOMO), comprehending it (SLOCOMO).
82
+
83
+ The detection is honest — regular expressions, Campbell's cognitive complexity,
84
+ Brysbaert's reading-speed research. The formulas are public. Only the units are
85
+ playful (coffee, therapy, GPU-hours of regret); the arithmetic behind them is
86
+ real.
87
+
88
+ ## The Report
89
+
90
+ ### Volume
91
+
92
+ [`scc`](https://github.com/boyter/scc) does the walking and counting: 366
93
+ languages (251 of them code), real `.gitignore` semantics (negations
94
+ included), shebang-based language detection. Every file is classified into
95
+ one of four kinds, and the kind decides its fate:
96
+
97
+ | Kind | Examples | Slop detection | Role in metrics |
98
+ |---|---|---|---|
99
+ | **code** | Python, C, Makefile, SQL | style + phrases in comments | the only source of SLOC, comments and complexity |
100
+ | **markdown** | `.md`, `.markdown` | bloat + phrases | docs lines; infected lines count toward SLOP |
101
+ | **prose** | plain text | phrases | words count toward reading time |
102
+ | **data** | JSON, YAML, TOML, XML… | — | not read; line counts still feed COCOMO/LOCOMO estimates |
103
+
104
+ The boundary is “does a human read this to understand the program”: a Makefile
105
+ is code (logic lives there), JSON is data (declarations live there). Want a
106
+ different classification for your project? See [`--rules`](#configuration).
107
+
108
+ ### Three ratios, three scales
109
+
110
+ | Ratio | Meaning | Scale |
111
+ |---|---|---|
112
+ | **SLOP/SLOC** | detected slop lines per line of code | CLEAN < 0.005 · TRACE < 0.10 · NOTICEABLE < 0.30 · HEAVY < 1.0 · INFESTED |
113
+ | **MD/SLOC** | markdown lines per line of code | HUMAN < 0.05 · NEURO_CLOUD < 0.35 · ESTABLISHED_SLOP < 0.50 · AGENT_SELF_SERVICE < 0.75 · AGENT_OCCUPATION < 1.00 · RECURSION |
114
+ | **comment/SLOC** | comment lines per line of code | ASCETIC < 0.05 · DOCUMENTED < 0.30 · CHATTY < 0.60 · LECTURE_NOTES < 1.00 · COMMENT_DRIVEN |
115
+
116
+ Scale bounds are calibrated against reference repositories: django sits at
117
+ MD/SLOC 0.0005 and flask at 0.015 (HUMAN), requests at 0.34 (NEURO_CLOUD). The
118
+ CLEAN/TRACE boundary for SLOP is the midpoint of the only measured gap between
119
+ human repositories (~0.0025) and slop references (~0.0071).
120
+
121
+ **How SLOP is counted:** unique `(file, line)` findings across the
122
+ prose/docs/style/history categories, plus `round(lines × 0.8)` for every
123
+ markdown file flagged as infected (giant: > 500 lines, or marker density
124
+ > 0.1 per line).
125
+
126
+ ### Comprehension effort (per person)
127
+
128
+ | Component | Formula |
129
+ |---|---|
130
+ | Reading documentation | words ÷ 238 wpm × 2.3 (technical-text penalty) |
131
+ | Reading code | 200 SLOC per hour |
132
+ | Reading comments | ~6 words per comment line |
133
+ | Cognitive processing | cognitive complexity × 0.5 min per point |
134
+
135
+ ### The cost ladder
136
+
137
+ | Model | Question | How |
138
+ |---|---|---|
139
+ | **COCOMO** | what would the whole tree cost to *write*? | classic 2.4·K^1.05 person-months over all lines |
140
+ | **LOCOMO** | what would it cost to *regenerate* with an LLM? | scc's token estimate (in+out): generation + review hours |
141
+ | **SLOCOMO** | what does it cost to *comprehend*? | reading hours × (1 + slop ratio) → person-months × wage × overhead; team cost scales 1–21 heads (Fibonacci) |
142
+
143
+ In dollars, writing usually dominates and regeneration is nearly free.
144
+ Comprehension sits between the two — but unlike the sunk cost of writing,
145
+ it repeats: every engineer and every agent who joins the project pays it
146
+ again.
147
+
148
+ ### Detected slop
149
+
150
+ Findings fall into five categories. Four count toward SLOP: **prose**
151
+ (comments/docstrings), **docs** (markdown), **style** (code style), **history**
152
+ (git). **Agency** markers (`CLAUDE.md`, `.claude/`, …) are reported but never
153
+ counted: agents living in a repository is a fact, not an accusation.
154
+
155
+ ## Example Output
156
+
157
+ Fixture project from the test suite
158
+ (`slopcount --lang en tests/fixtures/slop_project`):
159
+
160
+ ```
161
+ COCOMO write the whole tree (docs count as code) = $ 352 (0.0 person-months · 0.7 mo · 0.0 people)
162
+ LOCOMO regenerate it with an LLM = $ 0.01 (0.0 h + 0.0 h review)
163
+ SLOCOMO comprehend the project = $ 25 (0.1 h reading · 0.0 person-months) — per person
164
+
165
+ Slop-to-Code Ratio (SLOP/SLOC) = 2.000 [████████████████████] 200.0% INFESTED
166
+ ```
167
+
168
+ <details>
169
+ <summary>Full report</summary>
170
+
171
+ ```
172
+ PROJECT VOLUME
173
+ -------------------------------------------------------------------------------
174
+ Code by language: files SLOC
175
+ Python 2 8
176
+ -------------------------------------------------------------------------------
177
+ Total SLOC = 8
178
+ Files in scan = 4
179
+ Documentation = 12 lines (2 files)
180
+ Documentation-to-Code Ratio (MD/SLOC) = 1.500 [████████████████████] 150.0% RECURSION
181
+ You ran slopcount inside slop. Recursion
182
+ Comments = 12 lines
183
+ Comments-to-Code Ratio (comment/SLOC) = 1.500 [████████████████████] 150.0% COMMENT_DRIVEN
184
+ Comment-driven development. The code is an attachment
185
+ Detected SLOP = 16 lines
186
+ Slop-to-Code Ratio (SLOP/SLOC) = 2.000 [████████████████████] 200.0% INFESTED
187
+ Full slop infestation. Call the exterminators
188
+ COMPREHENSION EFFORT & COST
189
+ -------------------------------------------------------------------------------
190
+ Reading documentation = 0.0 h (43 words / 238.0 wpm × 2.3)
191
+ Reading code = 0.0 h (8 SLOC / 200.0 per hour)
192
+ Reading comments = 0.0 h (72 words, lines × 6 estimate)
193
+ Cognitive processing = 0.1 h (7 points × 0.5 min)
194
+ Total reading time = 0.1 h
195
+
196
+ -------------------------------------------------------------------------------
197
+ Cost Ladder (write / regenerate / comprehend)
198
+ -------------------------------------------------------------------------------
199
+ COCOMO write the whole tree (docs count as code) = $ 352 (0.0 person-months · 0.7 mo · 0.0 people)
200
+ docs 0.0 person-months · $ 170
201
+ source code 0.0 person-months · $ 170
202
+ code 0.0 person-months · $ 170
203
+ comments — (scc does not count comments)
204
+ data 0.0 person-months · $ 0
205
+ LOCOMO regenerate it with an LLM = $ 0.01 (0.0 h + 0.0 h review)
206
+ docs 0.0 h · $ 0.01
207
+ source code 0.0 h · $ 0.01
208
+ code 0.0 h · $ 0.01
209
+ comments — (scc does not count comments)
210
+ data 0.0 h · $ 0.00
211
+ SLOCOMO comprehend the project = $ 25 (0.1 h reading · 0.0 person-months) — per person
212
+ (a team multiplies by headcount — see the scale below)
213
+ docs 0.0 h · $ 2
214
+ source code 0.1 h · $ 23
215
+ code 0.1 h · $ 22 (incl. cognitive 0.1 h)
216
+ comments 0.0 h · $ 1
217
+ data — (not read)
218
+
219
+ Team Comprehension Cost (headcount × per person)
220
+ 1 person = 0.0 person-months · $ 25
221
+ 2 people = 0.0 person-months · $ 49
222
+ 3 people = 0.0 person-months · $ 74
223
+ 5 people = 0.0 person-months · $ 123
224
+ 8 people = 0.0 person-months · $ 196
225
+ 13 people = 0.0 person-months · $ 319
226
+ 21 people = 0.0 person-months · $ 515
227
+
228
+ Comprehension Tokens (LOCOMO round-trip: in + out) = 2,775
229
+ Context Windows Consumed = 0.0139 × 200K / 0.0028 × 1M
230
+ GPU-hours of Regret = 0.0077
231
+
232
+ Coffee Required = 1 cup ($ 4.00)
233
+ Therapy Recommended = 1 session ($ 150.00)
234
+ DETECTED SLOP
235
+ -------------------------------------------------------------------------------
236
+ Totals grouped by slop origin (dominant slop source first):
237
+ -------------------------------------------------------------------------------
238
+ Origin files slop lines slop % cognitivity
239
+ -------------------------------------------------------------------------------
240
+ Prose (comments/docstrings) 2 3 18.8 high
241
+ Markdown specs 1 2 12.5 medium
242
+ Code style 1 2 12.5 medium
243
+ Git history 0 0 0.0 low
244
+ Environment markers 1 — — —
245
+ -------------------------------------------------------------------------------
246
+ Top slop files: README.md 13 lines · src/defensive.py 2 lines · src/greeter.py 1 line
247
+ Agents detected (not counted as slop): CLAUDE.md
248
+ Run with --evidence to see every finding with its source line.
249
+ ```
250
+
251
+ </details>
252
+
253
+ ## Configuration
254
+
255
+ | Option | Meaning |
256
+ |---|---|
257
+ | `--lang en\|ru` | interface language |
258
+ | `--rules FILE` | TOML: `[languages]` reclassification + `[[rule]]` phrase entries (repeatable) |
259
+ | `--scc-path PATH` / `SLOPCOUNT_SCC_BIN` | custom `scc` binary |
260
+ | `--personcost USD` | monthly person cost for estimates (default 4690.5) |
261
+ | `--overhead X` | cost overhead multiplier, applied to COCOMO and SLOCOMO (default 2.4) |
262
+ | `--coffee-price USD`, `--no-therapy` | tune the joke units |
263
+ | `--history N` | also scan git history, N commits deep |
264
+ | `--perplexity` | GPT-2 perplexity detector (see below) |
265
+
266
+ Reclassify a language for your project without touching code:
267
+
268
+ ```toml
269
+ # my-rules.toml — run: slopcount --rules my-rules.toml .
270
+ [languages]
271
+ "AsciiDoc" = "markdown" # count as documentation
272
+ "SQL" = "data" # count as data, not code
273
+ ```
274
+
275
+ **Perplexity detector** (optional, local GPT-2): contextual per-sentence
276
+ perplexity; median < 40 flags machine-smooth prose:
277
+
278
+ ```bash
279
+ pipx install 'slopcount[perplexity]'
280
+ python -m slopcount.download_model # fetches GPT-2 once
281
+ slopcount . --perplexity
282
+ ```
283
+
284
+ ## CI Gate
285
+
286
+ `slopcount` exits 0 on success and 2 on errors; gate on the JSON report:
287
+
288
+ ```bash
289
+ slopcount . --json | jq -e '(.slop.ratio // 9) < 0.05' > /dev/null || echo "too much slop"
290
+ ```
291
+
292
+ The `// 9` matters: a docs-only repository maps `inf` to `null` in JSON, and a
293
+ bare `null < 0.05` would evaluate true — `// 9` sends null to 9, so the gate
294
+ fails closed.
295
+
296
+ ## Acknowledgements
297
+
298
+ All the grunt work — walking the tree, counting lines, comments and complexity,
299
+ COCOMO and LOCOMO — is delegated to [scc](https://github.com/boyter/scc) by Ben
300
+ Boyter. slopcount only exists because scc does. Our hat is off.
301
+
302
+ ## License
303
+
304
+ MIT — see [LICENSE](LICENSE).