ru-stylometry 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (51) hide show
  1. ru_stylometry-0.1.1/LICENSE +21 -0
  2. ru_stylometry-0.1.1/MANIFEST.in +1 -0
  3. ru_stylometry-0.1.1/PKG-INFO +274 -0
  4. ru_stylometry-0.1.1/README.md +228 -0
  5. ru_stylometry-0.1.1/pyproject.toml +92 -0
  6. ru_stylometry-0.1.1/setup.cfg +4 -0
  7. ru_stylometry-0.1.1/src/ru_stylometry/__init__.py +34 -0
  8. ru_stylometry-0.1.1/src/ru_stylometry/__main__.py +5 -0
  9. ru_stylometry-0.1.1/src/ru_stylometry/_lexicons.py +76 -0
  10. ru_stylometry-0.1.1/src/ru_stylometry/_morph.py +41 -0
  11. ru_stylometry-0.1.1/src/ru_stylometry/_stats.py +91 -0
  12. ru_stylometry-0.1.1/src/ru_stylometry/_text.py +99 -0
  13. ru_stylometry-0.1.1/src/ru_stylometry/cli.py +269 -0
  14. ru_stylometry-0.1.1/src/ru_stylometry/document.py +66 -0
  15. ru_stylometry-0.1.1/src/ru_stylometry/evaluation.py +267 -0
  16. ru_stylometry-0.1.1/src/ru_stylometry/explain.py +134 -0
  17. ru_stylometry-0.1.1/src/ru_stylometry/features/__init__.py +130 -0
  18. ru_stylometry-0.1.1/src/ru_stylometry/features/char.py +54 -0
  19. ru_stylometry-0.1.1/src/ru_stylometry/features/formatting.py +34 -0
  20. ru_stylometry-0.1.1/src/ru_stylometry/features/length.py +30 -0
  21. ru_stylometry-0.1.1/src/ru_stylometry/features/lexical.py +37 -0
  22. ru_stylometry-0.1.1/src/ru_stylometry/features/morph.py +109 -0
  23. ru_stylometry-0.1.1/src/ru_stylometry/features/readability.py +37 -0
  24. ru_stylometry-0.1.1/src/ru_stylometry/features/repetition.py +43 -0
  25. ru_stylometry-0.1.1/src/ru_stylometry/features/syntax.py +64 -0
  26. ru_stylometry-0.1.1/src/ru_stylometry/features/typography.py +57 -0
  27. ru_stylometry-0.1.1/src/ru_stylometry/model.py +196 -0
  28. ru_stylometry-0.1.1/src/ru_stylometry/perturb.py +116 -0
  29. ru_stylometry-0.1.1/src/ru_stylometry/py.typed +1 -0
  30. ru_stylometry-0.1.1/src/ru_stylometry/vectorizer.py +91 -0
  31. ru_stylometry-0.1.1/src/ru_stylometry.egg-info/PKG-INFO +274 -0
  32. ru_stylometry-0.1.1/src/ru_stylometry.egg-info/SOURCES.txt +49 -0
  33. ru_stylometry-0.1.1/src/ru_stylometry.egg-info/dependency_links.txt +1 -0
  34. ru_stylometry-0.1.1/src/ru_stylometry.egg-info/entry_points.txt +2 -0
  35. ru_stylometry-0.1.1/src/ru_stylometry.egg-info/requires.txt +20 -0
  36. ru_stylometry-0.1.1/src/ru_stylometry.egg-info/top_level.txt +1 -0
  37. ru_stylometry-0.1.1/tests/data/sample.txt +1 -0
  38. ru_stylometry-0.1.1/tests/data/toy.csv +41 -0
  39. ru_stylometry-0.1.1/tests/test_cli.py +52 -0
  40. ru_stylometry-0.1.1/tests/test_cli_models.py +96 -0
  41. ru_stylometry-0.1.1/tests/test_evaluation.py +139 -0
  42. ru_stylometry-0.1.1/tests/test_explain.py +101 -0
  43. ru_stylometry-0.1.1/tests/test_features.py +254 -0
  44. ru_stylometry-0.1.1/tests/test_max_words.py +65 -0
  45. ru_stylometry-0.1.1/tests/test_model.py +123 -0
  46. ru_stylometry-0.1.1/tests/test_morph_cache.py +23 -0
  47. ru_stylometry-0.1.1/tests/test_perturb.py +88 -0
  48. ru_stylometry-0.1.1/tests/test_registry.py +109 -0
  49. ru_stylometry-0.1.1/tests/test_stats.py +55 -0
  50. ru_stylometry-0.1.1/tests/test_text.py +50 -0
  51. ru_stylometry-0.1.1/tests/test_vectorizer.py +79 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Dinis
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1 @@
1
+ recursive-include tests *.py *.csv *.txt
@@ -0,0 +1,274 @@
1
+ Metadata-Version: 2.4
2
+ Name: ru-stylometry
3
+ Version: 0.1.1
4
+ Summary: Interpretable stylometric features for Russian text and tools for detecting AI-generated text.
5
+ Author: Dinis
6
+ License-Expression: MIT
7
+ Project-URL: Homepage, https://github.com/dinis-a/ru-stylometry
8
+ Project-URL: Repository, https://github.com/dinis-a/ru-stylometry
9
+ Project-URL: Issues, https://github.com/dinis-a/ru-stylometry/issues
10
+ Keywords: nlp,stylometry,russian,feature-extraction,text-classification,ai-detection,scikit-learn
11
+ Classifier: Development Status :: 3 - Alpha
12
+ Classifier: Intended Audience :: Developers
13
+ Classifier: Intended Audience :: Education
14
+ Classifier: Intended Audience :: Science/Research
15
+ Classifier: Natural Language :: Russian
16
+ Classifier: Operating System :: OS Independent
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.9
19
+ Classifier: Programming Language :: Python :: 3.10
20
+ Classifier: Programming Language :: Python :: 3.11
21
+ Classifier: Programming Language :: Python :: 3.12
22
+ Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
23
+ Classifier: Topic :: Text Processing :: Linguistic
24
+ Classifier: Typing :: Typed
25
+ Requires-Python: >=3.9
26
+ Description-Content-Type: text/markdown
27
+ License-File: LICENSE
28
+ Requires-Dist: numpy>=1.21
29
+ Requires-Dist: joblib>=1.1
30
+ Requires-Dist: scikit-learn>=1.2
31
+ Requires-Dist: pymorphy3>=1.2
32
+ Requires-Dist: pymorphy3-dicts-ru
33
+ Provides-Extra: fast
34
+ Requires-Dist: pymorphy3[fast]; extra == "fast"
35
+ Provides-Extra: experiments
36
+ Requires-Dist: pandas>=2.0; extra == "experiments"
37
+ Requires-Dist: pyarrow>=14; extra == "experiments"
38
+ Requires-Dist: huggingface_hub>=0.20; extra == "experiments"
39
+ Requires-Dist: tabulate; extra == "experiments"
40
+ Provides-Extra: dev
41
+ Requires-Dist: pytest>=7.0; extra == "dev"
42
+ Requires-Dist: isort>=5.0; extra == "dev"
43
+ Requires-Dist: black>=23.0; extra == "dev"
44
+ Requires-Dist: build>=1.0; extra == "dev"
45
+ Dynamic: license-file
46
+
47
+ # ru-stylometry
48
+
49
+ Interpretable stylometric features for Russian text, and a classifier that tells human-written text
50
+ from machine-generated text (and, if asked, the model family behind it) with an explanation of
51
+ every decision.
52
+
53
+ > **Status: alpha.** The feature extractor, the scikit-learn transformer, the classifier with
54
+ > explanations, the evaluation tools and the command-line interface are implemented and tested. The
55
+ > package is on [PyPI](https://pypi.org/project/ru-stylometry/); test builds are published to
56
+ > [TestPyPI](https://test.pypi.org/project/ru-stylometry/).
57
+
58
+ It continues [`stylometric-ai-detector`](https://pypi.org/project/stylometric-ai-detector/), an
59
+ English baseline built on 8 surface features. Those features are mostly raw counts that reflect
60
+ text length, and nothing in them is specific to Russian. `ru-stylometry` replaces them with a
61
+ documented vector of 66 features that are normalised by text length and use Russian morphology.
62
+
63
+ ## Installation
64
+
65
+ ```bash
66
+ pip install ru-stylometry # library and command-line tool
67
+ pip install "ru-stylometry[fast]" # + C backend that speeds up morphological analysis
68
+ pip install "ru-stylometry[experiments]" # + pandas, pyarrow, ...: reading Parquet tables
69
+ ```
70
+
71
+ The latest code can be installed straight from GitHub, and a clone of the repository can be
72
+ installed for development:
73
+
74
+ ```bash
75
+ pip install git+https://github.com/dinis-a/ru-stylometry.git # the latest code from GitHub
76
+ pip install -e .[dev] # from a clone: + pytest, black, isort, build
77
+ ```
78
+
79
+ Test builds come from TestPyPI; the second index supplies the dependencies:
80
+
81
+ ```bash
82
+ pip install --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ ru-stylometry
83
+ ```
84
+
85
+ Python 3.9+ is required. Dependencies: numpy, scikit-learn, joblib, pymorphy3.
86
+
87
+ ## Quick start
88
+
89
+ ### Features
90
+
91
+ ```python
92
+ from ru_stylometry import StylometricVectorizer, extract_features, feature_names
93
+
94
+ features = extract_features("Мама мыла раму. Рама была чистой!")
95
+ features["avg_sentence_len"] # 3.0
96
+ features["exclam_per100w"] # 16.67
97
+ features["noun_ratio"] # 0.67
98
+
99
+ len(feature_names()) # 66
100
+
101
+ # scikit-learn transformer: iterable of texts -> (n_texts, n_features) matrix
102
+ X = StylometricVectorizer(max_words=500, n_jobs=-1).fit_transform(["Первый текст.", "Второй текст."])
103
+ ```
104
+
105
+ Pick feature groups explicitly, for example for an ablation study:
106
+
107
+ ```python
108
+ extract_features(text, groups=["char", "lexical", "syntax"])
109
+ StylometricVectorizer(groups=["morph"])
110
+ ```
111
+
112
+ The order of the columns is fixed by the registry, not by the order of `groups`, so the same
113
+ selection always gives the same layout. Blank or non-string input gives an all-zero vector.
114
+
115
+ ### Classifier
116
+
117
+ ```python
118
+ from ru_stylometry import StylometricClassifier
119
+
120
+ model = StylometricClassifier(estimator="hgb", class_weight="balanced")
121
+ model.fit(train_texts, train_labels) # labels: any strings, two or more classes
122
+ model.predict(["Текст для проверки."])
123
+ model.predict_proba(["Текст для проверки."])
124
+
125
+ model.explain("Текст для проверки.", top_k=5)
126
+ # effect of every feature and feature group on the evidence for the predicted class
127
+
128
+ model.save("model.joblib")
129
+ StylometricClassifier.load("model.joblib") # load only files you trust: the format is pickle
130
+ ```
131
+
132
+ The *effect* of a feature is the fall of the class log-odds when the feature is replaced by its
133
+ typical value (the training median); positive effects push towards the class. Log-odds are used
134
+ because the probability of a confident decision is saturated and hides which features carry the
135
+ evidence. Effects are not additive, because correlated features share the credit; the effect of a
136
+ whole group is therefore reported separately.
137
+
138
+ Output for a machine-generated review from the LLMTrace test split, with a model trained on
139
+ 30,000 texts of the same corpus:
140
+
141
+ ```text
142
+ $ ru-stylometry explain --file review.txt -m model.joblib --top 3
143
+ label: ai (probability 0.998)
144
+ effect on the log-odds of class 'ai' (positive: towards it):
145
+ lemma_mattr_50 +0.82 value 0.934, typical 0.857 [morph]
146
+ conj_ratio +0.82 value 0.116, typical 0.084 [morph]
147
+ verb_pres_share +0.68 value 1.000, typical 0.429 [morph]
148
+ by group: morph +4.59, lexical +0.79, char +0.51, readability +0.43, syntax +0.21, ...
149
+ ```
150
+
151
+ ## Command line
152
+
153
+ ```bash
154
+ ru-stylometry features "Мама мыла раму." --groups length syntax # JSON to stdout
155
+ ru-stylometry list-features --format markdown # the feature catalogue
156
+
157
+ ru-stylometry train corpus.parquet -o model.joblib --estimator hgb --balanced --valid dev.parquet
158
+ ru-stylometry evaluate test.parquet -m model.joblib # metrics and time per document
159
+ ru-stylometry predict "Текст для проверки." -m model.joblib
160
+ ru-stylometry explain "Текст для проверки." -m model.joblib --top 10
161
+ ```
162
+
163
+ Tables may be `.csv`, `.jsonl` or `.parquet` with a text column and a label column
164
+ (`--text-column`, `--label-column`).
165
+
166
+ ## Feature groups
167
+
168
+ | Group | Features | Default | What it measures |
169
+ |---------------|---------:|:-------:|-------------------------------------------------------------------|
170
+ | `char` | 7 | yes | letters, digits, punctuation, symbols, capitals, Latin, use of ё |
171
+ | `typography` | 11 | yes | punctuation density; dash, quotation-mark and ellipsis habits |
172
+ | `lexical` | 7 | yes | word length; MATTR, MTLD, Yule's K, hapax share |
173
+ | `syntax` | 9 | yes | sentence length distribution, function words, negation, openers |
174
+ | `readability` | 3 | yes | syllables per word, polysyllabic words, LIX |
175
+ | `repetition` | 3 | yes | deflate compression ratio, repeated bigrams and trigrams |
176
+ | `morph` | 26 | yes | part of speech, case, tense, person, unknown words, lemma variety |
177
+ | `formatting` | 3 | no | line breaks, list items, Markdown symbols |
178
+ | `length` | 4 | no | raw counts of characters, words, sentences, lines |
179
+
180
+ Every feature has a name and a one-line definition; `ru-stylometry list-features --format markdown`
181
+ prints the whole catalogue.
182
+
183
+ `formatting` and `length` are opt-in because they depend on how a corpus was collected more than on
184
+ style: texts of different classes often differ in length and layout for reasons that have nothing
185
+ to do with authorship.
186
+
187
+ ## Evaluation, explanation and robustness
188
+
189
+ | module | what it offers |
190
+ |:--|:--|
191
+ | `ru_stylometry.evaluation` | accuracy, macro-F1, precision and recall of every class, confusion matrix, time per document, bootstrap intervals that resample groups of related texts |
192
+ | `ru_stylometry.explain` | effect of every feature and group on one decision; permutation importance of whole feature groups |
193
+ | `ru_stylometry.perturb` | text transformations for robustness checks: typography, Markdown, discourse markers, sentence order, typos |
194
+
195
+ Macro averaging counts every class equally, so a weak result on a rare source is not hidden by the
196
+ large classes.
197
+
198
+ ## How well it works
199
+
200
+ Measured by the authors (the experiment scripts are not part of this repository) on the Russian
201
+ part of the LLMTrace corpus ([arXiv:2509.21269](https://arxiv.org/abs/2509.21269): 340,197 texts by
202
+ humans and 31 language models in eight genres). The official test split has 52,521 texts that share
203
+ no topic with the training split; the classifier is gradient boosting on the first 500 words of
204
+ every text:
205
+
206
+ | features | number | accuracy | macro-F1 | AUC |
207
+ |:--|--:|--:|--:|--:|
208
+ | length of the text only | 2 | 0.705 | 0.684 | 0.754 |
209
+ | the 8 features of `stylometric-ai-detector` | 8 | 0.751 | 0.735 | 0.820 |
210
+ | **`ru-stylometry`, default groups** | 66 | **0.883** | **0.879** | **0.954** |
211
+
212
+ The 95% interval of the accuracy of the 66 features is [0.881, 0.886]; it resamples whole topics,
213
+ because texts about the same subject are not independent. For the 15 classes (humans and 14 model
214
+ families) the macro-F1 is 0.446 against 0.195 for the 8 legacy features.
215
+
216
+ - **A fine-tuned language model is stronger.** `rubert-base-cased` fine-tuned on the same split
217
+ reaches an accuracy of 0.953, about 7 points more. The stylometric classifier is a 1.2 MB file, needs
218
+ no GPU (about 3.4 ms per text on one CPU core with a warm cache, around 290 texts per second) and
219
+ explains every decision. Adding the 66 features to the outputs of the language model gave 1.4-2.6
220
+ points of macro-F1 over the model families and about 0.1 point of accuracy for human vs machine.
221
+ - **The decision threshold does not carry over to another genre.** Trained on LLMTrace and tested on
222
+ scientific abstracts (AINL-Eval 2025, [arXiv:2508.09622](https://arxiv.org/abs/2508.09622)), the
223
+ ranking of the texts partly transfers (AUC 0.769) but only 36% of the human abstracts are
224
+ recognised: formal human text is taken for machine text. Calibrate the threshold on the genre you
225
+ are going to classify.
226
+
227
+ ## Design notes
228
+
229
+ - **No raw counts by default.** Features are shares, densities per 100 words, means, or indices
230
+ designed to be comparable between texts of different length (MATTR and MTLD instead of the plain
231
+ type-token ratio).
232
+ - **Morphology comes from pymorphy3**, using the most probable parse of each word without context.
233
+ Analyses are cached per word, so a word is parsed once per process. Importing the package does
234
+ not load the dictionaries; the first use of the `morph` group does.
235
+ - **Sentence splitting is rule-based**: it knows common Russian abbreviations (т.д., г., ул.),
236
+ initials, ellipses, closing quotes and list items.
237
+ - **Lexicons are heuristics.** The lists of function words and discourse markers are hand-made and
238
+ small; treat the features built on them as interpretable markers rather than measurements.
239
+ - Flesch-type readability formulas are left out on purpose: they are linear combinations of
240
+ features that are already in the vector, and published Russian coefficients differ.
241
+
242
+ ## Limitations
243
+
244
+ - Short texts (below roughly 50 words) give noisy lexical-diversity and sentence-rhythm values.
245
+ - Homonymous word forms are resolved by frequency, not by context, so part-of-speech shares are
246
+ approximate.
247
+ - Stylometric features describe style, not truth or authorship in a legal sense: a classifier built
248
+ on them gives a probability that depends on the corpus it was trained on, and it should not be
249
+ used as the only evidence against a person.
250
+
251
+ ## Roadmap
252
+
253
+ - a stacking classifier as a class of the library: boosting on the outputs of a fine-tuned language
254
+ model and the 66 features. In the authors' experiments it did as well as or better than training
255
+ the two jointly inside one network, so it is the design to implement;
256
+ - an open-set decision for texts of unseen generators.
257
+
258
+ ## Development
259
+
260
+ ```bash
261
+ pip install -e .[dev]
262
+ pytest # the test suite in tests/
263
+ isort . && black . # style: black and isort, line length 100 (see pyproject.toml)
264
+ python -m build # sdist and wheel in dist/
265
+ ```
266
+
267
+ Pushing a version tag (`git tag v0.1.1 && git push origin v0.1.1`) starts the GitHub Actions
268
+ workflow: it checks the style, runs the tests on Python 3.9-3.12, builds the package and uploads it
269
+ to TestPyPI with the repository secret `TESTPYPI_TOKEN`. TestPyPI accepts a version only once, so
270
+ raise `__version__` in `src/ru_stylometry/__init__.py` before tagging.
271
+
272
+ ## License
273
+
274
+ MIT, see the `LICENSE` file.
@@ -0,0 +1,228 @@
1
+ # ru-stylometry
2
+
3
+ Interpretable stylometric features for Russian text, and a classifier that tells human-written text
4
+ from machine-generated text (and, if asked, the model family behind it) with an explanation of
5
+ every decision.
6
+
7
+ > **Status: alpha.** The feature extractor, the scikit-learn transformer, the classifier with
8
+ > explanations, the evaluation tools and the command-line interface are implemented and tested. The
9
+ > package is on [PyPI](https://pypi.org/project/ru-stylometry/); test builds are published to
10
+ > [TestPyPI](https://test.pypi.org/project/ru-stylometry/).
11
+
12
+ It continues [`stylometric-ai-detector`](https://pypi.org/project/stylometric-ai-detector/), an
13
+ English baseline built on 8 surface features. Those features are mostly raw counts that reflect
14
+ text length, and nothing in them is specific to Russian. `ru-stylometry` replaces them with a
15
+ documented vector of 66 features that are normalised by text length and use Russian morphology.
16
+
17
+ ## Installation
18
+
19
+ ```bash
20
+ pip install ru-stylometry # library and command-line tool
21
+ pip install "ru-stylometry[fast]" # + C backend that speeds up morphological analysis
22
+ pip install "ru-stylometry[experiments]" # + pandas, pyarrow, ...: reading Parquet tables
23
+ ```
24
+
25
+ The latest code can be installed straight from GitHub, and a clone of the repository can be
26
+ installed for development:
27
+
28
+ ```bash
29
+ pip install git+https://github.com/dinis-a/ru-stylometry.git # the latest code from GitHub
30
+ pip install -e .[dev] # from a clone: + pytest, black, isort, build
31
+ ```
32
+
33
+ Test builds come from TestPyPI; the second index supplies the dependencies:
34
+
35
+ ```bash
36
+ pip install --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ ru-stylometry
37
+ ```
38
+
39
+ Python 3.9+ is required. Dependencies: numpy, scikit-learn, joblib, pymorphy3.
40
+
41
+ ## Quick start
42
+
43
+ ### Features
44
+
45
+ ```python
46
+ from ru_stylometry import StylometricVectorizer, extract_features, feature_names
47
+
48
+ features = extract_features("Мама мыла раму. Рама была чистой!")
49
+ features["avg_sentence_len"] # 3.0
50
+ features["exclam_per100w"] # 16.67
51
+ features["noun_ratio"] # 0.67
52
+
53
+ len(feature_names()) # 66
54
+
55
+ # scikit-learn transformer: iterable of texts -> (n_texts, n_features) matrix
56
+ X = StylometricVectorizer(max_words=500, n_jobs=-1).fit_transform(["Первый текст.", "Второй текст."])
57
+ ```
58
+
59
+ Pick feature groups explicitly, for example for an ablation study:
60
+
61
+ ```python
62
+ extract_features(text, groups=["char", "lexical", "syntax"])
63
+ StylometricVectorizer(groups=["morph"])
64
+ ```
65
+
66
+ The order of the columns is fixed by the registry, not by the order of `groups`, so the same
67
+ selection always gives the same layout. Blank or non-string input gives an all-zero vector.
68
+
69
+ ### Classifier
70
+
71
+ ```python
72
+ from ru_stylometry import StylometricClassifier
73
+
74
+ model = StylometricClassifier(estimator="hgb", class_weight="balanced")
75
+ model.fit(train_texts, train_labels) # labels: any strings, two or more classes
76
+ model.predict(["Текст для проверки."])
77
+ model.predict_proba(["Текст для проверки."])
78
+
79
+ model.explain("Текст для проверки.", top_k=5)
80
+ # effect of every feature and feature group on the evidence for the predicted class
81
+
82
+ model.save("model.joblib")
83
+ StylometricClassifier.load("model.joblib") # load only files you trust: the format is pickle
84
+ ```
85
+
86
+ The *effect* of a feature is the fall of the class log-odds when the feature is replaced by its
87
+ typical value (the training median); positive effects push towards the class. Log-odds are used
88
+ because the probability of a confident decision is saturated and hides which features carry the
89
+ evidence. Effects are not additive, because correlated features share the credit; the effect of a
90
+ whole group is therefore reported separately.
91
+
92
+ Output for a machine-generated review from the LLMTrace test split, with a model trained on
93
+ 30,000 texts of the same corpus:
94
+
95
+ ```text
96
+ $ ru-stylometry explain --file review.txt -m model.joblib --top 3
97
+ label: ai (probability 0.998)
98
+ effect on the log-odds of class 'ai' (positive: towards it):
99
+ lemma_mattr_50 +0.82 value 0.934, typical 0.857 [morph]
100
+ conj_ratio +0.82 value 0.116, typical 0.084 [morph]
101
+ verb_pres_share +0.68 value 1.000, typical 0.429 [morph]
102
+ by group: morph +4.59, lexical +0.79, char +0.51, readability +0.43, syntax +0.21, ...
103
+ ```
104
+
105
+ ## Command line
106
+
107
+ ```bash
108
+ ru-stylometry features "Мама мыла раму." --groups length syntax # JSON to stdout
109
+ ru-stylometry list-features --format markdown # the feature catalogue
110
+
111
+ ru-stylometry train corpus.parquet -o model.joblib --estimator hgb --balanced --valid dev.parquet
112
+ ru-stylometry evaluate test.parquet -m model.joblib # metrics and time per document
113
+ ru-stylometry predict "Текст для проверки." -m model.joblib
114
+ ru-stylometry explain "Текст для проверки." -m model.joblib --top 10
115
+ ```
116
+
117
+ Tables may be `.csv`, `.jsonl` or `.parquet` with a text column and a label column
118
+ (`--text-column`, `--label-column`).
119
+
120
+ ## Feature groups
121
+
122
+ | Group | Features | Default | What it measures |
123
+ |---------------|---------:|:-------:|-------------------------------------------------------------------|
124
+ | `char` | 7 | yes | letters, digits, punctuation, symbols, capitals, Latin, use of ё |
125
+ | `typography` | 11 | yes | punctuation density; dash, quotation-mark and ellipsis habits |
126
+ | `lexical` | 7 | yes | word length; MATTR, MTLD, Yule's K, hapax share |
127
+ | `syntax` | 9 | yes | sentence length distribution, function words, negation, openers |
128
+ | `readability` | 3 | yes | syllables per word, polysyllabic words, LIX |
129
+ | `repetition` | 3 | yes | deflate compression ratio, repeated bigrams and trigrams |
130
+ | `morph` | 26 | yes | part of speech, case, tense, person, unknown words, lemma variety |
131
+ | `formatting` | 3 | no | line breaks, list items, Markdown symbols |
132
+ | `length` | 4 | no | raw counts of characters, words, sentences, lines |
133
+
134
+ Every feature has a name and a one-line definition; `ru-stylometry list-features --format markdown`
135
+ prints the whole catalogue.
136
+
137
+ `formatting` and `length` are opt-in because they depend on how a corpus was collected more than on
138
+ style: texts of different classes often differ in length and layout for reasons that have nothing
139
+ to do with authorship.
140
+
141
+ ## Evaluation, explanation and robustness
142
+
143
+ | module | what it offers |
144
+ |:--|:--|
145
+ | `ru_stylometry.evaluation` | accuracy, macro-F1, precision and recall of every class, confusion matrix, time per document, bootstrap intervals that resample groups of related texts |
146
+ | `ru_stylometry.explain` | effect of every feature and group on one decision; permutation importance of whole feature groups |
147
+ | `ru_stylometry.perturb` | text transformations for robustness checks: typography, Markdown, discourse markers, sentence order, typos |
148
+
149
+ Macro averaging counts every class equally, so a weak result on a rare source is not hidden by the
150
+ large classes.
151
+
152
+ ## How well it works
153
+
154
+ Measured by the authors (the experiment scripts are not part of this repository) on the Russian
155
+ part of the LLMTrace corpus ([arXiv:2509.21269](https://arxiv.org/abs/2509.21269): 340,197 texts by
156
+ humans and 31 language models in eight genres). The official test split has 52,521 texts that share
157
+ no topic with the training split; the classifier is gradient boosting on the first 500 words of
158
+ every text:
159
+
160
+ | features | number | accuracy | macro-F1 | AUC |
161
+ |:--|--:|--:|--:|--:|
162
+ | length of the text only | 2 | 0.705 | 0.684 | 0.754 |
163
+ | the 8 features of `stylometric-ai-detector` | 8 | 0.751 | 0.735 | 0.820 |
164
+ | **`ru-stylometry`, default groups** | 66 | **0.883** | **0.879** | **0.954** |
165
+
166
+ The 95% interval of the accuracy of the 66 features is [0.881, 0.886]; it resamples whole topics,
167
+ because texts about the same subject are not independent. For the 15 classes (humans and 14 model
168
+ families) the macro-F1 is 0.446 against 0.195 for the 8 legacy features.
169
+
170
+ - **A fine-tuned language model is stronger.** `rubert-base-cased` fine-tuned on the same split
171
+ reaches an accuracy of 0.953, about 7 points more. The stylometric classifier is a 1.2 MB file, needs
172
+ no GPU (about 3.4 ms per text on one CPU core with a warm cache, around 290 texts per second) and
173
+ explains every decision. Adding the 66 features to the outputs of the language model gave 1.4-2.6
174
+ points of macro-F1 over the model families and about 0.1 point of accuracy for human vs machine.
175
+ - **The decision threshold does not carry over to another genre.** Trained on LLMTrace and tested on
176
+ scientific abstracts (AINL-Eval 2025, [arXiv:2508.09622](https://arxiv.org/abs/2508.09622)), the
177
+ ranking of the texts partly transfers (AUC 0.769) but only 36% of the human abstracts are
178
+ recognised: formal human text is taken for machine text. Calibrate the threshold on the genre you
179
+ are going to classify.
180
+
181
+ ## Design notes
182
+
183
+ - **No raw counts by default.** Features are shares, densities per 100 words, means, or indices
184
+ designed to be comparable between texts of different length (MATTR and MTLD instead of the plain
185
+ type-token ratio).
186
+ - **Morphology comes from pymorphy3**, using the most probable parse of each word without context.
187
+ Analyses are cached per word, so a word is parsed once per process. Importing the package does
188
+ not load the dictionaries; the first use of the `morph` group does.
189
+ - **Sentence splitting is rule-based**: it knows common Russian abbreviations (т.д., г., ул.),
190
+ initials, ellipses, closing quotes and list items.
191
+ - **Lexicons are heuristics.** The lists of function words and discourse markers are hand-made and
192
+ small; treat the features built on them as interpretable markers rather than measurements.
193
+ - Flesch-type readability formulas are left out on purpose: they are linear combinations of
194
+ features that are already in the vector, and published Russian coefficients differ.
195
+
196
+ ## Limitations
197
+
198
+ - Short texts (below roughly 50 words) give noisy lexical-diversity and sentence-rhythm values.
199
+ - Homonymous word forms are resolved by frequency, not by context, so part-of-speech shares are
200
+ approximate.
201
+ - Stylometric features describe style, not truth or authorship in a legal sense: a classifier built
202
+ on them gives a probability that depends on the corpus it was trained on, and it should not be
203
+ used as the only evidence against a person.
204
+
205
+ ## Roadmap
206
+
207
+ - a stacking classifier as a class of the library: boosting on the outputs of a fine-tuned language
208
+ model and the 66 features. In the authors' experiments it did as well as or better than training
209
+ the two jointly inside one network, so it is the design to implement;
210
+ - an open-set decision for texts of unseen generators.
211
+
212
+ ## Development
213
+
214
+ ```bash
215
+ pip install -e .[dev]
216
+ pytest # the test suite in tests/
217
+ isort . && black . # style: black and isort, line length 100 (see pyproject.toml)
218
+ python -m build # sdist and wheel in dist/
219
+ ```
220
+
221
+ Pushing a version tag (`git tag v0.1.1 && git push origin v0.1.1`) starts the GitHub Actions
222
+ workflow: it checks the style, runs the tests on Python 3.9-3.12, builds the package and uploads it
223
+ to TestPyPI with the repository secret `TESTPYPI_TOKEN`. TestPyPI accepts a version only once, so
224
+ raise `__version__` in `src/ru_stylometry/__init__.py` before tagging.
225
+
226
+ ## License
227
+
228
+ MIT, see the `LICENSE` file.
@@ -0,0 +1,92 @@
1
+ [build-system]
2
+ requires = ["setuptools>=77", "wheel"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "ru-stylometry"
7
+ dynamic = ["version"]
8
+ description = "Interpretable stylometric features for Russian text and tools for detecting AI-generated text."
9
+ readme = "README.md"
10
+ license = "MIT"
11
+ license-files = ["LICENSE"]
12
+ requires-python = ">=3.9"
13
+ authors = [
14
+ {name = "Dinis"},
15
+ ]
16
+ keywords = [
17
+ "nlp",
18
+ "stylometry",
19
+ "russian",
20
+ "feature-extraction",
21
+ "text-classification",
22
+ "ai-detection",
23
+ "scikit-learn",
24
+ ]
25
+ classifiers = [
26
+ "Development Status :: 3 - Alpha",
27
+ "Intended Audience :: Developers",
28
+ "Intended Audience :: Education",
29
+ "Intended Audience :: Science/Research",
30
+ "Natural Language :: Russian",
31
+ "Operating System :: OS Independent",
32
+ "Programming Language :: Python :: 3",
33
+ "Programming Language :: Python :: 3.9",
34
+ "Programming Language :: Python :: 3.10",
35
+ "Programming Language :: Python :: 3.11",
36
+ "Programming Language :: Python :: 3.12",
37
+ "Topic :: Scientific/Engineering :: Artificial Intelligence",
38
+ "Topic :: Text Processing :: Linguistic",
39
+ "Typing :: Typed",
40
+ ]
41
+ dependencies = [
42
+ "numpy>=1.21",
43
+ "joblib>=1.1",
44
+ "scikit-learn>=1.2",
45
+ "pymorphy3>=1.2",
46
+ "pymorphy3-dicts-ru",
47
+ ]
48
+
49
+ [project.optional-dependencies]
50
+ # C extension that roughly halves the cost of parsing words missing from the cache.
51
+ fast = ["pymorphy3[fast]"]
52
+ # Packages used by the scripts in experiments/ and for reading Parquet files.
53
+ experiments = ["pandas>=2.0", "pyarrow>=14", "huggingface_hub>=0.20", "tabulate"]
54
+ dev = [
55
+ "pytest>=7.0",
56
+ "isort>=5.0",
57
+ "black>=23.0",
58
+ "build>=1.0",
59
+ ]
60
+
61
+ [project.urls]
62
+ Homepage = "https://github.com/dinis-a/ru-stylometry"
63
+ Repository = "https://github.com/dinis-a/ru-stylometry"
64
+ Issues = "https://github.com/dinis-a/ru-stylometry/issues"
65
+
66
+ [project.scripts]
67
+ ru-stylometry = "ru_stylometry.cli:main"
68
+
69
+ [tool.setuptools.dynamic]
70
+ version = {attr = "ru_stylometry.__version__"}
71
+
72
+ [tool.setuptools.packages.find]
73
+ where = ["src"]
74
+
75
+ [tool.setuptools.package-data]
76
+ ru_stylometry = ["py.typed"]
77
+
78
+ [tool.isort]
79
+ profile = "black"
80
+ line_length = 100
81
+ src_paths = ["src", "tests"]
82
+ # Notebook cells keep their own import order (some imports must follow sys.path changes).
83
+ skip_glob = ["notebooks/cells/*", ".cache/*"]
84
+
85
+ [tool.black]
86
+ line-length = 100
87
+ target-version = ["py39"]
88
+ # Tool caches and scratch files live in .cache/ and are not part of the project.
89
+ extend-exclude = "/\\.cache/"
90
+
91
+ [tool.pytest.ini_options]
92
+ testpaths = ["tests"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,34 @@
1
+ """ru-stylometry — interpretable stylometric features for Russian text.
2
+
3
+ Public API:
4
+ - :func:`extract_features` — named stylometric features of one text.
5
+ - :func:`feature_names` and :func:`describe_features` — the feature catalogue.
6
+ - :class:`StylometricVectorizer` — scikit-learn transformer: texts -> feature matrix.
7
+ - :class:`StylometricClassifier` — classifier over those features with explanations.
8
+
9
+ Features are grouped by the level of analysis (characters, typography, vocabulary, syntax,
10
+ readability, repetition, morphology); see :data:`ALL_GROUPS` and :data:`DEFAULT_GROUPS`.
11
+ The submodules ``evaluation``, ``explain`` and ``perturb`` hold the quality criteria, the
12
+ explanations and the text transformations used in robustness checks.
13
+ """
14
+
15
+ from ru_stylometry.features import (
16
+ ALL_GROUPS,
17
+ DEFAULT_GROUPS,
18
+ describe_features,
19
+ extract_features,
20
+ feature_names,
21
+ )
22
+ from ru_stylometry.model import StylometricClassifier
23
+ from ru_stylometry.vectorizer import StylometricVectorizer
24
+
25
+ __all__ = [
26
+ "ALL_GROUPS",
27
+ "DEFAULT_GROUPS",
28
+ "StylometricClassifier",
29
+ "StylometricVectorizer",
30
+ "describe_features",
31
+ "extract_features",
32
+ "feature_names",
33
+ ]
34
+ __version__ = "0.1.1"
@@ -0,0 +1,5 @@
1
+ """Allow ``python -m ru_stylometry``."""
2
+
3
+ from ru_stylometry.cli import main
4
+
5
+ raise SystemExit(main())