ru-stylometry 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- ru_stylometry-0.1.1/LICENSE +21 -0
- ru_stylometry-0.1.1/MANIFEST.in +1 -0
- ru_stylometry-0.1.1/PKG-INFO +274 -0
- ru_stylometry-0.1.1/README.md +228 -0
- ru_stylometry-0.1.1/pyproject.toml +92 -0
- ru_stylometry-0.1.1/setup.cfg +4 -0
- ru_stylometry-0.1.1/src/ru_stylometry/__init__.py +34 -0
- ru_stylometry-0.1.1/src/ru_stylometry/__main__.py +5 -0
- ru_stylometry-0.1.1/src/ru_stylometry/_lexicons.py +76 -0
- ru_stylometry-0.1.1/src/ru_stylometry/_morph.py +41 -0
- ru_stylometry-0.1.1/src/ru_stylometry/_stats.py +91 -0
- ru_stylometry-0.1.1/src/ru_stylometry/_text.py +99 -0
- ru_stylometry-0.1.1/src/ru_stylometry/cli.py +269 -0
- ru_stylometry-0.1.1/src/ru_stylometry/document.py +66 -0
- ru_stylometry-0.1.1/src/ru_stylometry/evaluation.py +267 -0
- ru_stylometry-0.1.1/src/ru_stylometry/explain.py +134 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/__init__.py +130 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/char.py +54 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/formatting.py +34 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/length.py +30 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/lexical.py +37 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/morph.py +109 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/readability.py +37 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/repetition.py +43 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/syntax.py +64 -0
- ru_stylometry-0.1.1/src/ru_stylometry/features/typography.py +57 -0
- ru_stylometry-0.1.1/src/ru_stylometry/model.py +196 -0
- ru_stylometry-0.1.1/src/ru_stylometry/perturb.py +116 -0
- ru_stylometry-0.1.1/src/ru_stylometry/py.typed +1 -0
- ru_stylometry-0.1.1/src/ru_stylometry/vectorizer.py +91 -0
- ru_stylometry-0.1.1/src/ru_stylometry.egg-info/PKG-INFO +274 -0
- ru_stylometry-0.1.1/src/ru_stylometry.egg-info/SOURCES.txt +49 -0
- ru_stylometry-0.1.1/src/ru_stylometry.egg-info/dependency_links.txt +1 -0
- ru_stylometry-0.1.1/src/ru_stylometry.egg-info/entry_points.txt +2 -0
- ru_stylometry-0.1.1/src/ru_stylometry.egg-info/requires.txt +20 -0
- ru_stylometry-0.1.1/src/ru_stylometry.egg-info/top_level.txt +1 -0
- ru_stylometry-0.1.1/tests/data/sample.txt +1 -0
- ru_stylometry-0.1.1/tests/data/toy.csv +41 -0
- ru_stylometry-0.1.1/tests/test_cli.py +52 -0
- ru_stylometry-0.1.1/tests/test_cli_models.py +96 -0
- ru_stylometry-0.1.1/tests/test_evaluation.py +139 -0
- ru_stylometry-0.1.1/tests/test_explain.py +101 -0
- ru_stylometry-0.1.1/tests/test_features.py +254 -0
- ru_stylometry-0.1.1/tests/test_max_words.py +65 -0
- ru_stylometry-0.1.1/tests/test_model.py +123 -0
- ru_stylometry-0.1.1/tests/test_morph_cache.py +23 -0
- ru_stylometry-0.1.1/tests/test_perturb.py +88 -0
- ru_stylometry-0.1.1/tests/test_registry.py +109 -0
- ru_stylometry-0.1.1/tests/test_stats.py +55 -0
- ru_stylometry-0.1.1/tests/test_text.py +50 -0
- ru_stylometry-0.1.1/tests/test_vectorizer.py +79 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Dinis
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
recursive-include tests *.py *.csv *.txt
|
|
@@ -0,0 +1,274 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: ru-stylometry
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Interpretable stylometric features for Russian text and tools for detecting AI-generated text.
|
|
5
|
+
Author: Dinis
|
|
6
|
+
License-Expression: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/dinis-a/ru-stylometry
|
|
8
|
+
Project-URL: Repository, https://github.com/dinis-a/ru-stylometry
|
|
9
|
+
Project-URL: Issues, https://github.com/dinis-a/ru-stylometry/issues
|
|
10
|
+
Keywords: nlp,stylometry,russian,feature-extraction,text-classification,ai-detection,scikit-learn
|
|
11
|
+
Classifier: Development Status :: 3 - Alpha
|
|
12
|
+
Classifier: Intended Audience :: Developers
|
|
13
|
+
Classifier: Intended Audience :: Education
|
|
14
|
+
Classifier: Intended Audience :: Science/Research
|
|
15
|
+
Classifier: Natural Language :: Russian
|
|
16
|
+
Classifier: Operating System :: OS Independent
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
22
|
+
Classifier: Topic :: Scientific/Engineering :: Artificial Intelligence
|
|
23
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
24
|
+
Classifier: Typing :: Typed
|
|
25
|
+
Requires-Python: >=3.9
|
|
26
|
+
Description-Content-Type: text/markdown
|
|
27
|
+
License-File: LICENSE
|
|
28
|
+
Requires-Dist: numpy>=1.21
|
|
29
|
+
Requires-Dist: joblib>=1.1
|
|
30
|
+
Requires-Dist: scikit-learn>=1.2
|
|
31
|
+
Requires-Dist: pymorphy3>=1.2
|
|
32
|
+
Requires-Dist: pymorphy3-dicts-ru
|
|
33
|
+
Provides-Extra: fast
|
|
34
|
+
Requires-Dist: pymorphy3[fast]; extra == "fast"
|
|
35
|
+
Provides-Extra: experiments
|
|
36
|
+
Requires-Dist: pandas>=2.0; extra == "experiments"
|
|
37
|
+
Requires-Dist: pyarrow>=14; extra == "experiments"
|
|
38
|
+
Requires-Dist: huggingface_hub>=0.20; extra == "experiments"
|
|
39
|
+
Requires-Dist: tabulate; extra == "experiments"
|
|
40
|
+
Provides-Extra: dev
|
|
41
|
+
Requires-Dist: pytest>=7.0; extra == "dev"
|
|
42
|
+
Requires-Dist: isort>=5.0; extra == "dev"
|
|
43
|
+
Requires-Dist: black>=23.0; extra == "dev"
|
|
44
|
+
Requires-Dist: build>=1.0; extra == "dev"
|
|
45
|
+
Dynamic: license-file
|
|
46
|
+
|
|
47
|
+
# ru-stylometry
|
|
48
|
+
|
|
49
|
+
Interpretable stylometric features for Russian text, and a classifier that tells human-written text
|
|
50
|
+
from machine-generated text (and, if asked, the model family behind it) with an explanation of
|
|
51
|
+
every decision.
|
|
52
|
+
|
|
53
|
+
> **Status: alpha.** The feature extractor, the scikit-learn transformer, the classifier with
|
|
54
|
+
> explanations, the evaluation tools and the command-line interface are implemented and tested. The
|
|
55
|
+
> package is on [PyPI](https://pypi.org/project/ru-stylometry/); test builds are published to
|
|
56
|
+
> [TestPyPI](https://test.pypi.org/project/ru-stylometry/).
|
|
57
|
+
|
|
58
|
+
It continues [`stylometric-ai-detector`](https://pypi.org/project/stylometric-ai-detector/), an
|
|
59
|
+
English baseline built on 8 surface features. Those features are mostly raw counts that reflect
|
|
60
|
+
text length, and nothing in them is specific to Russian. `ru-stylometry` replaces them with a
|
|
61
|
+
documented vector of 66 features that are normalised by text length and use Russian morphology.
|
|
62
|
+
|
|
63
|
+
## Installation
|
|
64
|
+
|
|
65
|
+
```bash
|
|
66
|
+
pip install ru-stylometry # library and command-line tool
|
|
67
|
+
pip install "ru-stylometry[fast]" # + C backend that speeds up morphological analysis
|
|
68
|
+
pip install "ru-stylometry[experiments]" # + pandas, pyarrow, ...: reading Parquet tables
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
The latest code can be installed straight from GitHub, and a clone of the repository can be
|
|
72
|
+
installed for development:
|
|
73
|
+
|
|
74
|
+
```bash
|
|
75
|
+
pip install git+https://github.com/dinis-a/ru-stylometry.git # the latest code from GitHub
|
|
76
|
+
pip install -e .[dev] # from a clone: + pytest, black, isort, build
|
|
77
|
+
```
|
|
78
|
+
|
|
79
|
+
Test builds come from TestPyPI; the second index supplies the dependencies:
|
|
80
|
+
|
|
81
|
+
```bash
|
|
82
|
+
pip install --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ ru-stylometry
|
|
83
|
+
```
|
|
84
|
+
|
|
85
|
+
Python 3.9+ is required. Dependencies: numpy, scikit-learn, joblib, pymorphy3.
|
|
86
|
+
|
|
87
|
+
## Quick start
|
|
88
|
+
|
|
89
|
+
### Features
|
|
90
|
+
|
|
91
|
+
```python
|
|
92
|
+
from ru_stylometry import StylometricVectorizer, extract_features, feature_names
|
|
93
|
+
|
|
94
|
+
features = extract_features("Мама мыла раму. Рама была чистой!")
|
|
95
|
+
features["avg_sentence_len"] # 3.0
|
|
96
|
+
features["exclam_per100w"] # 16.67
|
|
97
|
+
features["noun_ratio"] # 0.67
|
|
98
|
+
|
|
99
|
+
len(feature_names()) # 66
|
|
100
|
+
|
|
101
|
+
# scikit-learn transformer: iterable of texts -> (n_texts, n_features) matrix
|
|
102
|
+
X = StylometricVectorizer(max_words=500, n_jobs=-1).fit_transform(["Первый текст.", "Второй текст."])
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
Pick feature groups explicitly, for example for an ablation study:
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
extract_features(text, groups=["char", "lexical", "syntax"])
|
|
109
|
+
StylometricVectorizer(groups=["morph"])
|
|
110
|
+
```
|
|
111
|
+
|
|
112
|
+
The order of the columns is fixed by the registry, not by the order of `groups`, so the same
|
|
113
|
+
selection always gives the same layout. Blank or non-string input gives an all-zero vector.
|
|
114
|
+
|
|
115
|
+
### Classifier
|
|
116
|
+
|
|
117
|
+
```python
|
|
118
|
+
from ru_stylometry import StylometricClassifier
|
|
119
|
+
|
|
120
|
+
model = StylometricClassifier(estimator="hgb", class_weight="balanced")
|
|
121
|
+
model.fit(train_texts, train_labels) # labels: any strings, two or more classes
|
|
122
|
+
model.predict(["Текст для проверки."])
|
|
123
|
+
model.predict_proba(["Текст для проверки."])
|
|
124
|
+
|
|
125
|
+
model.explain("Текст для проверки.", top_k=5)
|
|
126
|
+
# effect of every feature and feature group on the evidence for the predicted class
|
|
127
|
+
|
|
128
|
+
model.save("model.joblib")
|
|
129
|
+
StylometricClassifier.load("model.joblib") # load only files you trust: the format is pickle
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
The *effect* of a feature is the fall of the class log-odds when the feature is replaced by its
|
|
133
|
+
typical value (the training median); positive effects push towards the class. Log-odds are used
|
|
134
|
+
because the probability of a confident decision is saturated and hides which features carry the
|
|
135
|
+
evidence. Effects are not additive, because correlated features share the credit; the effect of a
|
|
136
|
+
whole group is therefore reported separately.
|
|
137
|
+
|
|
138
|
+
Output for a machine-generated review from the LLMTrace test split, with a model trained on
|
|
139
|
+
30,000 texts of the same corpus:
|
|
140
|
+
|
|
141
|
+
```text
|
|
142
|
+
$ ru-stylometry explain --file review.txt -m model.joblib --top 3
|
|
143
|
+
label: ai (probability 0.998)
|
|
144
|
+
effect on the log-odds of class 'ai' (positive: towards it):
|
|
145
|
+
lemma_mattr_50 +0.82 value 0.934, typical 0.857 [morph]
|
|
146
|
+
conj_ratio +0.82 value 0.116, typical 0.084 [morph]
|
|
147
|
+
verb_pres_share +0.68 value 1.000, typical 0.429 [morph]
|
|
148
|
+
by group: morph +4.59, lexical +0.79, char +0.51, readability +0.43, syntax +0.21, ...
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
## Command line
|
|
152
|
+
|
|
153
|
+
```bash
|
|
154
|
+
ru-stylometry features "Мама мыла раму." --groups length syntax # JSON to stdout
|
|
155
|
+
ru-stylometry list-features --format markdown # the feature catalogue
|
|
156
|
+
|
|
157
|
+
ru-stylometry train corpus.parquet -o model.joblib --estimator hgb --balanced --valid dev.parquet
|
|
158
|
+
ru-stylometry evaluate test.parquet -m model.joblib # metrics and time per document
|
|
159
|
+
ru-stylometry predict "Текст для проверки." -m model.joblib
|
|
160
|
+
ru-stylometry explain "Текст для проверки." -m model.joblib --top 10
|
|
161
|
+
```
|
|
162
|
+
|
|
163
|
+
Tables may be `.csv`, `.jsonl` or `.parquet` with a text column and a label column
|
|
164
|
+
(`--text-column`, `--label-column`).
|
|
165
|
+
|
|
166
|
+
## Feature groups
|
|
167
|
+
|
|
168
|
+
| Group | Features | Default | What it measures |
|
|
169
|
+
|---------------|---------:|:-------:|-------------------------------------------------------------------|
|
|
170
|
+
| `char` | 7 | yes | letters, digits, punctuation, symbols, capitals, Latin, use of ё |
|
|
171
|
+
| `typography` | 11 | yes | punctuation density; dash, quotation-mark and ellipsis habits |
|
|
172
|
+
| `lexical` | 7 | yes | word length; MATTR, MTLD, Yule's K, hapax share |
|
|
173
|
+
| `syntax` | 9 | yes | sentence length distribution, function words, negation, openers |
|
|
174
|
+
| `readability` | 3 | yes | syllables per word, polysyllabic words, LIX |
|
|
175
|
+
| `repetition` | 3 | yes | deflate compression ratio, repeated bigrams and trigrams |
|
|
176
|
+
| `morph` | 26 | yes | part of speech, case, tense, person, unknown words, lemma variety |
|
|
177
|
+
| `formatting` | 3 | no | line breaks, list items, Markdown symbols |
|
|
178
|
+
| `length` | 4 | no | raw counts of characters, words, sentences, lines |
|
|
179
|
+
|
|
180
|
+
Every feature has a name and a one-line definition; `ru-stylometry list-features --format markdown`
|
|
181
|
+
prints the whole catalogue.
|
|
182
|
+
|
|
183
|
+
`formatting` and `length` are opt-in because they depend on how a corpus was collected more than on
|
|
184
|
+
style: texts of different classes often differ in length and layout for reasons that have nothing
|
|
185
|
+
to do with authorship.
|
|
186
|
+
|
|
187
|
+
## Evaluation, explanation and robustness
|
|
188
|
+
|
|
189
|
+
| module | what it offers |
|
|
190
|
+
|:--|:--|
|
|
191
|
+
| `ru_stylometry.evaluation` | accuracy, macro-F1, precision and recall of every class, confusion matrix, time per document, bootstrap intervals that resample groups of related texts |
|
|
192
|
+
| `ru_stylometry.explain` | effect of every feature and group on one decision; permutation importance of whole feature groups |
|
|
193
|
+
| `ru_stylometry.perturb` | text transformations for robustness checks: typography, Markdown, discourse markers, sentence order, typos |
|
|
194
|
+
|
|
195
|
+
Macro averaging counts every class equally, so a weak result on a rare source is not hidden by the
|
|
196
|
+
large classes.
|
|
197
|
+
|
|
198
|
+
## How well it works
|
|
199
|
+
|
|
200
|
+
Measured by the authors (the experiment scripts are not part of this repository) on the Russian
|
|
201
|
+
part of the LLMTrace corpus ([arXiv:2509.21269](https://arxiv.org/abs/2509.21269): 340,197 texts by
|
|
202
|
+
humans and 31 language models in eight genres). The official test split has 52,521 texts that share
|
|
203
|
+
no topic with the training split; the classifier is gradient boosting on the first 500 words of
|
|
204
|
+
every text:
|
|
205
|
+
|
|
206
|
+
| features | number | accuracy | macro-F1 | AUC |
|
|
207
|
+
|:--|--:|--:|--:|--:|
|
|
208
|
+
| length of the text only | 2 | 0.705 | 0.684 | 0.754 |
|
|
209
|
+
| the 8 features of `stylometric-ai-detector` | 8 | 0.751 | 0.735 | 0.820 |
|
|
210
|
+
| **`ru-stylometry`, default groups** | 66 | **0.883** | **0.879** | **0.954** |
|
|
211
|
+
|
|
212
|
+
The 95% interval of the accuracy of the 66 features is [0.881, 0.886]; it resamples whole topics,
|
|
213
|
+
because texts about the same subject are not independent. For the 15 classes (humans and 14 model
|
|
214
|
+
families) the macro-F1 is 0.446 against 0.195 for the 8 legacy features.
|
|
215
|
+
|
|
216
|
+
- **A fine-tuned language model is stronger.** `rubert-base-cased` fine-tuned on the same split
|
|
217
|
+
reaches an accuracy of 0.953, about 7 points more. The stylometric classifier is a 1.2 MB file, needs
|
|
218
|
+
no GPU (about 3.4 ms per text on one CPU core with a warm cache, around 290 texts per second) and
|
|
219
|
+
explains every decision. Adding the 66 features to the outputs of the language model gave 1.4-2.6
|
|
220
|
+
points of macro-F1 over the model families and about 0.1 point of accuracy for human vs machine.
|
|
221
|
+
- **The decision threshold does not carry over to another genre.** Trained on LLMTrace and tested on
|
|
222
|
+
scientific abstracts (AINL-Eval 2025, [arXiv:2508.09622](https://arxiv.org/abs/2508.09622)), the
|
|
223
|
+
ranking of the texts partly transfers (AUC 0.769) but only 36% of the human abstracts are
|
|
224
|
+
recognised: formal human text is taken for machine text. Calibrate the threshold on the genre you
|
|
225
|
+
are going to classify.
|
|
226
|
+
|
|
227
|
+
## Design notes
|
|
228
|
+
|
|
229
|
+
- **No raw counts by default.** Features are shares, densities per 100 words, means, or indices
|
|
230
|
+
designed to be comparable between texts of different length (MATTR and MTLD instead of the plain
|
|
231
|
+
type-token ratio).
|
|
232
|
+
- **Morphology comes from pymorphy3**, using the most probable parse of each word without context.
|
|
233
|
+
Analyses are cached per word, so a word is parsed once per process. Importing the package does
|
|
234
|
+
not load the dictionaries; the first use of the `morph` group does.
|
|
235
|
+
- **Sentence splitting is rule-based**: it knows common Russian abbreviations (т.д., г., ул.),
|
|
236
|
+
initials, ellipses, closing quotes and list items.
|
|
237
|
+
- **Lexicons are heuristics.** The lists of function words and discourse markers are hand-made and
|
|
238
|
+
small; treat the features built on them as interpretable markers rather than measurements.
|
|
239
|
+
- Flesch-type readability formulas are left out on purpose: they are linear combinations of
|
|
240
|
+
features that are already in the vector, and published Russian coefficients differ.
|
|
241
|
+
|
|
242
|
+
## Limitations
|
|
243
|
+
|
|
244
|
+
- Short texts (below roughly 50 words) give noisy lexical-diversity and sentence-rhythm values.
|
|
245
|
+
- Homonymous word forms are resolved by frequency, not by context, so part-of-speech shares are
|
|
246
|
+
approximate.
|
|
247
|
+
- Stylometric features describe style, not truth or authorship in a legal sense: a classifier built
|
|
248
|
+
on them gives a probability that depends on the corpus it was trained on, and it should not be
|
|
249
|
+
used as the only evidence against a person.
|
|
250
|
+
|
|
251
|
+
## Roadmap
|
|
252
|
+
|
|
253
|
+
- a stacking classifier as a class of the library: boosting on the outputs of a fine-tuned language
|
|
254
|
+
model and the 66 features. In the authors' experiments it did as well as or better than training
|
|
255
|
+
the two jointly inside one network, so it is the design to implement;
|
|
256
|
+
- an open-set decision for texts of unseen generators.
|
|
257
|
+
|
|
258
|
+
## Development
|
|
259
|
+
|
|
260
|
+
```bash
|
|
261
|
+
pip install -e .[dev]
|
|
262
|
+
pytest # the test suite in tests/
|
|
263
|
+
isort . && black . # style: black and isort, line length 100 (see pyproject.toml)
|
|
264
|
+
python -m build # sdist and wheel in dist/
|
|
265
|
+
```
|
|
266
|
+
|
|
267
|
+
Pushing a version tag (`git tag v0.1.1 && git push origin v0.1.1`) starts the GitHub Actions
|
|
268
|
+
workflow: it checks the style, runs the tests on Python 3.9-3.12, builds the package and uploads it
|
|
269
|
+
to TestPyPI with the repository secret `TESTPYPI_TOKEN`. TestPyPI accepts a version only once, so
|
|
270
|
+
raise `__version__` in `src/ru_stylometry/__init__.py` before tagging.
|
|
271
|
+
|
|
272
|
+
## License
|
|
273
|
+
|
|
274
|
+
MIT, see the `LICENSE` file.
|
|
@@ -0,0 +1,228 @@
|
|
|
1
|
+
# ru-stylometry
|
|
2
|
+
|
|
3
|
+
Interpretable stylometric features for Russian text, and a classifier that tells human-written text
|
|
4
|
+
from machine-generated text (and, if asked, the model family behind it) with an explanation of
|
|
5
|
+
every decision.
|
|
6
|
+
|
|
7
|
+
> **Status: alpha.** The feature extractor, the scikit-learn transformer, the classifier with
|
|
8
|
+
> explanations, the evaluation tools and the command-line interface are implemented and tested. The
|
|
9
|
+
> package is on [PyPI](https://pypi.org/project/ru-stylometry/); test builds are published to
|
|
10
|
+
> [TestPyPI](https://test.pypi.org/project/ru-stylometry/).
|
|
11
|
+
|
|
12
|
+
It continues [`stylometric-ai-detector`](https://pypi.org/project/stylometric-ai-detector/), an
|
|
13
|
+
English baseline built on 8 surface features. Those features are mostly raw counts that reflect
|
|
14
|
+
text length, and nothing in them is specific to Russian. `ru-stylometry` replaces them with a
|
|
15
|
+
documented vector of 66 features that are normalised by text length and use Russian morphology.
|
|
16
|
+
|
|
17
|
+
## Installation
|
|
18
|
+
|
|
19
|
+
```bash
|
|
20
|
+
pip install ru-stylometry # library and command-line tool
|
|
21
|
+
pip install "ru-stylometry[fast]" # + C backend that speeds up morphological analysis
|
|
22
|
+
pip install "ru-stylometry[experiments]" # + pandas, pyarrow, ...: reading Parquet tables
|
|
23
|
+
```
|
|
24
|
+
|
|
25
|
+
The latest code can be installed straight from GitHub, and a clone of the repository can be
|
|
26
|
+
installed for development:
|
|
27
|
+
|
|
28
|
+
```bash
|
|
29
|
+
pip install git+https://github.com/dinis-a/ru-stylometry.git # the latest code from GitHub
|
|
30
|
+
pip install -e .[dev] # from a clone: + pytest, black, isort, build
|
|
31
|
+
```
|
|
32
|
+
|
|
33
|
+
Test builds come from TestPyPI; the second index supplies the dependencies:
|
|
34
|
+
|
|
35
|
+
```bash
|
|
36
|
+
pip install --index-url https://test.pypi.org/simple/ --extra-index-url https://pypi.org/simple/ ru-stylometry
|
|
37
|
+
```
|
|
38
|
+
|
|
39
|
+
Python 3.9+ is required. Dependencies: numpy, scikit-learn, joblib, pymorphy3.
|
|
40
|
+
|
|
41
|
+
## Quick start
|
|
42
|
+
|
|
43
|
+
### Features
|
|
44
|
+
|
|
45
|
+
```python
|
|
46
|
+
from ru_stylometry import StylometricVectorizer, extract_features, feature_names
|
|
47
|
+
|
|
48
|
+
features = extract_features("Мама мыла раму. Рама была чистой!")
|
|
49
|
+
features["avg_sentence_len"] # 3.0
|
|
50
|
+
features["exclam_per100w"] # 16.67
|
|
51
|
+
features["noun_ratio"] # 0.67
|
|
52
|
+
|
|
53
|
+
len(feature_names()) # 66
|
|
54
|
+
|
|
55
|
+
# scikit-learn transformer: iterable of texts -> (n_texts, n_features) matrix
|
|
56
|
+
X = StylometricVectorizer(max_words=500, n_jobs=-1).fit_transform(["Первый текст.", "Второй текст."])
|
|
57
|
+
```
|
|
58
|
+
|
|
59
|
+
Pick feature groups explicitly, for example for an ablation study:
|
|
60
|
+
|
|
61
|
+
```python
|
|
62
|
+
extract_features(text, groups=["char", "lexical", "syntax"])
|
|
63
|
+
StylometricVectorizer(groups=["morph"])
|
|
64
|
+
```
|
|
65
|
+
|
|
66
|
+
The order of the columns is fixed by the registry, not by the order of `groups`, so the same
|
|
67
|
+
selection always gives the same layout. Blank or non-string input gives an all-zero vector.
|
|
68
|
+
|
|
69
|
+
### Classifier
|
|
70
|
+
|
|
71
|
+
```python
|
|
72
|
+
from ru_stylometry import StylometricClassifier
|
|
73
|
+
|
|
74
|
+
model = StylometricClassifier(estimator="hgb", class_weight="balanced")
|
|
75
|
+
model.fit(train_texts, train_labels) # labels: any strings, two or more classes
|
|
76
|
+
model.predict(["Текст для проверки."])
|
|
77
|
+
model.predict_proba(["Текст для проверки."])
|
|
78
|
+
|
|
79
|
+
model.explain("Текст для проверки.", top_k=5)
|
|
80
|
+
# effect of every feature and feature group on the evidence for the predicted class
|
|
81
|
+
|
|
82
|
+
model.save("model.joblib")
|
|
83
|
+
StylometricClassifier.load("model.joblib") # load only files you trust: the format is pickle
|
|
84
|
+
```
|
|
85
|
+
|
|
86
|
+
The *effect* of a feature is the fall of the class log-odds when the feature is replaced by its
|
|
87
|
+
typical value (the training median); positive effects push towards the class. Log-odds are used
|
|
88
|
+
because the probability of a confident decision is saturated and hides which features carry the
|
|
89
|
+
evidence. Effects are not additive, because correlated features share the credit; the effect of a
|
|
90
|
+
whole group is therefore reported separately.
|
|
91
|
+
|
|
92
|
+
Output for a machine-generated review from the LLMTrace test split, with a model trained on
|
|
93
|
+
30,000 texts of the same corpus:
|
|
94
|
+
|
|
95
|
+
```text
|
|
96
|
+
$ ru-stylometry explain --file review.txt -m model.joblib --top 3
|
|
97
|
+
label: ai (probability 0.998)
|
|
98
|
+
effect on the log-odds of class 'ai' (positive: towards it):
|
|
99
|
+
lemma_mattr_50 +0.82 value 0.934, typical 0.857 [morph]
|
|
100
|
+
conj_ratio +0.82 value 0.116, typical 0.084 [morph]
|
|
101
|
+
verb_pres_share +0.68 value 1.000, typical 0.429 [morph]
|
|
102
|
+
by group: morph +4.59, lexical +0.79, char +0.51, readability +0.43, syntax +0.21, ...
|
|
103
|
+
```
|
|
104
|
+
|
|
105
|
+
## Command line
|
|
106
|
+
|
|
107
|
+
```bash
|
|
108
|
+
ru-stylometry features "Мама мыла раму." --groups length syntax # JSON to stdout
|
|
109
|
+
ru-stylometry list-features --format markdown # the feature catalogue
|
|
110
|
+
|
|
111
|
+
ru-stylometry train corpus.parquet -o model.joblib --estimator hgb --balanced --valid dev.parquet
|
|
112
|
+
ru-stylometry evaluate test.parquet -m model.joblib # metrics and time per document
|
|
113
|
+
ru-stylometry predict "Текст для проверки." -m model.joblib
|
|
114
|
+
ru-stylometry explain "Текст для проверки." -m model.joblib --top 10
|
|
115
|
+
```
|
|
116
|
+
|
|
117
|
+
Tables may be `.csv`, `.jsonl` or `.parquet` with a text column and a label column
|
|
118
|
+
(`--text-column`, `--label-column`).
|
|
119
|
+
|
|
120
|
+
## Feature groups
|
|
121
|
+
|
|
122
|
+
| Group | Features | Default | What it measures |
|
|
123
|
+
|---------------|---------:|:-------:|-------------------------------------------------------------------|
|
|
124
|
+
| `char` | 7 | yes | letters, digits, punctuation, symbols, capitals, Latin, use of ё |
|
|
125
|
+
| `typography` | 11 | yes | punctuation density; dash, quotation-mark and ellipsis habits |
|
|
126
|
+
| `lexical` | 7 | yes | word length; MATTR, MTLD, Yule's K, hapax share |
|
|
127
|
+
| `syntax` | 9 | yes | sentence length distribution, function words, negation, openers |
|
|
128
|
+
| `readability` | 3 | yes | syllables per word, polysyllabic words, LIX |
|
|
129
|
+
| `repetition` | 3 | yes | deflate compression ratio, repeated bigrams and trigrams |
|
|
130
|
+
| `morph` | 26 | yes | part of speech, case, tense, person, unknown words, lemma variety |
|
|
131
|
+
| `formatting` | 3 | no | line breaks, list items, Markdown symbols |
|
|
132
|
+
| `length` | 4 | no | raw counts of characters, words, sentences, lines |
|
|
133
|
+
|
|
134
|
+
Every feature has a name and a one-line definition; `ru-stylometry list-features --format markdown`
|
|
135
|
+
prints the whole catalogue.
|
|
136
|
+
|
|
137
|
+
`formatting` and `length` are opt-in because they depend on how a corpus was collected more than on
|
|
138
|
+
style: texts of different classes often differ in length and layout for reasons that have nothing
|
|
139
|
+
to do with authorship.
|
|
140
|
+
|
|
141
|
+
## Evaluation, explanation and robustness
|
|
142
|
+
|
|
143
|
+
| module | what it offers |
|
|
144
|
+
|:--|:--|
|
|
145
|
+
| `ru_stylometry.evaluation` | accuracy, macro-F1, precision and recall of every class, confusion matrix, time per document, bootstrap intervals that resample groups of related texts |
|
|
146
|
+
| `ru_stylometry.explain` | effect of every feature and group on one decision; permutation importance of whole feature groups |
|
|
147
|
+
| `ru_stylometry.perturb` | text transformations for robustness checks: typography, Markdown, discourse markers, sentence order, typos |
|
|
148
|
+
|
|
149
|
+
Macro averaging counts every class equally, so a weak result on a rare source is not hidden by the
|
|
150
|
+
large classes.
|
|
151
|
+
|
|
152
|
+
## How well it works
|
|
153
|
+
|
|
154
|
+
Measured by the authors (the experiment scripts are not part of this repository) on the Russian
|
|
155
|
+
part of the LLMTrace corpus ([arXiv:2509.21269](https://arxiv.org/abs/2509.21269): 340,197 texts by
|
|
156
|
+
humans and 31 language models in eight genres). The official test split has 52,521 texts that share
|
|
157
|
+
no topic with the training split; the classifier is gradient boosting on the first 500 words of
|
|
158
|
+
every text:
|
|
159
|
+
|
|
160
|
+
| features | number | accuracy | macro-F1 | AUC |
|
|
161
|
+
|:--|--:|--:|--:|--:|
|
|
162
|
+
| length of the text only | 2 | 0.705 | 0.684 | 0.754 |
|
|
163
|
+
| the 8 features of `stylometric-ai-detector` | 8 | 0.751 | 0.735 | 0.820 |
|
|
164
|
+
| **`ru-stylometry`, default groups** | 66 | **0.883** | **0.879** | **0.954** |
|
|
165
|
+
|
|
166
|
+
The 95% interval of the accuracy of the 66 features is [0.881, 0.886]; it resamples whole topics,
|
|
167
|
+
because texts about the same subject are not independent. For the 15 classes (humans and 14 model
|
|
168
|
+
families) the macro-F1 is 0.446 against 0.195 for the 8 legacy features.
|
|
169
|
+
|
|
170
|
+
- **A fine-tuned language model is stronger.** `rubert-base-cased` fine-tuned on the same split
|
|
171
|
+
reaches an accuracy of 0.953, about 7 points more. The stylometric classifier is a 1.2 MB file, needs
|
|
172
|
+
no GPU (about 3.4 ms per text on one CPU core with a warm cache, around 290 texts per second) and
|
|
173
|
+
explains every decision. Adding the 66 features to the outputs of the language model gave 1.4-2.6
|
|
174
|
+
points of macro-F1 over the model families and about 0.1 point of accuracy for human vs machine.
|
|
175
|
+
- **The decision threshold does not carry over to another genre.** Trained on LLMTrace and tested on
|
|
176
|
+
scientific abstracts (AINL-Eval 2025, [arXiv:2508.09622](https://arxiv.org/abs/2508.09622)), the
|
|
177
|
+
ranking of the texts partly transfers (AUC 0.769) but only 36% of the human abstracts are
|
|
178
|
+
recognised: formal human text is taken for machine text. Calibrate the threshold on the genre you
|
|
179
|
+
are going to classify.
|
|
180
|
+
|
|
181
|
+
## Design notes
|
|
182
|
+
|
|
183
|
+
- **No raw counts by default.** Features are shares, densities per 100 words, means, or indices
|
|
184
|
+
designed to be comparable between texts of different length (MATTR and MTLD instead of the plain
|
|
185
|
+
type-token ratio).
|
|
186
|
+
- **Morphology comes from pymorphy3**, using the most probable parse of each word without context.
|
|
187
|
+
Analyses are cached per word, so a word is parsed once per process. Importing the package does
|
|
188
|
+
not load the dictionaries; the first use of the `morph` group does.
|
|
189
|
+
- **Sentence splitting is rule-based**: it knows common Russian abbreviations (т.д., г., ул.),
|
|
190
|
+
initials, ellipses, closing quotes and list items.
|
|
191
|
+
- **Lexicons are heuristics.** The lists of function words and discourse markers are hand-made and
|
|
192
|
+
small; treat the features built on them as interpretable markers rather than measurements.
|
|
193
|
+
- Flesch-type readability formulas are left out on purpose: they are linear combinations of
|
|
194
|
+
features that are already in the vector, and published Russian coefficients differ.
|
|
195
|
+
|
|
196
|
+
## Limitations
|
|
197
|
+
|
|
198
|
+
- Short texts (below roughly 50 words) give noisy lexical-diversity and sentence-rhythm values.
|
|
199
|
+
- Homonymous word forms are resolved by frequency, not by context, so part-of-speech shares are
|
|
200
|
+
approximate.
|
|
201
|
+
- Stylometric features describe style, not truth or authorship in a legal sense: a classifier built
|
|
202
|
+
on them gives a probability that depends on the corpus it was trained on, and it should not be
|
|
203
|
+
used as the only evidence against a person.
|
|
204
|
+
|
|
205
|
+
## Roadmap
|
|
206
|
+
|
|
207
|
+
- a stacking classifier as a class of the library: boosting on the outputs of a fine-tuned language
|
|
208
|
+
model and the 66 features. In the authors' experiments it did as well as or better than training
|
|
209
|
+
the two jointly inside one network, so it is the design to implement;
|
|
210
|
+
- an open-set decision for texts of unseen generators.
|
|
211
|
+
|
|
212
|
+
## Development
|
|
213
|
+
|
|
214
|
+
```bash
|
|
215
|
+
pip install -e .[dev]
|
|
216
|
+
pytest # the test suite in tests/
|
|
217
|
+
isort . && black . # style: black and isort, line length 100 (see pyproject.toml)
|
|
218
|
+
python -m build # sdist and wheel in dist/
|
|
219
|
+
```
|
|
220
|
+
|
|
221
|
+
Pushing a version tag (`git tag v0.1.1 && git push origin v0.1.1`) starts the GitHub Actions
|
|
222
|
+
workflow: it checks the style, runs the tests on Python 3.9-3.12, builds the package and uploads it
|
|
223
|
+
to TestPyPI with the repository secret `TESTPYPI_TOKEN`. TestPyPI accepts a version only once, so
|
|
224
|
+
raise `__version__` in `src/ru_stylometry/__init__.py` before tagging.
|
|
225
|
+
|
|
226
|
+
## License
|
|
227
|
+
|
|
228
|
+
MIT, see the `LICENSE` file.
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=77", "wheel"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "ru-stylometry"
|
|
7
|
+
dynamic = ["version"]
|
|
8
|
+
description = "Interpretable stylometric features for Russian text and tools for detecting AI-generated text."
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = "MIT"
|
|
11
|
+
license-files = ["LICENSE"]
|
|
12
|
+
requires-python = ">=3.9"
|
|
13
|
+
authors = [
|
|
14
|
+
{name = "Dinis"},
|
|
15
|
+
]
|
|
16
|
+
keywords = [
|
|
17
|
+
"nlp",
|
|
18
|
+
"stylometry",
|
|
19
|
+
"russian",
|
|
20
|
+
"feature-extraction",
|
|
21
|
+
"text-classification",
|
|
22
|
+
"ai-detection",
|
|
23
|
+
"scikit-learn",
|
|
24
|
+
]
|
|
25
|
+
classifiers = [
|
|
26
|
+
"Development Status :: 3 - Alpha",
|
|
27
|
+
"Intended Audience :: Developers",
|
|
28
|
+
"Intended Audience :: Education",
|
|
29
|
+
"Intended Audience :: Science/Research",
|
|
30
|
+
"Natural Language :: Russian",
|
|
31
|
+
"Operating System :: OS Independent",
|
|
32
|
+
"Programming Language :: Python :: 3",
|
|
33
|
+
"Programming Language :: Python :: 3.9",
|
|
34
|
+
"Programming Language :: Python :: 3.10",
|
|
35
|
+
"Programming Language :: Python :: 3.11",
|
|
36
|
+
"Programming Language :: Python :: 3.12",
|
|
37
|
+
"Topic :: Scientific/Engineering :: Artificial Intelligence",
|
|
38
|
+
"Topic :: Text Processing :: Linguistic",
|
|
39
|
+
"Typing :: Typed",
|
|
40
|
+
]
|
|
41
|
+
dependencies = [
|
|
42
|
+
"numpy>=1.21",
|
|
43
|
+
"joblib>=1.1",
|
|
44
|
+
"scikit-learn>=1.2",
|
|
45
|
+
"pymorphy3>=1.2",
|
|
46
|
+
"pymorphy3-dicts-ru",
|
|
47
|
+
]
|
|
48
|
+
|
|
49
|
+
[project.optional-dependencies]
|
|
50
|
+
# C extension that roughly halves the cost of parsing words missing from the cache.
|
|
51
|
+
fast = ["pymorphy3[fast]"]
|
|
52
|
+
# Packages used by the scripts in experiments/ and for reading Parquet files.
|
|
53
|
+
experiments = ["pandas>=2.0", "pyarrow>=14", "huggingface_hub>=0.20", "tabulate"]
|
|
54
|
+
dev = [
|
|
55
|
+
"pytest>=7.0",
|
|
56
|
+
"isort>=5.0",
|
|
57
|
+
"black>=23.0",
|
|
58
|
+
"build>=1.0",
|
|
59
|
+
]
|
|
60
|
+
|
|
61
|
+
[project.urls]
|
|
62
|
+
Homepage = "https://github.com/dinis-a/ru-stylometry"
|
|
63
|
+
Repository = "https://github.com/dinis-a/ru-stylometry"
|
|
64
|
+
Issues = "https://github.com/dinis-a/ru-stylometry/issues"
|
|
65
|
+
|
|
66
|
+
[project.scripts]
|
|
67
|
+
ru-stylometry = "ru_stylometry.cli:main"
|
|
68
|
+
|
|
69
|
+
[tool.setuptools.dynamic]
|
|
70
|
+
version = {attr = "ru_stylometry.__version__"}
|
|
71
|
+
|
|
72
|
+
[tool.setuptools.packages.find]
|
|
73
|
+
where = ["src"]
|
|
74
|
+
|
|
75
|
+
[tool.setuptools.package-data]
|
|
76
|
+
ru_stylometry = ["py.typed"]
|
|
77
|
+
|
|
78
|
+
[tool.isort]
|
|
79
|
+
profile = "black"
|
|
80
|
+
line_length = 100
|
|
81
|
+
src_paths = ["src", "tests"]
|
|
82
|
+
# Notebook cells keep their own import order (some imports must follow sys.path changes).
|
|
83
|
+
skip_glob = ["notebooks/cells/*", ".cache/*"]
|
|
84
|
+
|
|
85
|
+
[tool.black]
|
|
86
|
+
line-length = 100
|
|
87
|
+
target-version = ["py39"]
|
|
88
|
+
# Tool caches and scratch files live in .cache/ and are not part of the project.
|
|
89
|
+
extend-exclude = "/\\.cache/"
|
|
90
|
+
|
|
91
|
+
[tool.pytest.ini_options]
|
|
92
|
+
testpaths = ["tests"]
|
|
@@ -0,0 +1,34 @@
|
|
|
1
|
+
"""ru-stylometry — interpretable stylometric features for Russian text.
|
|
2
|
+
|
|
3
|
+
Public API:
|
|
4
|
+
- :func:`extract_features` — named stylometric features of one text.
|
|
5
|
+
- :func:`feature_names` and :func:`describe_features` — the feature catalogue.
|
|
6
|
+
- :class:`StylometricVectorizer` — scikit-learn transformer: texts -> feature matrix.
|
|
7
|
+
- :class:`StylometricClassifier` — classifier over those features with explanations.
|
|
8
|
+
|
|
9
|
+
Features are grouped by the level of analysis (characters, typography, vocabulary, syntax,
|
|
10
|
+
readability, repetition, morphology); see :data:`ALL_GROUPS` and :data:`DEFAULT_GROUPS`.
|
|
11
|
+
The submodules ``evaluation``, ``explain`` and ``perturb`` hold the quality criteria, the
|
|
12
|
+
explanations and the text transformations used in robustness checks.
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from ru_stylometry.features import (
|
|
16
|
+
ALL_GROUPS,
|
|
17
|
+
DEFAULT_GROUPS,
|
|
18
|
+
describe_features,
|
|
19
|
+
extract_features,
|
|
20
|
+
feature_names,
|
|
21
|
+
)
|
|
22
|
+
from ru_stylometry.model import StylometricClassifier
|
|
23
|
+
from ru_stylometry.vectorizer import StylometricVectorizer
|
|
24
|
+
|
|
25
|
+
__all__ = [
|
|
26
|
+
"ALL_GROUPS",
|
|
27
|
+
"DEFAULT_GROUPS",
|
|
28
|
+
"StylometricClassifier",
|
|
29
|
+
"StylometricVectorizer",
|
|
30
|
+
"describe_features",
|
|
31
|
+
"extract_features",
|
|
32
|
+
"feature_names",
|
|
33
|
+
]
|
|
34
|
+
__version__ = "0.1.1"
|