hsk30 0.1.1__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- hsk30-0.1.1/LICENSE +21 -0
- hsk30-0.1.1/PKG-INFO +258 -0
- hsk30-0.1.1/README.md +234 -0
- hsk30-0.1.1/pyproject.toml +40 -0
- hsk30-0.1.1/setup.cfg +4 -0
- hsk30-0.1.1/src/hsk30/__init__.py +50 -0
- hsk30-0.1.1/src/hsk30/cli.py +96 -0
- hsk30-0.1.1/src/hsk30/data/hsk2025_chars.tsv +3089 -0
- hsk30-0.1.1/src/hsk30/data/hsk2025_words.tsv +10897 -0
- hsk30-0.1.1/src/hsk30/data/hsk20_words.tsv +4992 -0
- hsk30-0.1.1/src/hsk30/data/hsk30_chars.tsv +3001 -0
- hsk30-0.1.1/src/hsk30/data/hsk30_words.tsv +10917 -0
- hsk30-0.1.1/src/hsk30/data.py +107 -0
- hsk30-0.1.1/src/hsk30/grade.py +327 -0
- hsk30-0.1.1/src/hsk30.egg-info/PKG-INFO +258 -0
- hsk30-0.1.1/src/hsk30.egg-info/SOURCES.txt +19 -0
- hsk30-0.1.1/src/hsk30.egg-info/dependency_links.txt +1 -0
- hsk30-0.1.1/src/hsk30.egg-info/entry_points.txt +2 -0
- hsk30-0.1.1/src/hsk30.egg-info/requires.txt +3 -0
- hsk30-0.1.1/src/hsk30.egg-info/top_level.txt +1 -0
- hsk30-0.1.1/tests/test_hsk30.py +299 -0
hsk30-0.1.1/LICENSE
ADDED
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 Alvaro Serrano
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
hsk30-0.1.1/PKG-INFO
ADDED
|
@@ -0,0 +1,258 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: hsk30
|
|
3
|
+
Version: 0.1.1
|
|
4
|
+
Summary: Grade Chinese text against either official document called HSK 3.0
|
|
5
|
+
License: MIT
|
|
6
|
+
Project-URL: Homepage, https://github.com/harukicoder/hsk30
|
|
7
|
+
Project-URL: Source, https://github.com/harukicoder/hsk30
|
|
8
|
+
Project-URL: Issues, https://github.com/harukicoder/hsk30/issues
|
|
9
|
+
Keywords: chinese,mandarin,hsk,readability,language-learning,cefr
|
|
10
|
+
Classifier: Development Status :: 4 - Beta
|
|
11
|
+
Classifier: Intended Audience :: Education
|
|
12
|
+
Classifier: Intended Audience :: Science/Research
|
|
13
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
14
|
+
Classifier: Natural Language :: Chinese (Simplified)
|
|
15
|
+
Classifier: Programming Language :: Python :: 3
|
|
16
|
+
Classifier: Programming Language :: Python :: 3.9
|
|
17
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
18
|
+
Requires-Python: >=3.9
|
|
19
|
+
Description-Content-Type: text/markdown
|
|
20
|
+
License-File: LICENSE
|
|
21
|
+
Provides-Extra: dev
|
|
22
|
+
Requires-Dist: pytest>=7; extra == "dev"
|
|
23
|
+
Dynamic: license-file
|
|
24
|
+
|
|
25
|
+
# hsk30
|
|
26
|
+
|
|
27
|
+
Grade Chinese text against HSK — against **either** document that gets called
|
|
28
|
+
"HSK 3.0", and it will tell you which one it used.
|
|
29
|
+
|
|
30
|
+
```bash
|
|
31
|
+
pip install hsk30
|
|
32
|
+
```
|
|
33
|
+
|
|
34
|
+
```python
|
|
35
|
+
import hsk30
|
|
36
|
+
|
|
37
|
+
hsk30.grade("我每天早上七点起床,然后去公园跑步。").label
|
|
38
|
+
# '3' ← graded against the 2026 examination syllabus, the default
|
|
39
|
+
|
|
40
|
+
hsk30.grade("...", standard="2021").label
|
|
41
|
+
# the GF0025-2021 national grading standard instead
|
|
42
|
+
```
|
|
43
|
+
|
|
44
|
+
> **"HSK 3.0" is two different documents, and the choice changes the answer.**
|
|
45
|
+
> They disagree on **41.5%** of shared vocabulary and **40.7%** of shared
|
|
46
|
+
> characters. Regrading 102 authentic graded readers against one rather than
|
|
47
|
+
> the other changes the level of **48% of them**, almost always upward. This
|
|
48
|
+
> library defaults to the examination syllabus in force since July 2026 and
|
|
49
|
+
> records `profile.standard` on every result. See [Versions](#versions).
|
|
50
|
+
|
|
51
|
+
```bash
|
|
52
|
+
$ hsk30 "我每天早上七点起床,然后去公园跑步。" --curve --target 2
|
|
53
|
+
HSK 3 (16 characters, 0 ungraded)
|
|
54
|
+
HSK 1 68.8% ############################
|
|
55
|
+
HSK 2 87.5% ###################################
|
|
56
|
+
HSK 3 100.0% ########################################
|
|
57
|
+
target HSK 2: misses the 95% bar (12.5% above target)
|
|
58
|
+
步 HSK 3 6.2% <- over budget on its own
|
|
59
|
+
每 HSK 3 6.2% <- over budget on its own
|
|
60
|
+
```
|
|
61
|
+
|
|
62
|
+
No dependencies. Python 3.9+.
|
|
63
|
+
|
|
64
|
+
## Why this exists
|
|
65
|
+
|
|
66
|
+
Most Chinese-learning tools report an "HSK 3.0 level" without saying which
|
|
67
|
+
document produced it. There are two, published four years apart:
|
|
68
|
+
|
|
69
|
+
| Comparison | Shared words | Same level | Moved |
|
|
70
|
+
| --- | ---: | ---: | ---: |
|
|
71
|
+
| HSK 2.0 → GF0025-2021 | 4,482 | 814 | 3,668 (81.8%) |
|
|
72
|
+
| HSK 2.0 → 2025 syllabus | 4,802 | 2,349 | 2,453 (51.1%) |
|
|
73
|
+
| **GF0025-2021 → 2025 syllabus** | **9,674** | **5,662** | **4,012 (41.5%)** |
|
|
74
|
+
|
|
75
|
+
Judged against the 2021 standard, HSK 2.0 looks almost entirely regraded.
|
|
76
|
+
Judged against the examination syllabus, barely half moved. The syllabus is
|
|
77
|
+
markedly more conservative, and it is the document learners are actually tested
|
|
78
|
+
on.
|
|
79
|
+
|
|
80
|
+
Every figure in this README is produced by `python3 scripts/reproduce.py`.
|
|
81
|
+
|
|
82
|
+
## What it does
|
|
83
|
+
|
|
84
|
+
Answers one question: **what HSK level does a reader need to read this text?**
|
|
85
|
+
The answer is the level at which cumulative character coverage reaches 95% —
|
|
86
|
+
the point at which a reader can follow a passage and infer the rest.
|
|
87
|
+
|
|
88
|
+
```python
|
|
89
|
+
p = hsk30.grade("这项研究揭示了神经网络的内在缺陷。")
|
|
90
|
+
p.level # 6
|
|
91
|
+
p.label # '6'
|
|
92
|
+
p.chars # 16
|
|
93
|
+
p.ungraded # characters outside the 3,000
|
|
94
|
+
p.curve() # cumulative coverage at every level
|
|
95
|
+
```
|
|
96
|
+
|
|
97
|
+
### Three decisions that matter
|
|
98
|
+
|
|
99
|
+
**Character-level, not word-level.** Word-level grading is unusable on
|
|
100
|
+
segmented Chinese. Real segmenters emit phrase tokens (我的, 七点, 蓝色) that
|
|
101
|
+
are not entries in any graded word list, pushing "unknown" past 95% at every
|
|
102
|
+
level and reporting ordinary beginner text as off-scale. HSK 3.0 grades 3,000
|
|
103
|
+
characters separately from its words precisely because the character inventory
|
|
104
|
+
is what gates reading.
|
|
105
|
+
|
|
106
|
+
**The official character list, not a derived one.** Deriving character levels
|
|
107
|
+
from the lowest-level word containing each character agrees with the 2021
|
|
108
|
+
official list on 2,962 of 2,969 characters — and gets the family terms wrong. 哥, 妈,
|
|
109
|
+
妹, 弟 are level-1 characters whose only listed words (哥哥, 妈妈) sit at
|
|
110
|
+
level 4. Also wrong: 王, 第, 零. This package ships the official list.
|
|
111
|
+
|
|
112
|
+
**Proper nouns are excluded when identifiable.** A reader does not need the
|
|
113
|
+
puppy's name in their vocabulary; it is glossed in place. Counting names as
|
|
114
|
+
difficulty graded a story called "My Puppy Doudou" at HSK 4 on a beginner
|
|
115
|
+
shelf, entirely on the strength of 豆豆. Detection needs pinyin, so it is
|
|
116
|
+
available through `grade_tokens`:
|
|
117
|
+
|
|
118
|
+
```python
|
|
119
|
+
hsk30.grade_tokens([
|
|
120
|
+
{"hz": "我", "py": "wǒ"},
|
|
121
|
+
{"hz": "李明。", "py": "Lǐ Míng"}, # excluded
|
|
122
|
+
]).chars # 1
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
### Levels 7, 8 and 9 are one band
|
|
126
|
+
|
|
127
|
+
Neither document splits them — in the 2021 standard they share a single
|
|
128
|
+
5,599-word list and 1,200 characters. This package carries the band as level `7` and renders it
|
|
129
|
+
`"7-9"`. A vendor advertising an "HSK 8 word list" invented the split.
|
|
130
|
+
|
|
131
|
+
### Character budgets
|
|
132
|
+
|
|
133
|
+
Reaching a 95% bar means keeping the above-target share under 5%, so a single
|
|
134
|
+
character over that budget blocks the target on its own:
|
|
135
|
+
|
|
136
|
+
```python
|
|
137
|
+
share, offenders = hsk30.budget_violations(text, target=3)
|
|
138
|
+
```
|
|
139
|
+
|
|
140
|
+
This is how a short passage silently regresses when an otherwise harmless edit
|
|
141
|
+
repeats one hard character a fourth time.
|
|
142
|
+
|
|
143
|
+
### Grading collections
|
|
144
|
+
|
|
145
|
+
```python
|
|
146
|
+
shelf = hsk30.profile_shelf([hsk30.grade(t) for t in texts])
|
|
147
|
+
shelf.label # median text — not the pooled figure
|
|
148
|
+
shelf.span_label # 'HSK 2-3', the interquartile range
|
|
149
|
+
```
|
|
150
|
+
|
|
151
|
+
Reports the **median text**. Pooling every character in a shelf lets a handful
|
|
152
|
+
of hard texts speak for all of them: it reported "HSK 3" for a beginner shelf
|
|
153
|
+
on which 16 of 22 texts individually read at HSK 1–2, describing nothing
|
|
154
|
+
actually on the shelf.
|
|
155
|
+
|
|
156
|
+
## What's in this repository
|
|
157
|
+
|
|
158
|
+
| Path | Contents |
|
|
159
|
+
| --- | --- |
|
|
160
|
+
| `src/hsk30/` | The library and its five graded lists (MIT) |
|
|
161
|
+
| `corpus/` | 102 aligned graded readers, 1,185 sentences (CC BY 4.0) |
|
|
162
|
+
| `benchmark/` | HSKBench — controlled-difficulty generation |
|
|
163
|
+
| `paper/` | The accompanying paper and its figures |
|
|
164
|
+
| `scripts/reproduce.py` | Recomputes every published figure |
|
|
165
|
+
| `scripts/extract_syllabus_2025.py` | Parses the official syllabus PDF |
|
|
166
|
+
| `corpus/syllabus2025/PROVENANCE.md` | Where the 2025 tables come from, and their rights position |
|
|
167
|
+
|
|
168
|
+
## HSKBench
|
|
169
|
+
|
|
170
|
+
Generating text *at* a level turns out to be much harder than grading it.
|
|
171
|
+
Human authors writing to an explicit target hit it **61.8%** of the time,
|
|
172
|
+
overshooting at the easy end and undershooting at the hard end. HSKBench scores
|
|
173
|
+
that task objectively — the grader is the metric, the way a compiler is the
|
|
174
|
+
metric for generated code. See [`benchmark/README.md`](https://github.com/harukicoder/hsk30/blob/main/benchmark/README.md).
|
|
175
|
+
|
|
176
|
+
## Versions
|
|
177
|
+
|
|
178
|
+
Three documents are routinely conflated, including by commercial HSK sites.
|
|
179
|
+
They are different, and it matters which one a tool grades against.
|
|
180
|
+
|
|
181
|
+
| `standard=` | Document | Date | Words | Characters |
|
|
182
|
+
| --- | --- | --- | ---: | ---: |
|
|
183
|
+
| `"2.0"` | HSK 2.0 exam lists | 2009–10 | 4,991 | — |
|
|
184
|
+
| `"2021"` | 《国际中文教育中文水平等级标准》 (GF0025-2021) | in force 1 Jul 2021 | 10,916 | 3,000 |
|
|
185
|
+
| `"2025"` *(default)* | 新版HSK考试大纲 | pub. Nov 2025, in force Jul 2026 | 10,896 | 3,088 |
|
|
186
|
+
|
|
187
|
+
The 2021 document is a national language standard (语言文字规范) from the
|
|
188
|
+
Ministry of Education and the State Language Commission. The 2025 document is
|
|
189
|
+
the examination syllabus from the Center for Language Education and Cooperation
|
|
190
|
+
(中外语言交流合作中心) and governs the test learners actually sit — which is why
|
|
191
|
+
it is the default.
|
|
192
|
+
|
|
193
|
+
HSK 2.0 graded no characters separately, so `characters("2.0")` raises.
|
|
194
|
+
|
|
195
|
+
**The 2025 lists are extracted from the official 406-page PDF** by
|
|
196
|
+
`scripts/extract_syllabus_2025.py`, which self-validates: parsed per-level entry
|
|
197
|
+
counts reproduce the published cumulative totals (300 / 500 / 1,000 / 2,000 /
|
|
198
|
+
3,600 / 5,400 / 11,000) exactly. Two notes from doing it — the syllabus numbers
|
|
199
|
+
11,000 *entries* but only 10,896 distinct words (homographs like 所/所2 get
|
|
200
|
+
their own rows), and it grades **3,088** recognition characters, not the 3,079
|
|
201
|
+
widely reported.
|
|
202
|
+
|
|
203
|
+
## Data sources
|
|
204
|
+
|
|
205
|
+
| Source | Provides | Licence |
|
|
206
|
+
| --- | --- | --- |
|
|
207
|
+
| [ivankra/hsk30](https://github.com/ivankra/hsk30) | HSK 3.0 word and character lists | MIT |
|
|
208
|
+
| [drkameleon/complete-hsk-vocabulary](https://github.com/drkameleon/complete-hsk-vocabulary) | HSK 2.0 levels, pinyin, glosses | MIT |
|
|
209
|
+
|
|
210
|
+
Both are transcriptions of 《国际中文教育中文水平等级标准》. Regenerate the
|
|
211
|
+
shipped tables with `python3 scripts/gen_data.py` (needs network).
|
|
212
|
+
|
|
213
|
+
## Limitations
|
|
214
|
+
|
|
215
|
+
- **Simplified characters only.** Convert traditional text with OpenCC first.
|
|
216
|
+
- **Coverage is not comprehension.** 95% character coverage is a necessary
|
|
217
|
+
condition for fluent reading, not a sufficient one; grammar, register and
|
|
218
|
+
world knowledge are not modelled.
|
|
219
|
+
- **Proper-noun detection needs pinyin.** `grade()` on a bare string cannot
|
|
220
|
+
identify names; pass them via `exclude=`, or use `grade_tokens()`.
|
|
221
|
+
- **CJK Extension A–F characters are treated as ungraded**, which is correct
|
|
222
|
+
under the standard but means literary text scores off-scale readily.
|
|
223
|
+
- **Authored segmentation** in the corpus groups some phrases a segmenter
|
|
224
|
+
would split.
|
|
225
|
+
|
|
226
|
+
## Development
|
|
227
|
+
|
|
228
|
+
```bash
|
|
229
|
+
git clone https://github.com/harukicoder/hsk30 && cd hsk30
|
|
230
|
+
pip install -e ".[dev]"
|
|
231
|
+
pytest # or: python3 tests/test_hsk30.py
|
|
232
|
+
python3 scripts/reproduce.py # every figure in the paper
|
|
233
|
+
```
|
|
234
|
+
|
|
235
|
+
The library is a port of the implementation that runs
|
|
236
|
+
[pinyora.com](https://pinyora.com). It reproduces that implementation's output
|
|
237
|
+
on all 102 corpus texts exactly (`test_python_reproduces_the_javascript_reference_exactly`),
|
|
238
|
+
with one deliberate fix: the original's ASCII-only `^[A-Z]` proper-noun test
|
|
239
|
+
missed names romanised with an accented capital — Ōuzhōu, Ōuyà, Ā Q Zhèngzhuàn
|
|
240
|
+
— and since 欧 and 洲 are both HSK 7–9 characters, missing one place name moved
|
|
241
|
+
a text two levels. The legacy behaviour remains available as
|
|
242
|
+
`is_proper_noun_ascii`.
|
|
243
|
+
|
|
244
|
+
## Citation
|
|
245
|
+
|
|
246
|
+
```bibtex
|
|
247
|
+
@misc{serrano2026hsk30,
|
|
248
|
+
title = {hsk30: Grading Chinese Text Against the HSK 3.0 Standard},
|
|
249
|
+
author = {Serrano, Alvaro},
|
|
250
|
+
year = {2026},
|
|
251
|
+
url = {https://github.com/harukicoder/hsk30}
|
|
252
|
+
}
|
|
253
|
+
```
|
|
254
|
+
|
|
255
|
+
## Licence
|
|
256
|
+
|
|
257
|
+
MIT for the code and the derived level tables; **CC BY 4.0** for the corpus
|
|
258
|
+
(see `corpus/LICENSE`).
|
hsk30-0.1.1/README.md
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
# hsk30
|
|
2
|
+
|
|
3
|
+
Grade Chinese text against HSK — against **either** document that gets called
|
|
4
|
+
"HSK 3.0", and it will tell you which one it used.
|
|
5
|
+
|
|
6
|
+
```bash
|
|
7
|
+
pip install hsk30
|
|
8
|
+
```
|
|
9
|
+
|
|
10
|
+
```python
|
|
11
|
+
import hsk30
|
|
12
|
+
|
|
13
|
+
hsk30.grade("我每天早上七点起床,然后去公园跑步。").label
|
|
14
|
+
# '3' ← graded against the 2026 examination syllabus, the default
|
|
15
|
+
|
|
16
|
+
hsk30.grade("...", standard="2021").label
|
|
17
|
+
# the GF0025-2021 national grading standard instead
|
|
18
|
+
```
|
|
19
|
+
|
|
20
|
+
> **"HSK 3.0" is two different documents, and the choice changes the answer.**
|
|
21
|
+
> They disagree on **41.5%** of shared vocabulary and **40.7%** of shared
|
|
22
|
+
> characters. Regrading 102 authentic graded readers against one rather than
|
|
23
|
+
> the other changes the level of **48% of them**, almost always upward. This
|
|
24
|
+
> library defaults to the examination syllabus in force since July 2026 and
|
|
25
|
+
> records `profile.standard` on every result. See [Versions](#versions).
|
|
26
|
+
|
|
27
|
+
```bash
|
|
28
|
+
$ hsk30 "我每天早上七点起床,然后去公园跑步。" --curve --target 2
|
|
29
|
+
HSK 3 (16 characters, 0 ungraded)
|
|
30
|
+
HSK 1 68.8% ############################
|
|
31
|
+
HSK 2 87.5% ###################################
|
|
32
|
+
HSK 3 100.0% ########################################
|
|
33
|
+
target HSK 2: misses the 95% bar (12.5% above target)
|
|
34
|
+
步 HSK 3 6.2% <- over budget on its own
|
|
35
|
+
每 HSK 3 6.2% <- over budget on its own
|
|
36
|
+
```
|
|
37
|
+
|
|
38
|
+
No dependencies. Python 3.9+.
|
|
39
|
+
|
|
40
|
+
## Why this exists
|
|
41
|
+
|
|
42
|
+
Most Chinese-learning tools report an "HSK 3.0 level" without saying which
|
|
43
|
+
document produced it. There are two, published four years apart:
|
|
44
|
+
|
|
45
|
+
| Comparison | Shared words | Same level | Moved |
|
|
46
|
+
| --- | ---: | ---: | ---: |
|
|
47
|
+
| HSK 2.0 → GF0025-2021 | 4,482 | 814 | 3,668 (81.8%) |
|
|
48
|
+
| HSK 2.0 → 2025 syllabus | 4,802 | 2,349 | 2,453 (51.1%) |
|
|
49
|
+
| **GF0025-2021 → 2025 syllabus** | **9,674** | **5,662** | **4,012 (41.5%)** |
|
|
50
|
+
|
|
51
|
+
Judged against the 2021 standard, HSK 2.0 looks almost entirely regraded.
|
|
52
|
+
Judged against the examination syllabus, barely half moved. The syllabus is
|
|
53
|
+
markedly more conservative, and it is the document learners are actually tested
|
|
54
|
+
on.
|
|
55
|
+
|
|
56
|
+
Every figure in this README is produced by `python3 scripts/reproduce.py`.
|
|
57
|
+
|
|
58
|
+
## What it does
|
|
59
|
+
|
|
60
|
+
Answers one question: **what HSK level does a reader need to read this text?**
|
|
61
|
+
The answer is the level at which cumulative character coverage reaches 95% —
|
|
62
|
+
the point at which a reader can follow a passage and infer the rest.
|
|
63
|
+
|
|
64
|
+
```python
|
|
65
|
+
p = hsk30.grade("这项研究揭示了神经网络的内在缺陷。")
|
|
66
|
+
p.level # 6
|
|
67
|
+
p.label # '6'
|
|
68
|
+
p.chars # 16
|
|
69
|
+
p.ungraded # characters outside the 3,000
|
|
70
|
+
p.curve() # cumulative coverage at every level
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
### Three decisions that matter
|
|
74
|
+
|
|
75
|
+
**Character-level, not word-level.** Word-level grading is unusable on
|
|
76
|
+
segmented Chinese. Real segmenters emit phrase tokens (我的, 七点, 蓝色) that
|
|
77
|
+
are not entries in any graded word list, pushing "unknown" past 95% at every
|
|
78
|
+
level and reporting ordinary beginner text as off-scale. HSK 3.0 grades 3,000
|
|
79
|
+
characters separately from its words precisely because the character inventory
|
|
80
|
+
is what gates reading.
|
|
81
|
+
|
|
82
|
+
**The official character list, not a derived one.** Deriving character levels
|
|
83
|
+
from the lowest-level word containing each character agrees with the 2021
|
|
84
|
+
official list on 2,962 of 2,969 characters — and gets the family terms wrong. 哥, 妈,
|
|
85
|
+
妹, 弟 are level-1 characters whose only listed words (哥哥, 妈妈) sit at
|
|
86
|
+
level 4. Also wrong: 王, 第, 零. This package ships the official list.
|
|
87
|
+
|
|
88
|
+
**Proper nouns are excluded when identifiable.** A reader does not need the
|
|
89
|
+
puppy's name in their vocabulary; it is glossed in place. Counting names as
|
|
90
|
+
difficulty graded a story called "My Puppy Doudou" at HSK 4 on a beginner
|
|
91
|
+
shelf, entirely on the strength of 豆豆. Detection needs pinyin, so it is
|
|
92
|
+
available through `grade_tokens`:
|
|
93
|
+
|
|
94
|
+
```python
|
|
95
|
+
hsk30.grade_tokens([
|
|
96
|
+
{"hz": "我", "py": "wǒ"},
|
|
97
|
+
{"hz": "李明。", "py": "Lǐ Míng"}, # excluded
|
|
98
|
+
]).chars # 1
|
|
99
|
+
```
|
|
100
|
+
|
|
101
|
+
### Levels 7, 8 and 9 are one band
|
|
102
|
+
|
|
103
|
+
Neither document splits them — in the 2021 standard they share a single
|
|
104
|
+
5,599-word list and 1,200 characters. This package carries the band as level `7` and renders it
|
|
105
|
+
`"7-9"`. A vendor advertising an "HSK 8 word list" invented the split.
|
|
106
|
+
|
|
107
|
+
### Character budgets
|
|
108
|
+
|
|
109
|
+
Reaching a 95% bar means keeping the above-target share under 5%, so a single
|
|
110
|
+
character over that budget blocks the target on its own:
|
|
111
|
+
|
|
112
|
+
```python
|
|
113
|
+
share, offenders = hsk30.budget_violations(text, target=3)
|
|
114
|
+
```
|
|
115
|
+
|
|
116
|
+
This is how a short passage silently regresses when an otherwise harmless edit
|
|
117
|
+
repeats one hard character a fourth time.
|
|
118
|
+
|
|
119
|
+
### Grading collections
|
|
120
|
+
|
|
121
|
+
```python
|
|
122
|
+
shelf = hsk30.profile_shelf([hsk30.grade(t) for t in texts])
|
|
123
|
+
shelf.label # median text — not the pooled figure
|
|
124
|
+
shelf.span_label # 'HSK 2-3', the interquartile range
|
|
125
|
+
```
|
|
126
|
+
|
|
127
|
+
Reports the **median text**. Pooling every character in a shelf lets a handful
|
|
128
|
+
of hard texts speak for all of them: it reported "HSK 3" for a beginner shelf
|
|
129
|
+
on which 16 of 22 texts individually read at HSK 1–2, describing nothing
|
|
130
|
+
actually on the shelf.
|
|
131
|
+
|
|
132
|
+
## What's in this repository
|
|
133
|
+
|
|
134
|
+
| Path | Contents |
|
|
135
|
+
| --- | --- |
|
|
136
|
+
| `src/hsk30/` | The library and its five graded lists (MIT) |
|
|
137
|
+
| `corpus/` | 102 aligned graded readers, 1,185 sentences (CC BY 4.0) |
|
|
138
|
+
| `benchmark/` | HSKBench — controlled-difficulty generation |
|
|
139
|
+
| `paper/` | The accompanying paper and its figures |
|
|
140
|
+
| `scripts/reproduce.py` | Recomputes every published figure |
|
|
141
|
+
| `scripts/extract_syllabus_2025.py` | Parses the official syllabus PDF |
|
|
142
|
+
| `corpus/syllabus2025/PROVENANCE.md` | Where the 2025 tables come from, and their rights position |
|
|
143
|
+
|
|
144
|
+
## HSKBench
|
|
145
|
+
|
|
146
|
+
Generating text *at* a level turns out to be much harder than grading it.
|
|
147
|
+
Human authors writing to an explicit target hit it **61.8%** of the time,
|
|
148
|
+
overshooting at the easy end and undershooting at the hard end. HSKBench scores
|
|
149
|
+
that task objectively — the grader is the metric, the way a compiler is the
|
|
150
|
+
metric for generated code. See [`benchmark/README.md`](https://github.com/harukicoder/hsk30/blob/main/benchmark/README.md).
|
|
151
|
+
|
|
152
|
+
## Versions
|
|
153
|
+
|
|
154
|
+
Three documents are routinely conflated, including by commercial HSK sites.
|
|
155
|
+
They are different, and it matters which one a tool grades against.
|
|
156
|
+
|
|
157
|
+
| `standard=` | Document | Date | Words | Characters |
|
|
158
|
+
| --- | --- | --- | ---: | ---: |
|
|
159
|
+
| `"2.0"` | HSK 2.0 exam lists | 2009–10 | 4,991 | — |
|
|
160
|
+
| `"2021"` | 《国际中文教育中文水平等级标准》 (GF0025-2021) | in force 1 Jul 2021 | 10,916 | 3,000 |
|
|
161
|
+
| `"2025"` *(default)* | 新版HSK考试大纲 | pub. Nov 2025, in force Jul 2026 | 10,896 | 3,088 |
|
|
162
|
+
|
|
163
|
+
The 2021 document is a national language standard (语言文字规范) from the
|
|
164
|
+
Ministry of Education and the State Language Commission. The 2025 document is
|
|
165
|
+
the examination syllabus from the Center for Language Education and Cooperation
|
|
166
|
+
(中外语言交流合作中心) and governs the test learners actually sit — which is why
|
|
167
|
+
it is the default.
|
|
168
|
+
|
|
169
|
+
HSK 2.0 graded no characters separately, so `characters("2.0")` raises.
|
|
170
|
+
|
|
171
|
+
**The 2025 lists are extracted from the official 406-page PDF** by
|
|
172
|
+
`scripts/extract_syllabus_2025.py`, which self-validates: parsed per-level entry
|
|
173
|
+
counts reproduce the published cumulative totals (300 / 500 / 1,000 / 2,000 /
|
|
174
|
+
3,600 / 5,400 / 11,000) exactly. Two notes from doing it — the syllabus numbers
|
|
175
|
+
11,000 *entries* but only 10,896 distinct words (homographs like 所/所2 get
|
|
176
|
+
their own rows), and it grades **3,088** recognition characters, not the 3,079
|
|
177
|
+
widely reported.
|
|
178
|
+
|
|
179
|
+
## Data sources
|
|
180
|
+
|
|
181
|
+
| Source | Provides | Licence |
|
|
182
|
+
| --- | --- | --- |
|
|
183
|
+
| [ivankra/hsk30](https://github.com/ivankra/hsk30) | HSK 3.0 word and character lists | MIT |
|
|
184
|
+
| [drkameleon/complete-hsk-vocabulary](https://github.com/drkameleon/complete-hsk-vocabulary) | HSK 2.0 levels, pinyin, glosses | MIT |
|
|
185
|
+
|
|
186
|
+
Both are transcriptions of 《国际中文教育中文水平等级标准》. Regenerate the
|
|
187
|
+
shipped tables with `python3 scripts/gen_data.py` (needs network).
|
|
188
|
+
|
|
189
|
+
## Limitations
|
|
190
|
+
|
|
191
|
+
- **Simplified characters only.** Convert traditional text with OpenCC first.
|
|
192
|
+
- **Coverage is not comprehension.** 95% character coverage is a necessary
|
|
193
|
+
condition for fluent reading, not a sufficient one; grammar, register and
|
|
194
|
+
world knowledge are not modelled.
|
|
195
|
+
- **Proper-noun detection needs pinyin.** `grade()` on a bare string cannot
|
|
196
|
+
identify names; pass them via `exclude=`, or use `grade_tokens()`.
|
|
197
|
+
- **CJK Extension A–F characters are treated as ungraded**, which is correct
|
|
198
|
+
under the standard but means literary text scores off-scale readily.
|
|
199
|
+
- **Authored segmentation** in the corpus groups some phrases a segmenter
|
|
200
|
+
would split.
|
|
201
|
+
|
|
202
|
+
## Development
|
|
203
|
+
|
|
204
|
+
```bash
|
|
205
|
+
git clone https://github.com/harukicoder/hsk30 && cd hsk30
|
|
206
|
+
pip install -e ".[dev]"
|
|
207
|
+
pytest # or: python3 tests/test_hsk30.py
|
|
208
|
+
python3 scripts/reproduce.py # every figure in the paper
|
|
209
|
+
```
|
|
210
|
+
|
|
211
|
+
The library is a port of the implementation that runs
|
|
212
|
+
[pinyora.com](https://pinyora.com). It reproduces that implementation's output
|
|
213
|
+
on all 102 corpus texts exactly (`test_python_reproduces_the_javascript_reference_exactly`),
|
|
214
|
+
with one deliberate fix: the original's ASCII-only `^[A-Z]` proper-noun test
|
|
215
|
+
missed names romanised with an accented capital — Ōuzhōu, Ōuyà, Ā Q Zhèngzhuàn
|
|
216
|
+
— and since 欧 and 洲 are both HSK 7–9 characters, missing one place name moved
|
|
217
|
+
a text two levels. The legacy behaviour remains available as
|
|
218
|
+
`is_proper_noun_ascii`.
|
|
219
|
+
|
|
220
|
+
## Citation
|
|
221
|
+
|
|
222
|
+
```bibtex
|
|
223
|
+
@misc{serrano2026hsk30,
|
|
224
|
+
title = {hsk30: Grading Chinese Text Against the HSK 3.0 Standard},
|
|
225
|
+
author = {Serrano, Alvaro},
|
|
226
|
+
year = {2026},
|
|
227
|
+
url = {https://github.com/harukicoder/hsk30}
|
|
228
|
+
}
|
|
229
|
+
```
|
|
230
|
+
|
|
231
|
+
## Licence
|
|
232
|
+
|
|
233
|
+
MIT for the code and the derived level tables; **CC BY 4.0** for the corpus
|
|
234
|
+
(see `corpus/LICENSE`).
|
|
@@ -0,0 +1,40 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=61"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "hsk30"
|
|
7
|
+
version = "0.1.1"
|
|
8
|
+
description = "Grade Chinese text against either official document called HSK 3.0"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
requires-python = ">=3.9"
|
|
11
|
+
license = { text = "MIT" }
|
|
12
|
+
keywords = ["chinese", "mandarin", "hsk", "readability", "language-learning", "cefr"]
|
|
13
|
+
classifiers = [
|
|
14
|
+
"Development Status :: 4 - Beta",
|
|
15
|
+
"Intended Audience :: Education",
|
|
16
|
+
"Intended Audience :: Science/Research",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Natural Language :: Chinese (Simplified)",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3.9",
|
|
21
|
+
"Topic :: Text Processing :: Linguistic",
|
|
22
|
+
]
|
|
23
|
+
dependencies = []
|
|
24
|
+
|
|
25
|
+
[project.optional-dependencies]
|
|
26
|
+
dev = ["pytest>=7"]
|
|
27
|
+
|
|
28
|
+
[project.urls]
|
|
29
|
+
Homepage = "https://github.com/harukicoder/hsk30"
|
|
30
|
+
Source = "https://github.com/harukicoder/hsk30"
|
|
31
|
+
Issues = "https://github.com/harukicoder/hsk30/issues"
|
|
32
|
+
|
|
33
|
+
[project.scripts]
|
|
34
|
+
hsk30 = "hsk30.cli:main"
|
|
35
|
+
|
|
36
|
+
[tool.setuptools.packages.find]
|
|
37
|
+
where = ["src"]
|
|
38
|
+
|
|
39
|
+
[tool.setuptools.package-data]
|
|
40
|
+
hsk30 = ["data/*.tsv"]
|
hsk30-0.1.1/setup.cfg
ADDED
|
@@ -0,0 +1,50 @@
|
|
|
1
|
+
"""Grade Chinese text against the HSK 3.0 standard.
|
|
2
|
+
|
|
3
|
+
>>> import hsk30
|
|
4
|
+
>>> hsk30.grade("我每天早上七点起床,然后去公园跑步。").label
|
|
5
|
+
'3'
|
|
6
|
+
|
|
7
|
+
Grades against 《国际中文教育中文水平等级标准》 (GF0025-2021), the Chinese
|
|
8
|
+
Proficiency Grading Standards for International Chinese Language Education,
|
|
9
|
+
issued by the Ministry of Education and the State Language Commission and in
|
|
10
|
+
force as a national language standard since **1 July 2021**.
|
|
11
|
+
|
|
12
|
+
It is not a renumbering of HSK 2.0: of the 4,482 words both list, only 814 keep
|
|
13
|
+
their level. Tooling calibrated to HSK 2.0 is therefore not approximately
|
|
14
|
+
right — it is wrong for four words in five.
|
|
15
|
+
|
|
16
|
+
NOTE ON VERSIONS. The 2021 grading standard is *not* the HSK 3.0 exam
|
|
17
|
+
syllabus. A separate 406-page exam syllabus (新版HSK考试大纲) was published in
|
|
18
|
+
November 2025 and takes effect in July 2026; it uses different lists —
|
|
19
|
+
cumulative word counts 300/500/1,000/2,000/3,600/5,400/11,000 against this
|
|
20
|
+
standard's 485/1,227/2,171/3,143/4,199/5,317/10,916, and roughly 3,079
|
|
21
|
+
recognition characters against this standard's 3,000. This package grades
|
|
22
|
+
against the 2021 standard. Do not describe its output as the exam syllabus.
|
|
23
|
+
"""
|
|
24
|
+
|
|
25
|
+
from .data import BAND, DEFAULT_STANDARD, LEVELS, characters, label, resolve, words
|
|
26
|
+
from .grade import (
|
|
27
|
+
DEFAULT_BUDGET,
|
|
28
|
+
DEFAULT_THRESHOLD,
|
|
29
|
+
Profile,
|
|
30
|
+
ShelfProfile,
|
|
31
|
+
budget_violations,
|
|
32
|
+
grade,
|
|
33
|
+
grade_tokens,
|
|
34
|
+
hanzi,
|
|
35
|
+
is_proper_noun,
|
|
36
|
+
is_proper_noun_ascii,
|
|
37
|
+
profile_shelf,
|
|
38
|
+
strip_punct,
|
|
39
|
+
)
|
|
40
|
+
|
|
41
|
+
__version__ = "0.1.1"
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"BAND", "LEVELS", "DEFAULT_BUDGET", "DEFAULT_THRESHOLD", "DEFAULT_STANDARD",
|
|
45
|
+
"resolve",
|
|
46
|
+
"Profile", "ShelfProfile",
|
|
47
|
+
"budget_violations", "characters", "grade", "grade_tokens", "hanzi",
|
|
48
|
+
"is_proper_noun", "is_proper_noun_ascii", "label", "profile_shelf", "strip_punct", "words",
|
|
49
|
+
"__version__",
|
|
50
|
+
]
|