hsk30 0.1.1__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
hsk30-0.1.1/LICENSE ADDED
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 Alvaro Serrano
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
hsk30-0.1.1/PKG-INFO ADDED
@@ -0,0 +1,258 @@
1
+ Metadata-Version: 2.4
2
+ Name: hsk30
3
+ Version: 0.1.1
4
+ Summary: Grade Chinese text against either official document called HSK 3.0
5
+ License: MIT
6
+ Project-URL: Homepage, https://github.com/harukicoder/hsk30
7
+ Project-URL: Source, https://github.com/harukicoder/hsk30
8
+ Project-URL: Issues, https://github.com/harukicoder/hsk30/issues
9
+ Keywords: chinese,mandarin,hsk,readability,language-learning,cefr
10
+ Classifier: Development Status :: 4 - Beta
11
+ Classifier: Intended Audience :: Education
12
+ Classifier: Intended Audience :: Science/Research
13
+ Classifier: License :: OSI Approved :: MIT License
14
+ Classifier: Natural Language :: Chinese (Simplified)
15
+ Classifier: Programming Language :: Python :: 3
16
+ Classifier: Programming Language :: Python :: 3.9
17
+ Classifier: Topic :: Text Processing :: Linguistic
18
+ Requires-Python: >=3.9
19
+ Description-Content-Type: text/markdown
20
+ License-File: LICENSE
21
+ Provides-Extra: dev
22
+ Requires-Dist: pytest>=7; extra == "dev"
23
+ Dynamic: license-file
24
+
25
+ # hsk30
26
+
27
+ Grade Chinese text against HSK — against **either** document that gets called
28
+ "HSK 3.0", and it will tell you which one it used.
29
+
30
+ ```bash
31
+ pip install hsk30
32
+ ```
33
+
34
+ ```python
35
+ import hsk30
36
+
37
+ hsk30.grade("我每天早上七点起床,然后去公园跑步。").label
38
+ # '3' ← graded against the 2026 examination syllabus, the default
39
+
40
+ hsk30.grade("...", standard="2021").label
41
+ # the GF0025-2021 national grading standard instead
42
+ ```
43
+
44
+ > **"HSK 3.0" is two different documents, and the choice changes the answer.**
45
+ > They disagree on **41.5%** of shared vocabulary and **40.7%** of shared
46
+ > characters. Regrading 102 authentic graded readers against one rather than
47
+ > the other changes the level of **48% of them**, almost always upward. This
48
+ > library defaults to the examination syllabus in force since July 2026 and
49
+ > records `profile.standard` on every result. See [Versions](#versions).
50
+
51
+ ```bash
52
+ $ hsk30 "我每天早上七点起床,然后去公园跑步。" --curve --target 2
53
+ HSK 3 (16 characters, 0 ungraded)
54
+ HSK 1 68.8% ############################
55
+ HSK 2 87.5% ###################################
56
+ HSK 3 100.0% ########################################
57
+ target HSK 2: misses the 95% bar (12.5% above target)
58
+ 步 HSK 3 6.2% <- over budget on its own
59
+ 每 HSK 3 6.2% <- over budget on its own
60
+ ```
61
+
62
+ No dependencies. Python 3.9+.
63
+
64
+ ## Why this exists
65
+
66
+ Most Chinese-learning tools report an "HSK 3.0 level" without saying which
67
+ document produced it. There are two, published four years apart:
68
+
69
+ | Comparison | Shared words | Same level | Moved |
70
+ | --- | ---: | ---: | ---: |
71
+ | HSK 2.0 → GF0025-2021 | 4,482 | 814 | 3,668 (81.8%) |
72
+ | HSK 2.0 → 2025 syllabus | 4,802 | 2,349 | 2,453 (51.1%) |
73
+ | **GF0025-2021 → 2025 syllabus** | **9,674** | **5,662** | **4,012 (41.5%)** |
74
+
75
+ Judged against the 2021 standard, HSK 2.0 looks almost entirely regraded.
76
+ Judged against the examination syllabus, barely half moved. The syllabus is
77
+ markedly more conservative, and it is the document learners are actually tested
78
+ on.
79
+
80
+ Every figure in this README is produced by `python3 scripts/reproduce.py`.
81
+
82
+ ## What it does
83
+
84
+ Answers one question: **what HSK level does a reader need to read this text?**
85
+ The answer is the level at which cumulative character coverage reaches 95% —
86
+ the point at which a reader can follow a passage and infer the rest.
87
+
88
+ ```python
89
+ p = hsk30.grade("这项研究揭示了神经网络的内在缺陷。")
90
+ p.level # 6
91
+ p.label # '6'
92
+ p.chars # 16
93
+ p.ungraded # characters outside the 3,000
94
+ p.curve() # cumulative coverage at every level
95
+ ```
96
+
97
+ ### Three decisions that matter
98
+
99
+ **Character-level, not word-level.** Word-level grading is unusable on
100
+ segmented Chinese. Real segmenters emit phrase tokens (我的, 七点, 蓝色) that
101
+ are not entries in any graded word list, pushing "unknown" past 95% at every
102
+ level and reporting ordinary beginner text as off-scale. HSK 3.0 grades 3,000
103
+ characters separately from its words precisely because the character inventory
104
+ is what gates reading.
105
+
106
+ **The official character list, not a derived one.** Deriving character levels
107
+ from the lowest-level word containing each character agrees with the 2021
108
+ official list on 2,962 of 2,969 characters — and gets the family terms wrong. 哥, 妈,
109
+ 妹, 弟 are level-1 characters whose only listed words (哥哥, 妈妈) sit at
110
+ level 4. Also wrong: 王, 第, 零. This package ships the official list.
111
+
112
+ **Proper nouns are excluded when identifiable.** A reader does not need the
113
+ puppy's name in their vocabulary; it is glossed in place. Counting names as
114
+ difficulty graded a story called "My Puppy Doudou" at HSK 4 on a beginner
115
+ shelf, entirely on the strength of 豆豆. Detection needs pinyin, so it is
116
+ available through `grade_tokens`:
117
+
118
+ ```python
119
+ hsk30.grade_tokens([
120
+ {"hz": "我", "py": "wǒ"},
121
+ {"hz": "李明。", "py": "Lǐ Míng"}, # excluded
122
+ ]).chars # 1
123
+ ```
124
+
125
+ ### Levels 7, 8 and 9 are one band
126
+
127
+ Neither document splits them — in the 2021 standard they share a single
128
+ 5,599-word list and 1,200 characters. This package carries the band as level `7` and renders it
129
+ `"7-9"`. A vendor advertising an "HSK 8 word list" invented the split.
130
+
131
+ ### Character budgets
132
+
133
+ Reaching a 95% bar means keeping the above-target share under 5%, so a single
134
+ character over that budget blocks the target on its own:
135
+
136
+ ```python
137
+ share, offenders = hsk30.budget_violations(text, target=3)
138
+ ```
139
+
140
+ This is how a short passage silently regresses when an otherwise harmless edit
141
+ repeats one hard character a fourth time.
142
+
143
+ ### Grading collections
144
+
145
+ ```python
146
+ shelf = hsk30.profile_shelf([hsk30.grade(t) for t in texts])
147
+ shelf.label # median text — not the pooled figure
148
+ shelf.span_label # 'HSK 2-3', the interquartile range
149
+ ```
150
+
151
+ Reports the **median text**. Pooling every character in a shelf lets a handful
152
+ of hard texts speak for all of them: it reported "HSK 3" for a beginner shelf
153
+ on which 16 of 22 texts individually read at HSK 1–2, describing nothing
154
+ actually on the shelf.
155
+
156
+ ## What's in this repository
157
+
158
+ | Path | Contents |
159
+ | --- | --- |
160
+ | `src/hsk30/` | The library and its five graded lists (MIT) |
161
+ | `corpus/` | 102 aligned graded readers, 1,185 sentences (CC BY 4.0) |
162
+ | `benchmark/` | HSKBench — controlled-difficulty generation |
163
+ | `paper/` | The accompanying paper and its figures |
164
+ | `scripts/reproduce.py` | Recomputes every published figure |
165
+ | `scripts/extract_syllabus_2025.py` | Parses the official syllabus PDF |
166
+ | `corpus/syllabus2025/PROVENANCE.md` | Where the 2025 tables come from, and their rights position |
167
+
168
+ ## HSKBench
169
+
170
+ Generating text *at* a level turns out to be much harder than grading it.
171
+ Human authors writing to an explicit target hit it **61.8%** of the time,
172
+ overshooting at the easy end and undershooting at the hard end. HSKBench scores
173
+ that task objectively — the grader is the metric, the way a compiler is the
174
+ metric for generated code. See [`benchmark/README.md`](https://github.com/harukicoder/hsk30/blob/main/benchmark/README.md).
175
+
176
+ ## Versions
177
+
178
+ Three documents are routinely conflated, including by commercial HSK sites.
179
+ They are different, and it matters which one a tool grades against.
180
+
181
+ | `standard=` | Document | Date | Words | Characters |
182
+ | --- | --- | --- | ---: | ---: |
183
+ | `"2.0"` | HSK 2.0 exam lists | 2009–10 | 4,991 | — |
184
+ | `"2021"` | 《国际中文教育中文水平等级标准》 (GF0025-2021) | in force 1 Jul 2021 | 10,916 | 3,000 |
185
+ | `"2025"` *(default)* | 新版HSK考试大纲 | pub. Nov 2025, in force Jul 2026 | 10,896 | 3,088 |
186
+
187
+ The 2021 document is a national language standard (语言文字规范) from the
188
+ Ministry of Education and the State Language Commission. The 2025 document is
189
+ the examination syllabus from the Center for Language Education and Cooperation
190
+ (中外语言交流合作中心) and governs the test learners actually sit — which is why
191
+ it is the default.
192
+
193
+ HSK 2.0 graded no characters separately, so `characters("2.0")` raises.
194
+
195
+ **The 2025 lists are extracted from the official 406-page PDF** by
196
+ `scripts/extract_syllabus_2025.py`, which self-validates: parsed per-level entry
197
+ counts reproduce the published cumulative totals (300 / 500 / 1,000 / 2,000 /
198
+ 3,600 / 5,400 / 11,000) exactly. Two notes from doing it — the syllabus numbers
199
+ 11,000 *entries* but only 10,896 distinct words (homographs like 所/所2 get
200
+ their own rows), and it grades **3,088** recognition characters, not the 3,079
201
+ widely reported.
202
+
203
+ ## Data sources
204
+
205
+ | Source | Provides | Licence |
206
+ | --- | --- | --- |
207
+ | [ivankra/hsk30](https://github.com/ivankra/hsk30) | HSK 3.0 word and character lists | MIT |
208
+ | [drkameleon/complete-hsk-vocabulary](https://github.com/drkameleon/complete-hsk-vocabulary) | HSK 2.0 levels, pinyin, glosses | MIT |
209
+
210
+ Both are transcriptions of 《国际中文教育中文水平等级标准》. Regenerate the
211
+ shipped tables with `python3 scripts/gen_data.py` (needs network).
212
+
213
+ ## Limitations
214
+
215
+ - **Simplified characters only.** Convert traditional text with OpenCC first.
216
+ - **Coverage is not comprehension.** 95% character coverage is a necessary
217
+ condition for fluent reading, not a sufficient one; grammar, register and
218
+ world knowledge are not modelled.
219
+ - **Proper-noun detection needs pinyin.** `grade()` on a bare string cannot
220
+ identify names; pass them via `exclude=`, or use `grade_tokens()`.
221
+ - **CJK Extension A–F characters are treated as ungraded**, which is correct
222
+ under the standard but means literary text scores off-scale readily.
223
+ - **Authored segmentation** in the corpus groups some phrases a segmenter
224
+ would split.
225
+
226
+ ## Development
227
+
228
+ ```bash
229
+ git clone https://github.com/harukicoder/hsk30 && cd hsk30
230
+ pip install -e ".[dev]"
231
+ pytest # or: python3 tests/test_hsk30.py
232
+ python3 scripts/reproduce.py # every figure in the paper
233
+ ```
234
+
235
+ The library is a port of the implementation that runs
236
+ [pinyora.com](https://pinyora.com). It reproduces that implementation's output
237
+ on all 102 corpus texts exactly (`test_python_reproduces_the_javascript_reference_exactly`),
238
+ with one deliberate fix: the original's ASCII-only `^[A-Z]` proper-noun test
239
+ missed names romanised with an accented capital — Ōuzhōu, Ōuyà, Ā Q Zhèngzhuàn
240
+ — and since 欧 and 洲 are both HSK 7–9 characters, missing one place name moved
241
+ a text two levels. The legacy behaviour remains available as
242
+ `is_proper_noun_ascii`.
243
+
244
+ ## Citation
245
+
246
+ ```bibtex
247
+ @misc{serrano2026hsk30,
248
+ title = {hsk30: Grading Chinese Text Against the HSK 3.0 Standard},
249
+ author = {Serrano, Alvaro},
250
+ year = {2026},
251
+ url = {https://github.com/harukicoder/hsk30}
252
+ }
253
+ ```
254
+
255
+ ## Licence
256
+
257
+ MIT for the code and the derived level tables; **CC BY 4.0** for the corpus
258
+ (see `corpus/LICENSE`).
hsk30-0.1.1/README.md ADDED
@@ -0,0 +1,234 @@
1
+ # hsk30
2
+
3
+ Grade Chinese text against HSK — against **either** document that gets called
4
+ "HSK 3.0", and it will tell you which one it used.
5
+
6
+ ```bash
7
+ pip install hsk30
8
+ ```
9
+
10
+ ```python
11
+ import hsk30
12
+
13
+ hsk30.grade("我每天早上七点起床,然后去公园跑步。").label
14
+ # '3' ← graded against the 2026 examination syllabus, the default
15
+
16
+ hsk30.grade("...", standard="2021").label
17
+ # the GF0025-2021 national grading standard instead
18
+ ```
19
+
20
+ > **"HSK 3.0" is two different documents, and the choice changes the answer.**
21
+ > They disagree on **41.5%** of shared vocabulary and **40.7%** of shared
22
+ > characters. Regrading 102 authentic graded readers against one rather than
23
+ > the other changes the level of **48% of them**, almost always upward. This
24
+ > library defaults to the examination syllabus in force since July 2026 and
25
+ > records `profile.standard` on every result. See [Versions](#versions).
26
+
27
+ ```bash
28
+ $ hsk30 "我每天早上七点起床,然后去公园跑步。" --curve --target 2
29
+ HSK 3 (16 characters, 0 ungraded)
30
+ HSK 1 68.8% ############################
31
+ HSK 2 87.5% ###################################
32
+ HSK 3 100.0% ########################################
33
+ target HSK 2: misses the 95% bar (12.5% above target)
34
+ 步 HSK 3 6.2% <- over budget on its own
35
+ 每 HSK 3 6.2% <- over budget on its own
36
+ ```
37
+
38
+ No dependencies. Python 3.9+.
39
+
40
+ ## Why this exists
41
+
42
+ Most Chinese-learning tools report an "HSK 3.0 level" without saying which
43
+ document produced it. There are two, published four years apart:
44
+
45
+ | Comparison | Shared words | Same level | Moved |
46
+ | --- | ---: | ---: | ---: |
47
+ | HSK 2.0 → GF0025-2021 | 4,482 | 814 | 3,668 (81.8%) |
48
+ | HSK 2.0 → 2025 syllabus | 4,802 | 2,349 | 2,453 (51.1%) |
49
+ | **GF0025-2021 → 2025 syllabus** | **9,674** | **5,662** | **4,012 (41.5%)** |
50
+
51
+ Judged against the 2021 standard, HSK 2.0 looks almost entirely regraded.
52
+ Judged against the examination syllabus, barely half moved. The syllabus is
53
+ markedly more conservative, and it is the document learners are actually tested
54
+ on.
55
+
56
+ Every figure in this README is produced by `python3 scripts/reproduce.py`.
57
+
58
+ ## What it does
59
+
60
+ Answers one question: **what HSK level does a reader need to read this text?**
61
+ The answer is the level at which cumulative character coverage reaches 95% —
62
+ the point at which a reader can follow a passage and infer the rest.
63
+
64
+ ```python
65
+ p = hsk30.grade("这项研究揭示了神经网络的内在缺陷。")
66
+ p.level # 6
67
+ p.label # '6'
68
+ p.chars # 16
69
+ p.ungraded # characters outside the 3,000
70
+ p.curve() # cumulative coverage at every level
71
+ ```
72
+
73
+ ### Three decisions that matter
74
+
75
+ **Character-level, not word-level.** Word-level grading is unusable on
76
+ segmented Chinese. Real segmenters emit phrase tokens (我的, 七点, 蓝色) that
77
+ are not entries in any graded word list, pushing "unknown" past 95% at every
78
+ level and reporting ordinary beginner text as off-scale. HSK 3.0 grades 3,000
79
+ characters separately from its words precisely because the character inventory
80
+ is what gates reading.
81
+
82
+ **The official character list, not a derived one.** Deriving character levels
83
+ from the lowest-level word containing each character agrees with the 2021
84
+ official list on 2,962 of 2,969 characters — and gets the family terms wrong. 哥, 妈,
85
+ 妹, 弟 are level-1 characters whose only listed words (哥哥, 妈妈) sit at
86
+ level 4. Also wrong: 王, 第, 零. This package ships the official list.
87
+
88
+ **Proper nouns are excluded when identifiable.** A reader does not need the
89
+ puppy's name in their vocabulary; it is glossed in place. Counting names as
90
+ difficulty graded a story called "My Puppy Doudou" at HSK 4 on a beginner
91
+ shelf, entirely on the strength of 豆豆. Detection needs pinyin, so it is
92
+ available through `grade_tokens`:
93
+
94
+ ```python
95
+ hsk30.grade_tokens([
96
+ {"hz": "我", "py": "wǒ"},
97
+ {"hz": "李明。", "py": "Lǐ Míng"}, # excluded
98
+ ]).chars # 1
99
+ ```
100
+
101
+ ### Levels 7, 8 and 9 are one band
102
+
103
+ Neither document splits them — in the 2021 standard they share a single
104
+ 5,599-word list and 1,200 characters. This package carries the band as level `7` and renders it
105
+ `"7-9"`. A vendor advertising an "HSK 8 word list" invented the split.
106
+
107
+ ### Character budgets
108
+
109
+ Reaching a 95% bar means keeping the above-target share under 5%, so a single
110
+ character over that budget blocks the target on its own:
111
+
112
+ ```python
113
+ share, offenders = hsk30.budget_violations(text, target=3)
114
+ ```
115
+
116
+ This is how a short passage silently regresses when an otherwise harmless edit
117
+ repeats one hard character a fourth time.
118
+
119
+ ### Grading collections
120
+
121
+ ```python
122
+ shelf = hsk30.profile_shelf([hsk30.grade(t) for t in texts])
123
+ shelf.label # median text — not the pooled figure
124
+ shelf.span_label # 'HSK 2-3', the interquartile range
125
+ ```
126
+
127
+ Reports the **median text**. Pooling every character in a shelf lets a handful
128
+ of hard texts speak for all of them: it reported "HSK 3" for a beginner shelf
129
+ on which 16 of 22 texts individually read at HSK 1–2, describing nothing
130
+ actually on the shelf.
131
+
132
+ ## What's in this repository
133
+
134
+ | Path | Contents |
135
+ | --- | --- |
136
+ | `src/hsk30/` | The library and its five graded lists (MIT) |
137
+ | `corpus/` | 102 aligned graded readers, 1,185 sentences (CC BY 4.0) |
138
+ | `benchmark/` | HSKBench — controlled-difficulty generation |
139
+ | `paper/` | The accompanying paper and its figures |
140
+ | `scripts/reproduce.py` | Recomputes every published figure |
141
+ | `scripts/extract_syllabus_2025.py` | Parses the official syllabus PDF |
142
+ | `corpus/syllabus2025/PROVENANCE.md` | Where the 2025 tables come from, and their rights position |
143
+
144
+ ## HSKBench
145
+
146
+ Generating text *at* a level turns out to be much harder than grading it.
147
+ Human authors writing to an explicit target hit it **61.8%** of the time,
148
+ overshooting at the easy end and undershooting at the hard end. HSKBench scores
149
+ that task objectively — the grader is the metric, the way a compiler is the
150
+ metric for generated code. See [`benchmark/README.md`](https://github.com/harukicoder/hsk30/blob/main/benchmark/README.md).
151
+
152
+ ## Versions
153
+
154
+ Three documents are routinely conflated, including by commercial HSK sites.
155
+ They are different, and it matters which one a tool grades against.
156
+
157
+ | `standard=` | Document | Date | Words | Characters |
158
+ | --- | --- | --- | ---: | ---: |
159
+ | `"2.0"` | HSK 2.0 exam lists | 2009–10 | 4,991 | — |
160
+ | `"2021"` | 《国际中文教育中文水平等级标准》 (GF0025-2021) | in force 1 Jul 2021 | 10,916 | 3,000 |
161
+ | `"2025"` *(default)* | 新版HSK考试大纲 | pub. Nov 2025, in force Jul 2026 | 10,896 | 3,088 |
162
+
163
+ The 2021 document is a national language standard (语言文字规范) from the
164
+ Ministry of Education and the State Language Commission. The 2025 document is
165
+ the examination syllabus from the Center for Language Education and Cooperation
166
+ (中外语言交流合作中心) and governs the test learners actually sit — which is why
167
+ it is the default.
168
+
169
+ HSK 2.0 graded no characters separately, so `characters("2.0")` raises.
170
+
171
+ **The 2025 lists are extracted from the official 406-page PDF** by
172
+ `scripts/extract_syllabus_2025.py`, which self-validates: parsed per-level entry
173
+ counts reproduce the published cumulative totals (300 / 500 / 1,000 / 2,000 /
174
+ 3,600 / 5,400 / 11,000) exactly. Two notes from doing it — the syllabus numbers
175
+ 11,000 *entries* but only 10,896 distinct words (homographs like 所/所2 get
176
+ their own rows), and it grades **3,088** recognition characters, not the 3,079
177
+ widely reported.
178
+
179
+ ## Data sources
180
+
181
+ | Source | Provides | Licence |
182
+ | --- | --- | --- |
183
+ | [ivankra/hsk30](https://github.com/ivankra/hsk30) | HSK 3.0 word and character lists | MIT |
184
+ | [drkameleon/complete-hsk-vocabulary](https://github.com/drkameleon/complete-hsk-vocabulary) | HSK 2.0 levels, pinyin, glosses | MIT |
185
+
186
+ Both are transcriptions of 《国际中文教育中文水平等级标准》. Regenerate the
187
+ shipped tables with `python3 scripts/gen_data.py` (needs network).
188
+
189
+ ## Limitations
190
+
191
+ - **Simplified characters only.** Convert traditional text with OpenCC first.
192
+ - **Coverage is not comprehension.** 95% character coverage is a necessary
193
+ condition for fluent reading, not a sufficient one; grammar, register and
194
+ world knowledge are not modelled.
195
+ - **Proper-noun detection needs pinyin.** `grade()` on a bare string cannot
196
+ identify names; pass them via `exclude=`, or use `grade_tokens()`.
197
+ - **CJK Extension A–F characters are treated as ungraded**, which is correct
198
+ under the standard but means literary text scores off-scale readily.
199
+ - **Authored segmentation** in the corpus groups some phrases a segmenter
200
+ would split.
201
+
202
+ ## Development
203
+
204
+ ```bash
205
+ git clone https://github.com/harukicoder/hsk30 && cd hsk30
206
+ pip install -e ".[dev]"
207
+ pytest # or: python3 tests/test_hsk30.py
208
+ python3 scripts/reproduce.py # every figure in the paper
209
+ ```
210
+
211
+ The library is a port of the implementation that runs
212
+ [pinyora.com](https://pinyora.com). It reproduces that implementation's output
213
+ on all 102 corpus texts exactly (`test_python_reproduces_the_javascript_reference_exactly`),
214
+ with one deliberate fix: the original's ASCII-only `^[A-Z]` proper-noun test
215
+ missed names romanised with an accented capital — Ōuzhōu, Ōuyà, Ā Q Zhèngzhuàn
216
+ — and since 欧 and 洲 are both HSK 7–9 characters, missing one place name moved
217
+ a text two levels. The legacy behaviour remains available as
218
+ `is_proper_noun_ascii`.
219
+
220
+ ## Citation
221
+
222
+ ```bibtex
223
+ @misc{serrano2026hsk30,
224
+ title = {hsk30: Grading Chinese Text Against the HSK 3.0 Standard},
225
+ author = {Serrano, Alvaro},
226
+ year = {2026},
227
+ url = {https://github.com/harukicoder/hsk30}
228
+ }
229
+ ```
230
+
231
+ ## Licence
232
+
233
+ MIT for the code and the derived level tables; **CC BY 4.0** for the corpus
234
+ (see `corpus/LICENSE`).
@@ -0,0 +1,40 @@
1
+ [build-system]
2
+ requires = ["setuptools>=61"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "hsk30"
7
+ version = "0.1.1"
8
+ description = "Grade Chinese text against either official document called HSK 3.0"
9
+ readme = "README.md"
10
+ requires-python = ">=3.9"
11
+ license = { text = "MIT" }
12
+ keywords = ["chinese", "mandarin", "hsk", "readability", "language-learning", "cefr"]
13
+ classifiers = [
14
+ "Development Status :: 4 - Beta",
15
+ "Intended Audience :: Education",
16
+ "Intended Audience :: Science/Research",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Natural Language :: Chinese (Simplified)",
19
+ "Programming Language :: Python :: 3",
20
+ "Programming Language :: Python :: 3.9",
21
+ "Topic :: Text Processing :: Linguistic",
22
+ ]
23
+ dependencies = []
24
+
25
+ [project.optional-dependencies]
26
+ dev = ["pytest>=7"]
27
+
28
+ [project.urls]
29
+ Homepage = "https://github.com/harukicoder/hsk30"
30
+ Source = "https://github.com/harukicoder/hsk30"
31
+ Issues = "https://github.com/harukicoder/hsk30/issues"
32
+
33
+ [project.scripts]
34
+ hsk30 = "hsk30.cli:main"
35
+
36
+ [tool.setuptools.packages.find]
37
+ where = ["src"]
38
+
39
+ [tool.setuptools.package-data]
40
+ hsk30 = ["data/*.tsv"]
hsk30-0.1.1/setup.cfg ADDED
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,50 @@
1
+ """Grade Chinese text against the HSK 3.0 standard.
2
+
3
+ >>> import hsk30
4
+ >>> hsk30.grade("我每天早上七点起床,然后去公园跑步。").label
5
+ '3'
6
+
7
+ Grades against 《国际中文教育中文水平等级标准》 (GF0025-2021), the Chinese
8
+ Proficiency Grading Standards for International Chinese Language Education,
9
+ issued by the Ministry of Education and the State Language Commission and in
10
+ force as a national language standard since **1 July 2021**.
11
+
12
+ It is not a renumbering of HSK 2.0: of the 4,482 words both list, only 814 keep
13
+ their level. Tooling calibrated to HSK 2.0 is therefore not approximately
14
+ right — it is wrong for four words in five.
15
+
16
+ NOTE ON VERSIONS. The 2021 grading standard is *not* the HSK 3.0 exam
17
+ syllabus. A separate 406-page exam syllabus (新版HSK考试大纲) was published in
18
+ November 2025 and takes effect in July 2026; it uses different lists —
19
+ cumulative word counts 300/500/1,000/2,000/3,600/5,400/11,000 against this
20
+ standard's 485/1,227/2,171/3,143/4,199/5,317/10,916, and roughly 3,079
21
+ recognition characters against this standard's 3,000. This package grades
22
+ against the 2021 standard. Do not describe its output as the exam syllabus.
23
+ """
24
+
25
+ from .data import BAND, DEFAULT_STANDARD, LEVELS, characters, label, resolve, words
26
+ from .grade import (
27
+ DEFAULT_BUDGET,
28
+ DEFAULT_THRESHOLD,
29
+ Profile,
30
+ ShelfProfile,
31
+ budget_violations,
32
+ grade,
33
+ grade_tokens,
34
+ hanzi,
35
+ is_proper_noun,
36
+ is_proper_noun_ascii,
37
+ profile_shelf,
38
+ strip_punct,
39
+ )
40
+
41
+ __version__ = "0.1.1"
42
+
43
+ __all__ = [
44
+ "BAND", "LEVELS", "DEFAULT_BUDGET", "DEFAULT_THRESHOLD", "DEFAULT_STANDARD",
45
+ "resolve",
46
+ "Profile", "ShelfProfile",
47
+ "budget_violations", "characters", "grade", "grade_tokens", "hanzi",
48
+ "is_proper_noun", "is_proper_noun_ascii", "label", "profile_shelf", "strip_punct", "words",
49
+ "__version__",
50
+ ]