worddael 0.1.0rc3__tar.gz
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- worddael-0.1.0rc3/LICENSE +21 -0
- worddael-0.1.0rc3/PKG-INFO +201 -0
- worddael-0.1.0rc3/README.md +167 -0
- worddael-0.1.0rc3/pyproject.toml +58 -0
- worddael-0.1.0rc3/setup.cfg +4 -0
- worddael-0.1.0rc3/src/worddael/__init__.py +92 -0
- worddael-0.1.0rc3/src/worddael/adapters.py +45 -0
- worddael-0.1.0rc3/src/worddael/chunkers.py +474 -0
- worddael-0.1.0rc3/src/worddael/cli.py +137 -0
- worddael-0.1.0rc3/src/worddael/compare.py +66 -0
- worddael-0.1.0rc3/src/worddael/counters.py +128 -0
- worddael-0.1.0rc3/src/worddael/embedders.py +151 -0
- worddael-0.1.0rc3/src/worddael/eval/__init__.py +21 -0
- worddael-0.1.0rc3/src/worddael/eval/judge_runner.py +258 -0
- worddael-0.1.0rc3/src/worddael/eval/llm_judge.py +123 -0
- worddael-0.1.0rc3/src/worddael/eval/qa_gen.py +239 -0
- worddael-0.1.0rc3/src/worddael/eval/report.py +366 -0
- worddael-0.1.0rc3/src/worddael/eval/retrieval.py +241 -0
- worddael-0.1.0rc3/src/worddael/eval/sample_data.py +537 -0
- worddael-0.1.0rc3/src/worddael/explain.py +83 -0
- worddael-0.1.0rc3/src/worddael/io_utils.py +44 -0
- worddael-0.1.0rc3/src/worddael/parent_child.py +96 -0
- worddael-0.1.0rc3/src/worddael/py.typed +0 -0
- worddael-0.1.0rc3/src/worddael/sentences.py +114 -0
- worddael-0.1.0rc3/src/worddael/types.py +36 -0
- worddael-0.1.0rc3/src/worddael.egg-info/PKG-INFO +201 -0
- worddael-0.1.0rc3/src/worddael.egg-info/SOURCES.txt +50 -0
- worddael-0.1.0rc3/src/worddael.egg-info/dependency_links.txt +1 -0
- worddael-0.1.0rc3/src/worddael.egg-info/entry_points.txt +2 -0
- worddael-0.1.0rc3/src/worddael.egg-info/requires.txt +10 -0
- worddael-0.1.0rc3/src/worddael.egg-info/top_level.txt +1 -0
- worddael-0.1.0rc3/tests/test_adapters.py +63 -0
- worddael-0.1.0rc3/tests/test_chunkers.py +154 -0
- worddael-0.1.0rc3/tests/test_cli_io.py +80 -0
- worddael-0.1.0rc3/tests/test_compare.py +43 -0
- worddael-0.1.0rc3/tests/test_counters.py +35 -0
- worddael-0.1.0rc3/tests/test_docs_examples.py +102 -0
- worddael-0.1.0rc3/tests/test_embedders.py +173 -0
- worddael-0.1.0rc3/tests/test_eval.py +102 -0
- worddael-0.1.0rc3/tests/test_explain.py +130 -0
- worddael-0.1.0rc3/tests/test_hygiene.py +98 -0
- worddael-0.1.0rc3/tests/test_iter7.py +90 -0
- worddael-0.1.0rc3/tests/test_judge_runner.py +146 -0
- worddael-0.1.0rc3/tests/test_offsets.py +101 -0
- worddael-0.1.0rc3/tests/test_parent_child.py +99 -0
- worddael-0.1.0rc3/tests/test_perf_structure.py +79 -0
- worddael-0.1.0rc3/tests/test_polish.py +92 -0
- worddael-0.1.0rc3/tests/test_qa_gen.py +86 -0
- worddael-0.1.0rc3/tests/test_r10.py +74 -0
- worddael-0.1.0rc3/tests/test_r11.py +49 -0
- worddael-0.1.0rc3/tests/test_r12.py +61 -0
- worddael-0.1.0rc3/tests/test_traditional.py +59 -0
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
MIT License
|
|
2
|
+
|
|
3
|
+
Copyright (c) 2026 worddael contributors
|
|
4
|
+
|
|
5
|
+
Permission is hereby granted, free of charge, to any person obtaining a copy
|
|
6
|
+
of this software and associated documentation files (the "Software"), to deal
|
|
7
|
+
in the Software without restriction, including without limitation the rights
|
|
8
|
+
to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
|
|
9
|
+
copies of the Software, and to permit persons to whom the Software is
|
|
10
|
+
furnished to do so, subject to the following conditions:
|
|
11
|
+
|
|
12
|
+
The above copyright notice and this permission notice shall be included in all
|
|
13
|
+
copies or substantial portions of the Software.
|
|
14
|
+
|
|
15
|
+
THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
|
|
16
|
+
IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
|
|
17
|
+
FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
|
|
18
|
+
AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
|
|
19
|
+
LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
|
|
20
|
+
OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
|
|
21
|
+
SOFTWARE.
|
|
@@ -0,0 +1,201 @@
|
|
|
1
|
+
Metadata-Version: 2.4
|
|
2
|
+
Name: worddael
|
|
3
|
+
Version: 0.1.0rc3
|
|
4
|
+
Summary: The Dale of Remembered Words (Worddael) | Chinese-aware RAG text chunking
|
|
5
|
+
Author: worddael contributors
|
|
6
|
+
License: MIT
|
|
7
|
+
Project-URL: Homepage, https://github.com/cloudydreamland/TheDaleOfRememberedWords
|
|
8
|
+
Project-URL: Repository, https://github.com/cloudydreamland/TheDaleOfRememberedWords
|
|
9
|
+
Project-URL: Issues, https://github.com/cloudydreamland/TheDaleOfRememberedWords/issues
|
|
10
|
+
Project-URL: Changelog, https://github.com/cloudydreamland/TheDaleOfRememberedWords/blob/main/CHANGELOG.md
|
|
11
|
+
Project-URL: Security, https://github.com/cloudydreamland/TheDaleOfRememberedWords/security/policy
|
|
12
|
+
Keywords: chinese,chunking,rag,nlp,text-splitting,retrieval
|
|
13
|
+
Classifier: Development Status :: 3 - Alpha
|
|
14
|
+
Classifier: Intended Audience :: Developers
|
|
15
|
+
Classifier: License :: OSI Approved :: MIT License
|
|
16
|
+
Classifier: Natural Language :: Chinese (Simplified)
|
|
17
|
+
Classifier: Programming Language :: Python :: 3
|
|
18
|
+
Classifier: Programming Language :: Python :: 3.10
|
|
19
|
+
Classifier: Programming Language :: Python :: 3.11
|
|
20
|
+
Classifier: Programming Language :: Python :: 3.12
|
|
21
|
+
Classifier: Programming Language :: Python :: 3.13
|
|
22
|
+
Classifier: Topic :: Text Processing :: Linguistic
|
|
23
|
+
Requires-Python: >=3.10
|
|
24
|
+
Description-Content-Type: text/markdown
|
|
25
|
+
License-File: LICENSE
|
|
26
|
+
Provides-Extra: jieba
|
|
27
|
+
Requires-Dist: jieba>=0.42; extra == "jieba"
|
|
28
|
+
Provides-Extra: tiktoken
|
|
29
|
+
Requires-Dist: tiktoken>=0.7; extra == "tiktoken"
|
|
30
|
+
Provides-Extra: dev
|
|
31
|
+
Requires-Dist: pytest>=8.0; extra == "dev"
|
|
32
|
+
Requires-Dist: ruff>=0.6; extra == "dev"
|
|
33
|
+
Dynamic: license-file
|
|
34
|
+
|
|
35
|
+
# The Dale of Remembered Words — Worddael
|
|
36
|
+
|
|
37
|
+
简体中文 · [English](README.en.md)
|
|
38
|
+
|
|
39
|
+
> 展示名 **The Dale of Remembered Words** 描绘一处安放文字的山谷;Worddael 是该项目的短名。
|
|
40
|
+
|
|
41
|
+
**中文优先的 RAG 文本切分库。把中文文本切成好用的块,并保留精确到字符的原文偏移量。**
|
|
42
|
+
|
|
43
|
+
[](.github/workflows/ci.yml)
|
|
44
|
+
[](pyproject.toml)
|
|
45
|
+
[](LICENSE)
|
|
46
|
+
|
|
47
|
+
## 为什么需要它 / Why
|
|
48
|
+
|
|
49
|
+
通用文本切分器可用于多种语言,但中文句界、标点归属和回引源文等细节往往需要应用自行处理。`worddael` 提供中文规则、可组合策略和原文偏移,让这些行为能在管线中检查:
|
|
50
|
+
|
|
51
|
+
- **中文句界原生正确**:`。!?;…` 加引号闭合格式感知;ASCII `.` 不切(保护 `3.14`、版本号、URL)
|
|
52
|
+
- **字符级偏移不变量**:每个 chunk 保证 `chunk.text == source[chunk.start:chunk.end]`,检索结果可直接引用回原文(fuzz 测试覆盖)
|
|
53
|
+
- **零必装依赖**:核心纯 Python;jieba / tiktoken / embedding 全部可选
|
|
54
|
+
- **自带评测**:内置 BM25 检索召回评测 + 可插拔 LLM 判分器(支持干跑,零成本先跑通)
|
|
55
|
+
- **CPU 即可用**:不需要 GPU,不需要下载模型
|
|
56
|
+
|
|
57
|
+
## Quickstart (English)
|
|
58
|
+
|
|
59
|
+
worddael (Worddael, "a cut-out chapter") chunks **Chinese** text for RAG, keeping exact
|
|
60
|
+
character offsets so every retrieved chunk can be cited back to its source.
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
from worddael import chunk
|
|
64
|
+
|
|
65
|
+
chunks = chunk(long_text, strategy="recursive", max_chars=500, overlap_chars=50)
|
|
66
|
+
assert all(c.text == long_text[c.start:c.end] for c in chunks) # invariant
|
|
67
|
+
```
|
|
68
|
+
|
|
69
|
+
- Chinese-aware sentence boundaries (。!?;… with quote-closer attachment;
|
|
70
|
+
ASCII `.` never splits decimals/versions)
|
|
71
|
+
- Markdown heading paths, atomic code fences, parent-child (small-to-big)
|
|
72
|
+
chunking, pluggable semantic chunking and token counters
|
|
73
|
+
- Zero required dependencies, CPU-only; optional extras: `worddael[jieba]`,
|
|
74
|
+
`worddael[tiktoken]`
|
|
75
|
+
- Built-in evaluation: BM25 recall/precision@k over an original 25-doc
|
|
76
|
+
benchmark, LLM answerability judging (dry-run needs no API key)
|
|
77
|
+
|
|
78
|
+
See [docs/status_quo.md](docs/status_quo.md) for reproducible failure cases
|
|
79
|
+
of default splitters on Chinese, and [benchmarks/results.md](benchmarks/results.md)
|
|
80
|
+
for current numbers.
|
|
81
|
+
|
|
82
|
+
## 安装 / Install
|
|
83
|
+
|
|
84
|
+
> 当前尚未发布到 PyPI;下方给出从 GitHub 获取并本地安装的命令。
|
|
85
|
+
|
|
86
|
+
```bash
|
|
87
|
+
git clone https://github.com/cloudydreamland/TheDaleOfRememberedWords.git
|
|
88
|
+
cd TheDaleOfRememberedWords
|
|
89
|
+
python -m pip install .
|
|
90
|
+
# PyPI 首发后:python -m pip install worddael
|
|
91
|
+
python -m pip install ".[jieba]"
|
|
92
|
+
```
|
|
93
|
+
|
|
94
|
+
## 快速开始 / Quickstart
|
|
95
|
+
|
|
96
|
+
```python
|
|
97
|
+
from worddael import chunk
|
|
98
|
+
|
|
99
|
+
text = open("manual.md", encoding="utf-8").read()
|
|
100
|
+
chunks = chunk(text, strategy="recursive", max_chars=500, overlap_chars=50)
|
|
101
|
+
|
|
102
|
+
for c in chunks:
|
|
103
|
+
print(c.seq, c.start, c.end, c.meta.get("headings"), c.text[:20])
|
|
104
|
+
assert c.text == text[c.start:c.end] # 永远成立,可放心引用
|
|
105
|
+
```
|
|
106
|
+
|
|
107
|
+
```python
|
|
108
|
+
from worddael import SemanticChunker, OpenAICompatibleEmbedder, HashingEmbedder
|
|
109
|
+
|
|
110
|
+
# 语义切分:接任意 OpenAI 兼容 embedding 接口(key 从环境变量读取,批量+超时可配)
|
|
111
|
+
|
|
112
|
+
embedder = OpenAICompatibleEmbedder(
|
|
113
|
+
base_url="https://open.bigmodel.cn/api/paas/v4",
|
|
114
|
+
model="embedding-3",
|
|
115
|
+
api_key_env="ZHIPUAI_API_KEY",
|
|
116
|
+
)
|
|
117
|
+
chunks = SemanticChunker(embed_fn=embedder, similarity_threshold=0.55,
|
|
118
|
+
max_chars=500, min_chars=80).chunk(long_text)
|
|
119
|
+
|
|
120
|
+
# 或用零依赖的 HashingEmbedder 先把管线跑通(玩具级,非质量结论)
|
|
121
|
+
|
|
122
|
+
chunks = SemanticChunker(embed_fn=HashingEmbedder()).chunk(long_text)
|
|
123
|
+
```
|
|
124
|
+
|
|
125
|
+
命令行:
|
|
126
|
+
|
|
127
|
+
```bash
|
|
128
|
+
worddael doc.md --stats
|
|
129
|
+
worddael manual.txt --strategy recursive --max-chars 400 --overlap-chars 50 --jsonl chunks.jsonl
|
|
130
|
+
```
|
|
131
|
+
|
|
132
|
+
评测(CPU、零 API 成本):
|
|
133
|
+
|
|
134
|
+
```bash
|
|
135
|
+
python -m worddael.eval.report --k 3 # 内置语料上对比各策略的 recall@k
|
|
136
|
+
```
|
|
137
|
+
|
|
138
|
+
## 切分策略 / Strategies
|
|
139
|
+
|
|
140
|
+
| strategy | 适用场景 | 说明 |
|
|
141
|
+
|---|---|---|
|
|
142
|
+
| `recursive` | 通用中文文本 | 段落→行→句号级→逗号级递归切分,中文标点优先 |
|
|
143
|
+
| `markdown` | .md 文档 | 按标题结构切,heading path 写入 meta,代码块保持完整 |
|
|
144
|
+
| `sentence` | 需要整句边界 | 按中文句界整句打包 |
|
|
145
|
+
| `token` | 对齐模型 token 预算 | 可插拔计数器(启发式 / jieba / tiktoken) |
|
|
146
|
+
| `semantic` | 主题边界敏感 | 自带 embedding 函数,相邻句相似度下降处切分,带防碎片最小块长 |
|
|
147
|
+
| `parent-child` | 小-大检索(RAG 常见手搓模式) | 子块精确匹配、父块给上下文;`chunk_families()` 返回 (父, 子),meta 双向链接,偏移不变量照旧 |
|
|
148
|
+
|
|
149
|
+
## 与现有方案的关系 / Landscape
|
|
150
|
+
|
|
151
|
+
| 方案 | 问题 |
|
|
152
|
+
|---|---|
|
|
153
|
+
| LangChain text splitters | 通用切分组件;Worddael 可作为中文切分选项,并提供自己的偏移与策略 API |
|
|
154
|
+
| Chonkie 等 ingestion 库 | 面向更广的文档摄取与切分场景;选择时应按格式覆盖、语言和集成需求比较 |
|
|
155
|
+
| RAGFlow 等 RAG 引擎 | 提供完整应用或管线;Worddael 是可嵌入现有 Python 项目的独立库 |
|
|
156
|
+
| jieba | 中文分词工具;与按句界、结构和长度做文档切分的目标不同 |
|
|
157
|
+
|
|
158
|
+
LangChain 默认参数切中文的**真实坏例子**(答案腰斩、标题孤立,可复现脚本)见 [docs/status_quo.md](docs/status_quo.md);详细论证见 [GAP_PROOF.md](GAP_PROOF.md)。
|
|
159
|
+
|
|
160
|
+
## 评测 / Evaluation
|
|
161
|
+
|
|
162
|
+
- 内置基准(25 篇 / 75 问)真实结果:[benchmarks/results.md](benchmarks/results.md)
|
|
163
|
+
- 当前快照:recall@1 recursive 0.96 vs fixed-window 0.907;小-大检索(子块 150 检索 / 父块 600 判分)0.987 vs 平铺 0.96
|
|
164
|
+
- 大规模 LLM 判分评测指南(拿到 key 后):[docs/eval_guide.md](docs/eval_guide.md)
|
|
165
|
+
- 选题论证(为什么这个缺口是真的):[GAP_PROOF.md](GAP_PROOF.md)
|
|
166
|
+
|
|
167
|
+
## 路线图 / Roadmap
|
|
168
|
+
|
|
169
|
+
见 [ROADMAP.md](ROADMAP.md)。当前 v0.1.0rc3:六个策略(含父子块)+ 可解释切分 + 标点卫生保证 + 评测框架(测试数以 CI 为准)。真实语料实测见 [docs/dogfood.md](docs/dogfood.md):90.6 万字《紅樓夢》,24.6 MB/s,卫生违例 0。
|
|
170
|
+
|
|
171
|
+
## Non-goals(明确不做)
|
|
172
|
+
|
|
173
|
+
- **模型/LLM 驱动切分**:红海方向且依赖 GPU/API,与零依赖纯 CPU 定位冲突;语义切分走可插拔 embedder
|
|
174
|
+
- **通用 RAG 框架**:只做切分与切分评测,通过 [adapters](#安装--install) 进入 LangChain 等既有管线
|
|
175
|
+
- **计费级 token 计数**:启发式计数只服务预算切分(误差数字见基准报告校准小节)
|
|
176
|
+
- **流式超大文件**:单文档需可入内存;GB 级请自行分片
|
|
177
|
+
|
|
178
|
+
架构与扩展指南见 [docs/architecture.md](docs/architecture.md)。
|
|
179
|
+
|
|
180
|
+
## FAQ
|
|
181
|
+
|
|
182
|
+
- **overlap 模式下拼接结果比原文长?** 设计如此:overlap 是上下文重复(上一块尾部回看)。`overlap_chars=0` 时拼接与原文逐字相等(有测试锁定)。
|
|
183
|
+
- **为什么有的块以换行结尾?** 分隔符归属前一块是偏移精确的前提;判断块质量请对 `rstrip()` 后的文本判断(内置统计均如此)。
|
|
184
|
+
- **预算为何偶尔超出 1-2 字?** 标点卫生会把下一块开头的悬垂标点吸收进前一块(上限 2 字),换来"块首无悬垂标点"的保证。
|
|
185
|
+
- **测试数是固定的吗?** 以 CI 最新运行为准;文档不写死数字。
|
|
186
|
+
|
|
187
|
+
## 开发 / Development
|
|
188
|
+
|
|
189
|
+
```bash
|
|
190
|
+
pip install -e ".[jieba,dev]"
|
|
191
|
+
pytest
|
|
192
|
+
ruff check src tests
|
|
193
|
+
```
|
|
194
|
+
|
|
195
|
+
## 反馈与参与
|
|
196
|
+
|
|
197
|
+
使用问题和功能建议可以在 [Discussions](https://github.com/cloudydreamland/TheDaleOfRememberedWords/discussions) 交流;可复现缺陷请提交 [Issue](https://github.com/cloudydreamland/TheDaleOfRememberedWords/issues)。请只附合成或脱敏后的最小样例,不要上传真实个人信息、API key 或业务原文。安全问题请按 [SECURITY.md](SECURITY.md) 私下报告。
|
|
198
|
+
|
|
199
|
+
## License
|
|
200
|
+
|
|
201
|
+
MIT
|
|
@@ -0,0 +1,167 @@
|
|
|
1
|
+
# The Dale of Remembered Words — Worddael
|
|
2
|
+
|
|
3
|
+
简体中文 · [English](README.en.md)
|
|
4
|
+
|
|
5
|
+
> 展示名 **The Dale of Remembered Words** 描绘一处安放文字的山谷;Worddael 是该项目的短名。
|
|
6
|
+
|
|
7
|
+
**中文优先的 RAG 文本切分库。把中文文本切成好用的块,并保留精确到字符的原文偏移量。**
|
|
8
|
+
|
|
9
|
+
[](.github/workflows/ci.yml)
|
|
10
|
+
[](pyproject.toml)
|
|
11
|
+
[](LICENSE)
|
|
12
|
+
|
|
13
|
+
## 为什么需要它 / Why
|
|
14
|
+
|
|
15
|
+
通用文本切分器可用于多种语言,但中文句界、标点归属和回引源文等细节往往需要应用自行处理。`worddael` 提供中文规则、可组合策略和原文偏移,让这些行为能在管线中检查:
|
|
16
|
+
|
|
17
|
+
- **中文句界原生正确**:`。!?;…` 加引号闭合格式感知;ASCII `.` 不切(保护 `3.14`、版本号、URL)
|
|
18
|
+
- **字符级偏移不变量**:每个 chunk 保证 `chunk.text == source[chunk.start:chunk.end]`,检索结果可直接引用回原文(fuzz 测试覆盖)
|
|
19
|
+
- **零必装依赖**:核心纯 Python;jieba / tiktoken / embedding 全部可选
|
|
20
|
+
- **自带评测**:内置 BM25 检索召回评测 + 可插拔 LLM 判分器(支持干跑,零成本先跑通)
|
|
21
|
+
- **CPU 即可用**:不需要 GPU,不需要下载模型
|
|
22
|
+
|
|
23
|
+
## Quickstart (English)
|
|
24
|
+
|
|
25
|
+
worddael (Worddael, "a cut-out chapter") chunks **Chinese** text for RAG, keeping exact
|
|
26
|
+
character offsets so every retrieved chunk can be cited back to its source.
|
|
27
|
+
|
|
28
|
+
```python
|
|
29
|
+
from worddael import chunk
|
|
30
|
+
|
|
31
|
+
chunks = chunk(long_text, strategy="recursive", max_chars=500, overlap_chars=50)
|
|
32
|
+
assert all(c.text == long_text[c.start:c.end] for c in chunks) # invariant
|
|
33
|
+
```
|
|
34
|
+
|
|
35
|
+
- Chinese-aware sentence boundaries (。!?;… with quote-closer attachment;
|
|
36
|
+
ASCII `.` never splits decimals/versions)
|
|
37
|
+
- Markdown heading paths, atomic code fences, parent-child (small-to-big)
|
|
38
|
+
chunking, pluggable semantic chunking and token counters
|
|
39
|
+
- Zero required dependencies, CPU-only; optional extras: `worddael[jieba]`,
|
|
40
|
+
`worddael[tiktoken]`
|
|
41
|
+
- Built-in evaluation: BM25 recall/precision@k over an original 25-doc
|
|
42
|
+
benchmark, LLM answerability judging (dry-run needs no API key)
|
|
43
|
+
|
|
44
|
+
See [docs/status_quo.md](docs/status_quo.md) for reproducible failure cases
|
|
45
|
+
of default splitters on Chinese, and [benchmarks/results.md](benchmarks/results.md)
|
|
46
|
+
for current numbers.
|
|
47
|
+
|
|
48
|
+
## 安装 / Install
|
|
49
|
+
|
|
50
|
+
> 当前尚未发布到 PyPI;下方给出从 GitHub 获取并本地安装的命令。
|
|
51
|
+
|
|
52
|
+
```bash
|
|
53
|
+
git clone https://github.com/cloudydreamland/TheDaleOfRememberedWords.git
|
|
54
|
+
cd TheDaleOfRememberedWords
|
|
55
|
+
python -m pip install .
|
|
56
|
+
# PyPI 首发后:python -m pip install worddael
|
|
57
|
+
python -m pip install ".[jieba]"
|
|
58
|
+
```
|
|
59
|
+
|
|
60
|
+
## 快速开始 / Quickstart
|
|
61
|
+
|
|
62
|
+
```python
|
|
63
|
+
from worddael import chunk
|
|
64
|
+
|
|
65
|
+
text = open("manual.md", encoding="utf-8").read()
|
|
66
|
+
chunks = chunk(text, strategy="recursive", max_chars=500, overlap_chars=50)
|
|
67
|
+
|
|
68
|
+
for c in chunks:
|
|
69
|
+
print(c.seq, c.start, c.end, c.meta.get("headings"), c.text[:20])
|
|
70
|
+
assert c.text == text[c.start:c.end] # 永远成立,可放心引用
|
|
71
|
+
```
|
|
72
|
+
|
|
73
|
+
```python
|
|
74
|
+
from worddael import SemanticChunker, OpenAICompatibleEmbedder, HashingEmbedder
|
|
75
|
+
|
|
76
|
+
# 语义切分:接任意 OpenAI 兼容 embedding 接口(key 从环境变量读取,批量+超时可配)
|
|
77
|
+
|
|
78
|
+
embedder = OpenAICompatibleEmbedder(
|
|
79
|
+
base_url="https://open.bigmodel.cn/api/paas/v4",
|
|
80
|
+
model="embedding-3",
|
|
81
|
+
api_key_env="ZHIPUAI_API_KEY",
|
|
82
|
+
)
|
|
83
|
+
chunks = SemanticChunker(embed_fn=embedder, similarity_threshold=0.55,
|
|
84
|
+
max_chars=500, min_chars=80).chunk(long_text)
|
|
85
|
+
|
|
86
|
+
# 或用零依赖的 HashingEmbedder 先把管线跑通(玩具级,非质量结论)
|
|
87
|
+
|
|
88
|
+
chunks = SemanticChunker(embed_fn=HashingEmbedder()).chunk(long_text)
|
|
89
|
+
```
|
|
90
|
+
|
|
91
|
+
命令行:
|
|
92
|
+
|
|
93
|
+
```bash
|
|
94
|
+
worddael doc.md --stats
|
|
95
|
+
worddael manual.txt --strategy recursive --max-chars 400 --overlap-chars 50 --jsonl chunks.jsonl
|
|
96
|
+
```
|
|
97
|
+
|
|
98
|
+
评测(CPU、零 API 成本):
|
|
99
|
+
|
|
100
|
+
```bash
|
|
101
|
+
python -m worddael.eval.report --k 3 # 内置语料上对比各策略的 recall@k
|
|
102
|
+
```
|
|
103
|
+
|
|
104
|
+
## 切分策略 / Strategies
|
|
105
|
+
|
|
106
|
+
| strategy | 适用场景 | 说明 |
|
|
107
|
+
|---|---|---|
|
|
108
|
+
| `recursive` | 通用中文文本 | 段落→行→句号级→逗号级递归切分,中文标点优先 |
|
|
109
|
+
| `markdown` | .md 文档 | 按标题结构切,heading path 写入 meta,代码块保持完整 |
|
|
110
|
+
| `sentence` | 需要整句边界 | 按中文句界整句打包 |
|
|
111
|
+
| `token` | 对齐模型 token 预算 | 可插拔计数器(启发式 / jieba / tiktoken) |
|
|
112
|
+
| `semantic` | 主题边界敏感 | 自带 embedding 函数,相邻句相似度下降处切分,带防碎片最小块长 |
|
|
113
|
+
| `parent-child` | 小-大检索(RAG 常见手搓模式) | 子块精确匹配、父块给上下文;`chunk_families()` 返回 (父, 子),meta 双向链接,偏移不变量照旧 |
|
|
114
|
+
|
|
115
|
+
## 与现有方案的关系 / Landscape
|
|
116
|
+
|
|
117
|
+
| 方案 | 问题 |
|
|
118
|
+
|---|---|
|
|
119
|
+
| LangChain text splitters | 通用切分组件;Worddael 可作为中文切分选项,并提供自己的偏移与策略 API |
|
|
120
|
+
| Chonkie 等 ingestion 库 | 面向更广的文档摄取与切分场景;选择时应按格式覆盖、语言和集成需求比较 |
|
|
121
|
+
| RAGFlow 等 RAG 引擎 | 提供完整应用或管线;Worddael 是可嵌入现有 Python 项目的独立库 |
|
|
122
|
+
| jieba | 中文分词工具;与按句界、结构和长度做文档切分的目标不同 |
|
|
123
|
+
|
|
124
|
+
LangChain 默认参数切中文的**真实坏例子**(答案腰斩、标题孤立,可复现脚本)见 [docs/status_quo.md](docs/status_quo.md);详细论证见 [GAP_PROOF.md](GAP_PROOF.md)。
|
|
125
|
+
|
|
126
|
+
## 评测 / Evaluation
|
|
127
|
+
|
|
128
|
+
- 内置基准(25 篇 / 75 问)真实结果:[benchmarks/results.md](benchmarks/results.md)
|
|
129
|
+
- 当前快照:recall@1 recursive 0.96 vs fixed-window 0.907;小-大检索(子块 150 检索 / 父块 600 判分)0.987 vs 平铺 0.96
|
|
130
|
+
- 大规模 LLM 判分评测指南(拿到 key 后):[docs/eval_guide.md](docs/eval_guide.md)
|
|
131
|
+
- 选题论证(为什么这个缺口是真的):[GAP_PROOF.md](GAP_PROOF.md)
|
|
132
|
+
|
|
133
|
+
## 路线图 / Roadmap
|
|
134
|
+
|
|
135
|
+
见 [ROADMAP.md](ROADMAP.md)。当前 v0.1.0rc3:六个策略(含父子块)+ 可解释切分 + 标点卫生保证 + 评测框架(测试数以 CI 为准)。真实语料实测见 [docs/dogfood.md](docs/dogfood.md):90.6 万字《紅樓夢》,24.6 MB/s,卫生违例 0。
|
|
136
|
+
|
|
137
|
+
## Non-goals(明确不做)
|
|
138
|
+
|
|
139
|
+
- **模型/LLM 驱动切分**:红海方向且依赖 GPU/API,与零依赖纯 CPU 定位冲突;语义切分走可插拔 embedder
|
|
140
|
+
- **通用 RAG 框架**:只做切分与切分评测,通过 [adapters](#安装--install) 进入 LangChain 等既有管线
|
|
141
|
+
- **计费级 token 计数**:启发式计数只服务预算切分(误差数字见基准报告校准小节)
|
|
142
|
+
- **流式超大文件**:单文档需可入内存;GB 级请自行分片
|
|
143
|
+
|
|
144
|
+
架构与扩展指南见 [docs/architecture.md](docs/architecture.md)。
|
|
145
|
+
|
|
146
|
+
## FAQ
|
|
147
|
+
|
|
148
|
+
- **overlap 模式下拼接结果比原文长?** 设计如此:overlap 是上下文重复(上一块尾部回看)。`overlap_chars=0` 时拼接与原文逐字相等(有测试锁定)。
|
|
149
|
+
- **为什么有的块以换行结尾?** 分隔符归属前一块是偏移精确的前提;判断块质量请对 `rstrip()` 后的文本判断(内置统计均如此)。
|
|
150
|
+
- **预算为何偶尔超出 1-2 字?** 标点卫生会把下一块开头的悬垂标点吸收进前一块(上限 2 字),换来"块首无悬垂标点"的保证。
|
|
151
|
+
- **测试数是固定的吗?** 以 CI 最新运行为准;文档不写死数字。
|
|
152
|
+
|
|
153
|
+
## 开发 / Development
|
|
154
|
+
|
|
155
|
+
```bash
|
|
156
|
+
pip install -e ".[jieba,dev]"
|
|
157
|
+
pytest
|
|
158
|
+
ruff check src tests
|
|
159
|
+
```
|
|
160
|
+
|
|
161
|
+
## 反馈与参与
|
|
162
|
+
|
|
163
|
+
使用问题和功能建议可以在 [Discussions](https://github.com/cloudydreamland/TheDaleOfRememberedWords/discussions) 交流;可复现缺陷请提交 [Issue](https://github.com/cloudydreamland/TheDaleOfRememberedWords/issues)。请只附合成或脱敏后的最小样例,不要上传真实个人信息、API key 或业务原文。安全问题请按 [SECURITY.md](SECURITY.md) 私下报告。
|
|
164
|
+
|
|
165
|
+
## License
|
|
166
|
+
|
|
167
|
+
MIT
|
|
@@ -0,0 +1,58 @@
|
|
|
1
|
+
[build-system]
|
|
2
|
+
requires = ["setuptools>=68"]
|
|
3
|
+
build-backend = "setuptools.build_meta"
|
|
4
|
+
|
|
5
|
+
[project]
|
|
6
|
+
name = "worddael"
|
|
7
|
+
version = "0.1.0rc3"
|
|
8
|
+
description = "The Dale of Remembered Words (Worddael) | Chinese-aware RAG text chunking"
|
|
9
|
+
readme = "README.md"
|
|
10
|
+
license = { text = "MIT" }
|
|
11
|
+
authors = [{ name = "worddael contributors" }]
|
|
12
|
+
requires-python = ">=3.10"
|
|
13
|
+
keywords = ["chinese", "chunking", "rag", "nlp", "text-splitting", "retrieval"]
|
|
14
|
+
classifiers = [
|
|
15
|
+
"Development Status :: 3 - Alpha",
|
|
16
|
+
"Intended Audience :: Developers",
|
|
17
|
+
"License :: OSI Approved :: MIT License",
|
|
18
|
+
"Natural Language :: Chinese (Simplified)",
|
|
19
|
+
"Programming Language :: Python :: 3",
|
|
20
|
+
"Programming Language :: Python :: 3.10",
|
|
21
|
+
"Programming Language :: Python :: 3.11",
|
|
22
|
+
"Programming Language :: Python :: 3.12",
|
|
23
|
+
"Programming Language :: Python :: 3.13",
|
|
24
|
+
"Topic :: Text Processing :: Linguistic",
|
|
25
|
+
]
|
|
26
|
+
dependencies = []
|
|
27
|
+
|
|
28
|
+
[project.optional-dependencies]
|
|
29
|
+
jieba = ["jieba>=0.42"]
|
|
30
|
+
tiktoken = ["tiktoken>=0.7"]
|
|
31
|
+
dev = ["pytest>=8.0", "ruff>=0.6"]
|
|
32
|
+
|
|
33
|
+
[project.scripts]
|
|
34
|
+
worddael = "worddael.cli:main"
|
|
35
|
+
|
|
36
|
+
[project.urls]
|
|
37
|
+
Homepage = "https://github.com/cloudydreamland/TheDaleOfRememberedWords"
|
|
38
|
+
Repository = "https://github.com/cloudydreamland/TheDaleOfRememberedWords"
|
|
39
|
+
Issues = "https://github.com/cloudydreamland/TheDaleOfRememberedWords/issues"
|
|
40
|
+
Changelog = "https://github.com/cloudydreamland/TheDaleOfRememberedWords/blob/main/CHANGELOG.md"
|
|
41
|
+
Security = "https://github.com/cloudydreamland/TheDaleOfRememberedWords/security/policy"
|
|
42
|
+
[tool.setuptools.packages.find]
|
|
43
|
+
where = ["src"]
|
|
44
|
+
|
|
45
|
+
[tool.setuptools.package-data]
|
|
46
|
+
worddael = ["py.typed"]
|
|
47
|
+
|
|
48
|
+
[tool.pytest.ini_options]
|
|
49
|
+
testpaths = ["tests"]
|
|
50
|
+
addopts = "-q"
|
|
51
|
+
|
|
52
|
+
[tool.ruff]
|
|
53
|
+
line-length = 100
|
|
54
|
+
target-version = "py310"
|
|
55
|
+
|
|
56
|
+
[tool.ruff.lint]
|
|
57
|
+
select = ["E", "F", "I", "W"]
|
|
58
|
+
ignore = ["E501"]
|
|
@@ -0,0 +1,92 @@
|
|
|
1
|
+
"""Worddael — Chinese-first text chunking for RAG.
|
|
2
|
+
|
|
3
|
+
把中文文本切成好用的块,同时保留精确到字符的原文偏移量,
|
|
4
|
+
让 RAG 的每一条检索结果都能被可靠引用。
|
|
5
|
+
|
|
6
|
+
Quickstart::
|
|
7
|
+
|
|
8
|
+
from worddael import chunk
|
|
9
|
+
|
|
10
|
+
chunks = chunk(long_text, strategy="recursive", max_chars=500, overlap_chars=50)
|
|
11
|
+
for c in chunks:
|
|
12
|
+
print(c.seq, c.start, c.end, c.text[:20])
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
from __future__ import annotations
|
|
16
|
+
|
|
17
|
+
from .chunkers import (
|
|
18
|
+
STRATEGIES,
|
|
19
|
+
BaseChunker,
|
|
20
|
+
MarkdownChunker,
|
|
21
|
+
RecursiveChunker,
|
|
22
|
+
SemanticChunker,
|
|
23
|
+
SentenceChunker,
|
|
24
|
+
TokenChunker,
|
|
25
|
+
)
|
|
26
|
+
from .counters import CharCounter, get_counter
|
|
27
|
+
from .embedders import (
|
|
28
|
+
Embedder,
|
|
29
|
+
EmbedderError,
|
|
30
|
+
HashingEmbedder,
|
|
31
|
+
OpenAICompatibleEmbedder,
|
|
32
|
+
)
|
|
33
|
+
from .io_utils import chunk_file, to_jsonl
|
|
34
|
+
from .parent_child import ParentChildChunker
|
|
35
|
+
from .types import Chunk
|
|
36
|
+
|
|
37
|
+
__version__ = "0.1.0rc3"
|
|
38
|
+
|
|
39
|
+
# parent-child families are produced by a dedicated module; register the
|
|
40
|
+
# strategy here so get_chunker("parent-child") works like any other.
|
|
41
|
+
STRATEGIES["parent-child"] = ParentChildChunker
|
|
42
|
+
|
|
43
|
+
__all__ = [
|
|
44
|
+
"BaseChunker",
|
|
45
|
+
"CharCounter",
|
|
46
|
+
"Chunk",
|
|
47
|
+
"Embedder",
|
|
48
|
+
"EmbedderError",
|
|
49
|
+
"HashingEmbedder",
|
|
50
|
+
"MarkdownChunker",
|
|
51
|
+
"OpenAICompatibleEmbedder",
|
|
52
|
+
"ParentChildChunker",
|
|
53
|
+
"RecursiveChunker",
|
|
54
|
+
"SemanticChunker",
|
|
55
|
+
"SentenceChunker",
|
|
56
|
+
"STRATEGIES",
|
|
57
|
+
"TokenChunker",
|
|
58
|
+
"__version__",
|
|
59
|
+
"chunk",
|
|
60
|
+
"chunk_file",
|
|
61
|
+
"get_chunker",
|
|
62
|
+
"get_counter",
|
|
63
|
+
"to_jsonl",
|
|
64
|
+
]
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def get_chunker(strategy: str = "recursive", **kwargs) -> BaseChunker:
|
|
68
|
+
"""Build a chunker by name. See ``STRATEGIES`` for available names."""
|
|
69
|
+
try:
|
|
70
|
+
cls = STRATEGIES[strategy]
|
|
71
|
+
except KeyError:
|
|
72
|
+
raise ValueError(
|
|
73
|
+
f"unknown strategy {strategy!r}; expected one of {sorted(STRATEGIES)}"
|
|
74
|
+
) from None
|
|
75
|
+
return cls(**kwargs)
|
|
76
|
+
|
|
77
|
+
|
|
78
|
+
def chunk(text: str, strategy: str = "recursive", **kwargs) -> list[Chunk]:
|
|
79
|
+
"""One-shot helper: ``chunk(text, strategy, **chunker_kwargs)``.
|
|
80
|
+
|
|
81
|
+
Uniform budget names: ``max_chars``/``overlap_chars`` are accepted by
|
|
82
|
+
every strategy. For ``strategy="token"`` they are interpreted as token
|
|
83
|
+
budgets (mapped to ``max_tokens``/``overlap_tokens``) — tokens are what
|
|
84
|
+
that strategy measures, and the char-named arguments keep call sites
|
|
85
|
+
strategy-agnostic.
|
|
86
|
+
"""
|
|
87
|
+
if strategy == "token":
|
|
88
|
+
if "max_tokens" not in kwargs and "max_chars" in kwargs:
|
|
89
|
+
kwargs["max_tokens"] = kwargs.pop("max_chars")
|
|
90
|
+
if "overlap_tokens" not in kwargs and "overlap_chars" in kwargs:
|
|
91
|
+
kwargs["overlap_tokens"] = kwargs.pop("overlap_chars")
|
|
92
|
+
return get_chunker(strategy, **kwargs).chunk(text)
|
|
@@ -0,0 +1,45 @@
|
|
|
1
|
+
"""Ecosystem adapters: plug worddael into frameworks users already run.
|
|
2
|
+
|
|
3
|
+
LangChain: :class:`LangChainChunker` subclasses
|
|
4
|
+
``langchain_text_splitters.TextSplitter`` (optional import), so it works
|
|
5
|
+
with ``split_documents`` / ``create_documents`` and drops into any existing
|
|
6
|
+
RAG pipeline as a drop-in replacement for RecursiveCharacterTextSplitter.
|
|
7
|
+
"""
|
|
8
|
+
|
|
9
|
+
from __future__ import annotations
|
|
10
|
+
|
|
11
|
+
from . import get_chunker
|
|
12
|
+
from .chunkers import BaseChunker
|
|
13
|
+
|
|
14
|
+
try: # optional dependency — the adapter degrades to a plain object otherwise
|
|
15
|
+
from langchain_text_splitters import TextSplitter as _LCTextSplitter
|
|
16
|
+
|
|
17
|
+
_HAS_LANGCHAIN = True
|
|
18
|
+
except ImportError: # pragma: no cover
|
|
19
|
+
_LCTextSplitter = object # type: ignore[assignment,misc]
|
|
20
|
+
_HAS_LANGCHAIN = False
|
|
21
|
+
|
|
22
|
+
__all__ = ["LangChainChunker", "has_langchain"]
|
|
23
|
+
|
|
24
|
+
|
|
25
|
+
def has_langchain() -> bool:
|
|
26
|
+
return _HAS_LANGCHAIN
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class LangChainChunker(_LCTextSplitter):
|
|
30
|
+
"""worddael chunking behind the LangChain ``TextSplitter`` interface.
|
|
31
|
+
|
|
32
|
+
Example::
|
|
33
|
+
|
|
34
|
+
from worddael.adapters import LangChainChunker
|
|
35
|
+
splitter = LangChainChunker(strategy="recursive", max_chars=500, overlap_chars=50)
|
|
36
|
+
docs = splitter.create_documents([long_text]) # full LC Document flow
|
|
37
|
+
"""
|
|
38
|
+
|
|
39
|
+
def __init__(self, strategy: str = "recursive", **chunker_kwargs) -> None:
|
|
40
|
+
if _HAS_LANGCHAIN:
|
|
41
|
+
super().__init__()
|
|
42
|
+
self._chunker: BaseChunker = get_chunker(strategy, **chunker_kwargs)
|
|
43
|
+
|
|
44
|
+
def split_text(self, text: str) -> list[str]:
|
|
45
|
+
return [c.text for c in self._chunker.chunk(text)]
|