worddael 0.1.0rc3__tar.gz

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (52) hide show
  1. worddael-0.1.0rc3/LICENSE +21 -0
  2. worddael-0.1.0rc3/PKG-INFO +201 -0
  3. worddael-0.1.0rc3/README.md +167 -0
  4. worddael-0.1.0rc3/pyproject.toml +58 -0
  5. worddael-0.1.0rc3/setup.cfg +4 -0
  6. worddael-0.1.0rc3/src/worddael/__init__.py +92 -0
  7. worddael-0.1.0rc3/src/worddael/adapters.py +45 -0
  8. worddael-0.1.0rc3/src/worddael/chunkers.py +474 -0
  9. worddael-0.1.0rc3/src/worddael/cli.py +137 -0
  10. worddael-0.1.0rc3/src/worddael/compare.py +66 -0
  11. worddael-0.1.0rc3/src/worddael/counters.py +128 -0
  12. worddael-0.1.0rc3/src/worddael/embedders.py +151 -0
  13. worddael-0.1.0rc3/src/worddael/eval/__init__.py +21 -0
  14. worddael-0.1.0rc3/src/worddael/eval/judge_runner.py +258 -0
  15. worddael-0.1.0rc3/src/worddael/eval/llm_judge.py +123 -0
  16. worddael-0.1.0rc3/src/worddael/eval/qa_gen.py +239 -0
  17. worddael-0.1.0rc3/src/worddael/eval/report.py +366 -0
  18. worddael-0.1.0rc3/src/worddael/eval/retrieval.py +241 -0
  19. worddael-0.1.0rc3/src/worddael/eval/sample_data.py +537 -0
  20. worddael-0.1.0rc3/src/worddael/explain.py +83 -0
  21. worddael-0.1.0rc3/src/worddael/io_utils.py +44 -0
  22. worddael-0.1.0rc3/src/worddael/parent_child.py +96 -0
  23. worddael-0.1.0rc3/src/worddael/py.typed +0 -0
  24. worddael-0.1.0rc3/src/worddael/sentences.py +114 -0
  25. worddael-0.1.0rc3/src/worddael/types.py +36 -0
  26. worddael-0.1.0rc3/src/worddael.egg-info/PKG-INFO +201 -0
  27. worddael-0.1.0rc3/src/worddael.egg-info/SOURCES.txt +50 -0
  28. worddael-0.1.0rc3/src/worddael.egg-info/dependency_links.txt +1 -0
  29. worddael-0.1.0rc3/src/worddael.egg-info/entry_points.txt +2 -0
  30. worddael-0.1.0rc3/src/worddael.egg-info/requires.txt +10 -0
  31. worddael-0.1.0rc3/src/worddael.egg-info/top_level.txt +1 -0
  32. worddael-0.1.0rc3/tests/test_adapters.py +63 -0
  33. worddael-0.1.0rc3/tests/test_chunkers.py +154 -0
  34. worddael-0.1.0rc3/tests/test_cli_io.py +80 -0
  35. worddael-0.1.0rc3/tests/test_compare.py +43 -0
  36. worddael-0.1.0rc3/tests/test_counters.py +35 -0
  37. worddael-0.1.0rc3/tests/test_docs_examples.py +102 -0
  38. worddael-0.1.0rc3/tests/test_embedders.py +173 -0
  39. worddael-0.1.0rc3/tests/test_eval.py +102 -0
  40. worddael-0.1.0rc3/tests/test_explain.py +130 -0
  41. worddael-0.1.0rc3/tests/test_hygiene.py +98 -0
  42. worddael-0.1.0rc3/tests/test_iter7.py +90 -0
  43. worddael-0.1.0rc3/tests/test_judge_runner.py +146 -0
  44. worddael-0.1.0rc3/tests/test_offsets.py +101 -0
  45. worddael-0.1.0rc3/tests/test_parent_child.py +99 -0
  46. worddael-0.1.0rc3/tests/test_perf_structure.py +79 -0
  47. worddael-0.1.0rc3/tests/test_polish.py +92 -0
  48. worddael-0.1.0rc3/tests/test_qa_gen.py +86 -0
  49. worddael-0.1.0rc3/tests/test_r10.py +74 -0
  50. worddael-0.1.0rc3/tests/test_r11.py +49 -0
  51. worddael-0.1.0rc3/tests/test_r12.py +61 -0
  52. worddael-0.1.0rc3/tests/test_traditional.py +59 -0
@@ -0,0 +1,21 @@
1
+ MIT License
2
+
3
+ Copyright (c) 2026 worddael contributors
4
+
5
+ Permission is hereby granted, free of charge, to any person obtaining a copy
6
+ of this software and associated documentation files (the "Software"), to deal
7
+ in the Software without restriction, including without limitation the rights
8
+ to use, copy, modify, merge, publish, distribute, sublicense, and/or sell
9
+ copies of the Software, and to permit persons to whom the Software is
10
+ furnished to do so, subject to the following conditions:
11
+
12
+ The above copyright notice and this permission notice shall be included in all
13
+ copies or substantial portions of the Software.
14
+
15
+ THE SOFTWARE IS PROVIDED "AS IS", WITHOUT WARRANTY OF ANY KIND, EXPRESS OR
16
+ IMPLIED, INCLUDING BUT NOT LIMITED TO THE WARRANTIES OF MERCHANTABILITY,
17
+ FITNESS FOR A PARTICULAR PURPOSE AND NONINFRINGEMENT. IN NO EVENT SHALL THE
18
+ AUTHORS OR COPYRIGHT HOLDERS BE LIABLE FOR ANY CLAIM, DAMAGES OR OTHER
19
+ LIABILITY, WHETHER IN AN ACTION OF CONTRACT, TORT OR OTHERWISE, ARISING FROM,
20
+ OUT OF OR IN CONNECTION WITH THE SOFTWARE OR THE USE OR OTHER DEALINGS IN THE
21
+ SOFTWARE.
@@ -0,0 +1,201 @@
1
+ Metadata-Version: 2.4
2
+ Name: worddael
3
+ Version: 0.1.0rc3
4
+ Summary: The Dale of Remembered Words (Worddael) | Chinese-aware RAG text chunking
5
+ Author: worddael contributors
6
+ License: MIT
7
+ Project-URL: Homepage, https://github.com/cloudydreamland/TheDaleOfRememberedWords
8
+ Project-URL: Repository, https://github.com/cloudydreamland/TheDaleOfRememberedWords
9
+ Project-URL: Issues, https://github.com/cloudydreamland/TheDaleOfRememberedWords/issues
10
+ Project-URL: Changelog, https://github.com/cloudydreamland/TheDaleOfRememberedWords/blob/main/CHANGELOG.md
11
+ Project-URL: Security, https://github.com/cloudydreamland/TheDaleOfRememberedWords/security/policy
12
+ Keywords: chinese,chunking,rag,nlp,text-splitting,retrieval
13
+ Classifier: Development Status :: 3 - Alpha
14
+ Classifier: Intended Audience :: Developers
15
+ Classifier: License :: OSI Approved :: MIT License
16
+ Classifier: Natural Language :: Chinese (Simplified)
17
+ Classifier: Programming Language :: Python :: 3
18
+ Classifier: Programming Language :: Python :: 3.10
19
+ Classifier: Programming Language :: Python :: 3.11
20
+ Classifier: Programming Language :: Python :: 3.12
21
+ Classifier: Programming Language :: Python :: 3.13
22
+ Classifier: Topic :: Text Processing :: Linguistic
23
+ Requires-Python: >=3.10
24
+ Description-Content-Type: text/markdown
25
+ License-File: LICENSE
26
+ Provides-Extra: jieba
27
+ Requires-Dist: jieba>=0.42; extra == "jieba"
28
+ Provides-Extra: tiktoken
29
+ Requires-Dist: tiktoken>=0.7; extra == "tiktoken"
30
+ Provides-Extra: dev
31
+ Requires-Dist: pytest>=8.0; extra == "dev"
32
+ Requires-Dist: ruff>=0.6; extra == "dev"
33
+ Dynamic: license-file
34
+
35
+ # The Dale of Remembered Words — Worddael
36
+
37
+ 简体中文 · [English](README.en.md)
38
+
39
+ > 展示名 **The Dale of Remembered Words** 描绘一处安放文字的山谷;Worddael 是该项目的短名。
40
+
41
+ **中文优先的 RAG 文本切分库。把中文文本切成好用的块,并保留精确到字符的原文偏移量。**
42
+
43
+ [![CI](https://github.com/cloudydreamland/TheDaleOfRememberedWords/actions/workflows/ci.yml/badge.svg)](.github/workflows/ci.yml)
44
+ [![Python](https://img.shields.io/badge/python-3.10%2B-blue)](pyproject.toml)
45
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
46
+
47
+ ## 为什么需要它 / Why
48
+
49
+ 通用文本切分器可用于多种语言,但中文句界、标点归属和回引源文等细节往往需要应用自行处理。`worddael` 提供中文规则、可组合策略和原文偏移,让这些行为能在管线中检查:
50
+
51
+ - **中文句界原生正确**:`。!?;…` 加引号闭合格式感知;ASCII `.` 不切(保护 `3.14`、版本号、URL)
52
+ - **字符级偏移不变量**:每个 chunk 保证 `chunk.text == source[chunk.start:chunk.end]`,检索结果可直接引用回原文(fuzz 测试覆盖)
53
+ - **零必装依赖**:核心纯 Python;jieba / tiktoken / embedding 全部可选
54
+ - **自带评测**:内置 BM25 检索召回评测 + 可插拔 LLM 判分器(支持干跑,零成本先跑通)
55
+ - **CPU 即可用**:不需要 GPU,不需要下载模型
56
+
57
+ ## Quickstart (English)
58
+
59
+ worddael (Worddael, "a cut-out chapter") chunks **Chinese** text for RAG, keeping exact
60
+ character offsets so every retrieved chunk can be cited back to its source.
61
+
62
+ ```python
63
+ from worddael import chunk
64
+
65
+ chunks = chunk(long_text, strategy="recursive", max_chars=500, overlap_chars=50)
66
+ assert all(c.text == long_text[c.start:c.end] for c in chunks) # invariant
67
+ ```
68
+
69
+ - Chinese-aware sentence boundaries (。!?;… with quote-closer attachment;
70
+ ASCII `.` never splits decimals/versions)
71
+ - Markdown heading paths, atomic code fences, parent-child (small-to-big)
72
+ chunking, pluggable semantic chunking and token counters
73
+ - Zero required dependencies, CPU-only; optional extras: `worddael[jieba]`,
74
+ `worddael[tiktoken]`
75
+ - Built-in evaluation: BM25 recall/precision@k over an original 25-doc
76
+ benchmark, LLM answerability judging (dry-run needs no API key)
77
+
78
+ See [docs/status_quo.md](docs/status_quo.md) for reproducible failure cases
79
+ of default splitters on Chinese, and [benchmarks/results.md](benchmarks/results.md)
80
+ for current numbers.
81
+
82
+ ## 安装 / Install
83
+
84
+ > 当前尚未发布到 PyPI;下方给出从 GitHub 获取并本地安装的命令。
85
+
86
+ ```bash
87
+ git clone https://github.com/cloudydreamland/TheDaleOfRememberedWords.git
88
+ cd TheDaleOfRememberedWords
89
+ python -m pip install .
90
+ # PyPI 首发后:python -m pip install worddael
91
+ python -m pip install ".[jieba]"
92
+ ```
93
+
94
+ ## 快速开始 / Quickstart
95
+
96
+ ```python
97
+ from worddael import chunk
98
+
99
+ text = open("manual.md", encoding="utf-8").read()
100
+ chunks = chunk(text, strategy="recursive", max_chars=500, overlap_chars=50)
101
+
102
+ for c in chunks:
103
+ print(c.seq, c.start, c.end, c.meta.get("headings"), c.text[:20])
104
+ assert c.text == text[c.start:c.end] # 永远成立,可放心引用
105
+ ```
106
+
107
+ ```python
108
+ from worddael import SemanticChunker, OpenAICompatibleEmbedder, HashingEmbedder
109
+
110
+ # 语义切分:接任意 OpenAI 兼容 embedding 接口(key 从环境变量读取,批量+超时可配)
111
+
112
+ embedder = OpenAICompatibleEmbedder(
113
+ base_url="https://open.bigmodel.cn/api/paas/v4",
114
+ model="embedding-3",
115
+ api_key_env="ZHIPUAI_API_KEY",
116
+ )
117
+ chunks = SemanticChunker(embed_fn=embedder, similarity_threshold=0.55,
118
+ max_chars=500, min_chars=80).chunk(long_text)
119
+
120
+ # 或用零依赖的 HashingEmbedder 先把管线跑通(玩具级,非质量结论)
121
+
122
+ chunks = SemanticChunker(embed_fn=HashingEmbedder()).chunk(long_text)
123
+ ```
124
+
125
+ 命令行:
126
+
127
+ ```bash
128
+ worddael doc.md --stats
129
+ worddael manual.txt --strategy recursive --max-chars 400 --overlap-chars 50 --jsonl chunks.jsonl
130
+ ```
131
+
132
+ 评测(CPU、零 API 成本):
133
+
134
+ ```bash
135
+ python -m worddael.eval.report --k 3 # 内置语料上对比各策略的 recall@k
136
+ ```
137
+
138
+ ## 切分策略 / Strategies
139
+
140
+ | strategy | 适用场景 | 说明 |
141
+ |---|---|---|
142
+ | `recursive` | 通用中文文本 | 段落→行→句号级→逗号级递归切分,中文标点优先 |
143
+ | `markdown` | .md 文档 | 按标题结构切,heading path 写入 meta,代码块保持完整 |
144
+ | `sentence` | 需要整句边界 | 按中文句界整句打包 |
145
+ | `token` | 对齐模型 token 预算 | 可插拔计数器(启发式 / jieba / tiktoken) |
146
+ | `semantic` | 主题边界敏感 | 自带 embedding 函数,相邻句相似度下降处切分,带防碎片最小块长 |
147
+ | `parent-child` | 小-大检索(RAG 常见手搓模式) | 子块精确匹配、父块给上下文;`chunk_families()` 返回 (父, 子),meta 双向链接,偏移不变量照旧 |
148
+
149
+ ## 与现有方案的关系 / Landscape
150
+
151
+ | 方案 | 问题 |
152
+ |---|---|
153
+ | LangChain text splitters | 通用切分组件;Worddael 可作为中文切分选项,并提供自己的偏移与策略 API |
154
+ | Chonkie 等 ingestion 库 | 面向更广的文档摄取与切分场景;选择时应按格式覆盖、语言和集成需求比较 |
155
+ | RAGFlow 等 RAG 引擎 | 提供完整应用或管线;Worddael 是可嵌入现有 Python 项目的独立库 |
156
+ | jieba | 中文分词工具;与按句界、结构和长度做文档切分的目标不同 |
157
+
158
+ LangChain 默认参数切中文的**真实坏例子**(答案腰斩、标题孤立,可复现脚本)见 [docs/status_quo.md](docs/status_quo.md);详细论证见 [GAP_PROOF.md](GAP_PROOF.md)。
159
+
160
+ ## 评测 / Evaluation
161
+
162
+ - 内置基准(25 篇 / 75 问)真实结果:[benchmarks/results.md](benchmarks/results.md)
163
+ - 当前快照:recall@1 recursive 0.96 vs fixed-window 0.907;小-大检索(子块 150 检索 / 父块 600 判分)0.987 vs 平铺 0.96
164
+ - 大规模 LLM 判分评测指南(拿到 key 后):[docs/eval_guide.md](docs/eval_guide.md)
165
+ - 选题论证(为什么这个缺口是真的):[GAP_PROOF.md](GAP_PROOF.md)
166
+
167
+ ## 路线图 / Roadmap
168
+
169
+ 见 [ROADMAP.md](ROADMAP.md)。当前 v0.1.0rc3:六个策略(含父子块)+ 可解释切分 + 标点卫生保证 + 评测框架(测试数以 CI 为准)。真实语料实测见 [docs/dogfood.md](docs/dogfood.md):90.6 万字《紅樓夢》,24.6 MB/s,卫生违例 0。
170
+
171
+ ## Non-goals(明确不做)
172
+
173
+ - **模型/LLM 驱动切分**:红海方向且依赖 GPU/API,与零依赖纯 CPU 定位冲突;语义切分走可插拔 embedder
174
+ - **通用 RAG 框架**:只做切分与切分评测,通过 [adapters](#安装--install) 进入 LangChain 等既有管线
175
+ - **计费级 token 计数**:启发式计数只服务预算切分(误差数字见基准报告校准小节)
176
+ - **流式超大文件**:单文档需可入内存;GB 级请自行分片
177
+
178
+ 架构与扩展指南见 [docs/architecture.md](docs/architecture.md)。
179
+
180
+ ## FAQ
181
+
182
+ - **overlap 模式下拼接结果比原文长?** 设计如此:overlap 是上下文重复(上一块尾部回看)。`overlap_chars=0` 时拼接与原文逐字相等(有测试锁定)。
183
+ - **为什么有的块以换行结尾?** 分隔符归属前一块是偏移精确的前提;判断块质量请对 `rstrip()` 后的文本判断(内置统计均如此)。
184
+ - **预算为何偶尔超出 1-2 字?** 标点卫生会把下一块开头的悬垂标点吸收进前一块(上限 2 字),换来"块首无悬垂标点"的保证。
185
+ - **测试数是固定的吗?** 以 CI 最新运行为准;文档不写死数字。
186
+
187
+ ## 开发 / Development
188
+
189
+ ```bash
190
+ pip install -e ".[jieba,dev]"
191
+ pytest
192
+ ruff check src tests
193
+ ```
194
+
195
+ ## 反馈与参与
196
+
197
+ 使用问题和功能建议可以在 [Discussions](https://github.com/cloudydreamland/TheDaleOfRememberedWords/discussions) 交流;可复现缺陷请提交 [Issue](https://github.com/cloudydreamland/TheDaleOfRememberedWords/issues)。请只附合成或脱敏后的最小样例,不要上传真实个人信息、API key 或业务原文。安全问题请按 [SECURITY.md](SECURITY.md) 私下报告。
198
+
199
+ ## License
200
+
201
+ MIT
@@ -0,0 +1,167 @@
1
+ # The Dale of Remembered Words — Worddael
2
+
3
+ 简体中文 · [English](README.en.md)
4
+
5
+ > 展示名 **The Dale of Remembered Words** 描绘一处安放文字的山谷;Worddael 是该项目的短名。
6
+
7
+ **中文优先的 RAG 文本切分库。把中文文本切成好用的块,并保留精确到字符的原文偏移量。**
8
+
9
+ [![CI](https://github.com/cloudydreamland/TheDaleOfRememberedWords/actions/workflows/ci.yml/badge.svg)](.github/workflows/ci.yml)
10
+ [![Python](https://img.shields.io/badge/python-3.10%2B-blue)](pyproject.toml)
11
+ [![License: MIT](https://img.shields.io/badge/license-MIT-green)](LICENSE)
12
+
13
+ ## 为什么需要它 / Why
14
+
15
+ 通用文本切分器可用于多种语言,但中文句界、标点归属和回引源文等细节往往需要应用自行处理。`worddael` 提供中文规则、可组合策略和原文偏移,让这些行为能在管线中检查:
16
+
17
+ - **中文句界原生正确**:`。!?;…` 加引号闭合格式感知;ASCII `.` 不切(保护 `3.14`、版本号、URL)
18
+ - **字符级偏移不变量**:每个 chunk 保证 `chunk.text == source[chunk.start:chunk.end]`,检索结果可直接引用回原文(fuzz 测试覆盖)
19
+ - **零必装依赖**:核心纯 Python;jieba / tiktoken / embedding 全部可选
20
+ - **自带评测**:内置 BM25 检索召回评测 + 可插拔 LLM 判分器(支持干跑,零成本先跑通)
21
+ - **CPU 即可用**:不需要 GPU,不需要下载模型
22
+
23
+ ## Quickstart (English)
24
+
25
+ worddael (Worddael, "a cut-out chapter") chunks **Chinese** text for RAG, keeping exact
26
+ character offsets so every retrieved chunk can be cited back to its source.
27
+
28
+ ```python
29
+ from worddael import chunk
30
+
31
+ chunks = chunk(long_text, strategy="recursive", max_chars=500, overlap_chars=50)
32
+ assert all(c.text == long_text[c.start:c.end] for c in chunks) # invariant
33
+ ```
34
+
35
+ - Chinese-aware sentence boundaries (。!?;… with quote-closer attachment;
36
+ ASCII `.` never splits decimals/versions)
37
+ - Markdown heading paths, atomic code fences, parent-child (small-to-big)
38
+ chunking, pluggable semantic chunking and token counters
39
+ - Zero required dependencies, CPU-only; optional extras: `worddael[jieba]`,
40
+ `worddael[tiktoken]`
41
+ - Built-in evaluation: BM25 recall/precision@k over an original 25-doc
42
+ benchmark, LLM answerability judging (dry-run needs no API key)
43
+
44
+ See [docs/status_quo.md](docs/status_quo.md) for reproducible failure cases
45
+ of default splitters on Chinese, and [benchmarks/results.md](benchmarks/results.md)
46
+ for current numbers.
47
+
48
+ ## 安装 / Install
49
+
50
+ > 当前尚未发布到 PyPI;下方给出从 GitHub 获取并本地安装的命令。
51
+
52
+ ```bash
53
+ git clone https://github.com/cloudydreamland/TheDaleOfRememberedWords.git
54
+ cd TheDaleOfRememberedWords
55
+ python -m pip install .
56
+ # PyPI 首发后:python -m pip install worddael
57
+ python -m pip install ".[jieba]"
58
+ ```
59
+
60
+ ## 快速开始 / Quickstart
61
+
62
+ ```python
63
+ from worddael import chunk
64
+
65
+ text = open("manual.md", encoding="utf-8").read()
66
+ chunks = chunk(text, strategy="recursive", max_chars=500, overlap_chars=50)
67
+
68
+ for c in chunks:
69
+ print(c.seq, c.start, c.end, c.meta.get("headings"), c.text[:20])
70
+ assert c.text == text[c.start:c.end] # 永远成立,可放心引用
71
+ ```
72
+
73
+ ```python
74
+ from worddael import SemanticChunker, OpenAICompatibleEmbedder, HashingEmbedder
75
+
76
+ # 语义切分:接任意 OpenAI 兼容 embedding 接口(key 从环境变量读取,批量+超时可配)
77
+
78
+ embedder = OpenAICompatibleEmbedder(
79
+ base_url="https://open.bigmodel.cn/api/paas/v4",
80
+ model="embedding-3",
81
+ api_key_env="ZHIPUAI_API_KEY",
82
+ )
83
+ chunks = SemanticChunker(embed_fn=embedder, similarity_threshold=0.55,
84
+ max_chars=500, min_chars=80).chunk(long_text)
85
+
86
+ # 或用零依赖的 HashingEmbedder 先把管线跑通(玩具级,非质量结论)
87
+
88
+ chunks = SemanticChunker(embed_fn=HashingEmbedder()).chunk(long_text)
89
+ ```
90
+
91
+ 命令行:
92
+
93
+ ```bash
94
+ worddael doc.md --stats
95
+ worddael manual.txt --strategy recursive --max-chars 400 --overlap-chars 50 --jsonl chunks.jsonl
96
+ ```
97
+
98
+ 评测(CPU、零 API 成本):
99
+
100
+ ```bash
101
+ python -m worddael.eval.report --k 3 # 内置语料上对比各策略的 recall@k
102
+ ```
103
+
104
+ ## 切分策略 / Strategies
105
+
106
+ | strategy | 适用场景 | 说明 |
107
+ |---|---|---|
108
+ | `recursive` | 通用中文文本 | 段落→行→句号级→逗号级递归切分,中文标点优先 |
109
+ | `markdown` | .md 文档 | 按标题结构切,heading path 写入 meta,代码块保持完整 |
110
+ | `sentence` | 需要整句边界 | 按中文句界整句打包 |
111
+ | `token` | 对齐模型 token 预算 | 可插拔计数器(启发式 / jieba / tiktoken) |
112
+ | `semantic` | 主题边界敏感 | 自带 embedding 函数,相邻句相似度下降处切分,带防碎片最小块长 |
113
+ | `parent-child` | 小-大检索(RAG 常见手搓模式) | 子块精确匹配、父块给上下文;`chunk_families()` 返回 (父, 子),meta 双向链接,偏移不变量照旧 |
114
+
115
+ ## 与现有方案的关系 / Landscape
116
+
117
+ | 方案 | 问题 |
118
+ |---|---|
119
+ | LangChain text splitters | 通用切分组件;Worddael 可作为中文切分选项,并提供自己的偏移与策略 API |
120
+ | Chonkie 等 ingestion 库 | 面向更广的文档摄取与切分场景;选择时应按格式覆盖、语言和集成需求比较 |
121
+ | RAGFlow 等 RAG 引擎 | 提供完整应用或管线;Worddael 是可嵌入现有 Python 项目的独立库 |
122
+ | jieba | 中文分词工具;与按句界、结构和长度做文档切分的目标不同 |
123
+
124
+ LangChain 默认参数切中文的**真实坏例子**(答案腰斩、标题孤立,可复现脚本)见 [docs/status_quo.md](docs/status_quo.md);详细论证见 [GAP_PROOF.md](GAP_PROOF.md)。
125
+
126
+ ## 评测 / Evaluation
127
+
128
+ - 内置基准(25 篇 / 75 问)真实结果:[benchmarks/results.md](benchmarks/results.md)
129
+ - 当前快照:recall@1 recursive 0.96 vs fixed-window 0.907;小-大检索(子块 150 检索 / 父块 600 判分)0.987 vs 平铺 0.96
130
+ - 大规模 LLM 判分评测指南(拿到 key 后):[docs/eval_guide.md](docs/eval_guide.md)
131
+ - 选题论证(为什么这个缺口是真的):[GAP_PROOF.md](GAP_PROOF.md)
132
+
133
+ ## 路线图 / Roadmap
134
+
135
+ 见 [ROADMAP.md](ROADMAP.md)。当前 v0.1.0rc3:六个策略(含父子块)+ 可解释切分 + 标点卫生保证 + 评测框架(测试数以 CI 为准)。真实语料实测见 [docs/dogfood.md](docs/dogfood.md):90.6 万字《紅樓夢》,24.6 MB/s,卫生违例 0。
136
+
137
+ ## Non-goals(明确不做)
138
+
139
+ - **模型/LLM 驱动切分**:红海方向且依赖 GPU/API,与零依赖纯 CPU 定位冲突;语义切分走可插拔 embedder
140
+ - **通用 RAG 框架**:只做切分与切分评测,通过 [adapters](#安装--install) 进入 LangChain 等既有管线
141
+ - **计费级 token 计数**:启发式计数只服务预算切分(误差数字见基准报告校准小节)
142
+ - **流式超大文件**:单文档需可入内存;GB 级请自行分片
143
+
144
+ 架构与扩展指南见 [docs/architecture.md](docs/architecture.md)。
145
+
146
+ ## FAQ
147
+
148
+ - **overlap 模式下拼接结果比原文长?** 设计如此:overlap 是上下文重复(上一块尾部回看)。`overlap_chars=0` 时拼接与原文逐字相等(有测试锁定)。
149
+ - **为什么有的块以换行结尾?** 分隔符归属前一块是偏移精确的前提;判断块质量请对 `rstrip()` 后的文本判断(内置统计均如此)。
150
+ - **预算为何偶尔超出 1-2 字?** 标点卫生会把下一块开头的悬垂标点吸收进前一块(上限 2 字),换来"块首无悬垂标点"的保证。
151
+ - **测试数是固定的吗?** 以 CI 最新运行为准;文档不写死数字。
152
+
153
+ ## 开发 / Development
154
+
155
+ ```bash
156
+ pip install -e ".[jieba,dev]"
157
+ pytest
158
+ ruff check src tests
159
+ ```
160
+
161
+ ## 反馈与参与
162
+
163
+ 使用问题和功能建议可以在 [Discussions](https://github.com/cloudydreamland/TheDaleOfRememberedWords/discussions) 交流;可复现缺陷请提交 [Issue](https://github.com/cloudydreamland/TheDaleOfRememberedWords/issues)。请只附合成或脱敏后的最小样例,不要上传真实个人信息、API key 或业务原文。安全问题请按 [SECURITY.md](SECURITY.md) 私下报告。
164
+
165
+ ## License
166
+
167
+ MIT
@@ -0,0 +1,58 @@
1
+ [build-system]
2
+ requires = ["setuptools>=68"]
3
+ build-backend = "setuptools.build_meta"
4
+
5
+ [project]
6
+ name = "worddael"
7
+ version = "0.1.0rc3"
8
+ description = "The Dale of Remembered Words (Worddael) | Chinese-aware RAG text chunking"
9
+ readme = "README.md"
10
+ license = { text = "MIT" }
11
+ authors = [{ name = "worddael contributors" }]
12
+ requires-python = ">=3.10"
13
+ keywords = ["chinese", "chunking", "rag", "nlp", "text-splitting", "retrieval"]
14
+ classifiers = [
15
+ "Development Status :: 3 - Alpha",
16
+ "Intended Audience :: Developers",
17
+ "License :: OSI Approved :: MIT License",
18
+ "Natural Language :: Chinese (Simplified)",
19
+ "Programming Language :: Python :: 3",
20
+ "Programming Language :: Python :: 3.10",
21
+ "Programming Language :: Python :: 3.11",
22
+ "Programming Language :: Python :: 3.12",
23
+ "Programming Language :: Python :: 3.13",
24
+ "Topic :: Text Processing :: Linguistic",
25
+ ]
26
+ dependencies = []
27
+
28
+ [project.optional-dependencies]
29
+ jieba = ["jieba>=0.42"]
30
+ tiktoken = ["tiktoken>=0.7"]
31
+ dev = ["pytest>=8.0", "ruff>=0.6"]
32
+
33
+ [project.scripts]
34
+ worddael = "worddael.cli:main"
35
+
36
+ [project.urls]
37
+ Homepage = "https://github.com/cloudydreamland/TheDaleOfRememberedWords"
38
+ Repository = "https://github.com/cloudydreamland/TheDaleOfRememberedWords"
39
+ Issues = "https://github.com/cloudydreamland/TheDaleOfRememberedWords/issues"
40
+ Changelog = "https://github.com/cloudydreamland/TheDaleOfRememberedWords/blob/main/CHANGELOG.md"
41
+ Security = "https://github.com/cloudydreamland/TheDaleOfRememberedWords/security/policy"
42
+ [tool.setuptools.packages.find]
43
+ where = ["src"]
44
+
45
+ [tool.setuptools.package-data]
46
+ worddael = ["py.typed"]
47
+
48
+ [tool.pytest.ini_options]
49
+ testpaths = ["tests"]
50
+ addopts = "-q"
51
+
52
+ [tool.ruff]
53
+ line-length = 100
54
+ target-version = "py310"
55
+
56
+ [tool.ruff.lint]
57
+ select = ["E", "F", "I", "W"]
58
+ ignore = ["E501"]
@@ -0,0 +1,4 @@
1
+ [egg_info]
2
+ tag_build =
3
+ tag_date = 0
4
+
@@ -0,0 +1,92 @@
1
+ """Worddael — Chinese-first text chunking for RAG.
2
+
3
+ 把中文文本切成好用的块,同时保留精确到字符的原文偏移量,
4
+ 让 RAG 的每一条检索结果都能被可靠引用。
5
+
6
+ Quickstart::
7
+
8
+ from worddael import chunk
9
+
10
+ chunks = chunk(long_text, strategy="recursive", max_chars=500, overlap_chars=50)
11
+ for c in chunks:
12
+ print(c.seq, c.start, c.end, c.text[:20])
13
+ """
14
+
15
+ from __future__ import annotations
16
+
17
+ from .chunkers import (
18
+ STRATEGIES,
19
+ BaseChunker,
20
+ MarkdownChunker,
21
+ RecursiveChunker,
22
+ SemanticChunker,
23
+ SentenceChunker,
24
+ TokenChunker,
25
+ )
26
+ from .counters import CharCounter, get_counter
27
+ from .embedders import (
28
+ Embedder,
29
+ EmbedderError,
30
+ HashingEmbedder,
31
+ OpenAICompatibleEmbedder,
32
+ )
33
+ from .io_utils import chunk_file, to_jsonl
34
+ from .parent_child import ParentChildChunker
35
+ from .types import Chunk
36
+
37
+ __version__ = "0.1.0rc3"
38
+
39
+ # parent-child families are produced by a dedicated module; register the
40
+ # strategy here so get_chunker("parent-child") works like any other.
41
+ STRATEGIES["parent-child"] = ParentChildChunker
42
+
43
+ __all__ = [
44
+ "BaseChunker",
45
+ "CharCounter",
46
+ "Chunk",
47
+ "Embedder",
48
+ "EmbedderError",
49
+ "HashingEmbedder",
50
+ "MarkdownChunker",
51
+ "OpenAICompatibleEmbedder",
52
+ "ParentChildChunker",
53
+ "RecursiveChunker",
54
+ "SemanticChunker",
55
+ "SentenceChunker",
56
+ "STRATEGIES",
57
+ "TokenChunker",
58
+ "__version__",
59
+ "chunk",
60
+ "chunk_file",
61
+ "get_chunker",
62
+ "get_counter",
63
+ "to_jsonl",
64
+ ]
65
+
66
+
67
+ def get_chunker(strategy: str = "recursive", **kwargs) -> BaseChunker:
68
+ """Build a chunker by name. See ``STRATEGIES`` for available names."""
69
+ try:
70
+ cls = STRATEGIES[strategy]
71
+ except KeyError:
72
+ raise ValueError(
73
+ f"unknown strategy {strategy!r}; expected one of {sorted(STRATEGIES)}"
74
+ ) from None
75
+ return cls(**kwargs)
76
+
77
+
78
+ def chunk(text: str, strategy: str = "recursive", **kwargs) -> list[Chunk]:
79
+ """One-shot helper: ``chunk(text, strategy, **chunker_kwargs)``.
80
+
81
+ Uniform budget names: ``max_chars``/``overlap_chars`` are accepted by
82
+ every strategy. For ``strategy="token"`` they are interpreted as token
83
+ budgets (mapped to ``max_tokens``/``overlap_tokens``) — tokens are what
84
+ that strategy measures, and the char-named arguments keep call sites
85
+ strategy-agnostic.
86
+ """
87
+ if strategy == "token":
88
+ if "max_tokens" not in kwargs and "max_chars" in kwargs:
89
+ kwargs["max_tokens"] = kwargs.pop("max_chars")
90
+ if "overlap_tokens" not in kwargs and "overlap_chars" in kwargs:
91
+ kwargs["overlap_tokens"] = kwargs.pop("overlap_chars")
92
+ return get_chunker(strategy, **kwargs).chunk(text)
@@ -0,0 +1,45 @@
1
+ """Ecosystem adapters: plug worddael into frameworks users already run.
2
+
3
+ LangChain: :class:`LangChainChunker` subclasses
4
+ ``langchain_text_splitters.TextSplitter`` (optional import), so it works
5
+ with ``split_documents`` / ``create_documents`` and drops into any existing
6
+ RAG pipeline as a drop-in replacement for RecursiveCharacterTextSplitter.
7
+ """
8
+
9
+ from __future__ import annotations
10
+
11
+ from . import get_chunker
12
+ from .chunkers import BaseChunker
13
+
14
+ try: # optional dependency — the adapter degrades to a plain object otherwise
15
+ from langchain_text_splitters import TextSplitter as _LCTextSplitter
16
+
17
+ _HAS_LANGCHAIN = True
18
+ except ImportError: # pragma: no cover
19
+ _LCTextSplitter = object # type: ignore[assignment,misc]
20
+ _HAS_LANGCHAIN = False
21
+
22
+ __all__ = ["LangChainChunker", "has_langchain"]
23
+
24
+
25
+ def has_langchain() -> bool:
26
+ return _HAS_LANGCHAIN
27
+
28
+
29
+ class LangChainChunker(_LCTextSplitter):
30
+ """worddael chunking behind the LangChain ``TextSplitter`` interface.
31
+
32
+ Example::
33
+
34
+ from worddael.adapters import LangChainChunker
35
+ splitter = LangChainChunker(strategy="recursive", max_chars=500, overlap_chars=50)
36
+ docs = splitter.create_documents([long_text]) # full LC Document flow
37
+ """
38
+
39
+ def __init__(self, strategy: str = "recursive", **chunker_kwargs) -> None:
40
+ if _HAS_LANGCHAIN:
41
+ super().__init__()
42
+ self._chunker: BaseChunker = get_chunker(strategy, **chunker_kwargs)
43
+
44
+ def split_text(self, text: str) -> list[str]:
45
+ return [c.text for c in self._chunker.chunk(text)]