@yottameta/yotta-humanize 0.1.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +13 -0
- package/LICENSE +21 -0
- package/NOTICE +11 -0
- package/README.md +157 -0
- package/SKILL.md +79 -0
- package/assets/banner.png +0 -0
- package/bin/install.js +163 -0
- package/install.sh +132 -0
- package/package.json +32 -0
- package/references/patterns.md +60 -0
- package/references/rewriting.md +48 -0
- package/references/scoring.md +50 -0
- package/scripts/humanize_rules.py +281 -0
- package/scripts/test_yotta_humanize.py +265 -0
- package/scripts/yotta_humanize.py +633 -0
|
@@ -0,0 +1,265 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
# -*- coding: utf-8 -*-
|
|
3
|
+
"""test_yotta_humanize.py — 元真(yotta-humanize)测试。
|
|
4
|
+
|
|
5
|
+
覆盖:24 类规则命中 / 干净文本低分 / 统计量 / 确定性改写 / CLI 子命令 /
|
|
6
|
+
退出码 / JSON 输出 / 控制台编码。纯标准库,无 pytest 依赖。
|
|
7
|
+
|
|
8
|
+
运行:python scripts/test_yotta_humanize.py
|
|
9
|
+
"""
|
|
10
|
+
import json
|
|
11
|
+
import os
|
|
12
|
+
import subprocess
|
|
13
|
+
import sys
|
|
14
|
+
import tempfile
|
|
15
|
+
from pathlib import Path
|
|
16
|
+
|
|
17
|
+
_HERE = Path(__file__).resolve().parent
|
|
18
|
+
sys.path.insert(0, str(_HERE))
|
|
19
|
+
|
|
20
|
+
import humanize_rules as HR # noqa: E402
|
|
21
|
+
import yotta_humanize as YH # noqa: E402
|
|
22
|
+
|
|
23
|
+
PASS = 0
|
|
24
|
+
FAIL = 0
|
|
25
|
+
FAILED = []
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def check(name, cond, detail=""):
|
|
29
|
+
global PASS, FAIL
|
|
30
|
+
if cond:
|
|
31
|
+
PASS += 1
|
|
32
|
+
else:
|
|
33
|
+
FAIL += 1
|
|
34
|
+
FAILED.append(name)
|
|
35
|
+
print(" FAIL: %s %s" % (name, detail))
|
|
36
|
+
|
|
37
|
+
|
|
38
|
+
def detect_ids(text):
|
|
39
|
+
return [f["rule_id"] for f in YH.detect(text)]
|
|
40
|
+
|
|
41
|
+
|
|
42
|
+
def test_rules_table():
|
|
43
|
+
ids = [r["id"] for r in HR.RULES]
|
|
44
|
+
check("规则数量 = 24", len(HR.RULES) == 24, "got %d" % len(HR.RULES))
|
|
45
|
+
check("规则 id 唯一", len(ids) == len(set(ids)))
|
|
46
|
+
check("规则 id 命名 HZ-01..HZ-24",
|
|
47
|
+
ids == ["HZ-%02d" % i for i in range(1, 25)], str(ids))
|
|
48
|
+
for r in HR.RULES:
|
|
49
|
+
check("规则 %s 字段齐全" % r["id"],
|
|
50
|
+
all(k in r for k in ("id", "name", "group", "weight",
|
|
51
|
+
"regexes", "words", "suggestion")),
|
|
52
|
+
str(r.keys()))
|
|
53
|
+
check("规则 %s group 合法" % r["id"], r["group"] in HR.GROUPS)
|
|
54
|
+
check("规则 %s weight 1-5" % r["id"], 1 <= r["weight"] <= 5)
|
|
55
|
+
check("规则 %s 有检测手段" % r["id"],
|
|
56
|
+
bool(r["regexes"]) or bool(r["words"]))
|
|
57
|
+
|
|
58
|
+
|
|
59
|
+
def test_detect_24():
|
|
60
|
+
cases = [
|
|
61
|
+
("HZ-01", "这一举措标志着新时代的到来,意义重大。"),
|
|
62
|
+
("HZ-02", "随着人工智能技术的发展,很多行业都变了。"),
|
|
63
|
+
("HZ-03", "综上所述,未来可期,让我们拭目以待。"),
|
|
64
|
+
("HZ-04", "说白了,这件事并不难。"),
|
|
65
|
+
("HZ-05", "专家指出,长期熬夜有害健康。"),
|
|
66
|
+
("HZ-06", "毫无疑问,这个方案一定会成功。"),
|
|
67
|
+
("HZ-07", "我们要给业务赋能,找到真正的抓手。"),
|
|
68
|
+
("HZ-08", "这个问题非常非常重要。"),
|
|
69
|
+
("HZ-09", "我们需要完善和优化现有流程。"),
|
|
70
|
+
("HZ-10", "我们坚信明天会更好。"),
|
|
71
|
+
("HZ-11", "问题的解决需要时间。"),
|
|
72
|
+
("HZ-12", "我们要更加深入地理解需求。"),
|
|
73
|
+
("HZ-13", "它不仅速度快,而且很稳定,更让人放心。"),
|
|
74
|
+
("HZ-14", "首先,我们要准备。其次,我们要行动。"),
|
|
75
|
+
("HZ-15", "难道不是最好的选择吗?"),
|
|
76
|
+
("HZ-16", "第一段——第二段——第三段——第四段。"),
|
|
77
|
+
("HZ-17", "这就是所谓的“赋能”。"),
|
|
78
|
+
("HZ-18", "太好了!真棒!加油!"),
|
|
79
|
+
("HZ-19", "希望对你有所帮助。"),
|
|
80
|
+
("HZ-20", "很好的问题。答案很简单。"),
|
|
81
|
+
("HZ-21", "限于篇幅,这里不再赘述。"),
|
|
82
|
+
("HZ-22", "其实,这事很简单。"),
|
|
83
|
+
("HZ-23", "话说回来,还是要小心。"),
|
|
84
|
+
("HZ-24", "值得注意的是,价格在上涨。"),
|
|
85
|
+
]
|
|
86
|
+
for rid, text in cases:
|
|
87
|
+
got = detect_ids(text)
|
|
88
|
+
check("命中 %s" % rid, rid in got, "text=%r got=%s" % (text, got))
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def test_clean_text_low_score():
|
|
92
|
+
clean = ("我们上周把接口切到了新网关,压测 5000 并发没有报错。"
|
|
93
|
+
"旧方案在凌晨会偶发超时,查了三天日志,最后发现是连接池太小。"
|
|
94
|
+
"这周先观察线上,再决定要不要调参数。")
|
|
95
|
+
res = YH.analyze(clean)
|
|
96
|
+
check("干净文本评分 < 20", res["score"] < 20, "score=%d" % res["score"])
|
|
97
|
+
check("干净文本规则命中少", res["rule_types"] <= 1,
|
|
98
|
+
"rules=%s" % [f["rule_id"] for f in res["findings"]])
|
|
99
|
+
|
|
100
|
+
|
|
101
|
+
def test_ai_text_high_score():
|
|
102
|
+
ai = ("众所周知,随着人工智能技术的飞速发展,各行各业都在发生深刻变革。"
|
|
103
|
+
"这一举措标志着新时代的到来,意义重大,充分体现了我们的价值。"
|
|
104
|
+
"我们要给业务赋能,找到抓手,形成闭环。"
|
|
105
|
+
"首先,我们要完善和优化流程。其次,我们要加强和完善团队建设。"
|
|
106
|
+
"专家指出,未来可期。值得注意的是,我们坚信明天会更好。"
|
|
107
|
+
"希望对你有所帮助。")
|
|
108
|
+
res = YH.analyze(ai)
|
|
109
|
+
check("AI 文本评分 >= 45", res["score"] >= 45, "score=%d" % res["score"])
|
|
110
|
+
check("AI 文本多类命中", res["rule_types"] >= 8,
|
|
111
|
+
"rules=%d" % res["rule_types"])
|
|
112
|
+
|
|
113
|
+
|
|
114
|
+
def test_stats():
|
|
115
|
+
uniform = "今天天气很好。明天天气也很好。后天天气依然很好。"
|
|
116
|
+
varied = ("昨晚下雨了。"
|
|
117
|
+
"雨停之后我下楼,看见小区门口那棵老槐树被风刮断了一根大枝,"
|
|
118
|
+
"横在路中间,几个保安正拿锯子一点点把它锯开。")
|
|
119
|
+
su = YH.compute_stats(uniform)
|
|
120
|
+
sv = YH.compute_stats(varied)
|
|
121
|
+
check("均匀文本 CV < 起伏文本 CV", su["cv"] < sv["cv"],
|
|
122
|
+
"%s vs %s" % (su["cv"], sv["cv"]))
|
|
123
|
+
check("起伏文本突发性 >= 均匀文本", sv["burstiness"] >= su["burstiness"],
|
|
124
|
+
"%s vs %s" % (sv["burstiness"], su["burstiness"]))
|
|
125
|
+
check("stats_score 短文本为 0", YH.stats_score(
|
|
126
|
+
YH.compute_stats("很短。")) == 0)
|
|
127
|
+
|
|
128
|
+
|
|
129
|
+
def test_rewrite_mechanical():
|
|
130
|
+
r = YH.rewrite("众所周知,AI 很重要。综上所述,未来可期。希望对你有所帮助。")
|
|
131
|
+
for bad in ("众所周知", "综上所述", "未来可期", "希望对你有所帮助"):
|
|
132
|
+
check("改写删除「%s」" % bad, bad not in r["text"],
|
|
133
|
+
"text=%r" % r["text"])
|
|
134
|
+
check("改写有 fixes 记录", len(r["fixes"]) >= 3, str(r["fixes"]))
|
|
135
|
+
|
|
136
|
+
r2 = YH.rewrite("我们要给业务赋能,找到抓手。")
|
|
137
|
+
check("改写黑话 赋能→支持", "赋能" not in r2["text"] and "支持" in r2["text"],
|
|
138
|
+
"text=%r" % r2["text"])
|
|
139
|
+
check("改写黑话 抓手→切入点", "抓手" not in r2["text"] and "切入点" in r2["text"],
|
|
140
|
+
"text=%r" % r2["text"])
|
|
141
|
+
|
|
142
|
+
r3 = YH.rewrite("第一段——第二段——第三段——第四段。")
|
|
143
|
+
check("改写破折号限 2 处", r3["text"].count("——") <= 2,
|
|
144
|
+
"text=%r" % r3["text"])
|
|
145
|
+
|
|
146
|
+
r4 = YH.rewrite("太好了!真棒!加油!")
|
|
147
|
+
check("改写感叹号限 2 处", r4["text"].count("!") <= 2,
|
|
148
|
+
"text=%r" % r4["text"])
|
|
149
|
+
|
|
150
|
+
r5b = YH.rewrite("我们要形成完整闭环。")
|
|
151
|
+
check("改写闭环不重复「完整」", "完整完整" not in r5b["text"],
|
|
152
|
+
"text=%r" % r5b["text"])
|
|
153
|
+
|
|
154
|
+
r5 = YH.rewrite("我们需要完善和优化流程。")
|
|
155
|
+
check("改写同义堆叠", "完善和优化" not in r5["text"],
|
|
156
|
+
"text=%r" % r5["text"])
|
|
157
|
+
|
|
158
|
+
# 改写后评分应下降(AI 腔文本)
|
|
159
|
+
ai = ("众所周知,随着人工智能技术的飞速发展,各行各业都在发生深刻变革。"
|
|
160
|
+
"这一举措标志着新时代的到来,意义重大。"
|
|
161
|
+
"我们要给业务赋能,找到抓手,形成闭环。"
|
|
162
|
+
"专家指出,未来可期。值得注意的是,我们坚信明天会更好。"
|
|
163
|
+
"希望对你有所帮助。")
|
|
164
|
+
r6 = YH.rewrite(ai)
|
|
165
|
+
check("改写后评分下降", r6["after"] < r6["before"],
|
|
166
|
+
"before=%d after=%d" % (r6["before"], r6["after"]))
|
|
167
|
+
|
|
168
|
+
|
|
169
|
+
def _run(args, inp=None):
|
|
170
|
+
return subprocess.run([sys.executable, str(_HERE / "yotta_humanize.py")] + args,
|
|
171
|
+
input=inp, capture_output=True, text=True,
|
|
172
|
+
encoding="utf-8", errors="replace")
|
|
173
|
+
|
|
174
|
+
|
|
175
|
+
def test_cli():
|
|
176
|
+
ai = "众所周知,未来可期。我们希望对你有所帮助。"
|
|
177
|
+
r = _run(["score"], inp=ai)
|
|
178
|
+
check("score 文本输出为数字", r.returncode == 0 and r.stdout.strip().isdigit(),
|
|
179
|
+
"rc=%d out=%r" % (r.returncode, r.stdout))
|
|
180
|
+
|
|
181
|
+
r = _run(["score", "--gate", "--threshold", "1"], inp=ai)
|
|
182
|
+
check("score --gate 高分退出码 1", r.returncode == 1, "rc=%d" % r.returncode)
|
|
183
|
+
|
|
184
|
+
r = _run(["score", "--gate", "--threshold", "99"], inp=ai)
|
|
185
|
+
check("score --gate 低分退出码 0", r.returncode == 0, "rc=%d" % r.returncode)
|
|
186
|
+
|
|
187
|
+
r = _run(["score", "--json"], inp=ai)
|
|
188
|
+
try:
|
|
189
|
+
obj = json.loads(r.stdout)
|
|
190
|
+
check("score --json 可解析且含 score",
|
|
191
|
+
r.returncode == 0 and "score" in obj, r.stdout[:80])
|
|
192
|
+
except Exception as e: # noqa: BLE001
|
|
193
|
+
check("score --json 可解析", False, str(e))
|
|
194
|
+
|
|
195
|
+
with tempfile.TemporaryDirectory() as td:
|
|
196
|
+
f = Path(td) / "article.txt"
|
|
197
|
+
f.write_text(ai, encoding="utf-8")
|
|
198
|
+
r = _run(["analyze", "-f", str(f)])
|
|
199
|
+
check("analyze -f 输出含评分", r.returncode == 0 and "评分" in r.stdout,
|
|
200
|
+
"rc=%d" % r.returncode)
|
|
201
|
+
r = _run(["report", "-f", str(f)])
|
|
202
|
+
check("report -f 输出 markdown 标题",
|
|
203
|
+
r.returncode == 0 and r.stdout.startswith("# AI 腔检测报告"),
|
|
204
|
+
"rc=%d head=%r" % (r.returncode, r.stdout[:40]))
|
|
205
|
+
r = _run(["suggest", "-f", str(f)])
|
|
206
|
+
check("suggest -f 输出建议", r.returncode == 0 and "建议" in r.stdout,
|
|
207
|
+
"rc=%d" % r.returncode)
|
|
208
|
+
r = _run(["rewrite", "-f", str(f)])
|
|
209
|
+
check("rewrite -f 输出修复与文本",
|
|
210
|
+
r.returncode == 0 and "改写后评分" in r.stdout,
|
|
211
|
+
"rc=%d" % r.returncode)
|
|
212
|
+
r = _run(["rewrite", "--json", "-f", str(f)])
|
|
213
|
+
try:
|
|
214
|
+
obj = json.loads(r.stdout)
|
|
215
|
+
check("rewrite --json 含 text/fixes",
|
|
216
|
+
"text" in obj and "fixes" in obj and "after" in obj,
|
|
217
|
+
r.stdout[:80])
|
|
218
|
+
except Exception as e: # noqa: BLE001
|
|
219
|
+
check("rewrite --json 可解析", False, str(e))
|
|
220
|
+
|
|
221
|
+
r = _run(["analyze", "-f", "不存在的文件.txt"])
|
|
222
|
+
check("analyze 缺文件退出码 4", r.returncode == 4, "rc=%d" % r.returncode)
|
|
223
|
+
|
|
224
|
+
r = _run(["version"])
|
|
225
|
+
check("version 输出版本", r.returncode == 0 and YH.VERSION in r.stdout,
|
|
226
|
+
"rc=%d out=%r" % (r.returncode, r.stdout))
|
|
227
|
+
|
|
228
|
+
r = _run(["badcmd"])
|
|
229
|
+
check("未知子命令退出码 2(argparse)", r.returncode == 2,
|
|
230
|
+
"rc=%d" % r.returncode)
|
|
231
|
+
|
|
232
|
+
|
|
233
|
+
def test_gbk_console():
|
|
234
|
+
# Windows GBK 控制台:引擎 stdout 已重配 UTF-8,中文输出不报错
|
|
235
|
+
ai = "众所周知,未来可期。"
|
|
236
|
+
env = dict(os.environ)
|
|
237
|
+
env["PYTHONIOENCODING"] = "gbk"
|
|
238
|
+
r = subprocess.run(
|
|
239
|
+
[sys.executable, str(_HERE / "yotta_humanize.py"), "score"],
|
|
240
|
+
input=ai, capture_output=True, text=True, encoding="gbk",
|
|
241
|
+
errors="replace", env=env)
|
|
242
|
+
check("GBK 控制台中文输出不炸", r.returncode == 0, "rc=%d err=%r" % (r.returncode, r.stderr[:100]))
|
|
243
|
+
|
|
244
|
+
|
|
245
|
+
def main():
|
|
246
|
+
print("元真(yotta-humanize)测试开始…")
|
|
247
|
+
test_rules_table()
|
|
248
|
+
test_detect_24()
|
|
249
|
+
test_clean_text_low_score()
|
|
250
|
+
test_ai_text_high_score()
|
|
251
|
+
test_stats()
|
|
252
|
+
test_rewrite_mechanical()
|
|
253
|
+
test_cli()
|
|
254
|
+
test_gbk_console()
|
|
255
|
+
print("")
|
|
256
|
+
print("通过 %d 项,失败 %d 项" % (PASS, FAIL))
|
|
257
|
+
if FAILED:
|
|
258
|
+
print("失败清单:")
|
|
259
|
+
for name in FAILED:
|
|
260
|
+
print(" - " + name)
|
|
261
|
+
sys.exit(1 if FAIL else 0)
|
|
262
|
+
|
|
263
|
+
|
|
264
|
+
if __name__ == "__main__":
|
|
265
|
+
main()
|