@yottameta/yotta-humanize 0.1.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,633 @@
1
+ #!/usr/bin/env python3
2
+ # -*- coding: utf-8 -*-
3
+ """yotta_humanize.py — YottaMeta 元真(yotta-humanize):去 AI 味中文写作编辑引擎。
4
+
5
+ 检测器引擎(24 类规则 + 词表 + 统计突发性)识别 AI 腔文本,并给出确定性改写。
6
+ 纯 Python 3.8+ 标准库,零外部依赖,Windows + Linux + macOS 通用,跨智能体可用。
7
+
8
+ 子命令:
9
+ score 输出 AI 腔评分(0-100,越高越像 AI 写的)
10
+ analyze 详细检测报告(文本)
11
+ report Markdown 检测报告
12
+ suggest 按优先级分组的改写建议
13
+ rewrite 确定性改写文本 + 修复清单
14
+ version 打印版本
15
+
16
+ 退出码(与元安 / 元审 / 元盾家族一致):
17
+ 0 = 成功(评分低于阈值,或未启用 --gate)
18
+ 1 = score --gate 且评分 >= 阈值(检测到明显 AI 腔)
19
+ 4 = 用法错误 / 致命异常
20
+
21
+ 用法示例:
22
+ python3 yotta_humanize.py score -f article.md
23
+ python3 yotta_humanize.py analyze -f article.md
24
+ python3 yotta_humanize.py report -f article.md > report.md
25
+ python3 yotta_humanize.py suggest -f article.md
26
+ python3 yotta_humanize.py rewrite -f article.md
27
+ echo "..." | python3 yotta_humanize.py score --stdin
28
+ """
29
+ import argparse
30
+ import json
31
+ import locale
32
+ import math
33
+ import re
34
+ import sys
35
+ from pathlib import Path
36
+
37
+ try:
38
+ sys.stdout.reconfigure(encoding="utf-8")
39
+ except Exception:
40
+ pass
41
+ try:
42
+ sys.stderr.reconfigure(encoding="utf-8")
43
+ except Exception:
44
+ pass
45
+
46
+ _HERE = Path(__file__).resolve().parent
47
+ sys.path.insert(0, str(_HERE))
48
+ import humanize_rules as HR # noqa: E402
49
+
50
+ VERSION = "0.1.0"
51
+ TOOL_NAME = "yotta-humanize"
52
+ TOOL_CN = "元真"
53
+ DEFAULT_THRESHOLD = 45
54
+
55
+ # ── 中文文本统计 ────────────────────────────────────────────────────────────
56
+
57
+ _CJK = re.compile(r"[\u4e00-\u9fff\u3400-\u4dbf]")
58
+ _ALNUM = re.compile(r"[0-9a-zA-Z]")
59
+ _SENT_SPLIT = re.compile(r"[。!?;!?;\n]+")
60
+ _INTENSIFIERS = re.compile(r"非常|十分|极其|特别|相当|无比|异常|格外")
61
+
62
+
63
+ def cjk_len(s):
64
+ """CJK 字数 + 数字/英文单词数(标点与空格不计)。"""
65
+ n = len(_CJK.findall(s))
66
+ n += len(re.findall(r"[0-9a-zA-Z]+", s))
67
+ return n
68
+
69
+
70
+ def split_sentences(text):
71
+ """按中文句末标点切句,返回(句子列表,分隔符列表)。"""
72
+ parts = _SENT_SPLIT.split(text)
73
+ seps = _SENT_SPLIT.findall(text)
74
+ sentences = []
75
+ for p in parts:
76
+ p = p.strip()
77
+ if p:
78
+ sentences.append(p)
79
+ return sentences, seps
80
+
81
+
82
+ def bigrams(text):
83
+ """把连续 CJK 字符切成双字单元,作为词汇多样性统计的伪词。"""
84
+ runs = re.findall(r"[\u4e00-\u9fff\u3400-\u4dbf]+", text)
85
+ out = []
86
+ for run in runs:
87
+ for i in range(len(run) - 1):
88
+ out.append(run[i:i + 2])
89
+ return out
90
+
91
+
92
+ def compute_stats(text):
93
+ """计算中文风格统计量(突发性 / 句长均匀度 / 词汇多样性)。"""
94
+ sentences, _ = split_sentences(text)
95
+ lengths = [cjk_len(s) for s in sentences if cjk_len(s) > 0]
96
+ n_sent = len(lengths)
97
+ chars = cjk_len(text)
98
+ toks = bigrams(text)
99
+ n_tok = len(toks)
100
+ uniq = len(set(toks))
101
+ ttr = (uniq / n_tok) if n_tok else 0.0
102
+
103
+ mean = 0.0
104
+ std = 0.0
105
+ cv = 0.0
106
+ burst = 0.0
107
+ if n_sent > 1:
108
+ mean = sum(lengths) / n_sent
109
+ var = sum((x - mean) ** 2 for x in lengths) / n_sent
110
+ std = math.sqrt(var)
111
+ cv = std / mean if mean else 0.0
112
+ diffs = [abs(lengths[i] - lengths[i - 1]) for i in range(1, n_sent)]
113
+ burst = (sum(diffs) / len(diffs)) / mean if mean else 0.0
114
+
115
+ # 逗号密度(每 100 字的逗号数,过高说明一逗到底)
116
+ comma_density = (text.count(",") + text.count(",")) / chars * 100 if chars else 0.0
117
+ # 连接词密度(每 100 字)
118
+ conj = sum(len(re.findall(w, text)) for w in
119
+ ("然而", "但是", "同时", "此外", "另外", "更重要的是"))
120
+ conj_density = conj / chars * 100 if chars else 0.0
121
+
122
+ return {
123
+ "char_count": chars,
124
+ "sentence_count": n_sent,
125
+ "avg_sentence_len": round(mean, 2),
126
+ "sentence_std": round(std, 2),
127
+ "cv": round(cv, 4),
128
+ "burstiness": round(burst, 4),
129
+ "type_token_ratio": round(ttr, 4),
130
+ "comma_density": round(comma_density, 2),
131
+ "conjunction_density": round(conj_density, 2),
132
+ }
133
+
134
+
135
+ def _clamp01(x):
136
+ return max(0.0, min(1.0, x))
137
+
138
+
139
+ def stats_score(stats):
140
+ """统计量 → 0-100 的 AI 腔分数(句长越均匀 / 突发性越低 / 词汇越贫乏越像 AI)。"""
141
+ if stats["char_count"] < 40 or stats["sentence_count"] < 3:
142
+ return 0
143
+ cv = stats["cv"]
144
+ burst = stats["burstiness"]
145
+ ttr = stats["type_token_ratio"]
146
+ u = _clamp01((0.5 - cv) / 0.35) # cv 高 = 句长变化大 = 人味 → 分低
147
+ b = _clamp01((0.6 - burst) / 0.5) # 突发性低 = 均匀 → 分高
148
+ t = _clamp01((0.6 - ttr) / 0.35) # 词汇多样性低 → 分高
149
+ return round(100 * (0.4 * u + 0.35 * b + 0.25 * t))
150
+
151
+
152
+ # ── 规则检测 ────────────────────────────────────────────────────────────────
153
+
154
+ def find_all(text, pattern):
155
+ """返回 pattern 在 text 中所有命中(match, start, end)。"""
156
+ out = []
157
+ for m in pattern.finditer(text):
158
+ out.append((m.group(0), m.start(), m.end()))
159
+ return out
160
+
161
+
162
+ def detect(text, only_ids=None, config=None):
163
+ """运行全部(或指定)规则,返回 findings 列表。"""
164
+ comp = HR.compiled()
165
+ findings = []
166
+ for rule in HR.RULES:
167
+ rid = rule["id"]
168
+ if only_ids is not None and rid not in only_ids:
169
+ continue
170
+ pats = comp[rid]
171
+ seen = set()
172
+ hits = []
173
+ for p in pats:
174
+ for match, start, end in find_all(text, p):
175
+ key = (match, start)
176
+ if key in seen:
177
+ continue
178
+ seen.add(key)
179
+ hits.append({"match": match, "start": start, "end": end})
180
+ if not hits:
181
+ continue
182
+ findings.append({
183
+ "rule_id": rid,
184
+ "name": rule["name"],
185
+ "group": rule["group"],
186
+ "weight": rule["weight"],
187
+ "matches": hits,
188
+ "suggestion": rule["suggestion"],
189
+ })
190
+ return findings
191
+
192
+
193
+ def pattern_score(findings, char_count):
194
+ """规则命中 → 0-100(密度 + 种类广度 + 分组多样性)。"""
195
+ if not findings or char_count == 0:
196
+ return 0
197
+ weighted = sum(f["weight"] * len(f["matches"]) for f in findings)
198
+ density = weighted / char_count * 100
199
+ density_score = min(math.log2(density + 1) * 14, 60)
200
+ breadth = min(len(findings) * 2, 20)
201
+ groups = len(set(f["group"] for f in findings))
202
+ cat = min(groups * 4, 16)
203
+ return min(round(density_score + breadth + cat), 100)
204
+
205
+
206
+ def analyze(text, opts=None):
207
+ """完整分析:{score, pattern_score, stats_score, total_matches, ...}。"""
208
+ opts = opts or {}
209
+ only_ids = opts.get("patterns")
210
+ stats = compute_stats(text)
211
+ findings = detect(text, only_ids=only_ids)
212
+ total_matches = sum(len(f["matches"]) for f in findings)
213
+ pscore = pattern_score(findings, stats["char_count"])
214
+ sscore = stats_score(stats)
215
+ if pscore == 0 and sscore == 0:
216
+ score = 0
217
+ elif not findings:
218
+ score = min(round(sscore * 0.12), 12)
219
+ else:
220
+ score = min(round(pscore * 0.7 + sscore * 0.3), 100)
221
+
222
+ categories = {}
223
+ for g in HR.GROUPS:
224
+ fs = [f for f in findings if f["group"] == g]
225
+ categories[g] = {
226
+ "label": HR.GROUP_LABELS[g],
227
+ "matches": sum(len(f["matches"]) for f in fs),
228
+ "rules": [f["name"] for f in fs],
229
+ }
230
+ return {
231
+ "score": score,
232
+ "pattern_score": pscore,
233
+ "stats_score": sscore,
234
+ "total_matches": total_matches,
235
+ "rule_types": len(findings),
236
+ "stats": stats,
237
+ "categories": categories,
238
+ "findings": findings,
239
+ "summary": build_summary(score, total_matches, findings, stats),
240
+ }
241
+
242
+
243
+ def build_summary(score, total_matches, findings, stats):
244
+ level = "明显 AI 腔" if score >= 70 else (
245
+ "中度 AI 腔" if score >= 45 else (
246
+ "轻度 AI 痕迹" if score >= 20 else "基本像人写的"))
247
+ top = sorted(findings, key=lambda f: f["weight"] * len(f["matches"]),
248
+ reverse=True)[:3]
249
+ top_names = "、".join(f["name"] for f in top)
250
+ msg = "评分 %d/100(%s):命中 %d 处、%d 类规则。" % (
251
+ score, level, total_matches, len(findings))
252
+ if top_names:
253
+ msg += " 主要问题:" + top_names + "。"
254
+ if stats["sentence_count"] > 3 and stats["cv"] < 0.25:
255
+ msg += " 句长过于均匀,节奏像机器。"
256
+ if stats["char_count"] > 100 and stats["type_token_ratio"] < 0.4:
257
+ msg += " 用词重复度偏高。"
258
+ return msg
259
+
260
+
261
+ # ── 确定性改写 ─────────────────────────────────────────────────────────────
262
+
263
+ def _apply_fix(text, spec):
264
+ """spec 形如 'pattern|replacement'(pattern 视为正则)。返回 (new_text, count)。"""
265
+ if "|" not in spec:
266
+ return text, 0
267
+ pat, rep = spec.split("|", 1)
268
+ if not pat:
269
+ return text, 0
270
+ new, n = re.subn(pat, rep, text)
271
+ return new, n
272
+
273
+
274
+ def rewrite(text):
275
+ """按规则 fix 映射做确定性机械改写,返回 {text, fixes, before, after}。"""
276
+ out = text
277
+ fixes = []
278
+ before = analyze(text)
279
+ for rule in HR.RULES:
280
+ for spec in rule.get("fix", []):
281
+ new, n = _apply_fix(out, spec)
282
+ if n > 0:
283
+ fixes.append({
284
+ "rule_id": rule["id"],
285
+ "name": rule["name"],
286
+ "count": n,
287
+ "detail": spec.split("|")[0],
288
+ })
289
+ out = new
290
+
291
+ # 结构类修复(依赖计数,需单独处理)
292
+ out, n1 = _limit_dash(out)
293
+ if n1:
294
+ fixes.append({"rule_id": "HZ-16", "name": "破折号滥用",
295
+ "count": n1, "detail": "多余破折号→逗号"})
296
+ out, n2 = _limit_bang(out)
297
+ if n2:
298
+ fixes.append({"rule_id": "HZ-18", "name": "感叹号滥用",
299
+ "count": n2, "detail": "多余感叹号→句号"})
300
+ out, n3 = _trim_blank(out)
301
+ if n3:
302
+ fixes.append({"rule_id": "HZ-22", "name": "废话填充",
303
+ "count": n3, "detail": "清理空壳标点"})
304
+ after = analyze(out)
305
+ return {
306
+ "text": out,
307
+ "fixes": fixes,
308
+ "before": before["score"],
309
+ "after": after["score"],
310
+ }
311
+
312
+
313
+ def _limit_dash(text):
314
+ """破折号超过 2 处时,把第 3 处起换成逗号。"""
315
+ count = text.count("——")
316
+ if count <= 2:
317
+ return text, 0
318
+ if count <= 2:
319
+ return text, 0
320
+ out = []
321
+ seen = 0
322
+ i = 0
323
+ n = 0
324
+ while i < len(text):
325
+ if text.startswith("——", i):
326
+ if seen >= 2:
327
+ out.append(",")
328
+ n += 1
329
+ else:
330
+ out.append("——")
331
+ seen += 1
332
+ i += 2
333
+ else:
334
+ out.append(text[i])
335
+ i += 1
336
+ return "".join(out), n
337
+
338
+
339
+ def _limit_bang(text):
340
+ """感叹号超过 2 个时,第 3 个起换成句号。"""
341
+ if text.count("!") <= 2:
342
+ return text, 0
343
+ out = []
344
+ seen = 0
345
+ n = 0
346
+ for ch in text:
347
+ if ch == "!":
348
+ if seen >= 2:
349
+ out.append("。")
350
+ n += 1
351
+ else:
352
+ out.append(ch)
353
+ seen += 1
354
+ else:
355
+ out.append(ch)
356
+ return "".join(out), n
357
+
358
+
359
+ def _trim_blank(text):
360
+ """清理改写后产生的空壳标点(如“,,”“。。”“,。”)。"""
361
+ before = text
362
+ text = re.sub(r"[,,]{2,}", ",", text)
363
+ text = re.sub(r"[。]{2,}", "。", text)
364
+ text = re.sub(r"[,,][。]", "。", text)
365
+ text = re.sub(r"^[,,。;;\s]+", "", text)
366
+ return text, (0 if before == text else 1)
367
+
368
+
369
+ # ── 报告格式化 ─────────────────────────────────────────────────────────────
370
+
371
+ def _snippet(text, start, end, radius=14):
372
+ lo = max(0, start - radius)
373
+ hi = min(len(text), end + radius)
374
+ return text[lo:hi].replace("\n", " ")
375
+
376
+
377
+ def score_label(s):
378
+ if s >= 70:
379
+ return "明显 AI 腔"
380
+ if s >= 45:
381
+ return "中度 AI 腔"
382
+ if s >= 20:
383
+ return "轻度 AI 痕迹"
384
+ return "基本像人写的"
385
+
386
+
387
+ def format_analyze(text, result):
388
+ lines = []
389
+ s = result
390
+ lines.append("")
391
+ lines.append("=" * 46)
392
+ lines.append(" 元真(yotta-humanize)AI 腔检测报告")
393
+ lines.append("=" * 46)
394
+ bar = "█" * round(s["score"] / 5) + "░" * (20 - round(s["score"] / 5))
395
+ lines.append(" 评分:%d/100 [%s](%s)" % (s["score"], bar, score_label(s["score"])))
396
+ lines.append(" 字数 %d | 句子 %d | 命中 %d 处 / %d 类规则"
397
+ % (s["stats"]["char_count"], s["stats"]["sentence_count"],
398
+ s["total_matches"], s["rule_types"]))
399
+ lines.append(" 规则分 %d | 统计分 %d" % (s["pattern_score"], s["stats_score"]))
400
+ lines.append(" %s" % s["summary"])
401
+ lines.append("")
402
+ st = s["stats"]
403
+ lines.append("── 文本统计 ───────────────────────────────────────")
404
+ lines.append(" 平均句长 %s 字 | 句长波动 σ=%s(CV %s)"
405
+ % (st["avg_sentence_len"], st["sentence_std"], st["cv"]))
406
+ lines.append(" 突发性 %s | 词汇多样性 %s | 逗号密度 %s/百字"
407
+ % (st["burstiness"], st["type_token_ratio"], st["comma_density"]))
408
+ lines.append("")
409
+ if s["findings"]:
410
+ lines.append("── 命中规则 ──────────────────────────────────────")
411
+ for f in s["findings"]:
412
+ lines.append("")
413
+ lines.append(" [%s] %s(×%d,权重 %d)"
414
+ % (f["rule_id"], f["name"], len(f["matches"]), f["weight"]))
415
+ for m in f["matches"][:5]:
416
+ lines.append(" · %s" % _snippet(text, m["start"], m["end"]))
417
+ if len(f["matches"]) > 5:
418
+ lines.append(" … 还有 %d 处" % (len(f["matches"]) - 5))
419
+ lines.append(" 建议:%s" % f["suggestion"])
420
+ lines.append("")
421
+ lines.append("=" * 46)
422
+ return "\n".join(lines)
423
+
424
+
425
+ def format_report(result):
426
+ lines = []
427
+ s = result
428
+ lines.append("# AI 腔检测报告")
429
+ lines.append("")
430
+ lines.append("**评分:%d/100**(%s)" % (s["score"], score_label(s["score"])))
431
+ lines.append("")
432
+ lines.append("字数 %d | 句子 %d | 命中 %d 处 / %d 类规则 | 规则分 %d | 统计分 %d"
433
+ % (s["stats"]["char_count"], s["stats"]["sentence_count"],
434
+ s["total_matches"], s["rule_types"],
435
+ s["pattern_score"], s["stats_score"]))
436
+ lines.append("")
437
+ lines.append(s["summary"])
438
+ lines.append("")
439
+ st = s["stats"]
440
+ lines.append("## 文本统计")
441
+ lines.append("")
442
+ lines.append("| 指标 | 数值 | 说明 |")
443
+ lines.append("|---|---|---|")
444
+ lines.append("| 平均句长 | %s 字 | %s |" % (
445
+ st["avg_sentence_len"],
446
+ "偏长" if st["avg_sentence_len"] > 30 else (
447
+ "偏短" if st["avg_sentence_len"] < 10 else "适中")))
448
+ lines.append("| 句长波动 CV | %s | %s |" % (
449
+ st["cv"], "均匀(AI 味)" if st["cv"] < 0.25 else (
450
+ "有起伏(人味)" if st["cv"] >= 0.45 else "中等")))
451
+ lines.append("| 突发性 | %s | %s |" % (
452
+ st["burstiness"],
453
+ "低(节奏单调)" if st["burstiness"] < 0.3 else (
454
+ "高(有起伏)" if st["burstiness"] >= 0.6 else "中等")))
455
+ lines.append("| 词汇多样性 | %s | %s |" % (
456
+ st["type_token_ratio"],
457
+ "低(用词重复)" if st["type_token_ratio"] < 0.4 else (
458
+ "高" if st["type_token_ratio"] >= 0.6 else "中等")))
459
+ lines.append("| 逗号密度 | %s/百字 | %s |" % (
460
+ st["comma_density"],
461
+ "一逗到底" if st["comma_density"] > 12 else "正常"))
462
+ lines.append("")
463
+ if s["findings"]:
464
+ lines.append("## 命中规则")
465
+ lines.append("")
466
+ for f in s["findings"]:
467
+ lines.append("### %s. %s(×%d)" % (f["rule_id"], f["name"], len(f["matches"])))
468
+ lines.append("")
469
+ lines.append(f["suggestion"])
470
+ lines.append("")
471
+ return "\n".join(lines)
472
+
473
+
474
+ def format_suggest(text, result):
475
+ levels = [
476
+ (5, "高优先级(一眼 AI,必须改)"),
477
+ (4, "中高优先级(明显 AI 腔)"),
478
+ (3, "中优先级(有 AI 痕迹)"),
479
+ (2, "低优先级(轻微)"),
480
+ ]
481
+ lines = []
482
+ lines.append("按优先级分组的改写建议:")
483
+ for w, label in levels:
484
+ fs = [f for f in result["findings"] if f["weight"] == w]
485
+ if not fs:
486
+ continue
487
+ lines.append("")
488
+ lines.append("■ %s" % label)
489
+ for f in fs:
490
+ lines.append(" [%s] %s ×%d" % (f["rule_id"], f["name"], len(f["matches"])))
491
+ lines.append(" " + f["suggestion"])
492
+ for m in f["matches"][:3]:
493
+ lines.append(" · %s" % _snippet(text, m["start"], m["end"]))
494
+ if not result["findings"]:
495
+ lines.append(" 未命中规则;如需进一步,可关注统计分与句长节奏。")
496
+ return "\n".join(lines)
497
+
498
+
499
+ # ── CLI ─────────────────────────────────────────────────────────────────────
500
+
501
+ def _read_stdin():
502
+ """读 stdin 原始字节,优先按 UTF-8 解码,失败回退到本机编码。"""
503
+ try:
504
+ data = sys.stdin.buffer.read()
505
+ except AttributeError:
506
+ return sys.stdin.read()
507
+ try:
508
+ return data.decode("utf-8")
509
+ except UnicodeDecodeError:
510
+ enc = locale.getpreferredencoding(False)
511
+ return data.decode(enc, errors="replace")
512
+
513
+
514
+ def _read_text(args):
515
+ if getattr(args, "file", None):
516
+ p = Path(args.file)
517
+ if not p.exists():
518
+ raise SystemExit("文件不存在:%s" % p)
519
+ return p.read_text(encoding="utf-8")
520
+ if getattr(args, "stdin", False):
521
+ return _read_stdin()
522
+ if getattr(args, "text", None):
523
+ return args.text
524
+ if not sys.stdin.isatty():
525
+ return _read_stdin()
526
+ raise SystemExit("缺少输入:用 -f 指定文件、--stdin 读管道,或直接传文本参数。")
527
+
528
+
529
+ def _add_input_args(ap):
530
+ ap.add_argument("-f", "--file", help="输入文件(UTF-8)")
531
+ ap.add_argument("--stdin", action="store_true", help="从 stdin 读取")
532
+ ap.add_argument("text", nargs="*", help="直接传入的文本")
533
+
534
+
535
+ def main(argv=None):
536
+ ap = argparse.ArgumentParser(
537
+ prog=TOOL_NAME,
538
+ description="元真(yotta-humanize):去 AI 味中文写作编辑引擎(24 类规则 + 词表 + 统计突发性)。")
539
+ ap.add_argument("--version", action="version", version="%s %s" % (TOOL_NAME, VERSION))
540
+ sub = ap.add_subparsers(dest="command", required=True)
541
+
542
+ p_ver = sub.add_parser("version", help="打印版本")
543
+ p_ver.set_defaults(cmd_version=True)
544
+
545
+ p_score = sub.add_parser("score", help="输出 AI 腔评分(0-100)")
546
+ _add_input_args(p_score)
547
+ p_score.add_argument("--threshold", type=int, default=DEFAULT_THRESHOLD,
548
+ help="--gate 判定阈值(默认 %d)" % DEFAULT_THRESHOLD)
549
+ p_score.add_argument("--gate", action="store_true",
550
+ help="评分 >= 阈值时退出码 1(可用于 CI 拦截)")
551
+ p_score.add_argument("--json", action="store_true", help="输出 JSON")
552
+
553
+ p_an = sub.add_parser("analyze", help="详细检测报告(文本)")
554
+ _add_input_args(p_an)
555
+ p_an.add_argument("--json", action="store_true", help="输出 JSON")
556
+
557
+ p_rep = sub.add_parser("report", help="Markdown 检测报告")
558
+ _add_input_args(p_rep)
559
+
560
+ p_sug = sub.add_parser("suggest", help="按优先级分组的改写建议")
561
+ _add_input_args(p_sug)
562
+
563
+ p_rw = sub.add_parser("rewrite", help="确定性改写文本")
564
+ _add_input_args(p_rw)
565
+ p_rw.add_argument("--json", action="store_true", help="输出 JSON")
566
+
567
+ args = ap.parse_args(argv)
568
+ try:
569
+ if args.command == "version":
570
+ print("%s %s" % (TOOL_NAME, VERSION))
571
+ return 0
572
+ text = _read_text(args)
573
+ except SystemExit as e:
574
+ print(str(e), file=sys.stderr)
575
+ return 4
576
+
577
+ try:
578
+ if args.command == "score":
579
+ res = analyze(text)
580
+ if args.json:
581
+ print(json.dumps({
582
+ "score": res["score"], "threshold": args.threshold,
583
+ "pattern_score": res["pattern_score"],
584
+ "stats_score": res["stats_score"],
585
+ "total_matches": res["total_matches"],
586
+ "summary": res["summary"],
587
+ }, ensure_ascii=False, indent=2))
588
+ else:
589
+ print(res["score"])
590
+ if args.gate and res["score"] >= args.threshold:
591
+ return 1
592
+ return 0
593
+ if args.command == "analyze":
594
+ res = analyze(text)
595
+ if args.json:
596
+ print(json.dumps(res, ensure_ascii=False, indent=2,
597
+ default=str))
598
+ else:
599
+ print(format_analyze(text, res))
600
+ return 0
601
+ if args.command == "report":
602
+ print(format_report(analyze(text)))
603
+ return 0
604
+ if args.command == "suggest":
605
+ print(format_suggest(text, analyze(text)))
606
+ return 0
607
+ if args.command == "rewrite":
608
+ res = rewrite(text)
609
+ if args.json:
610
+ print(json.dumps(res, ensure_ascii=False, indent=2))
611
+ else:
612
+ print("改写前评分:%d/100 → 改写后评分:%d/100" % (res["before"], res["after"]))
613
+ if res["fixes"]:
614
+ print("修复 %d 处:" % len(res["fixes"]))
615
+ for fx in res["fixes"]:
616
+ print(" [%s] %s ×%d(%s)"
617
+ % (fx["rule_id"], fx["name"], fx["count"], fx["detail"]))
618
+ else:
619
+ print("无需机械改写;如需人工润色,请运行 suggest 查看建议。")
620
+ print("")
621
+ print(res["text"])
622
+ return 0
623
+ except Exception as e: # noqa: BLE001
624
+ print("错误:%s" % e, file=sys.stderr)
625
+ return 4
626
+ return 4
627
+
628
+
629
+ if __name__ == "__main__":
630
+ try:
631
+ sys.exit(main())
632
+ except SystemExit:
633
+ raise