master-skill 0.12.10 → 0.12.12
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.claude-plugin/marketplace.json +1 -1
- package/.claude-plugin/plugin.json +5 -1
- package/.codex/INSTALL.md +23 -16
- package/.cursor-plugin/plugin.json +1 -1
- package/.opencode/INSTALL.md +29 -16
- package/README.md +2 -2
- package/README_EN.md +2 -2
- package/SKILL.md +2 -2
- package/bin/cli.mjs +96 -0
- package/gemini-extension.json +1 -1
- package/hooks/session_start.py +20 -7
- package/package.json +2 -2
- package/prebuilt/master-ajahn-chah/SKILL.md +4 -4
- package/prebuilt/master-ajahn-chah/sources/sutta-excerpts.md +3 -3
- package/prebuilt/master-atisha/SKILL.md +1 -1
- package/prebuilt/master-buddhaghosa/SKILL.md +1 -1
- package/prebuilt/master-huineng/references/teaching.md +2 -2
- package/prebuilt/master-huineng/references/voice.md +3 -3
- package/prebuilt/master-mahasi-sayadaw/SKILL.md +2 -2
- package/prebuilt/master-mahasi-sayadaw/sources/teachings-excerpts.md +2 -2
- package/prebuilt/master-milarepa/SKILL.md +4 -4
- package/prebuilt/master-ouyi/SKILL.md +1 -1
- package/prebuilt/master-tsongkhapa/SKILL.md +3 -3
- package/prebuilt/master-xuyun/SKILL.md +1 -1
- package/prebuilt/master-yinguang/sources/INDEX.md +1 -1
- package/prebuilt/master-yinguang/sources/yihanbianfu-excerpts.md +6 -5
- package/prebuilt/master-zhiyi/SKILL.md +2 -2
- package/references/workflow-details.md +27 -9
- package/scripts/check-gate-liveness.py +227 -0
- package/scripts/validate-quote-attribution.py +126 -0
- package/scripts/validate-section-references.py +162 -0
- package/tools/compiled-teaching-sources.json +89 -0
- package/tools/master_builder.py +88 -4
- package/tools/verify_sources.py +245 -2
- package/tools/version_manager.py +13 -3
|
@@ -0,0 +1,89 @@
|
|
|
1
|
+
{
|
|
2
|
+
"_comment": "CBETA 不收、但有免费全文可取的编集语录。周检 3i 用它把人设里当原话引的句子拿到原书里逐字找。3h 只能查 CBETA,查不到时对这些祖师一律记『未判定』——2026-09-15 修掉的那批拼接引文(虚云、印光)正是长在这个盲区里。coverage 决定这一步能不能判错:complete 表示该祖师声明的编集语录全都在下面取得到,找不到即伪造;partial 表示还有声明了却取不到全文的,找不到只记未判定,绝不报错。宁可少判,不可错判。",
|
|
3
|
+
"corpora": [
|
|
4
|
+
{
|
|
5
|
+
"master": "master-yinguang",
|
|
6
|
+
"corpus_title": "《印光法师文钞》",
|
|
7
|
+
"coverage": "complete",
|
|
8
|
+
"coverage_reason": "meta.json 声明的编集语录是正编、续编、三编,外加一个只作总名解析用的『Yinguang:Wenchao』(正续三编的合称,非另一部书)。这三编就是《文钞》的全部,且都在殆知阁语料库里。2026-09-16 核验:三个文件在 commit e9e11f1 下分别为 1285781 / 1410227 / 1973347 字节,且 voice.md 现有三条引文逐字命中(正编 2 条、续编 1 条)。故查不到即可判错。",
|
|
9
|
+
"verified_on": "2026-09-16",
|
|
10
|
+
"not_a_separate_book": [
|
|
11
|
+
"Yinguang:Wenchao"
|
|
12
|
+
],
|
|
13
|
+
"not_a_separate_book_reason": "『Yinguang:Wenchao』是正编、续编、三编的合称,meta.json 里留着它只为解析只写总名的引用,并非第四部书。列在这里,coverage=complete 才是可核验的声明而不是一句断言——测试会要求每个声明过的编集来源要么有全文、要么在这里说明为什么不需要。",
|
|
14
|
+
"texts": [
|
|
15
|
+
{
|
|
16
|
+
"id": "Yinguang:WenchaoZhengbian",
|
|
17
|
+
"title": "印光法师文钞正编",
|
|
18
|
+
"url": "https://raw.githubusercontent.com/daizhige-org/daizhigev20/e9e11f19b7e6bd9c1284bbdab71d2a9cb94d637c/佛藏/藏外/印光法师文钞.md",
|
|
19
|
+
"encoding": "utf-8"
|
|
20
|
+
},
|
|
21
|
+
{
|
|
22
|
+
"id": "Yinguang:WenchaoXubian",
|
|
23
|
+
"title": "印光法师文钞续编",
|
|
24
|
+
"url": "https://raw.githubusercontent.com/daizhige-org/daizhigev20/e9e11f19b7e6bd9c1284bbdab71d2a9cb94d637c/佛藏/藏外/印光法师文钞续编.md",
|
|
25
|
+
"encoding": "utf-8"
|
|
26
|
+
},
|
|
27
|
+
{
|
|
28
|
+
"id": "Yinguang:WenchaoSanbian",
|
|
29
|
+
"title": "印光法师文钞三编",
|
|
30
|
+
"url": "https://raw.githubusercontent.com/daizhige-org/daizhigev20/e9e11f19b7e6bd9c1284bbdab71d2a9cb94d637c/佛藏/藏外/印光法师文钞三编.md",
|
|
31
|
+
"encoding": "utf-8"
|
|
32
|
+
}
|
|
33
|
+
],
|
|
34
|
+
"source_note": "殆知阁古代文献 v2.0(github.com/daizhige-org/daizhigev20),默认分支是 data 不是 master——用 master 取会 404,2026-09-16 就这样踩过一次。地址钉在 commit e9e11f1(2026-09-12)上,分支移动不会悄悄换掉被核对的底本;哪天该 commit 取不到,3i 记未判定而不是判错。url 一律写字面字符、不要预先百分号编码:编码由 fetch_compiled_text 统一做,清单里再编一次就成了 %25 开头的双重编码,一取就 404 —— 2026-09-16 首次实跑正是这样让印光三部全部『取不到』的。"
|
|
35
|
+
},
|
|
36
|
+
{
|
|
37
|
+
"master": "master-xuyun",
|
|
38
|
+
"corpus_title": "《虚云和尚法汇》《虚云老和尚年谱》",
|
|
39
|
+
"coverage": "partial",
|
|
40
|
+
"coverage_reason": "meta.json 声明了开示录、法汇、年谱三种,BFNN 上只有岑学吕编的法汇六部与年谱,没有《虚云老和尚开示录》——后者是净慧所编,《虚云和尚全集》相对岑本新增约六十余万字,仅开示就多出 110 余则。也就是说虚云的真引文完全可能出自这里取不到的那部分,所以这部语料只能用来确认,不能用来判错。哪天开示录有了可取的免费全文,再改成 complete。",
|
|
41
|
+
"verified_on": "2026-09-16",
|
|
42
|
+
"texts": [
|
|
43
|
+
{
|
|
44
|
+
"id": "Xuyun:Nianpu",
|
|
45
|
+
"title": "虚云和尚年谱",
|
|
46
|
+
"url": "http://bookgb.bfnn.org/books2/1184.htm",
|
|
47
|
+
"encoding": "gb18030"
|
|
48
|
+
},
|
|
49
|
+
{
|
|
50
|
+
"id": "Xuyun:Fahui",
|
|
51
|
+
"title": "虚云和尚法汇—法语",
|
|
52
|
+
"url": "http://bookgb.bfnn.org/books2/1185.htm",
|
|
53
|
+
"encoding": "gb18030"
|
|
54
|
+
},
|
|
55
|
+
{
|
|
56
|
+
"id": "Xuyun:Fahui",
|
|
57
|
+
"title": "虚云和尚法汇—开示",
|
|
58
|
+
"url": "http://bookgb.bfnn.org/books2/1186.htm",
|
|
59
|
+
"encoding": "gb18030"
|
|
60
|
+
},
|
|
61
|
+
{
|
|
62
|
+
"id": "Xuyun:Fahui",
|
|
63
|
+
"title": "虚云和尚法汇—书问",
|
|
64
|
+
"url": "http://bookgb.bfnn.org/books2/1187.htm",
|
|
65
|
+
"encoding": "gb18030"
|
|
66
|
+
},
|
|
67
|
+
{
|
|
68
|
+
"id": "Xuyun:Fahui",
|
|
69
|
+
"title": "虚云和尚法汇—文记",
|
|
70
|
+
"url": "http://bookgb.bfnn.org/books2/1188.htm",
|
|
71
|
+
"encoding": "gb18030"
|
|
72
|
+
},
|
|
73
|
+
{
|
|
74
|
+
"id": "Xuyun:Fahui",
|
|
75
|
+
"title": "虚云和尚法汇—规约",
|
|
76
|
+
"url": "http://bookgb.bfnn.org/books2/1189.htm",
|
|
77
|
+
"encoding": "gb18030"
|
|
78
|
+
},
|
|
79
|
+
{
|
|
80
|
+
"id": "Xuyun:Fahui",
|
|
81
|
+
"title": "虚云和尚法汇—诗歌偈赞",
|
|
82
|
+
"url": "http://bookgb.bfnn.org/books2/1190.htm",
|
|
83
|
+
"encoding": "gb18030"
|
|
84
|
+
}
|
|
85
|
+
],
|
|
86
|
+
"source_note": "BFNN(bookgb.bfnn.org)转录岑学吕编本,gb18030 编码,默认 Python-urllib 即可取(不像 FoJin 会挡 UA)。2026-09-16 核验:voice.md 现有三条引文逐字命中 1186《法汇—开示》。"
|
|
87
|
+
}
|
|
88
|
+
]
|
|
89
|
+
}
|
package/tools/master_builder.py
CHANGED
|
@@ -309,6 +309,76 @@ def build_from_spec(spec: dict, output_dir: str) -> dict:
|
|
|
309
309
|
}
|
|
310
310
|
|
|
311
311
|
|
|
312
|
+
_SKILL_NAME = re.compile(r"^name:[ \t]*(\S+)[ \t]*$", re.MULTILINE)
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def register_teacher(teacher_dir: str, skills_dir: str) -> dict:
|
|
316
|
+
"""Make a generated persona invocable by linking it into a skills directory.
|
|
317
|
+
|
|
318
|
+
The generator writes to `${CLAUDE_SKILL_DIR}/masters/master-{slug}/`, so that
|
|
319
|
+
`master-skill update` can carry generated personas across runtime updates.
|
|
320
|
+
Claude Code does not look there: it loads `<skills dir>/<name>/SKILL.md` and
|
|
321
|
+
scans no deeper. Measured 2026-09-16 with Claude Code 2.1.273 in an isolated
|
|
322
|
+
config, `/skills` listed a probe skill at `~/.claude/skills/master-control/`
|
|
323
|
+
and not one at `~/.claude/skills/create-master/masters/master-probe/`; a
|
|
324
|
+
directory symlink `~/.claude/skills/master-probe` made it appear in the same
|
|
325
|
+
session, without a restart.
|
|
326
|
+
|
|
327
|
+
An existing entry of the same name is never replaced: a user regenerating a
|
|
328
|
+
prebuilt master (`master-huineng`) must not overwrite the installed one.
|
|
329
|
+
"""
|
|
330
|
+
teacher = Path(teacher_dir).resolve()
|
|
331
|
+
skill_md = teacher / "SKILL.md"
|
|
332
|
+
if not skill_md.is_file():
|
|
333
|
+
raise ValueError(f"{teacher} has no SKILL.md")
|
|
334
|
+
text = skill_md.read_text(encoding="utf-8")
|
|
335
|
+
frontmatter = text.split("---", 2)[1] if text.startswith("---") else ""
|
|
336
|
+
match = _SKILL_NAME.search(frontmatter)
|
|
337
|
+
if not match or match.group(1) != teacher.name:
|
|
338
|
+
raise ValueError(
|
|
339
|
+
f"SKILL.md name must equal the directory name {teacher.name!r} — "
|
|
340
|
+
"Claude Code invokes the skill by that name"
|
|
341
|
+
)
|
|
342
|
+
|
|
343
|
+
skills = Path(skills_dir).expanduser()
|
|
344
|
+
created_skills_dir = not skills.is_dir()
|
|
345
|
+
skills.mkdir(parents=True, exist_ok=True)
|
|
346
|
+
link = skills / teacher.name
|
|
347
|
+
|
|
348
|
+
if link.exists() or link.is_symlink():
|
|
349
|
+
if link.exists() and link.resolve() == teacher:
|
|
350
|
+
return {
|
|
351
|
+
"registered": str(link),
|
|
352
|
+
"target": str(teacher),
|
|
353
|
+
"invoke": f"/{teacher.name}",
|
|
354
|
+
"already_registered": True,
|
|
355
|
+
"restart_required": False,
|
|
356
|
+
}
|
|
357
|
+
raise ValueError(
|
|
358
|
+
f"{link} already exists and is not this persona; not replacing it. "
|
|
359
|
+
f"Rename the generated persona, or remove {link} yourself."
|
|
360
|
+
)
|
|
361
|
+
|
|
362
|
+
try:
|
|
363
|
+
os.symlink(teacher, link, target_is_directory=True)
|
|
364
|
+
except OSError:
|
|
365
|
+
if os.name != "nt":
|
|
366
|
+
raise
|
|
367
|
+
import _winapi # symlinks need a privilege on Windows; junctions do not
|
|
368
|
+
|
|
369
|
+
_winapi.CreateJunction(str(teacher), str(link))
|
|
370
|
+
|
|
371
|
+
return {
|
|
372
|
+
"registered": str(link),
|
|
373
|
+
"target": str(teacher),
|
|
374
|
+
"invoke": f"/{teacher.name}",
|
|
375
|
+
"already_registered": False,
|
|
376
|
+
# Claude Code watches skill directories that existed when the session
|
|
377
|
+
# started; a skills directory created now is seen after a restart.
|
|
378
|
+
"restart_required": created_skills_dir,
|
|
379
|
+
}
|
|
380
|
+
|
|
381
|
+
|
|
312
382
|
def main(argv: list[str] | None = None) -> int:
|
|
313
383
|
parser = argparse.ArgumentParser(
|
|
314
384
|
description="Build a create-master persona from an explicit generation spec"
|
|
@@ -320,15 +390,29 @@ def main(argv: list[str] | None = None) -> int:
|
|
|
320
390
|
action="store_true",
|
|
321
391
|
help="run a deterministic no-network generation smoke",
|
|
322
392
|
)
|
|
323
|
-
|
|
393
|
+
modes.add_argument(
|
|
394
|
+
"--register",
|
|
395
|
+
metavar="TEACHER_DIR",
|
|
396
|
+
help="link a generated persona into --skills-dir so it can be invoked",
|
|
397
|
+
)
|
|
398
|
+
parser.add_argument("--output", help="master output directory (with --spec / --offline-smoke)")
|
|
399
|
+
parser.add_argument(
|
|
400
|
+
"--skills-dir",
|
|
401
|
+
default=os.path.join("~", ".claude", "skills"),
|
|
402
|
+
help="skills directory to register into (with --register; default ~/.claude/skills)",
|
|
403
|
+
)
|
|
324
404
|
args = parser.parse_args(argv)
|
|
405
|
+
if not args.register and not args.output:
|
|
406
|
+
parser.error("--output is required with --spec / --offline-smoke")
|
|
325
407
|
|
|
326
408
|
try:
|
|
327
|
-
if args.
|
|
328
|
-
|
|
409
|
+
if args.register:
|
|
410
|
+
summary = register_teacher(args.register, args.skills_dir)
|
|
411
|
+
elif args.offline_smoke:
|
|
412
|
+
summary = build_from_spec(_offline_smoke_spec(), args.output)
|
|
329
413
|
else:
|
|
330
414
|
spec = json.loads(Path(args.spec).read_text(encoding="utf-8"))
|
|
331
|
-
|
|
415
|
+
summary = build_from_spec(spec, args.output)
|
|
332
416
|
except (OSError, json.JSONDecodeError, ValueError) as exc:
|
|
333
417
|
print(f"ERROR: {exc}", file=sys.stderr)
|
|
334
418
|
return 1
|
package/tools/verify_sources.py
CHANGED
|
@@ -673,6 +673,80 @@ def collect_excerpt_quotes() -> list[tuple[str, str, str, int | None]]:
|
|
|
673
673
|
return quotes
|
|
674
674
|
|
|
675
675
|
|
|
676
|
+
def collect_compiled_excerpt_blocks() -> list[tuple[str, str, str]]:
|
|
677
|
+
"""(位置, 引文, 所引篇名):「原典」块中引用格式指向 CBETA 之外编集语录的那些。
|
|
678
|
+
|
|
679
|
+
`collect_excerpt_quotes` 只收引用格式带 CBETA 经号的块,而《文钞》没有经号,
|
|
680
|
+
于是 master-yinguang 的五个「原典」块对 3f 不可见;它们的 `>` 行又是裸行文、
|
|
681
|
+
不带引号,`collect_persona_quotes` 同样收不到。2026-09-16 核出其中两块是用
|
|
682
|
+
真语拼接的改写,却一直以「原典」示人 —— 没有任何一步检查看得见它们。
|
|
683
|
+
"""
|
|
684
|
+
base = Path(PREBUILT_DIR)
|
|
685
|
+
blocks: list[tuple[str, str, str]] = []
|
|
686
|
+
for path in sorted(base.glob("*/sources/*-excerpts.md")):
|
|
687
|
+
where = path.relative_to(base).as_posix()
|
|
688
|
+
label_line, lines = 0, []
|
|
689
|
+
for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
|
|
690
|
+
if line.startswith("原典"):
|
|
691
|
+
label_line, lines = number, []
|
|
692
|
+
elif not label_line:
|
|
693
|
+
continue
|
|
694
|
+
elif line.startswith(">"):
|
|
695
|
+
lines.append(line[1:].strip())
|
|
696
|
+
elif line.startswith("#"):
|
|
697
|
+
label_line, lines = 0, []
|
|
698
|
+
elif "引用格式" in line:
|
|
699
|
+
citation = _DOC_CITATION.search(line)
|
|
700
|
+
text = citation.group(1) if citation else ""
|
|
701
|
+
if text and not _DOC_CBETA_ID.search(text) and any(lines):
|
|
702
|
+
blocks.append((f"{where}:{label_line}", "\n".join(lines), text))
|
|
703
|
+
label_line, lines = 0, []
|
|
704
|
+
return blocks
|
|
705
|
+
|
|
706
|
+
|
|
707
|
+
def classify_compiled_excerpt_blocks(
|
|
708
|
+
blocks: list[tuple[str, str, str]],
|
|
709
|
+
corpora: dict[str, dict],
|
|
710
|
+
fetch,
|
|
711
|
+
) -> tuple[list[tuple[str, str, str]], list[tuple[str, str]], list[tuple[str, str]]]:
|
|
712
|
+
"""「原典」块是不是所引编集语录的原文;省略号分段,每段都要在原书里找得到。
|
|
713
|
+
|
|
714
|
+
与 3i 判引文行同理:原书取不到一律记未判定,接口不通不是证据。语料不全
|
|
715
|
+
(coverage=partial)也只能确认、不能定罪 —— 判「原书没有」需要读得到全部。
|
|
716
|
+
"""
|
|
717
|
+
mismatched: list[tuple[str, str, str]] = []
|
|
718
|
+
verified: list[tuple[str, str]] = []
|
|
719
|
+
unknown: list[tuple[str, str]] = []
|
|
720
|
+
bodies: dict[str, str | None] = {}
|
|
721
|
+
for where, quote, citation in blocks:
|
|
722
|
+
master = where.split("/", 1)[0]
|
|
723
|
+
corpus = corpora.get(master)
|
|
724
|
+
if not corpus:
|
|
725
|
+
unknown.append((where, f"{master} declares no fetchable corpus"))
|
|
726
|
+
continue
|
|
727
|
+
texts = corpus.get("texts") or []
|
|
728
|
+
for text in texts:
|
|
729
|
+
url = str(text.get("url"))
|
|
730
|
+
if url not in bodies:
|
|
731
|
+
bodies[url] = fetch(url, text.get("encoding") or "utf-8")
|
|
732
|
+
readable = [bodies.get(str(t.get("url"))) for t in texts]
|
|
733
|
+
if not any(body for body in readable):
|
|
734
|
+
unknown.append((where, "could not read the declared full texts"))
|
|
735
|
+
continue
|
|
736
|
+
segments = [s for s in re.split(r"…+", quote) if len(_han_only(s)) >= EXCERPT_MIN_CLAUSE * 2]
|
|
737
|
+
if not segments:
|
|
738
|
+
unknown.append((where, "no segment long enough to search"))
|
|
739
|
+
continue
|
|
740
|
+
absent = [s for s in segments if not any(b and _han_only(s) in b for b in readable)]
|
|
741
|
+
if not absent:
|
|
742
|
+
verified.append((where, citation))
|
|
743
|
+
elif corpus.get("coverage") == "complete":
|
|
744
|
+
mismatched.append((where, absent[0].strip(), citation))
|
|
745
|
+
else:
|
|
746
|
+
unknown.append((where, f"{master}'s free full texts do not cover every declared compilation"))
|
|
747
|
+
return mismatched, verified, unknown
|
|
748
|
+
|
|
749
|
+
|
|
676
750
|
def cbeta_juan_plain_text(html: str) -> str:
|
|
677
751
|
"""`/stable/juans` 返回的 HTML → 正文。
|
|
678
752
|
|
|
@@ -963,11 +1037,35 @@ CBETA_SEARCH_URL = "https://cbdata.dila.edu.tw/stable/search"
|
|
|
963
1037
|
_QUOTE_SAMPLE = re.compile(r'^\s*\d+\.\s*[“"「『]([^”"」』\n]{8,200})')
|
|
964
1038
|
_QUOTE_BLOCK = re.compile(r'^\s*>\s*[“"「『]([^”"」』\n]{8,200})')
|
|
965
1039
|
_QUOTE_SAID = re.compile(r'(?:云|曰|偈云|经云|论云)\s*[::]?\s*[“"「『]([^”"」』\n]{8,200})')
|
|
1040
|
+
# 具名引出 + **冒号**:「佛说:""」「神秀偈:""」「慧能曰:""」「达摩祖师偈:""」。
|
|
1041
|
+
# 冒号是把「引原典」与「人设自己的话」分开的判别式 —— 后者写作「常说"看看那个想要
|
|
1042
|
+
# 解决问题的心"」「先问"为什么想读?"」,一律没有冒号。2026-09-16 量过:这一条能收进
|
|
1043
|
+
# 慧能的风幡、神秀与达摩的偈、阿姜查所引三段巴利经文,而不碰任何一句话术示例。
|
|
1044
|
+
_QUOTE_ATTRIBUTED = re.compile(
|
|
1045
|
+
r"(?:佛|世尊|[㐀-鿿]{2,6}(?:祖师|大师|尊者|菩萨|长老|禅师|居士)?)"
|
|
1046
|
+
r'\s*(?:偈曰|偈云|偈|曰|说)\s*[::]\s*[“"「『]([^”"」』\n]{8,200})'
|
|
1047
|
+
)
|
|
1048
|
+
# 《书名》同行引文,不需要动词:「《金刚经》"一切有为法…"」「闻《金刚经》至"应无所住
|
|
1049
|
+
# 而生其心"」。原先只认「云/曰」,这类引文一条都进不来。
|
|
1050
|
+
_QUOTE_TITLED = re.compile(r'《[^》\n]{2,30}》[^“"「『\n]{0,10}[“"「『]([^”"」』\n]{8,200})')
|
|
966
1051
|
_QUOTE_BOILER = re.compile(
|
|
967
1052
|
r"具格上师|亲近善知识|不可由文字|网络传授|须依止|本平台|不得对个体|面对面访谈"
|
|
968
1053
|
r"|如需深入学习|SuttaCentral|BDRC|fojin"
|
|
1054
|
+
# 书单与指引句:「汉译可参《菩提道灯论》(任杰译)」「《清净道论》汉译:叶均居士
|
|
1055
|
+
# 译本」「…可在 ajahnchah.org 免费下载」。它们写在引号里,却不是谁说过的话,
|
|
1056
|
+
# 送去全文检索只会变成查无此句。2026-09-16 量出 22 条这样的行。
|
|
1057
|
+
r"|可参|可详参|可阅|可查|查阅|译本|出版社|下载|开示全集|不可不读|逐句观照"
|
|
969
1058
|
)
|
|
970
1059
|
_QUOTE_PARAPHRASE = re.compile(r"转述|非原文|主旨|整理|概括|要旨|讲解")
|
|
1060
|
+
# 讲「这句话该不该引、该怎么标」的行,本身不是引文:纠错说明(「常被当作龙树的话
|
|
1061
|
+
# 引用,但《大智度论》中没有」)、禁用示例(「不可用宗喀巴的精确分判作为阿底峡立场」)。
|
|
1062
|
+
# 收了它们,周检会对一条文档已经查明并改正的记录拉响假警报。
|
|
1063
|
+
# 标记必须是关于**引用行为**的成句短语:试过「勿」「不可用」这类泛词,会误伤《坛经》
|
|
1064
|
+
# 「勿使惹尘埃」、罗什「且勿急」、虚云「不可用意识思量卜度」这些真引文(2026-09-16 实测)。
|
|
1065
|
+
# 「中没有」同样要紧跟书名号,否则撞上「心中没有」之类的寻常行文。
|
|
1066
|
+
_QUOTE_META = re.compile(
|
|
1067
|
+
r"常被当作|误传|讹传|应保守表述|不要把|不得加引号|引用规范|disclaimer|》中没有|》中查无"
|
|
1068
|
+
)
|
|
971
1069
|
_QUOTE_HAN = re.compile(r"[\u3400-\u9fff]")
|
|
972
1070
|
|
|
973
1071
|
|
|
@@ -987,9 +1085,20 @@ def collect_persona_quotes() -> list[tuple[str, str, str]]:
|
|
|
987
1085
|
):
|
|
988
1086
|
where_base = path.relative_to(base).as_posix()
|
|
989
1087
|
for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
|
|
990
|
-
if
|
|
1088
|
+
if (
|
|
1089
|
+
"出处" in line
|
|
1090
|
+
or "引用格式" in line
|
|
1091
|
+
or _QUOTE_PARAPHRASE.search(line)
|
|
1092
|
+
or _QUOTE_META.search(line)
|
|
1093
|
+
):
|
|
991
1094
|
continue
|
|
992
|
-
for pattern, needs_title in (
|
|
1095
|
+
for pattern, needs_title in (
|
|
1096
|
+
(_QUOTE_SAMPLE, False),
|
|
1097
|
+
(_QUOTE_BLOCK, False),
|
|
1098
|
+
(_QUOTE_SAID, True),
|
|
1099
|
+
(_QUOTE_ATTRIBUTED, False),
|
|
1100
|
+
(_QUOTE_TITLED, False),
|
|
1101
|
+
):
|
|
993
1102
|
match = pattern.search(line)
|
|
994
1103
|
if not match:
|
|
995
1104
|
continue
|
|
@@ -1101,6 +1210,100 @@ def classify_persona_quotes(
|
|
|
1101
1210
|
return mismatched, unknown
|
|
1102
1211
|
|
|
1103
1212
|
|
|
1213
|
+
COMPILED_SOURCES_FILE = Path(__file__).parent / "compiled-teaching-sources.json"
|
|
1214
|
+
|
|
1215
|
+
|
|
1216
|
+
def compiled_teaching_corpora() -> dict[str, dict]:
|
|
1217
|
+
"""{祖师目录: 该祖师在 CBETA 之外、有免费全文可取的编集语录}。
|
|
1218
|
+
|
|
1219
|
+
3h 只能查 CBETA,所以虚云的《法汇》、印光的《文钞》一律落进「未判定」——
|
|
1220
|
+
2026-09-15 修掉的那批拼接引文正是长在这个盲区里。清单把原书地址登记下来,
|
|
1221
|
+
3i 就能真的进原书逐字找。
|
|
1222
|
+
"""
|
|
1223
|
+
if not COMPILED_SOURCES_FILE.exists():
|
|
1224
|
+
return {}
|
|
1225
|
+
data = json.loads(COMPILED_SOURCES_FILE.read_text(encoding="utf-8"))
|
|
1226
|
+
return {str(entry["master"]): entry for entry in data.get("corpora") or []}
|
|
1227
|
+
|
|
1228
|
+
|
|
1229
|
+
def _han_only(text: str) -> str:
|
|
1230
|
+
"""只留汉字。两边都这么归一,标点和空白的写法差异就不会造成假的对不上。"""
|
|
1231
|
+
return "".join(_QUOTE_HAN.findall(re.sub(r"<[^>]+>", "\n", text)))
|
|
1232
|
+
|
|
1233
|
+
|
|
1234
|
+
def fetch_compiled_text(url: str, encoding: str = "utf-8") -> str | None:
|
|
1235
|
+
"""取一部编集语录的全文,归一成纯汉字;取不到返回 None(未知,不是「原书没有」)。"""
|
|
1236
|
+
import urllib.error
|
|
1237
|
+
import urllib.parse
|
|
1238
|
+
import urllib.request
|
|
1239
|
+
|
|
1240
|
+
# urllib 不像 curl 会自己处理非 ASCII 路径,原样传中文路径会抛 UnicodeEncodeError。
|
|
1241
|
+
split = urllib.parse.urlsplit(url)
|
|
1242
|
+
safe = urllib.parse.urlunsplit(split._replace(path=urllib.parse.quote(split.path)))
|
|
1243
|
+
try:
|
|
1244
|
+
with urllib.request.urlopen(safe, timeout=90) as response:
|
|
1245
|
+
raw = response.read()
|
|
1246
|
+
except (urllib.error.URLError, OSError, ValueError):
|
|
1247
|
+
return None
|
|
1248
|
+
return _han_only(raw.decode(encoding, errors="replace"))
|
|
1249
|
+
|
|
1250
|
+
|
|
1251
|
+
def classify_compiled_teaching_quotes(
|
|
1252
|
+
quotes: list[tuple[str, str, str]],
|
|
1253
|
+
corpora: dict[str, dict],
|
|
1254
|
+
fetch,
|
|
1255
|
+
) -> tuple[list[tuple[str, str, str]], list[tuple[str, str]], list[tuple[str, str]], list[str]]:
|
|
1256
|
+
"""到编集语录原书里逐字找人设当原话引的句子。
|
|
1257
|
+
|
|
1258
|
+
判「原书没有这句」只对 coverage 标 `complete` 的语料成立:正编、续编、三编就是
|
|
1259
|
+
《文钞》的全部,都取得到,找不到即伪造。虚云标 `partial` —— 净慧编的《开示录》
|
|
1260
|
+
比岑学吕的《法汇》多出六十余万字,BFNN 上没有,找不到只说明这一步够不着。
|
|
1261
|
+
任何一部取不到,也一律记未判定:接口不通不是证据。
|
|
1262
|
+
|
|
1263
|
+
第四个返回值是「整部语料一篇都没取到」的祖师 —— 那说明这一步对他什么也没检查。
|
|
1264
|
+
不把它单独报出来,一个取数早就坏掉的 3i 会年复一年地绿着,跟没有这道门禁一样。
|
|
1265
|
+
"""
|
|
1266
|
+
mismatched: list[tuple[str, str, str]] = []
|
|
1267
|
+
verified: list[tuple[str, str]] = []
|
|
1268
|
+
unknown: list[tuple[str, str]] = []
|
|
1269
|
+
bodies: dict[str, str | None] = {}
|
|
1270
|
+
touched: set[str] = set()
|
|
1271
|
+
for where, master, quote in quotes:
|
|
1272
|
+
corpus = corpora.get(master)
|
|
1273
|
+
if not corpus:
|
|
1274
|
+
continue
|
|
1275
|
+
touched.add(master)
|
|
1276
|
+
wanted = _han_only(quote)
|
|
1277
|
+
if len(wanted) < 8:
|
|
1278
|
+
unknown.append((where, "quote too short to search"))
|
|
1279
|
+
continue
|
|
1280
|
+
found_in, unreachable = None, []
|
|
1281
|
+
for text in corpus.get("texts") or []:
|
|
1282
|
+
url = str(text.get("url"))
|
|
1283
|
+
if url not in bodies:
|
|
1284
|
+
bodies[url] = fetch(url, text.get("encoding") or "utf-8")
|
|
1285
|
+
body = bodies[url]
|
|
1286
|
+
if body is None:
|
|
1287
|
+
unreachable.append(str(text.get("title")))
|
|
1288
|
+
elif wanted in body:
|
|
1289
|
+
found_in = str(text.get("title"))
|
|
1290
|
+
break
|
|
1291
|
+
if found_in:
|
|
1292
|
+
verified.append((where, found_in))
|
|
1293
|
+
elif unreachable:
|
|
1294
|
+
unknown.append((where, f"could not read {', '.join(sorted(set(unreachable)))}"))
|
|
1295
|
+
elif corpus.get("coverage") == "complete":
|
|
1296
|
+
mismatched.append((where, quote, str(corpus.get("corpus_title") or master)))
|
|
1297
|
+
else:
|
|
1298
|
+
unknown.append((where, f"{master}'s free full texts do not cover every declared compilation"))
|
|
1299
|
+
unreadable = sorted(
|
|
1300
|
+
str(corpora[master].get("corpus_title") or master)
|
|
1301
|
+
for master in touched
|
|
1302
|
+
if all(bodies.get(str(t.get("url"))) is None for t in corpora[master].get("texts") or [])
|
|
1303
|
+
)
|
|
1304
|
+
return mismatched, verified, unknown, unreadable
|
|
1305
|
+
|
|
1306
|
+
|
|
1104
1307
|
def verify_ids(bridge, cbeta_map: dict[str, list[str]], titles: dict[str, str]) -> dict[str, dict]:
|
|
1105
1308
|
"""Verify all CBETA IDs and return {full_cbeta_id: {text_id, short_id, title, ...}}.
|
|
1106
1309
|
|
|
@@ -1435,6 +1638,43 @@ def _run_legacy_link_verification(*, fix: bool) -> int:
|
|
|
1435
1638
|
if not quote_line_mismatched and not quote_line_unknown:
|
|
1436
1639
|
print(f" All {len(persona_quotes)} quoted lines are in CBETA, in a work the persona declares")
|
|
1437
1640
|
|
|
1641
|
+
# Step 3i: 3h 够不着的那些 —— 祖师自己的语录本就在 CBETA 之外,进原书逐字找
|
|
1642
|
+
# (见 classify_compiled_teaching_quotes)。
|
|
1643
|
+
print("\n[3i/4] Checking quoted lines against compiled teachings CBETA does not hold...")
|
|
1644
|
+
compiled_corpora = compiled_teaching_corpora()
|
|
1645
|
+
compiled_mismatched, compiled_verified, compiled_unknown, compiled_unreadable = (
|
|
1646
|
+
classify_compiled_teaching_quotes(persona_quotes, compiled_corpora, fetch_compiled_text)
|
|
1647
|
+
)
|
|
1648
|
+
for where, quote, corpus_title in compiled_mismatched:
|
|
1649
|
+
print(f" [WRONG] {where}: {corpus_title} has no 「{quote[:40]}」")
|
|
1650
|
+
for title in compiled_unreadable:
|
|
1651
|
+
print(f" [BROKEN] {title}: not one declared full text loaded — this step checked nothing")
|
|
1652
|
+
if compiled_verified:
|
|
1653
|
+
print(f" Verified {len(compiled_verified)} quoted line(s) in the compiled teachings:")
|
|
1654
|
+
for where, title in compiled_verified:
|
|
1655
|
+
print(f" {where}: {title}")
|
|
1656
|
+
if compiled_unknown:
|
|
1657
|
+
print(f" Could not check {len(compiled_unknown)} quoted line(s) — unknown, not wrong:")
|
|
1658
|
+
for where, reason in compiled_unknown:
|
|
1659
|
+
print(f" {where}: {reason}")
|
|
1660
|
+
if not compiled_corpora:
|
|
1661
|
+
print(" No compiled-teaching corpora are declared")
|
|
1662
|
+
|
|
1663
|
+
# 同一步里的第二类:「原典」块本身。3f 只认带 CBETA 经号的引用格式,《文钞》
|
|
1664
|
+
# 没有经号,这些块此前对每一步检查都不可见(见 collect_compiled_excerpt_blocks)。
|
|
1665
|
+
compiled_blocks = collect_compiled_excerpt_blocks()
|
|
1666
|
+
block_mismatched, block_verified, block_unknown = classify_compiled_excerpt_blocks(
|
|
1667
|
+
compiled_blocks, compiled_corpora, fetch_compiled_text
|
|
1668
|
+
)
|
|
1669
|
+
for where, segment, citation in block_mismatched:
|
|
1670
|
+
print(f" [WRONG] {where}: {citation} has no 「{segment[:40]}」")
|
|
1671
|
+
if block_verified:
|
|
1672
|
+
print(f" Verified {len(block_verified)} 「原典」 block(s) word for word in the compiled teachings")
|
|
1673
|
+
if block_unknown:
|
|
1674
|
+
print(f" Could not check {len(block_unknown)} 「原典」 block(s) — unknown, not wrong:")
|
|
1675
|
+
for where, reason in block_unknown:
|
|
1676
|
+
print(f" {where}: {reason}")
|
|
1677
|
+
|
|
1438
1678
|
# Step 4: Update URLs
|
|
1439
1679
|
# Build replacement map: full_cbeta_id -> str(internal_text_id)
|
|
1440
1680
|
id_replacement_map: dict[str, str] = {}
|
|
@@ -1483,6 +1723,9 @@ def _run_legacy_link_verification(*, fix: bool) -> int:
|
|
|
1483
1723
|
print(f" Excerpt quotes not in the cited text: {len(quote_mismatched)}")
|
|
1484
1724
|
print(f" BDRC records that do not match: {len(bdrc_mismatched)}")
|
|
1485
1725
|
print(f" Quoted lines CBETA does not have: {len(quote_line_mismatched)}")
|
|
1726
|
+
print(f" Quoted lines the compiled teachings do not have: {len(compiled_mismatched)}")
|
|
1727
|
+
print(f" Compiled teaching corpora that could not be read: {len(compiled_unreadable)}")
|
|
1728
|
+
print(f" Excerpt blocks the compiled teachings do not have: {len(block_mismatched)}")
|
|
1486
1729
|
if unknown_to_cbeta:
|
|
1487
1730
|
print(f" CBETA unreachable for: {len(unknown_to_cbeta)} (not counted as wrong)")
|
|
1488
1731
|
if dry_run and all_changes:
|
package/tools/version_manager.py
CHANGED
|
@@ -67,7 +67,14 @@ def rollback(teacher_dir: str, target_version: str) -> bool:
|
|
|
67
67
|
|
|
68
68
|
|
|
69
69
|
def cleanup_old_versions(teacher_dir: str) -> int:
|
|
70
|
-
"""Remove
|
|
70
|
+
"""Remove archived versions beyond MAX_VERSIONS, oldest first.
|
|
71
|
+
|
|
72
|
+
Prunes only what `list_versions` reports — directories named `v…` — and never a
|
|
73
|
+
`_before_rollback` copy. The first version selected *every* directory under
|
|
74
|
+
`versions/`, so anything else a maintainer kept there was deleted once the
|
|
75
|
+
archive passed the limit, and the backup `rollback` writes so that a bad
|
|
76
|
+
rollback can be undone was evicted by age like an ordinary version.
|
|
77
|
+
"""
|
|
71
78
|
versions_dir = os.path.join(teacher_dir, "versions")
|
|
72
79
|
if not os.path.exists(versions_dir):
|
|
73
80
|
return 0
|
|
@@ -75,8 +82,11 @@ def cleanup_old_versions(teacher_dir: str) -> int:
|
|
|
75
82
|
entries = []
|
|
76
83
|
for entry in os.listdir(versions_dir):
|
|
77
84
|
entry_path = os.path.join(versions_dir, entry)
|
|
78
|
-
if os.path.isdir(entry_path):
|
|
79
|
-
|
|
85
|
+
if not os.path.isdir(entry_path):
|
|
86
|
+
continue
|
|
87
|
+
if not entry.startswith("v") or entry.endswith("_before_rollback"):
|
|
88
|
+
continue
|
|
89
|
+
entries.append((entry_path, os.path.getmtime(entry_path)))
|
|
80
90
|
|
|
81
91
|
entries.sort(key=lambda x: x[1], reverse=True)
|
|
82
92
|
|