master-skill 0.12.2 → 0.12.3

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (38) hide show
  1. package/.claude-plugin/marketplace.json +1 -1
  2. package/.claude-plugin/plugin.json +1 -1
  3. package/.cursor-plugin/plugin.json +1 -1
  4. package/gemini-extension.json +1 -1
  5. package/package.json +1 -1
  6. package/prebuilt/master-fazang/SKILL.md +6 -3
  7. package/prebuilt/master-fazang/meta.json +5 -0
  8. package/prebuilt/master-fazang/references/teaching.md +10 -5
  9. package/prebuilt/master-fazang/sources/INDEX.md +1 -1
  10. package/prebuilt/master-fazang/sources/jinshizi-excerpts.md +12 -8
  11. package/prebuilt/master-fazang/sources/wujiao-zhang-excerpts.md +20 -12
  12. package/prebuilt/master-fazang/tests/fidelity.jsonl +2 -2
  13. package/prebuilt/master-kumarajiva/SKILL.md +3 -0
  14. package/prebuilt/master-kumarajiva/meta.json +5 -0
  15. package/prebuilt/master-kumarajiva/references/teaching.md +4 -2
  16. package/prebuilt/master-kumarajiva/references/voice.md +1 -1
  17. package/prebuilt/master-kumarajiva/sources/zhonglun-excerpts.md +6 -2
  18. package/prebuilt/master-nagarjuna/references/teaching.md +5 -3
  19. package/prebuilt/master-nagarjuna/sources/dazhidulun-excerpts.md +13 -19
  20. package/prebuilt/master-nagarjuna/sources/shizhu-yixing-excerpts.md +5 -5
  21. package/prebuilt/master-nagarjuna/sources/zhonglun-excerpts.md +22 -1
  22. package/prebuilt/master-ouyi/SKILL.md +3 -0
  23. package/prebuilt/master-ouyi/meta.json +5 -0
  24. package/prebuilt/master-ouyi/references/teaching.md +2 -2
  25. package/prebuilt/master-ouyi/sources/jiaoguan-gangzong-excerpts.md +9 -7
  26. package/prebuilt/master-ouyi/sources/mituo-yaojie-excerpts.md +7 -6
  27. package/prebuilt/master-xuanzang/sources/chengweishi-excerpts.md +2 -2
  28. package/prebuilt/master-xuyun/SKILL.md +6 -0
  29. package/prebuilt/master-xuyun/meta.json +17 -1
  30. package/prebuilt/master-xuyun/references/teaching.md +5 -5
  31. package/prebuilt/master-xuyun/sources/lengyanjing-excerpts.md +1 -1
  32. package/prebuilt/master-zhiyi/meta.json +1 -1
  33. package/prebuilt/master-zhiyi/references/teaching.md +3 -3
  34. package/prebuilt/master-zhiyi/references/voice.md +1 -1
  35. package/prebuilt/master-zhiyi/sources/INDEX.md +1 -1
  36. package/prebuilt/master-zhiyi/sources/fahua-xuanyi-excerpts.md +18 -15
  37. package/prebuilt/master-zhiyi/sources/mohezhiguan-excerpts.md +6 -2
  38. package/tools/verify_sources.py +339 -0
@@ -39,7 +39,7 @@
39
39
 
40
40
  原典(节选):
41
41
 
42
- > 大圆镜智相应心品,谓此心品离诸分别,所缘行相微细难知,不妄不愚一切境相,性相清净离诸杂染。
42
+ > 大圆镜智相应心品,谓此心品离诸分别,所缘行相微细难知,不忘不愚一切境相,性相清净离诸杂染。
43
43
 
44
44
  **引用格式:**【《成唯识论》卷十,T31n1585】→ https://fojin.app/texts/44
45
45
 
@@ -56,7 +56,7 @@
56
56
 
57
57
  原典(节选):
58
58
 
59
- > 菩萨于定位,观影唯是心。义想既灭除,审观唯自想。如是住内识,知所取非有。次能取亦无,后触无所得。
59
+ > 菩萨于定位,观影唯是心。义相既灭除,审观唯自想。如是住内心,知所取非有。次能取亦无,后触无所得。
60
60
 
61
61
  **引用格式:**【《成唯识论》卷九,T31n1585】→ https://fojin.app/texts/44
62
62
 
@@ -18,6 +18,12 @@ sources:
18
18
  - title: 大方廣圓覺修多羅了義經
19
19
  cbeta_id: T17n0842
20
20
  fojin_text_id: 64
21
+ - title: 虛雲老和尚開示錄
22
+ teaching_id: Xuyun:Kaishilu
23
+ - title: 虛雲和尚法彙
24
+ teaching_id: Xuyun:Fahui
25
+ - title: 虛雲老和尚年譜
26
+ teaching_id: Xuyun:Nianpu
21
27
  citation_format: "【《{title}》卷{juan},{cbeta_id}】"
22
28
  verified_by: xr843
23
29
  verified_at: 2026-04-06
@@ -5,7 +5,7 @@
5
5
  "version": 1,
6
6
  "claim_policy": "declared_sources_only",
7
7
  "required_for": ["doctrinal_claim", "practice_guidance", "text_interpretation"],
8
- "allowed_source_types": ["cbeta"],
8
+ "allowed_source_types": ["cbeta", "compiled_teaching"],
9
9
  "minimum_claim_coverage": 0.9,
10
10
  "live_retrieval_allowed": true
11
11
  },
@@ -36,6 +36,22 @@
36
36
  "type": "cbeta",
37
37
  "id": "T17n0842",
38
38
  "title": "大方广圆觉修多罗了义经"
39
+ },
40
+ {
41
+ "type": "compiled_teaching",
42
+ "id": "Xuyun:Kaishilu",
43
+ "title": "虚云老和尚开示录(虛雲老和尚開示錄)"
44
+ },
45
+ {
46
+ "type": "compiled_teaching",
47
+ "id": "Xuyun:Fahui",
48
+ "title": "虚云和尚法汇(虛雲和尚法彙)"
49
+ },
50
+ {
51
+ "type": "compiled_teaching",
52
+ "id": "Xuyun:Nianpu",
53
+ "title": "虚云老和尚年谱(虛雲老和尚年譜)",
54
+ "note": "开示录、法汇、年谱均为近代编集,CBETA 与 FoJin 均未收录(2026-09-15 核查),故无经号、无在线链接。"
39
55
  }
40
56
  ],
41
57
  "version": "1.0.0",
@@ -14,31 +14,31 @@
14
14
 
15
15
  参禅之法,在于起疑情。话头者,一念未生之际也。参"念佛是谁",将此疑情抱定不放,不可用意识思量卜度,只管疑去。疑到山穷水尽处,自有桶底脱落之时。
16
16
 
17
- > 出处:【《虚云老和尚开示录》】→ https://fojin.app/texts/65
17
+ > 出处:【《虚云老和尚开示录》】
18
18
 
19
19
  ### 2. 老实修行,不求奇特
20
20
 
21
21
  修行无别巧,只要老老实实,脚踏实地。不可贪求神通感应,不可妄想一朝顿悟。古人修行数十年方有消息,今人坐三日便要开悟,此大病也。
22
22
 
23
- > 出处:【《虚云和尚法汇》】→ https://fojin.app/texts/58
23
+ > 出处:【《虚云和尚法汇》】
24
24
 
25
25
  ### 3. 禅净双修
26
26
 
27
27
  参禅与念佛,本非对立。参禅者直究心源,念佛者亦须一心不乱。禅净双修,互不妨碍,末法时期尤为契机。参禅念佛,殊途同归,皆归一心。
28
28
 
29
- > 出处:【《虚云老和尚开示录》】→ https://fojin.app/texts/65
29
+ > 出处:【《虚云老和尚开示录》】
30
30
 
31
31
  ### 4. 持戒为本
32
32
 
33
33
  修行第一要紧持戒。戒为无上菩提之本。若不持戒而修禅定,犹如煮沙成饭,纵经尘劫,终不能成。出家人以戒为师,在家人亦当守五戒十善。
34
34
 
35
- > 出处:【《虚云和尚法汇》】→ https://fojin.app/texts/65
35
+ > 出处:【《虚云和尚法汇》】
36
36
 
37
37
  ### 5. 发长远心
38
38
 
39
39
  修行须发长远心,不可急于求成。古德云:"修行如钻木取火,未热先止则前功尽弃。"今人之病,在于心浮气躁,坐不住、忍不下。须知道业未成,生死不了,当生惭愧,发精进心,尽形寿不退转。
40
40
 
41
- > 出处:【《虚云老和尚年谱》】→ https://fojin.app/texts/58
41
+ > 出处:【《虚云老和尚年谱》】
42
42
 
43
43
  ## 精通经典
44
44
 
@@ -7,7 +7,7 @@
7
7
 
8
8
  原典(节选):
9
9
 
10
- > 阿难,汝今欲知奢摩他路,愿出生死,今复问汝:即时如来举金色臂,屈五轮指,语阿难言,汝今见不?阿难言见。佛言,汝何所见?阿难言,我见如来举臂屈指为光明拳,耀我心目。佛言,汝将谁见?
10
+ > 阿难,汝今欲知奢摩他路,愿出生死,今复问汝:即时如来举金色臂,屈五轮指,语阿难言,汝今见不?阿难言见。佛言,汝何所见?阿难言,我见如来举臂屈指为光明拳,曜我心目。佛言,汝将谁见?
11
11
 
12
12
  **引用格式:**【《大佛頂首楞嚴經》卷一,T19n0945】→ https://fojin.app/texts/65
13
13
 
@@ -135,7 +135,7 @@
135
135
  },
136
136
  {
137
137
  "keys": ["一心三观", "三谛", "空假中"],
138
- "content": "一空一切空,无假无中而不空,总空观也;一假一切假,无空无中而不假,总假观也;一中一切中,无空无假而不中,总中观也。即中论所说不可思议一心三观。於一念心同时观空假中三谛,非次第而修。",
138
+ "content": "一空一切空,无假中而不空,总空观也。一假一切假,无空中而不假,总假观也。一中一切中,无空假而不中,总中观也。即《中论》所说不可思议一心三观。——於一念心同时观空假中三谛,非次第而修。",
139
139
  "source_ref": "T46n1911#卷五上"
140
140
  }
141
141
  ],
@@ -20,13 +20,13 @@
20
20
 
21
21
  空、假、中三谛非前后次第,非一非异,三谛互具互融。一空一切空,假中皆空;一假一切假,空中皆假;一中一切中,空假皆中。三谛圆融无碍,即是实相。此为天台教观的理论基石。
22
22
 
23
- > 出处:【《法华玄义》卷二下】→ https://fojin.app/texts/52
23
+ > 出处:【《法华玄义》卷二下】→ https://fojin.app/texts/7889
24
24
 
25
25
  ### 3. 五时八教判教体系
26
26
 
27
27
  五时:华严时、阿含时、方等时、般若时、法华涅槃时。八教分化仪四教(顿、渐、秘密、不定)与化法四教(藏、通、别、圆)。以此体系统摄佛陀一代时教,判定诸经深浅先后,法华为纯圆独妙,会三归一。
28
28
 
29
- > 出处:【《法华玄义》卷一上】→ https://fojin.app/texts/52
29
+ > 出处:【《法华玄义》卷一上】→ https://fojin.app/texts/7889
30
30
 
31
31
  ### 4. 止观双修(一心三观)
32
32
 
@@ -46,7 +46,7 @@
46
46
  |------|------|------|
47
47
  | 《妙法莲华经》 | 天台宗根本经典,开权显实,会三归一 | [阅读原文](https://fojin.app/texts/6513) |
48
48
  | 《摩诃止观》 | 天台三大部之一,圆顿止观之集大成 | [阅读原文](https://fojin.app/texts/53) |
49
- | 《妙法莲华经玄义》 | 天台三大部之一,法华经深义之系统阐发 | [阅读原文](https://fojin.app/texts/52) |
49
+ | 《妙法莲华经玄义》 | 天台三大部之一,法华经深义之系统阐发 | [阅读原文](https://fojin.app/texts/7889) |
50
50
  | 《妙法莲华经文句》 | 天台三大部之一,法华经逐句注释 | [阅读原文](https://fojin.app/texts/52) |
51
51
  | 《修习止观坐禅法要》(小止观) | 止观入门之作,初学必读 | [阅读原文](https://fojin.app/texts/8085) |
52
52
  | 《观音玄义》 | 天台五小部之一,性具善恶之重要出处 | [阅读原文](https://fojin.app/texts/7898) |
@@ -92,5 +92,5 @@
92
92
 
93
93
  - "建议阅读《修习止观坐禅法要》(小止观)作为入门 → [FoJin 原文](https://fojin.app/texts/8085)"
94
94
  - "可参考《摩诃止观》深入圆顿止观之理 → [FoJin 原文](https://fojin.app/texts/53)"
95
- - "关于法华经义理,可详阅《法华玄义》 → [FoJin 原文](https://fojin.app/texts/52)"
95
+ - "关于法华经义理,可详阅《法华玄义》 → [FoJin 原文](https://fojin.app/texts/7889)"
96
96
  - "可在 FoJin 词典中查阅相关佛学术语的详解"
@@ -7,7 +7,7 @@
7
7
  | 文件 | 来源经典 | CBETA | FoJin | 覆盖主题 |
8
8
  |---|---|---|---|---|
9
9
  | `mohezhiguan-excerpts.md` | 《摩訶止觀》 | T1911 | [53](https://fojin.app/texts/53) | 一念三千、一心三观、二十五方便、十境十乘、六即佛、圆顿止观 |
10
- | `fahua-xuanyi-excerpts.md` | 《妙法蓮華經玄義》 | T1716 | [52](https://fojin.app/texts/52) | 五重玄义、五时判教、八教、三谛圆融、开权显实、火宅三车 |
10
+ | `fahua-xuanyi-excerpts.md` | 《妙法蓮華經玄義》 | T1716 | [7889](https://fojin.app/texts/7889) | 五重玄义、五时判教、八教、三谛圆融、开权显实、火宅三车 |
11
11
 
12
12
  ## 引用规范
13
13
 
@@ -1,13 +1,13 @@
1
1
  # 《妙法蓮華經玄義》关键片段
2
2
 
3
- > 智顗大师说,弟子灌顶记。CBETA ID: T1716。FoJin: https://fojin.app/texts/52
3
+ > 智顗大师说,弟子灌顶记。CBETA ID: T1716。FoJin: https://fojin.app/texts/7889
4
4
  > 本文件为教学引用用,节选自 CBETA 公开资料。完整经文请访问 FoJin 或 CBETA。
5
5
 
6
6
  ## 五重玄义(卷一上)
7
7
 
8
8
  解经总纲——**释名、辨体、明宗、论用、判教相**。
9
9
 
10
- **引用格式:**【《法華玄義》卷一上,T1716】→ https://fojin.app/texts/52
10
+ **引用格式:**【《法華玄義》卷一上,T1716】→ https://fojin.app/texts/7889
11
11
 
12
12
  **教义要点:**
13
13
  - 释名:解经题
@@ -20,11 +20,13 @@
20
20
 
21
21
  ## 五时判教(卷十上)
22
22
 
23
- 原典要义:
23
+ 要义(整理,非原文):
24
24
 
25
- > 佛一代時教,判為五時:華嚴時、阿含時(鹿苑)、方等時、般若時、法華涅槃時。如乳、酪、生酥、熟酥、醍醐五味次第。
25
+ 五时:华严时、阿含时(鹿苑)、方等时、般若时、法华涅槃时,配乳、酪、生酥、熟酥、醍醐五味。
26
26
 
27
- **引用格式:**【《法華玄義》卷十上,T1716】→ https://fojin.app/texts/52
27
+ **引用格式:**【《法華玄義》卷十上,T1716】→ https://fojin.app/texts/7889
28
+
29
+ 注:上面一句是整理,不是《法华玄义》原文,不可加引号作引文。玄义卷十以五味讨论判教;下列年数是后世天台家的通说。
28
30
 
29
31
  **教义要点:**
30
32
  - **华严时**(21日):顿说大法,如日出先照高山
@@ -41,7 +43,7 @@
41
43
  **化仪四教**(佛说法的方式):顿、渐、秘密、不定
42
44
  **化法四教**(所说内容的深浅):藏、通、别、圆
43
45
 
44
- **引用格式:**【《法華玄義》卷十上,T1716】→ https://fojin.app/texts/52
46
+ **引用格式:**【《法華玄義》卷十上,T1716】→ https://fojin.app/texts/7889
45
47
 
46
48
  **教义要点:**
47
49
  - 化仪如药方(怎么说)
@@ -53,28 +55,29 @@
53
55
 
54
56
  ---
55
57
 
56
- ## 三谛圆融(卷二下)
58
+ ## 三谛圆融(卷二)
57
59
 
58
60
  原典(节选):
59
61
 
60
- > 即空即假即中,一空一切空,無假無中而不空;一假一切假,無空無中而不假;一中一切中,無空無假而不中。三諦圓融,不縱不橫。
62
+ > 圓三諦者,非但中道具足佛法,真、俗亦然。三諦圓融,一三、三一,如《止觀》中說云云。
61
63
 
62
- **引用格式:**【《法華玄義》卷二下,T1716】→ https://fojin.app/texts/52
64
+ **引用格式:**【《法華玄義》卷二,T1716】→ https://fojin.app/texts/7889
63
65
 
64
66
  **教义要点:**
65
67
  - 空、假、中三谛非前后次第
66
- - 三谛互具互融、一即一切、一切即一
68
+ - 三谛互具互融,一三、三一
69
+ - 玄义此处指向《摩诃止观》详说;「一空一切空……」一段见 `mohezhiguan-excerpts.md` §一心三观
67
70
  - 为天台教观的理论基石
68
71
 
69
72
  ---
70
73
 
71
- ## 开权显实(卷一上)
74
+ ## 开权显实(卷一)
72
75
 
73
- 原典要义:
76
+ 原典(节选):
74
77
 
75
- > 《法華》為眾經之王,開權顯實,會三歸一。
78
+ > 九、開權顯實者,一切諸法莫不皆妙,一色一香無非中道,眾生情隔於妙耳。大悲順物,不與世諍,是故明諸權、實不同。
76
79
 
77
- **引用格式:**【《法華玄義》卷一上,T1716】→ https://fojin.app/texts/52
80
+ **引用格式:**【《法華玄義》卷一,T1716】→ https://fojin.app/texts/7889
78
81
 
79
82
  **教义要点:**
80
83
  - "权":方便权教(三乘、诸经权说)
@@ -87,7 +90,7 @@
87
90
 
88
91
  ## 火宅三车(引法华经譬喻品)
89
92
 
90
- **引用格式:**【《法華玄義》卷八下引《法華經》卷二,T1716/T0262】→ https://fojin.app/texts/52
93
+ **引用格式:**【《法華玄義》卷八下引《法華經》卷二,T1716/T0262】→ https://fojin.app/texts/7889
91
94
 
92
95
  **教义要点:**
93
96
  - 长者以羊车(声闻)、鹿车(缘觉)、牛车(菩萨)诱子出火宅
@@ -23,7 +23,7 @@
23
23
 
24
24
  原典(节选):
25
25
 
26
- > 一空一切空,无假无中而不空,总空观也;一假一切假,无空无中而不假,总假观也;一中一切中,无空无假而不中,总中观也。即中论所说不可思议一心三观。
26
+ > 一空一切空,无假中而不空,总空观也。一假一切假,无空中而不假,总假观也。一中一切中,无空假而不中,总中观也。即《中论》所说不可思议一心三观。
27
27
 
28
28
  **引用格式:**【《摩訶止觀》卷五上,T1911】→ https://fojin.app/texts/53
29
29
 
@@ -99,10 +99,14 @@
99
99
 
100
100
  原典(节选):
101
101
 
102
- > 止觀明靜,前代未聞。智者大師,承南岳之教,在瓦官寺說圓頓止觀。功在漸次,證在圓融。
102
+ > 止观明静,前代未闻。智者,大隋开皇十四年四月二十六日,于荆州玉泉寺,一夏敷扬、二时慈霔……
103
+ >
104
+ > 「圆顿」者,初缘实相,造境即中,无不真实。系缘法界,一念法界,一色一香无非中道。己界及佛界、众生界亦然。……法性寂然名「止」,寂而常照名「观」。虽言初后,无二无别,是名「圆顿止观」。
103
105
 
104
106
  **引用格式:**【《摩訶止觀》卷一上,T1911】→ https://fojin.app/texts/53
105
107
 
108
+ 注:第一段是灌顶序,记智者大师于隋开皇十四年(594)在荆州玉泉寺讲出本书。
109
+
106
110
  **教义要点:**
107
111
  - "圆顿"区别于渐次、不定三种止观
108
112
  - 不历次第,初心即观实相
@@ -505,6 +505,301 @@ def classify_frontmatter_fojin_ids(
505
505
  return mismatched, sorted(unknown)
506
506
 
507
507
 
508
+ _DOC_CITATION = re.compile(r"【([^】]*)】")
509
+ _DOC_FOJIN_LINK = re.compile(r"https://fojin\.app/texts/([0-9]+)")
510
+ _DOC_CBETA_ID = re.compile(r"(?<![0-9A-Za-z])([TXJ])(?:[0-9]{1,3}n)?(B?[0-9]{3,5})[a-z]?(?![0-9A-Za-z])")
511
+ _DOC_TEMPLATE = re.compile(r"\{|卷N|[A-Za-z][xX]{3,}")
512
+
513
+
514
+ def _cbeta_work(cid: str) -> tuple[str, str] | None:
515
+ """`T33n1716` / `T1716` → ("T", "1716"),经号去零;不是 CBETA 号 → None。"""
516
+ m = _DOC_CBETA_ID.fullmatch(cid.strip())
517
+ if not m:
518
+ return None
519
+ number = m.group(2)
520
+ return m.group(1), ("B" + str(int(number[1:]))) if number.startswith("B") else str(int(number))
521
+
522
+
523
+ def collect_doc_citation_links() -> list[tuple[str, str, list[str], str | None]]:
524
+ """(位置, text_id, 引文里的 CBETA 号, 引文书名):人设文档里每个后面同一行跟着
525
+ FoJin 数字链接的引文块。
526
+
527
+ 只认同一行、且在下一个引文块之前的链接。第一次核查用 120 字符窗口,把
528
+ master-yinguang 一条没有链接的引文和两行之后表格里《佛說阿彌陀經》的链接
529
+ 配成了一对。格式模板(`{title}`、`卷N`、`Wxxxxx`)不是引文,跳过。
530
+ """
531
+ pairs: list[tuple[str, str, list[str], str | None]] = []
532
+ base = Path(PREBUILT_DIR)
533
+ files = sorted(base.glob("*/SKILL.md")) + sorted(base.glob("*/references/*.md")) + sorted(base.glob("*/sources/*.md"))
534
+ for path in files:
535
+ where = path.relative_to(base).as_posix()
536
+ for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
537
+ for block in _DOC_CITATION.finditer(line):
538
+ if _DOC_TEMPLATE.search(block.group(1)):
539
+ continue
540
+ rest = line[block.end():]
541
+ following = _DOC_CITATION.search(rest)
542
+ link = _DOC_FOJIN_LINK.search(rest[: following.start()] if following else rest)
543
+ if not link:
544
+ continue
545
+ ids = [m.group(0) for m in _DOC_CBETA_ID.finditer(block.group(1))]
546
+ title = re.search(r"《([^》]+)》", block.group(1))
547
+ pairs.append((f"{where}:{number}", link.group(1), ids, title.group(1) if title else None))
548
+ return pairs
549
+
550
+
551
+ def classify_doc_citation_links(
552
+ pairs: list[tuple[str, str, list[str], str | None]], records: dict[str, dict | None]
553
+ ) -> tuple[list[tuple[str, str, str]], list[str]]:
554
+ """哪些文档链接打开的不是引文所说的那部书;FoJin 没给出记录的记为未知。
555
+
556
+ 比两样:链接文本的经号是否是引文里的某个号,书名(截掉「·品名」)是否与
557
+ `title_zh` 读音对得上(`titles_agree`)。
558
+ """
559
+ mismatched: list[tuple[str, str, str]] = []
560
+ unknown: list[str] = []
561
+ for where, tid, ids, title in pairs:
562
+ record = records.get(tid)
563
+ if not record:
564
+ unknown.append(where)
565
+ continue
566
+ problems = []
567
+ linked = record.get("cbeta_id")
568
+ wanted = {_cbeta_work(c) for c in ids} - {None}
569
+ if wanted and linked and _cbeta_work(str(linked)) not in wanted:
570
+ problems.append(f"texts/{tid} 是 {linked},不是 {'/'.join(ids)}")
571
+ linked_title = record.get("title_zh")
572
+ if title and linked_title:
573
+ book = re.split(r"[·・‧〈<]", title, maxsplit=1)[0].strip()
574
+ if book and titles_agree(book, str(linked_title)) is False:
575
+ problems.append(f"《{title}》对不上 texts/{tid} 的《{linked_title}》")
576
+ if problems:
577
+ mismatched.append((where, tid, ";".join(problems)))
578
+ return mismatched, unknown
579
+
580
+
581
+ # Step 3f:摘录里的「原典」引文是不是所引那一卷的原文。
582
+ #
583
+ # 前面几步核经号、卷号、题名和链接,都看不见引文本身。2026-09-15 逐句比对
584
+ # `sources/*-excerpts.md` 与 lore_triggers:63 段里 19 段有分句不在所引的那一卷,
585
+ # 其中「宁起有见如须弥山」挂在《大智度论》名下、四法界挂在《五教章》名下、
586
+ # 《摩诃止观》序文写错了讲经的寺名,而人设把这些当原文引给用户。
587
+
588
+ CBETA_JUANS_URL = "https://cbdata.dila.edu.tw/stable/juans"
589
+ # 引文没标卷次时,卷数不超过此数的书整部读;更长的记为未知,请补卷次。
590
+ EXCERPT_WHOLE_WORK_MAX_JUANS = 30
591
+ # 短于此数的分句(「第七」「华严经」)哪里都可能出现,不拿来判对错。
592
+ EXCERPT_MIN_CLAUSE = 4
593
+
594
+ _HAN = re.compile(r"[㐀-鿿豈-﫿]")
595
+ _HAN_RUN = re.compile(r"[㐀-鿿豈-﫿]+")
596
+ _CN_DIGITS = {"〇": 0, "零": 0, "一": 1, "二": 2, "三": 3, "四": 4, "五": 5, "六": 6, "七": 7, "八": 8, "九": 9}
597
+ _READINGS: dict[str, frozenset[str]] = {}
598
+
599
+
600
+ def _chinese_number(text: str) -> int | None:
601
+ """「五」「二十一」「一百零八」「一一二」「31」→ 整数;认不出 → None。"""
602
+ if text.isdigit():
603
+ return int(text)
604
+ if text and all(ch in _CN_DIGITS for ch in text):
605
+ return int("".join(str(_CN_DIGITS[ch]) for ch in text))
606
+ total, digit = 0, 0
607
+ for ch in text:
608
+ if ch in _CN_DIGITS:
609
+ digit = _CN_DIGITS[ch]
610
+ elif ch in "十百":
611
+ total += (digit or 1) * (10 if ch == "十" else 100)
612
+ digit = 0
613
+ else:
614
+ return None
615
+ return total + digit or None
616
+
617
+
618
+ def cited_juan(detail: str) -> int | None:
619
+ """引文里「卷五上」「卷31」「卷5·易行品」的卷次;没写或是区间(卷五至卷十、卷3-4)→ None。"""
620
+ match = re.search(r"卷([〇零一二三四五六七八九十百0-9]+)", detail)
621
+ if not match or re.match(r"[上中下]?\s*(?:至|[-–~~、])", detail[match.end():]):
622
+ return None
623
+ return _chinese_number(match.group(1))
624
+
625
+
626
+ def _cbeta_api_work(cid: str) -> str | None:
627
+ """CBETA API 的 work 参数:`T46n1911` / `T1911` → `T1911`,`J36nB348` → `JB348`。"""
628
+ parts = _cbeta_work(cid)
629
+ if not parts:
630
+ return None
631
+ canon, number = parts
632
+ return canon + (number if number.startswith("B") else number.zfill(4))
633
+
634
+
635
+ def collect_excerpt_quotes() -> list[tuple[str, str, str, int | None]]:
636
+ """(位置, 引文, CBETA 号, 卷次):摘录文件里每个「原典」块,与每条 source_ref
637
+ 是 CBETA 号的 lore_triggers。
638
+
639
+ 块的形状是一行以「原典」开头的标签、若干 `>` 行、再一行带【《书名》卷N,经号】
640
+ 的「引用格式」。标成「要义」之类的整理文字不是引文,不收。lore 条目只取「——」
641
+ 之前的部分:CONTRIBUTING §6 允许在原文后用「——」接一句浅释。
642
+ """
643
+ base = Path(PREBUILT_DIR)
644
+ quotes: list[tuple[str, str, str, int | None]] = []
645
+ for path in sorted(base.glob("*/sources/*-excerpts.md")):
646
+ where = path.relative_to(base).as_posix()
647
+ label_line, lines = 0, []
648
+ for number, line in enumerate(path.read_text(encoding="utf-8").splitlines(), 1):
649
+ if line.startswith("原典"):
650
+ label_line, lines = number, []
651
+ elif not label_line:
652
+ continue
653
+ elif line.startswith(">"):
654
+ lines.append(line[1:].strip())
655
+ elif line.startswith("#"):
656
+ label_line, lines = 0, []
657
+ elif "引用格式" in line:
658
+ citation = _DOC_CITATION.search(line)
659
+ cid = _DOC_CBETA_ID.search(citation.group(1)) if citation else None
660
+ if cid and any(lines):
661
+ quotes.append((f"{where}:{label_line}", "\n".join(lines), cid.group(0), cited_juan(citation.group(1))))
662
+ label_line, lines = 0, []
663
+ for path in sorted(base.glob("*/meta.json")):
664
+ where = path.relative_to(base).as_posix()
665
+ entries = json.loads(path.read_text(encoding="utf-8")).get("lore_triggers") or []
666
+ for index, entry in enumerate(entries):
667
+ cid, _, anchor = str(entry.get("source_ref") or "").partition("#")
668
+ if _cbeta_api_work(cid):
669
+ quote = str(entry.get("content") or "").split("——", 1)[0]
670
+ quotes.append((f"{where}:lore_triggers[{index}]", quote, cid, cited_juan(anchor)))
671
+ return quotes
672
+
673
+
674
+ def cbeta_juan_plain_text(html: str) -> str:
675
+ """`/stable/juans` 返回的 HTML → 正文。
676
+
677
+ 校勘注在末尾的 footnote 区,先切掉:注里的异读(如「含【甲】」)不是这一卷
678
+ 的正文,不能让一段改写的引文靠它对上。
679
+ """
680
+ body = re.split(r"<div[^>]*class=['\"][^'\"]*footnote", html, maxsplit=1)[0]
681
+ return re.sub(r"<[^>]+>", "", body)
682
+
683
+
684
+ def fetch_cbeta_juan_count(work: str) -> int | None:
685
+ """CBETA 记这部书有几卷;问不到记 None(未知,不是不符)。"""
686
+ import urllib.error
687
+ import urllib.parse
688
+ import urllib.request
689
+
690
+ url = f"{CBETA_WORKS_URL}?{urllib.parse.urlencode({'work': work})}"
691
+ try:
692
+ with urllib.request.urlopen(url, timeout=CBETA_TIMEOUT) as resp:
693
+ results = json.loads(resp.read().decode("utf-8")).get("results") or []
694
+ return int(results[0]["juan"]) if results else None
695
+ except (urllib.error.URLError, OSError, ValueError, KeyError, TypeError, AttributeError):
696
+ return None
697
+
698
+
699
+ def fetch_cbeta_juan_text(work: str, juan: int) -> str | None:
700
+ """CBETA 某部某卷的正文;问不到记 None(未知,不是不符)。"""
701
+ import urllib.error
702
+ import urllib.parse
703
+ import urllib.request
704
+
705
+ url = f"{CBETA_JUANS_URL}?{urllib.parse.urlencode({'work': work, 'juan': juan})}"
706
+ try:
707
+ with urllib.request.urlopen(url, timeout=CBETA_TIMEOUT) as resp:
708
+ results = json.loads(resp.read().decode("utf-8")).get("results") or []
709
+ except (urllib.error.URLError, OSError, ValueError, AttributeError):
710
+ return None
711
+ html = "".join(r for r in results if isinstance(r, str))
712
+ return cbeta_juan_plain_text(html) if html else None
713
+
714
+
715
+ def quote_clauses(quote: str) -> list[str]:
716
+ """按标点、省略号切成分句;不足 EXCERPT_MIN_CLAUSE 个汉字的不收。"""
717
+ return [run for run in _HAN_RUN.findall(quote) if len(run) >= EXCERPT_MIN_CLAUSE]
718
+
719
+
720
+ def _readings(text: str) -> list[frozenset[str]]:
721
+ """逐个汉字的全部读音,不看上下文:繁简同音即对得上,与 titles_agree 同理。"""
722
+ from pypinyin import Style, pinyin
723
+
724
+ out: list[frozenset[str]] = []
725
+ for ch in _HAN.findall(text):
726
+ if ch not in _READINGS:
727
+ _READINGS[ch] = frozenset(pinyin(ch, style=Style.NORMAL, heteronym=True)[0])
728
+ out.append(_READINGS[ch])
729
+ return out
730
+
731
+
732
+ def _reading_index(text: str) -> tuple[list[frozenset[str]], dict[str, list[int]]]:
733
+ sequence = _readings(text)
734
+ positions: dict[str, list[int]] = {}
735
+ for i, readings in enumerate(sequence):
736
+ for reading in readings:
737
+ positions.setdefault(reading, []).append(i)
738
+ return sequence, positions
739
+
740
+
741
+ def _clause_found(clause: str, index: tuple[list[frozenset[str]], dict[str, list[int]]]) -> bool:
742
+ wanted = _readings(clause)
743
+ sequence, positions = index
744
+ starts: set[int] = set()
745
+ for reading in wanted[0]:
746
+ starts.update(positions.get(reading, ()))
747
+ return any(
748
+ i + len(wanted) <= len(sequence) and all(w & sequence[i + k] for k, w in enumerate(wanted))
749
+ for i in starts
750
+ )
751
+
752
+
753
+ def excerpt_fascicles(juan: int | None, total: int | None) -> list[int] | None:
754
+ """该读哪几卷。标了卷读那一卷;没标而书不长读整部;否则 None(没法核)。
755
+
756
+ 标的卷超出全书卷数时返回 []:那是卷次写错了,不是没法核。
757
+ """
758
+ if not total:
759
+ return None
760
+ if juan is not None:
761
+ return [juan] if 1 <= juan <= total else []
762
+ return list(range(1, total + 1)) if total <= EXCERPT_WHOLE_WORK_MAX_JUANS else None
763
+
764
+
765
+ def classify_excerpt_quotes(
766
+ quotes: list[tuple[str, str, str, int | None]],
767
+ juan_counts: dict[str, int | None],
768
+ juan_texts: dict[tuple[str, int], str | None],
769
+ ) -> tuple[list[tuple[str, str, list[str]]], list[tuple[str, str]]]:
770
+ """哪些引文有分句不在所引的卷里;没法核的记为未知,不算错。
771
+
772
+ 逐分句找,不要求整段连续:《金师子章》本文在 T45n1880 里与净源注文逐句交错,
773
+ 照抄本文整段是找不到的。按读音比,繁简同音即对得上。已知边界:同音字替换
774
+ 看不出来(「不妄不愚」对得上「不忘不愚」);这道检查抓的是改写、增字、换序
775
+ 与张冠李戴,不是错别字。
776
+ """
777
+ mismatched: list[tuple[str, str, list[str]]] = []
778
+ unknown: list[tuple[str, str]] = []
779
+ indexes: dict[tuple[str, tuple[int, ...]], tuple[list[frozenset[str]], dict[str, list[int]]]] = {}
780
+ for where, quote, cid, juan in quotes:
781
+ work = _cbeta_api_work(cid)
782
+ total = juan_counts.get(work) if work else None
783
+ fascicles = excerpt_fascicles(juan, total)
784
+ if fascicles is None:
785
+ unknown.append((where, f"没标卷次,{work} 共 {total} 卷" if total else f"CBETA 没有返回 {cid} 的卷数"))
786
+ continue
787
+ if not fascicles:
788
+ mismatched.append((where, f"{work} 只有 {total} 卷,引文标的是卷{juan}", []))
789
+ continue
790
+ texts = [juan_texts.get((work, j)) for j in fascicles]
791
+ if any(text is None for text in texts):
792
+ unknown.append((where, f"CBETA 没有返回 {work} 的卷文"))
793
+ continue
794
+ key = (work, tuple(fascicles))
795
+ if key not in indexes:
796
+ indexes[key] = _reading_index("\n".join(texts))
797
+ missing = [clause for clause in quote_clauses(quote) if not _clause_found(clause, indexes[key])]
798
+ if missing:
799
+ mismatched.append((where, f"{work} 卷{juan}" if juan is not None else work, missing))
800
+ return mismatched, unknown
801
+
802
+
508
803
  def verify_ids(bridge, cbeta_map: dict[str, list[str]], titles: dict[str, str]) -> dict[str, dict]:
509
804
  """Verify all CBETA IDs and return {full_cbeta_id: {text_id, short_id, title, ...}}.
510
805
 
@@ -768,6 +1063,48 @@ def _run_legacy_link_verification(*, fix: bool) -> int:
768
1063
  if not fm_mismatched and not fm_unknown:
769
1064
  print(" Every frontmatter fojin_text_id matches what FoJin resolves")
770
1065
 
1066
+ # Step 3e: 人设文档里引文后面的 FoJin 链接打开的是不是那部书(见 collect_doc_citation_links)。
1067
+ print("\n[3e/4] Checking FoJin links after citations in persona docs...")
1068
+ doc_pairs = collect_doc_citation_links()
1069
+ doc_records: dict[str, dict | None] = {}
1070
+ for tid in sorted({tid for _, tid, _, _ in doc_pairs}, key=int):
1071
+ try:
1072
+ record = bridge.get_text(tid)
1073
+ except Exception: # noqa: BLE001 — 查不到是「未知」,不是「不符」
1074
+ record = None
1075
+ doc_records[tid] = record if isinstance(record, dict) and record else None
1076
+ doc_mismatched, doc_unknown = classify_doc_citation_links(doc_pairs, doc_records)
1077
+ for where, tid, problem in doc_mismatched:
1078
+ print(f" [WRONG] {where}: {problem}")
1079
+ if doc_unknown:
1080
+ print(f" FoJin returned nothing for {len(doc_unknown)} doc link(s) — "
1081
+ "unknown, not wrong: " + ", ".join(doc_unknown))
1082
+ if not doc_mismatched and not doc_unknown:
1083
+ print(f" All {len(doc_pairs)} citation links in persona docs open the cited work")
1084
+
1085
+ # Step 3f: 摘录里的「原典」引文是否真在所引那一卷(见 classify_excerpt_quotes)。
1086
+ print("\n[3f/4] Checking excerpt quotations against the cited CBETA fascicle...")
1087
+ quotes = collect_excerpt_quotes()
1088
+ quote_works = sorted({work for work in (_cbeta_api_work(cid) for _, _, cid, _ in quotes) if work})
1089
+ juan_counts = {work: fetch_cbeta_juan_count(work) for work in quote_works}
1090
+ wanted_juans = sorted({
1091
+ (work, fascicle)
1092
+ for _, _, cid, juan in quotes
1093
+ for work in [_cbeta_api_work(cid)]
1094
+ if work
1095
+ for fascicle in (excerpt_fascicles(juan, juan_counts.get(work)) or [])
1096
+ })
1097
+ juan_texts = {key: fetch_cbeta_juan_text(*key) for key in wanted_juans}
1098
+ quote_mismatched, quote_unknown = classify_excerpt_quotes(quotes, juan_counts, juan_texts)
1099
+ for where, cited, missing in quote_mismatched:
1100
+ print(f" [WRONG] {where}: {cited}" + (f" has no 「{'」「'.join(missing)}」" if missing else ""))
1101
+ if quote_unknown:
1102
+ print(f" Could not check {len(quote_unknown)} quotation(s) — unknown, not wrong:")
1103
+ for where, reason in quote_unknown:
1104
+ print(f" {where}: {reason}")
1105
+ if not quote_mismatched and not quote_unknown:
1106
+ print(f" All {len(quotes)} quotations appear clause by clause in the cited fascicle")
1107
+
771
1108
  # Step 4: Update URLs
772
1109
  # Build replacement map: full_cbeta_id -> str(internal_text_id)
773
1110
  id_replacement_map: dict[str, str] = {}
@@ -812,6 +1149,8 @@ def _run_legacy_link_verification(*, fix: bool) -> int:
812
1149
  print(f" CBETA id mismatches: {len(mismatched)}")
813
1150
  print(f" CBETA title mismatches: {len(title_mismatched)}")
814
1151
  print(f" Frontmatter FoJin id mismatches: {len(fm_mismatched)}")
1152
+ print(f" Doc citation links to another work: {len(doc_mismatched)}")
1153
+ print(f" Excerpt quotes not in the cited text: {len(quote_mismatched)}")
815
1154
  if unknown_to_cbeta:
816
1155
  print(f" CBETA unreachable for: {len(unknown_to_cbeta)} (not counted as wrong)")
817
1156
  if dry_run and all_changes: