@furongjun1999/dsh-memory 0.4.11 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (580) hide show
  1. package/README.md +552 -465
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +143 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
  15. package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
  16. package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
  17. package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
  18. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  19. package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
  20. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  21. package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
  22. package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
  23. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
  24. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
  25. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
  26. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
  27. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
  28. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
  29. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
  30. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
  31. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
  32. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
  33. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
  34. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
  35. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
  36. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
  37. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
  38. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
  39. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
  40. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  41. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  42. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  43. package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
  44. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  45. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  46. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  47. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  48. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +519 -503
  49. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  50. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  51. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  52. package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
  53. package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
  54. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  55. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  56. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  57. package/docs/mdcg/release_v0.4.11.md +49 -0
  58. package/docs/mdcg/release_v0.4.5.md +55 -55
  59. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  60. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  61. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  62. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  63. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  64. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  65. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  66. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  67. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  68. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  69. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  70. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  71. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  72. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  73. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  74. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  75. package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
  76. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  77. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  78. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  79. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  80. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  81. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  82. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  83. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  84. package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
  85. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  86. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  87. package/dsh/README.md +82 -82
  88. package/dsh/cordis.yml.example +139 -139
  89. package/dsh/update-lingshu.bat +11 -11
  90. package/lib/bridge.d.ts +9 -0
  91. package/lib/bridge.js +35 -0
  92. package/lib/hooks.js +36 -2
  93. package/lib/index.js +7 -1
  94. package/lib/lib/roleplay_web.js +116 -29
  95. package/lib/lib/token_store.d.ts +7 -1
  96. package/lib/lib/token_store.js +12 -3
  97. package/md_cg/__init__.py +7 -7
  98. package/md_cg/audit.py +379 -368
  99. package/md_cg/autonomy.py +287 -287
  100. package/md_cg/backfill.py +1328 -1327
  101. package/md_cg/backfill_bigdomain.py +34 -34
  102. package/md_cg/backfill_bucket_zh.py +35 -0
  103. package/md_cg/bench6_arms.py +410 -410
  104. package/md_cg/bench6_common.py +230 -230
  105. package/md_cg/bench6_competitors.py +212 -212
  106. package/md_cg/bench_axis_domain.py +257 -257
  107. package/md_cg/bench_blind_comp.py +308 -308
  108. package/md_cg/bench_e2e_judge.py +532 -0
  109. package/md_cg/bench_e2e_locomo_qa.py +368 -0
  110. package/md_cg/bench_e2e_qa.py +256 -0
  111. package/md_cg/bench_en_atoms_public.py +230 -230
  112. package/md_cg/bench_governance.py +348 -348
  113. package/md_cg/bench_lme_zh.py +410 -410
  114. package/md_cg/bench_locomo.py +121 -121
  115. package/md_cg/bench_locomo_zh.py +450 -450
  116. package/md_cg/bench_locomo_zh_public.py +147 -147
  117. package/md_cg/bench_longmem.py +112 -112
  118. package/md_cg/bench_membench.py +632 -632
  119. package/md_cg/bench_p0.py +149 -149
  120. package/md_cg/bench_progressive.py +287 -287
  121. package/md_cg/bench_role_views.py +238 -238
  122. package/md_cg/bench_task_ab.py +243 -243
  123. package/md_cg/bench_task_ab_llm.py +408 -408
  124. package/md_cg/bench_unified_en.py +204 -204
  125. package/md_cg/bench_zh_mad.py +601 -601
  126. package/md_cg/blindspot_tickets.py +123 -123
  127. package/md_cg/branches.py +301 -285
  128. package/md_cg/build_postings.py +73 -73
  129. package/md_cg/ccgc.py +1006 -948
  130. package/md_cg/census.py +132 -132
  131. package/md_cg/chain.py +315 -300
  132. package/md_cg/codeindex.py +531 -531
  133. package/md_cg/coldverify.py +292 -292
  134. package/md_cg/comment_gate.py +337 -337
  135. package/md_cg/cond_compose.py +190 -190
  136. package/md_cg/cond_facts.py +154 -154
  137. package/md_cg/cond_template.json +106 -106
  138. package/md_cg/condition_anchor.py +142 -142
  139. package/md_cg/conformance.py +726 -726
  140. package/md_cg/consistency.py +717 -717
  141. package/md_cg/consolidate.py +1537 -1439
  142. package/md_cg/corpus.py +110 -110
  143. package/md_cg/crosscheck.py +1098 -1097
  144. package/md_cg/crypto.py +3 -1
  145. package/md_cg/d_meta.py +310 -310
  146. package/md_cg/datapath.py +78 -18
  147. package/md_cg/docindex.py +473 -473
  148. package/md_cg/eval_common.py +575 -575
  149. package/md_cg/evidence.py +4 -2
  150. package/md_cg/evolution.py +477 -477
  151. package/md_cg/export.py +222 -220
  152. package/md_cg/forgetting.py +581 -581
  153. package/md_cg/fsutil.py +377 -329
  154. package/md_cg/hotcache.py +48 -7
  155. package/md_cg/hyperedge.py +251 -251
  156. package/md_cg/identity.py +390 -390
  157. package/md_cg/insight.py +500 -500
  158. package/md_cg/interop.py +338 -0
  159. package/md_cg/judgment_manifest.py +177 -0
  160. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  161. package/md_cg/lexicon/build_standard_en.py +171 -171
  162. package/md_cg/lexicon/expand_en_zh.py +211 -211
  163. package/md_cg/lifecycle.py +272 -272
  164. package/md_cg/linkref.py +280 -280
  165. package/md_cg/links.py +140 -107
  166. package/md_cg/mcp_server.py +403 -55
  167. package/md_cg/md_whitebox.py +345 -345
  168. package/md_cg/mdcg.py +783 -233
  169. package/md_cg/mdcos.py +500 -70
  170. package/md_cg/metacognition.py +591 -591
  171. package/md_cg/migrate.py +119 -119
  172. package/md_cg/migrate_aeis.py +221 -221
  173. package/md_cg/migrate_roleplay.py +293 -293
  174. package/md_cg/migrate_wisdom_graph.py +360 -360
  175. package/md_cg/mreview/__init__.py +25 -25
  176. package/md_cg/mreview/__main__.py +110 -110
  177. package/md_cg/mreview/bundle.py +178 -178
  178. package/md_cg/mreview/candidates.py +262 -262
  179. package/md_cg/mreview/govern.py +694 -693
  180. package/md_cg/mreview/locate.py +939 -939
  181. package/md_cg/mreview/pipeline.py +728 -728
  182. package/md_cg/mreview/rules/duplication.json +21 -21
  183. package/md_cg/mreview/rules/field_coverage.json +54 -54
  184. package/md_cg/mreview/rules/source_license.json +21 -21
  185. package/md_cg/mreview/rules/template_flow.json +21 -21
  186. package/md_cg/mreview/ruleset.py +252 -252
  187. package/md_cg/nodefile.py +575 -575
  188. package/md_cg/pooling.py +484 -472
  189. package/md_cg/postings.py +300 -298
  190. package/md_cg/predict.py +1100 -1100
  191. package/md_cg/progressive.py +123 -123
  192. package/md_cg/protect.py +272 -272
  193. package/md_cg/protocol/md_cg_gate.proto +33 -33
  194. package/md_cg/protocol.py +372 -372
  195. package/md_cg/provenance.py +582 -582
  196. package/md_cg/reach.py +453 -453
  197. package/md_cg/readcache.py +143 -0
  198. package/md_cg/reconcile.py +228 -0
  199. package/md_cg/refindex.py +833 -833
  200. package/md_cg/refine.py +604 -604
  201. package/md_cg/review_cli.py +170 -0
  202. package/md_cg/roleviews.py +89 -89
  203. package/md_cg/routing.py +393 -365
  204. package/md_cg/run_tests.py +211 -0
  205. package/md_cg/scrub.py +13 -3
  206. package/md_cg/security.py +128 -18
  207. package/md_cg/self_state.py +1029 -1029
  208. package/md_cg/selfreport.py +152 -151
  209. package/md_cg/semantic/__init__.py +10 -10
  210. package/md_cg/semantic/canonical.py +122 -122
  211. package/md_cg/semantic/en_normalizer.py +364 -364
  212. package/md_cg/semantic/en_zh_map.json +28694 -0
  213. package/md_cg/semantic/export_en_zh_map.py +64 -0
  214. package/md_cg/semantic/unify.py +45 -0
  215. package/md_cg/semantic/zh_en_atoms.py +139 -139
  216. package/md_cg/signer.py +7 -4
  217. package/md_cg/sources.py +816 -582
  218. package/md_cg/statushdr.py +179 -179
  219. package/md_cg/stg.py +54 -37
  220. package/md_cg/subgraph.py +729 -729
  221. package/md_cg/sustain.py +35 -5
  222. package/md_cg/tasks.py +470 -470
  223. package/md_cg/test_access_hints.py +147 -0
  224. package/md_cg/test_action_derive.py +203 -203
  225. package/md_cg/test_audit_rotate.py +270 -270
  226. package/md_cg/test_autonomy.py +143 -143
  227. package/md_cg/test_bench_governance.py +102 -102
  228. package/md_cg/test_blindspot_tickets.py +166 -166
  229. package/md_cg/test_branch_discard_tombstone.py +136 -0
  230. package/md_cg/test_branches.py +13 -3
  231. package/md_cg/test_ccg_perturb.py +184 -184
  232. package/md_cg/test_ccgc.py +433 -433
  233. package/md_cg/test_census_prune.py +81 -81
  234. package/md_cg/test_chain_read_isolate.py +168 -0
  235. package/md_cg/test_cond_compose_anchors.py +76 -76
  236. package/md_cg/test_cond_match.py +165 -165
  237. package/md_cg/test_condition_anchor.py +81 -81
  238. package/md_cg/test_d_meta.py +412 -412
  239. package/md_cg/test_datapath_device_name.py +203 -0
  240. package/md_cg/test_datapath_root.py +199 -199
  241. package/md_cg/test_emit_negtail_cache.py +156 -0
  242. package/md_cg/test_en_pipeline.py +22 -2
  243. package/md_cg/test_gain_gate.py +212 -212
  244. package/md_cg/test_govern_directread.py +421 -0
  245. package/md_cg/test_health_scale.py +173 -173
  246. package/md_cg/test_hive_ingest.py +285 -0
  247. package/md_cg/test_hot_cold.py +215 -215
  248. package/md_cg/test_hyperedge.py +245 -245
  249. package/md_cg/test_i26_empty_first_write.py +116 -0
  250. package/md_cg/test_i27_e041_identity.py +128 -0
  251. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  252. package/md_cg/test_i32_hotcache_env_key.py +218 -0
  253. package/md_cg/test_identity_attribution.py +96 -15
  254. package/md_cg/test_index_durability.py +17 -3
  255. package/md_cg/test_interop.py +95 -0
  256. package/md_cg/test_interop_judgment.py +228 -0
  257. package/md_cg/test_issue39_utf8_stdio.py +273 -0
  258. package/md_cg/test_lifecycle.py +309 -309
  259. package/md_cg/test_linkref.py +306 -306
  260. package/md_cg/test_links_concurrent_write.py +188 -0
  261. package/md_cg/test_lock.py +43 -43
  262. package/md_cg/test_md_access_parity.py +255 -255
  263. package/md_cg/test_md_writepath.py +345 -345
  264. package/md_cg/test_mdstore_search_parity.py +160 -0
  265. package/md_cg/test_merge_upsert.py +168 -0
  266. package/md_cg/test_mr_m2.py +587 -587
  267. package/md_cg/test_mr_m3.py +710 -710
  268. package/md_cg/test_mr_m4.py +485 -485
  269. package/md_cg/test_n123_derive_expiry_chain.py +205 -0
  270. package/md_cg/test_n130_verify_falsified_protect.py +185 -0
  271. package/md_cg/test_n131_merge_gate.py +205 -0
  272. package/md_cg/test_p0.py +250 -250
  273. package/md_cg/test_p1.py +316 -316
  274. package/md_cg/test_p10_identity.py +173 -173
  275. package/md_cg/test_p11_consistency.py +233 -233
  276. package/md_cg/test_p12_metacognition.py +212 -212
  277. package/md_cg/test_p13_encryption.py +241 -241
  278. package/md_cg/test_p14_sustain.py +249 -249
  279. package/md_cg/test_p15_scrub.py +280 -280
  280. package/md_cg/test_p16_self_state.py +301 -301
  281. package/md_cg/test_p17_predict.py +354 -354
  282. package/md_cg/test_p18_whitebox.py +171 -171
  283. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  284. package/md_cg/test_p1x_ref_root.py +160 -0
  285. package/md_cg/test_p20_evolution.py +315 -315
  286. package/md_cg/test_p21_tokens.py +293 -270
  287. package/md_cg/test_p22_theory.py +175 -175
  288. package/md_cg/test_p23_links.py +311 -311
  289. package/md_cg/test_p24_evidence.py +227 -227
  290. package/md_cg/test_p25_weights.py +156 -156
  291. package/md_cg/test_p26_refindex.py +416 -416
  292. package/md_cg/test_p27_docindex.py +16 -7
  293. package/md_cg/test_p28_refcheck.py +305 -305
  294. package/md_cg/test_p29_session_ingest_export.py +354 -333
  295. package/md_cg/test_p2_mcp.py +3 -0
  296. package/md_cg/test_p3.py +11 -2
  297. package/md_cg/test_p30_maintain.py +330 -330
  298. package/md_cg/test_p31_insight.py +534 -534
  299. package/md_cg/test_p32_backfill.py +7 -1
  300. package/md_cg/test_p33_ccg_wiring.py +293 -293
  301. package/md_cg/test_p34_crosscheck.py +331 -331
  302. package/md_cg/test_p35_conditioned_claim.py +252 -252
  303. package/md_cg/test_p36_kp_align.py +230 -230
  304. package/md_cg/test_p37_condition_space.py +248 -248
  305. package/md_cg/test_p38_concurrent_flush.py +102 -0
  306. package/md_cg/test_p38_contextualize.py +273 -273
  307. package/md_cg/test_p39_verify_flow.py +153 -0
  308. package/md_cg/test_p39_vision_evidence.py +369 -369
  309. package/md_cg/test_p40_refine_worklist.py +241 -241
  310. package/md_cg/test_p41_evolve_patrol.py +224 -224
  311. package/md_cg/test_p42_provenance.py +269 -269
  312. package/md_cg/test_p43_pooling.py +412 -398
  313. package/md_cg/test_p44_md_whitebox.py +231 -231
  314. package/md_cg/test_p45_session_identity.py +219 -219
  315. package/md_cg/test_p46_unit_scope.py +272 -272
  316. package/md_cg/test_p47_session_view.py +316 -0
  317. package/md_cg/test_p4_fuzzy.py +223 -223
  318. package/md_cg/test_p5_semantic.py +226 -226
  319. package/md_cg/test_p6_consolidate.py +440 -387
  320. package/md_cg/test_p7_goals_recent.py +202 -202
  321. package/md_cg/test_p8_subgraph_chain.py +200 -200
  322. package/md_cg/test_p9_forget_protect.py +231 -231
  323. package/md_cg/test_predict_beta.py +135 -135
  324. package/md_cg/test_preflight_failclosed.py +100 -100
  325. package/md_cg/test_progressive.py +146 -146
  326. package/md_cg/test_propose_tail_index.py +157 -0
  327. package/md_cg/test_protocol.py +243 -243
  328. package/md_cg/test_reach.py +378 -378
  329. package/md_cg/test_reach_keys.py +201 -201
  330. package/md_cg/test_read_clip.py +141 -141
  331. package/md_cg/test_read_scope_b27.py +277 -0
  332. package/md_cg/test_readcache_default_on.py +168 -0
  333. package/md_cg/test_readcache_precise_inval.py +270 -0
  334. package/md_cg/test_readcache_prodpath.py +203 -0
  335. package/md_cg/test_reconcile_v0.py +294 -0
  336. package/md_cg/test_retr_gates_prodpath.py +140 -0
  337. package/md_cg/test_retr_s1.py +344 -340
  338. package/md_cg/test_retr_s1b.py +276 -209
  339. package/md_cg/test_retr_s3.py +194 -194
  340. package/md_cg/test_retr_s4.py +163 -163
  341. package/md_cg/test_retr_s5.py +200 -200
  342. package/md_cg/test_retr_s6.py +157 -157
  343. package/md_cg/test_retr_s7.py +392 -384
  344. package/md_cg/test_retr_s8_time.py +369 -316
  345. package/md_cg/test_retr_s9_edges.py +286 -286
  346. package/md_cg/test_retr_s9_entity_ctx.py +9 -3
  347. package/md_cg/test_retr_score_once.py +208 -0
  348. package/md_cg/test_review_conformance.py +367 -367
  349. package/md_cg/test_review_onepass.py +170 -0
  350. package/md_cg/test_role_views.py +354 -354
  351. package/md_cg/test_rrf_graph_seed_cache.py +154 -0
  352. package/md_cg/test_security_audit.py +155 -0
  353. package/md_cg/test_security_audit_b26.py +161 -0
  354. package/md_cg/test_security_audit_v21.py +250 -0
  355. package/md_cg/test_sem_noise.py +242 -242
  356. package/md_cg/test_semantic_canonical.py +16 -2
  357. package/md_cg/test_session_isolation.py +168 -0
  358. package/md_cg/test_snapshot_autoclose.py +187 -0
  359. package/md_cg/test_subproc_encoding.py +192 -192
  360. package/md_cg/test_sustain_mutual.py +153 -153
  361. package/md_cg/test_tail_watermark_race.py +208 -0
  362. package/md_cg/test_tasks.py +409 -409
  363. package/md_cg/test_tenant_env_override_warn.py +139 -0
  364. package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
  365. package/md_cg/test_tool_face.py +189 -189
  366. package/md_cg/test_transfer.py +180 -180
  367. package/md_cg/test_trust.py +361 -361
  368. package/md_cg/test_twophase.py +286 -286
  369. package/md_cg/test_v14_fixes.py +38 -20
  370. package/md_cg/test_validity_filter.py +280 -280
  371. package/md_cg/test_verify_answer.py +138 -138
  372. package/md_cg/test_verify_dirty_reconcile.py +157 -0
  373. package/md_cg/test_wisdom_md_store.py +292 -292
  374. package/md_cg/test_writelimit.py +197 -197
  375. package/md_cg/test_writepipe.py +214 -214
  376. package/md_cg/theory.py +6 -3
  377. package/md_cg/tokens.py +85 -14
  378. package/md_cg/tool_face.py +260 -260
  379. package/md_cg/trust.py +986 -950
  380. package/md_cg/twophase.py +231 -231
  381. package/md_cg/units.py +668 -667
  382. package/md_cg/vision_evidence.py +667 -666
  383. package/md_cg/weights.py +624 -624
  384. package/md_cg/whitebox.py +527 -527
  385. package/md_cg/whitebox_kb/__init__.py +37 -37
  386. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  387. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  388. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  389. package/md_cg/whitebox_kb/engine.py +310 -310
  390. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  391. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  392. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  393. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  394. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  395. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  396. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  397. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  398. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  399. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  400. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  401. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  402. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  403. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  404. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  405. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  406. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  407. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  408. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  409. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  410. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  411. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  412. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  413. package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
  414. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  415. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  416. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  417. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  418. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  419. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  420. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  421. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  422. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  423. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  424. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  425. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  426. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  427. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  428. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  429. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  430. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  431. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  432. package/md_cg/writelimit.py +356 -356
  433. package/md_cg/writepipe.py +20 -8
  434. package/package.json +101 -96
  435. package/skills/plugin.json +54 -54
  436. package/skills/skills/designer-perspective/SKILL.md +158 -158
  437. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  438. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  439. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  440. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  441. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  442. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  443. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  444. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  445. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  446. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  447. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  488. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  489. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  490. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  491. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  492. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  493. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  494. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  495. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  496. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  497. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  498. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  499. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  500. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  501. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  502. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  503. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  504. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  505. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  506. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  507. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  508. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  509. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  510. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  511. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  512. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  513. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  514. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  515. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  516. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  517. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  518. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  519. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  520. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  521. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  522. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  523. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  524. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  525. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  526. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  527. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  528. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  529. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  530. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  531. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  532. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  533. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  534. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  535. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  536. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  537. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  538. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  539. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  540. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  541. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  542. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  543. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  544. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  545. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  546. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  547. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  548. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  549. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  550. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  551. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  552. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  553. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  554. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  555. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  556. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  557. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  558. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  559. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  560. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  561. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  562. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  563. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  564. package/skills/skills/lingshu-net/SKILL.md +48 -48
  565. package/skills/skills/lingshu-os/SKILL.md +64 -64
  566. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  567. package/src/bridge.ts +33 -0
  568. package/src/hooks.ts +38 -2
  569. package/src/index.ts +526 -518
  570. package/src/lib/datapath.ts +326 -326
  571. package/src/lib/mdcg_client.ts +413 -413
  572. package/src/lib/mutual.ts +428 -428
  573. package/src/lib/prompt_safety.ts +62 -62
  574. package/src/lib/python_path.ts +71 -71
  575. package/src/lib/roleplay_web.ts +116 -29
  576. package/src/lib/token_store.ts +13 -3
  577. package/src/tools.ts +212 -212
  578. package/zcode/AGENTS.md +11 -3
  579. package/zcode/README.md +41 -41
  580. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
@@ -1,1440 +1,1538 @@
1
- # -*- coding: utf-8 -*-
2
- """md_cg · 离线固化:LLM 补 CCG 四要素 → 确定性验证 → 固化为 md 字段
3
-
4
- 为什么是「离线固化」而不是「在线向量」:
5
- 白箱第 1 篇:相似度可以产生候选,但**不授予执行资格**;资格必须由条件证据裁决。
6
- LLM 是黑箱,它的输出只能是**候选条件**,不能直接成为检索依据——否则在线检索
7
- 就被黑箱污染,CCG 28%→88% 的改进会退化回去。故本工具把 LLM 严格限制在
8
- **离线一次性的固化工序**里:
9
-
10
- 读节点 → LLM 产出四要素候选 → 确定性验证 → 通过才写进 md 字段
11
-
12
- 在线检索(search / recall / _path_semantic)仍然只读 md 里已固化的字段,全程白箱。
13
-
14
- 理论 / 纪律对齐(docs/工作纪律_认知图条目_v1.1.json):
15
- · 第 3 条 白箱方法:不猜测;**未验证不写入**。
16
- · 第 5 条 验证纪律:**未经验证不固化**——入库前必须走验证(回放 / 断言 / 回归)。
17
- · 第 13 条 访谈澄清:节点四要素 = 条件 / 子内容 / 如何执行 / 不适用条件
18
- ——本工具固化的正是这四个字段(对齐 CCG 的生效条件 / 子功能 / 执行 / 不适用条件)。
19
- · 《智能的认知过程》:新条件能否**稳定解释误差**?成立 → 纳入知识结构;
20
- 不成立 → **不固化**,标记为待验证。
21
- 故 verdict 三态对齐白箱资格判定:ACCEPT(固化)/ REJECT(丢弃)/ DEFER(只存候选)。
22
-
23
- 验证分三段闸门——前两段确定性零 LLM,第三段是「双模型交叉验证」:
24
-
25
- 闸门 1 · grounding 支撑度(确定性):候选短语必须能在节点正文里找到字符级依据,
26
- 否则判为幻觉 → REJECT。(对应「不猜测」)
27
- 闸门 2 · replay 回放(确定性):把候选条件当作查询,回放生产检索路径的判定:
28
- · pos_recall 以「生效条件」为查询 → 本节点应被召回,且不被自身负条件挡住;
29
- · neg_separated 以「不适用条件」为查询 → 应触发条件级负路由,且负条件与正文
30
- 低相关(负条件必须是「域外」的,不能把知识本身否定掉);
31
- · no_conflict 生效条件与不适用条件不得互相覆盖。
32
- 三者同时成立才算「条件稳定」。(对应「回放 / 断言 / 回归」)
33
- 闸门 3 · 验证单元(GLM,独立模型):逐条核验候选是否有正文依据、负条件是否真域外。
34
- 硬约束:**验证单元只能否决,不能新增/改写**——它没有产出权,
35
- 否则验证环节自己就成了新的幻觉源。
36
-
37
- 回放器复用 _path_semantic 的同一批原语(_declared_conditions / _neg_hit /
38
- _weighted_coverage / expand_query_terms_weighted),并由 P6 测试与真实
39
- MdCGOS._path_semantic 做一致性回归,保证不漂移。
40
-
41
- 双模型角色分工(用户配置,可用环境变量覆盖):
42
- 反思单元 reflect → 默认 deepseek-v4.1-flash-expires-on-0910
43
- (IDE 显示名 DeepSeek-V4.1-Flash;限时模型,见常量注释)
44
- 验证单元 verify → 默认 glm-5.3-flash (IDE 显示名 GLM-5.3-flash)
45
- 环境变量:MDCG_REFLECT_MODEL/BASE/KEY、MDCG_VERIFY_MODEL/BASE/KEY。
46
- 注意 IDE 显示名 ≠ API 模型 id;`--check` 可零 token 探测各网关真实 id。
47
- 两个模型分属不同厂商,避免同源模型的系统性偏见互相印证(交叉验证的本意)。
48
-
49
- · 验证单元不可用(未配 key)时,默认 **不固化**(DEFER)——纪律 5「未经验证不固化」;
50
- 确需单模型跑通可显式 --no-verify(provenance 记为 skipped)或 --self-verify
51
- (同模型自审,provenance 记为 self_verify=true,属于降级模式)。
52
- · 「验证方式」是 CCG 必需要素,其值 = 声明的验证基底。本工具可写
53
- `# 验证方式:<声明>`(--verification-basis,默认即上面的双模型声明);
54
- frontmatter.verification_basis 只能取 nodefile 的枚举
55
- (compiler/test/measurement/formal_proof/data/other),双 LLM 交叉验证对应 "other"。
56
- 纯文本声明无需 LLM → --basis-only 可零成本补齐全库(纪律 3:不猜测)。
57
-
58
- 零第三方依赖(D-005):HTTP 走标准库 urllib.request;reflect_fn / verify_fn 可注入
59
- (离线可测)。默认 --dry-run,只有 --apply 才写盘(纪律 6:改动可核对)。
60
- 写盘只动 CCG 字段与 provenance,**不覆盖已有非空字段**(除非 --overwrite)。
61
- """
62
- from __future__ import annotations
63
-
64
- import argparse
65
- import hashlib
66
- import json
67
- import math
68
- import os
69
- import re
70
- import sys
71
- import time
72
- import urllib.error
73
- import urllib.request
74
-
75
- from . import crypto, evolution, nodefile, routing
76
- from .fsutil import append_jsonl
77
- from .mdcg import BUCKETED_LAYERS, bigrams, expand_query_terms_weighted
78
- from .mdcos import (MdCGOS, _ccg_field, _declared_conditions, _neg_hit, _sig,
79
- _weighted_coverage)
80
-
81
- # ---- 常量 ----------------------------------------------------------------
82
-
83
- # 与工作纪律第 13 条「节点四要素」同构:
84
- # 生效条件 ↔ conditions(什么时候适用)
85
- # 子功能 ↔ subgraph / depends_on(子内容)
86
- # 执行 ↔ execution(如何执行)
87
- # 不适用条件 ↔ negative(什么时候不适用)
88
- CCG_FIELDS = ("生效条件", "子功能", "执行", "不适用条件")
89
- MULTI_FIELDS = ("生效条件", "子功能", "不适用条件") # 列表型
90
- SINGLE_FIELDS = ("执行",) # 单值型
91
-
92
- # LLM 常把「子功能」写成「子内容」,别名容错(固化时统一落到标准字段名)
93
- FIELD_ALIASES = {
94
- "生效条件": ("生效条件", "适用条件", "conditions", "condition"),
95
- "子功能": ("子功能", "子内容", "子流程", "subgraph", "sub"),
96
- "执行": ("执行", "如何执行", "执行方式", "execution", "how"),
97
- "不适用条件": ("不适用条件", "不适用", "负条件", "negative", "reject"),
98
- }
99
-
100
- # 默认 grounding 阈值:不适用条件描述的是「域外」情境,与正文天然低相关,
101
- # 故阈值放宽;其余三要素必须能在正文里找到实打实的依据。
102
- DEFAULT_GROUNDING = {"生效条件": 0.5, "子功能": 0.5, "执行": 0.5, "不适用条件": 0.34}
103
-
104
- MAX_BODY_CHARS = 3000 # 正文截断(控制 token,且条件主要来自开头)
105
- MAX_TERMS = 8 # 单字段候选条数上限
106
- MAX_TERM_LEN = 40 # 单条候选长度上限
107
-
108
- # CCG 声明行(`# 生效条件:…` 等)——它们不是正文,grounding/replay 必须把它们剥掉,
109
- # 否则已写入的「不适用条件」会在二次运行时被当成正文依据,导致节点自我否定。
110
- _CCG_LINE_RE = re.compile(
111
- r"^\s*#\s*(功能名|生效条件|子功能|执行|验证方式|不适用条件)\s*[::]")
112
-
113
-
114
- # 生效条件:给定 content,返回剔除所有匹配 _CCG_LINE_RE 的行后以换行连接的非声明正文;content 为 None 时按空串处理。
115
- def body_text(content: str) -> str:
116
- """剥掉 CCG 声明行后的正文——验证只认正文,不认已写下的声明。"""
117
- return "\n".join(l for l in (content or "").split("\n")
118
- if not _CCG_LINE_RE.match(l))
119
-
120
- # ---- 双模型角色(反思单元 / 验证单元)-------------------------------------
121
-
122
- REFLECT_ROLE = "reflect" # 反思单元:产出候选
123
- VERIFY_ROLE = "verify" # 验证单元:否决候选(无产出权)
124
- ROLES = (REFLECT_ROLE, VERIFY_ROLE)
125
-
126
- # 推荐模型(真实 API id,经实际调用确认;可用 MDCG_<ROLE>_MODEL 覆盖)
127
- # 注意:reflect 的 id 带过期标记(expires-on-0910),属**限时模型**——过期后 /models
128
- # 列表会下架该 id,届时改用 deepseek-v4-flash 或用 MDCG_REFLECT_MODEL 覆盖。
129
- ROLE_DEFAULT_MODEL = {REFLECT_ROLE: "deepseek-v4.1-flash-expires-on-0910",
130
- VERIFY_ROLE: "glm-5.3-flash"}
131
- ROLE_DEFAULT_BASE = {REFLECT_ROLE: "https://api.deepseek.com",
132
- VERIFY_ROLE: "https://open.bigmodel.cn/api/paas/v4"}
133
- _ROLE_ENV = {REFLECT_ROLE: ("MDCG_REFLECT_MODEL", "MDCG_REFLECT_BASE", "MDCG_REFLECT_KEY"),
134
- VERIFY_ROLE: ("MDCG_VERIFY_MODEL", "MDCG_VERIFY_BASE", "MDCG_VERIFY_KEY")}
135
- # 验证单元 key 的常见别名(智谱系)
136
- VERIFY_KEY_ALIASES = ("ZHIPU_API_KEY", "ZHIPUAI_API_KEY", "GLM_API_KEY", "BIGMODEL_API_KEY")
137
-
138
- # 「验证方式」的声明文本(可 --verification-basis 覆盖 / --no-basis 关闭)
139
- BASIS_TEMPLATE = "双模型交叉验证(反思单元={reflect},验证单元={verify})"
140
- # frontmatter.verification_basis 只能取 nodefile 的枚举;双 LLM 交叉验证 → other
141
- BASIS_ENUM_DEFAULT = "other"
142
-
143
- REFLECT_PROMPT = (
144
- "你是认知图节点的**反思单元**。给定一个知识节点的标题与正文,反思并抽取四要素。\n"
145
- "只输出一个 JSON 对象,不要任何解释或代码围栏。\n"
146
- "字段含义:\n"
147
- ' "生效条件": 什么查询/情境下该知识**适用**(短语数组,2~5 条)\n'
148
- ' "子功能": 该知识包含的子内容/子步骤(短语数组,2~5 条)\n'
149
- ' "执行": 如何执行/如何使用该知识(单个字符串)\n'
150
- ' "不适用条件": 什么查询/情境下该知识**不**适用(短语数组,1~3 条)\n'
151
- "硬约束:\n"
152
- " 1. 每条短语必须能在正文中找到依据,禁止编造正文里没有的工具/概念;\n"
153
- " 2. 不适用条件必须是**正文之外的邻近易混情境**,不得与生效条件语义重叠;\n"
154
- " 3. 短语要短(不超过 20 字),不要写完整句子。\n"
155
- "输出格式:"
156
- '{{"生效条件": ["..."], "子功能": ["..."], "执行": "...", "不适用条件": ["..."]}}\n'
157
- "标题:{title}\n正文:\n{body}"
158
- )
159
-
160
- VERIFY_PROMPT = (
161
- "你是认知图节点的**验证单元**。你的职责是**否决**,不是补充。\n"
162
- "只能从候选里删除不成立的条目,**绝不允许新增或改写任何条目**。\n"
163
- "给定标题、正文与反思单元给出的候选四要素,逐条核验:\n"
164
- " · 该条目是否真的能在正文中找到依据?找不到依据 → 删除;\n"
165
- " · 不适用条件是否真的域外?若它其实是该节点的适用情境 → 删除;\n"
166
- " · 生效条件与不适用条件是否语义重叠?重叠者删除其一(保留更贴合正文的那个)。\n"
167
- "只输出一个 JSON 对象,键为字段名,值为 "
168
- '{{"keep": ["保留的条目"], "drop": ["删除的条目"], "reason": "一句话理由"}}。\n'
169
- "候选:{cand}\n标题:{title}\n正文:\n{body}"
170
- )
171
-
172
-
173
- # ---- LLM 侧(黑箱只在离线工序,产出候选)--------------------------------
174
-
175
- # 生效条件:给定 role,按显式参数、角色环境变量、通用兜底依次解析并返回 (model, base, key);key 不落 DEEPSEEK_API_KEY 除非 role 为 REFLECT_ROLE。
176
- def role_config(role: str, model: str = None, base: str = None,
177
- key: str = None) -> tuple:
178
- """解析某角色的 (model, base, key):显式参数 > 角色环境变量 > 通用兜底。
179
-
180
- key 刻意**不**让验证单元回落到 DEEPSEEK_API_KEY——跨厂商混用会把一个厂商的
181
- 凭证发到另一个厂商的网关,既必然失败又构成凭证外泄。
182
- """
183
- m_env, b_env, k_env = _ROLE_ENV[role]
184
- if key is None:
185
- key = os.environ.get(k_env)
186
- if key is None and role == VERIFY_ROLE:
187
- for alias in VERIFY_KEY_ALIASES:
188
- key = os.environ.get(alias)
189
- if key:
190
- break
191
- if key is None:
192
- key = os.environ.get("MDCG_LLM_KEY")
193
- if key is None and role == REFLECT_ROLE:
194
- key = os.environ.get("DEEPSEEK_API_KEY")
195
- model = (model or os.environ.get(m_env) or os.environ.get("MDCG_LLM_MODEL")
196
- or ROLE_DEFAULT_MODEL[role])
197
- base = (base or os.environ.get(b_env) or os.environ.get("MDCG_LLM_BASE")
198
- or ROLE_DEFAULT_BASE[role])
199
- return model, base, key
200
-
201
-
202
- # 生效条件:给定 prompt 且 role 解析或通用兜底得到非空 key 时,向 base 的 /chat/completions 发 POST 并返回首个 choice 的 message.content;key 为空则抛 RuntimeError。
203
- def http_llm(prompt: str, model: str = None, base: str = None, key: str = None,
204
- role: str = None, timeout: int = 120, max_tokens: int = 1200) -> str:
205
- """标准库 HTTP 调 LLM(OpenAI 兼容 /chat/completions)。零第三方依赖。
206
-
207
- role 给定时按该角色配置解析(reflect / verify),否则走通用配置。
208
- """
209
- if role:
210
- model, base, key = role_config(role, model, base, key)
211
- else:
212
- model = (model or os.environ.get("MDCG_LLM_MODEL")
213
- or ROLE_DEFAULT_MODEL[REFLECT_ROLE])
214
- base = (base or os.environ.get("MDCG_LLM_BASE")
215
- or ROLE_DEFAULT_BASE[REFLECT_ROLE])
216
- key = (key or os.environ.get("MDCG_LLM_KEY")
217
- or os.environ.get("DEEPSEEK_API_KEY"))
218
- if not key:
219
- raise RuntimeError(
220
- f"未配置 {role or 'llm'} 的 API key"
221
- f"({_ROLE_ENV[role][2] if role in _ROLE_ENV else 'MDCG_LLM_KEY'})")
222
- payload = json.dumps({
223
- "model": model,
224
- "messages": [{"role": "user", "content": prompt}],
225
- "max_tokens": max_tokens,
226
- }).encode("utf-8")
227
- req = urllib.request.Request(
228
- base.rstrip("/") + "/chat/completions", data=payload,
229
- headers={"Authorization": f"Bearer {key}",
230
- "Content-Type": "application/json"})
231
- try:
232
- with urllib.request.urlopen(req, timeout=timeout) as resp:
233
- data = json.loads(resp.read().decode("utf-8"))
234
- except urllib.error.HTTPError as exc:
235
- detail = exc.read().decode("utf-8", "replace")[:300]
236
- raise RuntimeError(f"HTTP {exc.code} model={model} base={base} :: {detail}") from None
237
- return data["choices"][0]["message"]["content"]
238
-
239
-
240
- # 生效条件:给定 role,若 role_config 得到非空 key 则 GET base/models 并返回含 ok/model_available/models 的字典;无 key 或请求异常则返回 ok=False 及错误信息。
241
- def probe_models(role: str, timeout: int = 20) -> dict:
242
- """零 token 探测:列出该角色网关的可用模型 id(GET /models)。"""
243
- model, base, key = role_config(role)
244
- if not key:
245
- return {"role": role, "model": model, "base": base, "ok": False,
246
- "error": "no_key", "models": []}
247
- req = urllib.request.Request(
248
- base.rstrip("/") + "/models",
249
- headers={"Authorization": f"Bearer {key}"})
250
- try:
251
- with urllib.request.urlopen(req, timeout=timeout) as resp:
252
- data = json.loads(resp.read().decode("utf-8"))
253
- ids = [m.get("id") for m in (data.get("data") or []) if m.get("id")]
254
- except Exception as exc: # noqa: BLE001 —— 探测要抗单点
255
- return {"role": role, "model": model, "base": base, "ok": False,
256
- "error": f"{type(exc).__name__}: {exc}"[:200], "models": []}
257
- res = {"role": role, "model": model, "base": base, "ok": True,
258
- "model_available": model in ids, "models": ids}
259
- if not res["model_available"]:
260
- res["note"] = ("该 id 未出现在 /models 列表:可能是限时/按需模型,或已下架;"
261
- "以实际 /chat/completions 调用结果为准")
262
- return res
263
-
264
-
265
- # 生效条件:raw 为 None 或 strip 后不含 "{"(i<0)、或末个 "}" 的位置 j<=i 时返回 None;否则对 s 从首个 "{" 到末个 "}" 的切片 json.loads,成功则返回解析结果,抛 ValueError 时返回 None。
266
- def _extract_json_obj(raw: str):
267
- """从 LLM 输出里抠出第一个 JSON 对象(容忍代码围栏 / 前后废话)。"""
268
- s = (raw or "").strip()
269
- i, j = s.find("{"), s.rfind("}")
270
- if i < 0 or j <= i:
271
- return None
272
- try:
273
- return json.loads(s[i:j + 1])
274
- except ValueError:
275
- return None
276
-
277
-
278
- # 生效条件:v 为 None 返回 [];否则按 v 是 str 取 [v]、是 list/tuple 取逐项、其他取 [str(v)],逐项 strip 并去两端包裹标点后跳过空串及长度超 MAX_TERM_LEN 的项,未出现过的才 append,每次 append 后若 len(out) >= limit 即 break 返回 out(故 limit<=0 且存在有效项时仍返回 1 项)。
279
- def _as_terms(v, limit: int = MAX_TERMS):
280
- """把 LLM 给的值规范成去重、限长的短语列表。"""
281
- if v is None:
282
- return []
283
- if isinstance(v, str):
284
- items = [v]
285
- elif isinstance(v, (list, tuple)):
286
- items = list(v)
287
- else:
288
- items = [str(v)]
289
- out = []
290
- for x in items:
291
- s = str(x).strip().strip(",。;;、,.;\"'“”")
292
- if not s or len(s) > MAX_TERM_LEN:
293
- continue
294
- if s not in out:
295
- out.append(s)
296
- if len(out) >= limit:
297
- break
298
- return out
299
-
300
-
301
- # 生效条件:给定 raw,若 _extract_json_obj 解析出 dict,则按 CCG_FIELDS 与 FIELD_ALIASES 提取非空字段并规范为列表或单值返回字典;否则返回 {}。
302
- def parse_candidate(raw: str) -> dict:
303
- """LLM 原始输出 → {字段: 列表/字符串};解析失败返回 {}。"""
304
- obj = _extract_json_obj(raw)
305
- if not isinstance(obj, dict):
306
- return {}
307
- out = {}
308
- for field in CCG_FIELDS:
309
- val = None
310
- for alias in FIELD_ALIASES[field]:
311
- if alias in obj and obj[alias] not in (None, "", [], {}):
312
- val = obj[alias]
313
- break
314
- if val is None:
315
- continue
316
- if field in SINGLE_FIELDS:
317
- terms = _as_terms(val, limit=1)
318
- if terms:
319
- out[field] = terms[0]
320
- else:
321
- terms = _as_terms(val)
322
- if terms:
323
- out[field] = terms
324
- return out
325
-
326
-
327
- # 生效条件:给定 raw,若解析出 dict,则按 CCG_FIELDS 提取 keep/drop/has_keep/reason 结构返回字典;否则返回 {}。
328
- def parse_verdict(raw: str) -> dict:
329
- """验证单元输出 → {字段: {keep, drop, has_keep, reason}};解析失败返回 {}。"""
330
- obj = _extract_json_obj(raw)
331
- if not isinstance(obj, dict):
332
- return {}
333
- out = {}
334
- for field in CCG_FIELDS:
335
- val = None
336
- for alias in FIELD_ALIASES[field]:
337
- if alias in obj and obj[alias] not in (None, "", [], {}):
338
- val = obj[alias]
339
- break
340
- if val is None:
341
- continue
342
- if isinstance(val, list): # 容忍只给 keep 数组
343
- out[field] = {"keep": _as_terms(val), "drop": [], "has_keep": True,
344
- "reason": ""}
345
- elif isinstance(val, dict):
346
- out[field] = {"keep": _as_terms(val.get("keep")),
347
- "drop": _as_terms(val.get("drop")),
348
- "has_keep": "keep" in val,
349
- "reason": str(val.get("reason") or "")[:200]}
350
- return out
351
-
352
-
353
- # 生效条件:给定 kept 与 verdict,若 verdict 为空则返回 (dict(kept), {});否则按 drop 与 has_keep 收窄候选并返回 (收窄后候选, 被剔除明细)。
354
- def narrow_by_verdict(kept: dict, verdict: dict):
355
- """按验证单元裁决收窄候选——**只能否决,不能新增**。
356
-
357
- · 验证单元未表态的字段 → 保留(沉默不等于否决)
358
- · has_keep=True → 取「候选 ∩ keep」;否则只按 drop 剔除
359
- 返回 (收窄后候选, 被剔除明细)。
360
- """
361
- if not verdict:
362
- return dict(kept), {}
363
- out, dropped = {}, {}
364
- for field, val in kept.items():
365
- terms = val if isinstance(val, list) else [val]
366
- vd = verdict.get(field)
367
- if vd is None:
368
- out[field] = val
369
- continue
370
- dropset = set(vd.get("drop") or [])
371
- keepset = set(vd.get("keep") or [])
372
- surv, gone = [], []
373
- for t in terms:
374
- if t in dropset:
375
- gone.append(t)
376
- elif vd.get("has_keep") and t not in keepset:
377
- gone.append(t)
378
- else:
379
- surv.append(t)
380
- if gone:
381
- dropped[field] = {"terms": gone, "reason": vd.get("reason") or ""}
382
- if surv:
383
- out[field] = surv if field in MULTI_FIELDS else surv[0]
384
- return out, dropped
385
-
386
-
387
- # ---- 确定性验证(零 LLM)-------------------------------------------------
388
-
389
- # 生效条件:给定 term 与 body,若 term 的 bigram 序列非空则返回命中 bigram 数除以总 bigram 数,否则返回 0.0。
390
- def grounding_score(term: str, body: str) -> float:
391
- """候选短语在正文里的字符级支撑度 = 命中 bigram 数 / 总 bigram 数。"""
392
- bg = bigrams(term or "")
393
- if not bg:
394
- return 0.0
395
- hit = sum(1 for g in bg if g in (body or ""))
396
- return hit / len(bg)
397
-
398
-
399
- # 生效条件:给定 cand 与 body,按 thresholds 更新 DEFAULT_GROUNDING 后逐字段过滤候选,返回 (达标 kept, detail);不达标者丢弃。
400
- def grounding_filter(cand: dict, body: str, thresholds: dict = None):
401
- """逐字段过滤候选:返回 (kept, detail)。不达标者丢弃(对应「不猜测」)。"""
402
- th = dict(DEFAULT_GROUNDING)
403
- th.update(thresholds or {})
404
- kept, detail = {}, {}
405
- for field, val in cand.items():
406
- terms = val if isinstance(val, list) else [val]
407
- ok_terms, scores = [], {}
408
- for t in terms:
409
- g = grounding_score(t, body)
410
- scores[t] = round(g, 3)
411
- if g >= th.get(field, 0.5):
412
- ok_terms.append(t)
413
- detail[field] = {"scores": scores, "kept": len(ok_terms)}
414
- if ok_terms:
415
- kept[field] = ok_terms if field in MULTI_FIELDS else ok_terms[0]
416
- return kept, detail
417
-
418
-
419
- # 生效条件:给定 pos_terms、neg_terms、body,返回含 pos_recall、neg_separated、no_conflict、ok 的回放判定字典。
420
- def replay_check(pos_terms, neg_terms, body: str) -> dict:
421
- """回放生产判定:正例召回 + 负例剔除 + 无自相矛盾。
422
-
423
- 复用 _path_semantic 的同一批原语,保证与生产路同源(P6 与真实路做一致性回归)。
424
- """
425
- pos_text = " ".join(pos_terms or [])
426
- neg_text = " ".join(neg_terms or [])
427
- tw_pos = expand_query_terms_weighted(pos_text) if pos_text else {}
428
- tw_neg = expand_query_terms_weighted(neg_text) if neg_text else {}
429
-
430
- # 1. 正例:以生效条件为查询,本节点正文应被命中,且不被自身负条件挡住
431
- pos_recall = bool(pos_text) and _weighted_coverage(tw_pos, body) > 0.0 \
432
- and not _neg_hit(tw_pos, neg_terms)
433
-
434
- # 2. 负例:以不适用条件为查询,应触发条件级负路由;且负条件与正文低相关
435
- # (负条件必须是「域外」的,若与正文强相关,等于让知识否定自己)
436
- if neg_terms:
437
- neg_separated = _neg_hit(tw_neg, neg_terms) \
438
- and _weighted_coverage(tw_neg, body) < 0.5
439
- else:
440
- neg_separated = True
441
-
442
- # 3. 生效条件与不适用条件不得互相覆盖
443
- no_conflict = (not pos_text) or (not neg_text) \
444
- or _weighted_coverage(tw_pos, neg_text) < 0.5
445
-
446
- ok = pos_recall and neg_separated and no_conflict
447
- return {"pos_recall": pos_recall, "neg_separated": neg_separated,
448
- "no_conflict": no_conflict, "ok": ok}
449
-
450
-
451
- # ---- 写盘(固化)---------------------------------------------------------
452
-
453
- # 生效条件:给定 content 与 field,当 content 含 "# field:" 或 "# field:" 时返回 True,否则 False。
454
- def _has_ccg_line(content: str, field: str) -> bool:
455
- return f"# {field}:" in (content or "") or f"# {field}:" in (content or "")
456
-
457
-
458
- # 生效条件:给定 fm 与 content,对每个 CCG_FIELDS,若 frontmatter.comment 值非空或正文含对应 CCG 行则记入,返回已有字段字典。
459
- def existing_fields(fm: dict, content: str) -> dict:
460
- """节点当前已有的四要素:正文 CCG 行 或 frontmatter.comment 任一存在即算有。"""
461
- comment = (fm.get("state_attributes") or {}).get("comment") or {}
462
- out = {}
463
- for field in CCG_FIELDS:
464
- v = comment.get(field)
465
- if v not in (None, "", [], {}):
466
- out[field] = v
467
- elif _has_ccg_line(content, field):
468
- out[field] = True
469
- return out
470
-
471
-
472
- # 生效条件:给定 content、field、value,若已有 "# field:" 行则替换并返回新正文;否则插在 "# 功能名" 之后,若无则该行前置。
473
- def _upsert_ccg_line(content: str, field: str, value: str) -> str:
474
- """在正文里写入/替换 `# <字段>:<值>`,优先插在「# 功能名」之后。"""
475
- lines = (content or "").split("\n")
476
- for i, ln in enumerate(lines):
477
- s = ln.strip()
478
- if not s.startswith("#") or field not in s:
479
- continue
480
- name = s.lstrip("#").strip().split(":")[0].split(":")[0].strip()
481
- if name == field:
482
- lines[i] = f"# {field}:{value}"
483
- return "\n".join(lines)
484
- newline = f"# {field}:{value}"
485
- for i, ln in enumerate(lines):
486
- if ln.strip().startswith("# 功能名"):
487
- lines.insert(i + 1, newline)
488
- return "\n".join(lines)
489
- return newline + "\n" + (content or "")
490
-
491
-
492
- # 生效条件:给定 kept 字段字典,返回一句话规律字符串,列出缺失字段名并声明补齐后可路由。
493
- def _evo_pattern(kept: dict) -> str:
494
- """规律(一句话):这一类节点反复缺的正是这批条件。"""
495
- names = "、".join(kept.keys())
496
- return f"缺「{names}」的节点条件不可判;补齐后四要素完整、可路由"
497
-
498
-
499
- # 生效条件:给定 prov 字典,拼接 reflect/verify 模型、grounding、replay、verification_basis 中存在的证据项并返回。
500
- def _evo_evidence(prov: dict) -> str:
501
- """证据:本次固化凭什么成立(模型 / 闸门 / 回放)。"""
502
- parts = []
503
- rf = (prov.get("reflect") or {}).get("model") or ""
504
- vf = (prov.get("verify") or {}).get("model") or ""
505
- if rf:
506
- parts.append(f"reflect={rf}")
507
- if vf:
508
- parts.append(f"verify={vf}")
509
- if prov.get("grounding"):
510
- parts.append("grounding通过")
511
- if prov.get("replay"):
512
- parts.append("replay通过")
513
- vb = prov.get("verification_basis") or ""
514
- if vb:
515
- parts.append(vb)
516
- return " · ".join(parts)
517
-
518
-
519
- # 生效条件:给定 cg、e、fm、content、kept、prov,将 kept 字段写入正文 CCG 行与 frontmatter.comment,不适用条件同步 non_applicable_conditions,并写 llm_consolidation 与演化记录,返回 None。
520
- def _apply_node(cg, e, fm: dict, content: str, kept: dict, prov: dict,
521
- basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT):
522
- """把通过验证的字段固化进 md:正文 CCG 行 + frontmatter.comment + 负条件 + provenance。
523
-
524
- 固化 = 对一条缺失条件的补充 → 同步落一条演化条目(md 账本,可回滚)。
525
- """
526
- nid = e.get("id") or os.path.basename(e["path"])[:-3]
527
- before = evolution.state_of(cg, nid) or {}
528
- comment = (fm.get("state_attributes") or {}).get("comment")
529
- if not isinstance(comment, dict):
530
- fm["state_attributes"] = dict(fm.get("state_attributes") or {})
531
- fm["state_attributes"]["comment"] = {}
532
- comment = fm["state_attributes"]["comment"]
533
- for field, val in kept.items():
534
- text = ";".join(val) if isinstance(val, list) else str(val)
535
- content = _upsert_ccg_line(content, field, text)
536
- comment[field] = text
537
- if field == "不适用条件":
538
- # 同步 frontmatter.non_applicable_conditions(引擎负路由读它)
539
- cur = [str(x) for x in (fm.get("non_applicable_conditions") or [])]
540
- for t in (val if isinstance(val, list) else [val]):
541
- if t not in cur:
542
- cur.append(t)
543
- fm["non_applicable_conditions"] = cur
544
- if basis:
545
- content = _upsert_ccg_line(content, "验证方式", basis)
546
- comment["验证方式"] = basis
547
- if not nodefile.verification_basis_valid(fm):
548
- # 枚举里没有「LLM 交叉验证」这一档,只能落到 other(声明文本在 CCG 行里)
549
- fm["verification_basis"] = basis_enum
550
- fm["llm_consolidation"] = prov
551
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
552
- durable=True)
553
- # 每一次修改都是对缺失条件的补充:记录规律 + 状态,不记录实现。
554
- evolution.record(
555
- cg, node_id=nid,
556
- pattern=_evo_pattern(kept),
557
- missing="、".join(kept.keys()),
558
- action="补齐 CCG 字段:" + "、".join(kept.keys()),
559
- evidence=_evo_evidence(prov),
560
- source="consolidate", kind=evolution.KIND_CONDITION_GAP,
561
- before=before, after=evolution.state_of(cg, nid) or {})
562
-
563
-
564
- # 生效条件:给定 root,扫描正排层节点并执行反思→白箱闸门→验证→固化,返回报表 rep;require_verify=True 且无 verify_fn 时全部 DEFER。
565
- def consolidate(root: str, layer: str = None, limit: int = None, apply: bool = False,
566
- overwrite: bool = False, llm_fn=None, reflect_fn=None,
567
- verify_fn=None, reflect_model: str = "", verify_model: str = "",
568
- verification_basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT,
569
- require_verify: bool = True, thresholds: dict = None,
570
- verbose: bool = True) -> dict:
571
- """对正排层节点做「反思单元产出候选 → 白箱闸门 → 验证单元否决 → 固化」。
572
-
573
- llm_fn 是 reflect_fn 的旧名(向后兼容,单模型模式)。
574
- require_verify=True 且无 verify_fn → 一律 DEFER(纪律 5:未经验证不固化)。
575
- """
576
- reflect_fn = reflect_fn or llm_fn
577
- cg = MdCGOS(root)
578
- entries = cg._candidates(layer=layer)
579
- t0 = time.time()
580
- rep = {"root": root, "layer": layer, "dry_run": not apply,
581
- "reflect_model": reflect_model, "verify_model": verify_model,
582
- "reflect": bool(reflect_fn), "verify": bool(verify_fn), "llm": bool(reflect_fn),
583
- "require_verify": require_verify,
584
- "verification_basis": verification_basis,
585
- "nodes_scanned": len(entries),
586
- "targeted": 0, "accepted": 0, "rejected": 0, "deferred": 0,
587
- "skipped_complete": 0, "written": 0, "reasons": {},
588
- "per_field": {f: 0 for f in CCG_FIELDS}, "verify_dropped": 0,
589
- "verification_basis_missing": 0, "samples": []}
590
-
591
- # 生效条件:以 reason 为键写入闭包 rep["reasons"],计数按 rep["reasons"].get(reason, 0) + 1 递增(键缺失从 0 起算),无返回值。
592
- def _bump(reason):
593
- rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
594
-
595
- for e in entries:
596
- if limit is not None and rep["targeted"] >= limit:
597
- break
598
- nid = os.path.basename(e["path"])[:-3]
599
- fm, content = cg._read(e)
600
- if fm is None:
601
- _bump("read_failed")
602
- continue
603
- if crypto.is_encrypted(content):
604
- _bump("locked") # 无密钥 → fail-closed:绝不改写密文
605
- continue
606
- have = existing_fields(fm, content)
607
- missing = [f for f in CCG_FIELDS if f not in have]
608
- if not nodefile.verification_basis_valid(fm):
609
- rep["verification_basis_missing"] += 1
610
- if not missing:
611
- rep["skipped_complete"] += 1
612
- continue
613
- rep["targeted"] += 1
614
-
615
- if not reflect_fn:
616
- rep["deferred"] += 1
617
- _bump("no_llm")
618
- continue
619
-
620
- # 1) 反思单元:产出候选(黑箱,唯一产出权)
621
- body = body_text(content)[:MAX_BODY_CHARS]
622
- title = _ccg_field(content, "功能名") or nid
623
- prompt = REFLECT_PROMPT.format(title=title, body=body)
624
- try:
625
- raw = reflect_fn(prompt)
626
- except Exception as exc: # noqa: BLE001 —— 离线批处理要抗单点失败
627
- rep["deferred"] += 1
628
- _bump(f"reflect_error:{type(exc).__name__}")
629
- continue
630
- cand = parse_candidate(raw)
631
- if not cand:
632
- rep["deferred"] += 1
633
- _bump("parse_failed")
634
- continue
635
-
636
- # 2) 白箱闸门:grounding + replay(零 LLM,先跑,省调用)
637
- kept, gdetail = grounding_filter(cand, body, thresholds)
638
- pos = kept.get("生效条件") or []
639
- neg = kept.get("不适用条件") or []
640
- replay = replay_check(pos, neg, body)
641
- if not kept or not replay["ok"]:
642
- rep["rejected"] += 1
643
- _bump("replay_failed" if kept else "grounding_failed")
644
- if verbose and len(rep["samples"]) < 8:
645
- rep["samples"].append({"id": nid, "verdict": "REJECT",
646
- "stage": "whitebox", "grounding": gdetail,
647
- "replay": replay})
648
- continue
649
-
650
- # 3) 验证单元:逐条核验,只能否决、不能新增
651
- dropped, vprompt, vd = {}, "", None
652
- if verify_fn:
653
- vprompt = VERIFY_PROMPT.format(
654
- cand=json.dumps(kept, ensure_ascii=False), title=title, body=body)
655
- try:
656
- vd = parse_verdict(verify_fn(vprompt))
657
- kept, dropped = narrow_by_verdict(kept, vd)
658
- except Exception as exc: # noqa: BLE001
659
- rep["deferred"] += 1
660
- _bump(f"verify_error:{type(exc).__name__}")
661
- continue
662
- if not kept:
663
- rep["rejected"] += 1
664
- _bump("verify_rejected")
665
- if verbose and len(rep["samples"]) < 8:
666
- rep["samples"].append({"id": nid, "verdict": "REJECT",
667
- "stage": "verify", "dropped": dropped})
668
- continue
669
- rep["verify_dropped"] += sum(len(d["terms"]) for d in dropped.values())
670
- elif require_verify:
671
- # 验证单元不可用 → 不固化(纪律 5:未经验证不固化)
672
- rep["deferred"] += 1
673
- _bump("verify_unavailable")
674
- continue
675
-
676
- # 3) 不覆盖已有非空字段(保护人工既有知识)
677
- if not overwrite:
678
- kept = {f: v for f, v in kept.items() if f not in have}
679
- if not kept:
680
- rep["skipped_complete"] += 1
681
- continue
682
-
683
- prov = {"at": round(time.time(), 3), "verdict": "ACCEPT",
684
- "source_hash": _sig(content),
685
- "reflect": {"model": reflect_model, "prompt_hash": _sig(prompt),
686
- "fields": sorted(kept)},
687
- "verify": ({"model": verify_model, "prompt_hash": _sig(vprompt),
688
- "dropped": dropped, "verdict_fields": sorted(vd or {}),
689
- "self_verify": verify_fn is reflect_fn}
690
- if verify_fn else {"model": "", "status": "skipped"}),
691
- "grounding": gdetail, "replay": replay,
692
- "verification_basis": verification_basis}
693
- rep["accepted"] += 1
694
- for f in kept:
695
- rep["per_field"][f] += 1
696
- if apply:
697
- _apply_node(cg, e, fm, content, kept, prov,
698
- verification_basis, basis_enum)
699
- append_jsonl(os.path.join(cg.root, "_consolidate.jsonl"),
700
- {"t": time.time(), "id": nid, "verdict": "ACCEPT",
701
- "fields": sorted(kept),
702
- "reflect_model": reflect_model,
703
- "verify_model": verify_model, "dropped": dropped,
704
- "source_hash": prov["source_hash"], "replay": replay})
705
- rep["written"] += 1
706
- if verbose and len(rep["samples"]) < 8:
707
- rep["samples"].append({"id": nid, "verdict": "ACCEPT",
708
- "fields": sorted(kept), "dropped": dropped,
709
- "replay": replay})
710
-
711
- if apply and rep["written"]:
712
- # 正文新增了 `# 不适用条件:` / `# 验证方式:` → 索引字段变了
713
- cg.rebuild_index()
714
- rep["elapsed_sec"] = round(time.time() - t0, 3)
715
- return rep
716
-
717
-
718
- # 生效条件:给定 root 与 basis,对缺 "# 验证方式" 行的节点补写验证方式并在需要时写入 basis_enum,返回统计 rep。
719
- def fill_verification_basis(root: str, basis: str, layer: str = None,
720
- limit: int = None, apply: bool = False,
721
- basis_enum: str = BASIS_ENUM_DEFAULT) -> dict:
722
- """只补「验证方式」——声明文本是常量,不需要黑箱生成,零 LLM 成本。
723
-
724
- 对应纪律 3「不猜测」:验证基底必须由人/流程声明,而不是让模型编出来。
725
- """
726
- cg = MdCGOS(root)
727
- entries = cg._candidates(layer=layer)
728
- rep = {"root": root, "layer": layer, "dry_run": not apply, "basis": basis,
729
- "basis_enum": basis_enum, "nodes_scanned": len(entries),
730
- "targeted": 0, "skipped_present": 0, "skipped_locked": 0,
731
- "written": 0}
732
- for e in entries:
733
- if limit is not None and rep["written"] >= limit:
734
- break
735
- fm, content = cg._read(e)
736
- if fm is None:
737
- continue
738
- if crypto.is_encrypted(content):
739
- rep["skipped_locked"] += 1 # 无密钥 → fail-closed:绝不改写密文
740
- continue
741
- if _has_ccg_line(content, "验证方式"):
742
- rep["skipped_present"] += 1
743
- continue
744
- rep["targeted"] += 1
745
- if not apply:
746
- continue
747
- comment = (fm.get("state_attributes") or {}).get("comment")
748
- if not isinstance(comment, dict):
749
- fm["state_attributes"] = dict(fm.get("state_attributes") or {})
750
- fm["state_attributes"]["comment"] = {}
751
- comment = fm["state_attributes"]["comment"]
752
- content = _upsert_ccg_line(content, "验证方式", basis)
753
- comment["验证方式"] = basis
754
- if not nodefile.verification_basis_valid(fm):
755
- fm["verification_basis"] = basis_enum
756
- nid = e.get("id") or os.path.basename(e["path"])[:-3]
757
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
758
- durable=True)
759
- rep["written"] += 1
760
- if apply and rep["written"]:
761
- cg.rebuild_index()
762
- return rep
763
-
764
-
765
- # ==========================================================================
766
- # 情境层批量提升(consolidate.promote)
767
- # ==========================================================================
768
- #
769
- # 场景:情境层(contextual)里有些记忆被反复命中/并入——它们已经不是「一次情境」,
770
- # 而是稳定的规律。本动作把它们提升为长期知识(knowledge),并保留:
771
- # · 双向可追溯:promoted_from + 演化账本(KIND_LAYER_SHIFT);
772
- # · 条件门槛:四要素(CCG)不全者**不提升**(未可判定就不该升格为长期知识);
773
- # · 可预演:apply=False 只出报表;可留痕:`_maintain.jsonl`。
774
-
775
- MAINTAIN_LOG = "_maintain.jsonl"
776
- CCG_REQUIRED = ("生效条件", "子功能", "执行", "不适用条件")
777
-
778
-
779
- # 生效条件:给定 cg、nid、e、fm、content、target_layer,把节点写入目标层(必要时按 routing 分桶)并删除旧路径,返回新相对路径与 bucket。
780
- def _relocate_layer(cg, nid, e, fm, content, target_layer):
781
- """把节点正文迁到目标层的正确目录(含分桶),删除旧文件。返回新相对路径。"""
782
- d = os.path.join(cg.root, target_layer)
783
- bucket = None
784
- if target_layer in BUCKETED_LAYERS:
785
- bucket = routing.bucket_dir(routing.route_key(fm.get("condition_space"),
786
- fm.get("tags")))
787
- d = os.path.join(d, bucket)
788
- os.makedirs(d, exist_ok=True)
789
- new_path = os.path.join(d, f"{nid}.md")
790
- old_path = os.path.join(cg.root, e.get("path") or f"{nid}.md")
791
- cg._write_node(nid, new_path, fm, content, durable=True)
792
- if os.path.abspath(old_path) != os.path.abspath(new_path) and os.path.exists(old_path):
793
- os.remove(old_path)
794
- return {"path": os.path.relpath(new_path, cg.root).replace("\\", "/"),
795
- "bucket": bucket}
796
-
797
-
798
- # 生效条件:给定 root,把 source_layer 中命中次数不小于 min_merge 或 importance 不小于 min_importance 且条件完整的节点提升到 target_layer,返回统计 rep。
799
- def promote_memories(root, source_layer="contextual", target_layer="knowledge",
800
- min_merge=2, min_importance=0.6, require_conditions=True,
801
- limit=None, apply=False, actor="maintain") -> dict:
802
- """把反复命中的情境记忆批量提升为长期知识(可预演 / 可留痕 / 可追溯)。"""
803
- cg = MdCGOS(root)
804
- entries = cg._candidates(layer=source_layer)
805
- rep = {"root": root, "source_layer": source_layer, "target_layer": target_layer,
806
- "dry_run": not apply, "nodes_scanned": len(entries), "targeted": 0,
807
- "skipped_locked": 0, "skipped_incomplete": 0, "skipped_not_hot": 0,
808
- "written": 0, "promoted": [], "samples": [],
809
- "min_merge": min_merge, "min_importance": min_importance,
810
- "require_conditions": bool(require_conditions)}
811
- batch = time.strftime("%Y%m%d-%H%M%S")
812
- for e in entries:
813
- if limit is not None and rep["written"] >= int(limit):
814
- break
815
- fm, content = cg._read(e)
816
- if fm is None:
817
- continue
818
- if crypto.is_encrypted(content):
819
- rep["skipped_locked"] += 1 # 无密钥 → fail-closed,绝不解密回写
820
- continue
821
- nid = e.get("id") or os.path.basename(e["path"])[:-3]
822
- hits = max(int(fm.get("merge_count") or 0),
823
- int(fm.get("access_count") or 0),
824
- int(fm.get("recall_count") or 0))
825
- imp = float(fm.get("importance") or e.get("importance") or 0.0)
826
- complete = all(_has_ccg_line(content, f) for f in CCG_REQUIRED)
827
- if require_conditions and not complete:
828
- rep["skipped_incomplete"] += 1 # 四要素不全 → 不可判定,不升格
829
- continue
830
- hot = hits >= int(min_merge)
831
- if not hot and imp < float(min_importance):
832
- rep["skipped_not_hot"] += 1
833
- continue
834
- rep["targeted"] += 1
835
- item = {"id": nid, "hits": hits, "importance": round(imp, 4),
836
- "conditions_complete": complete,
837
- "basis": fm.get("verification_basis")}
838
- if len(rep["samples"]) < 8:
839
- rep["samples"].append(item)
840
- if not apply:
841
- continue
842
- before = evolution.state_of(cg, nid) or {}
843
- fm["layer"] = target_layer
844
- fm["promoted_from"] = source_layer
845
- fm["promoted_at"] = time.time()
846
- fm["promotion_basis"] = {"hits": hits, "importance": round(imp, 4),
847
- "conditions_complete": complete, "batch": batch,
848
- "actor": actor}
849
- moved = _relocate_layer(cg, nid, e, fm, content, target_layer)
850
- evolution.record(
851
- cg, node_id=nid,
852
- pattern="情境记忆反复命中/并入 → 提升为长期知识",
853
- missing="", action=f"层迁移 {source_layer}→{target_layer}",
854
- evidence=f"hits={hits} importance={imp:.2f} conditions_complete={complete}",
855
- source="consolidate", kind=evolution.KIND_LAYER_SHIFT,
856
- before=before, after=evolution.state_of(cg, nid) or {})
857
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
858
- "t": time.time(), "action": "promote", "batch": batch, "id": nid,
859
- "from": source_layer, "to": target_layer, "hits": hits,
860
- "importance": round(imp, 4), "path": moved["path"], "actor": actor})
861
- rep["promoted"].append(nid)
862
- rep["written"] += 1
863
- if apply and rep["written"]:
864
- cg.rebuild_index()
865
- rep["note"] = ("dry-run:未写盘;apply=True 才迁移层"
866
- if not apply else f"已提升 {rep['written']} 个节点到 {target_layer}")
867
- return rep
868
-
869
-
870
- # 生效条件:给定 root,按 _maintain.jsonl 中 action=promote 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
871
- def rollback_promotion(root, node_ids=None, batch=None, actor="maintain") -> dict:
872
- """回滚情境提升:把 promoted_from 层迁回,并记一条演化条目。"""
873
- cg = MdCGOS(root)
874
- recs = [r for r in _read_maintain(root)
875
- if r.get("action") == "promote"
876
- and (not batch or r.get("batch") == batch)
877
- and (not node_ids or str(r.get("id")) in {str(x) for x in node_ids})]
878
- if not recs:
879
- return {"ok": False, "error": "no_records", "reverted": 0}
880
- reverted, ids = 0, []
881
- for rec in recs:
882
- nid = rec["id"]
883
- e = (cg.index.get("nodes") or {}).get(nid)
884
- if not e:
885
- continue
886
- fm, content = cg._read(e)
887
- if fm is None or crypto.is_encrypted(content):
888
- continue
889
- back = rec.get("from") or "contextual"
890
- before = evolution.state_of(cg, nid) or {}
891
- fm["layer"] = back
892
- fm["promoted_from"] = None
893
- fm["promotion_basis"] = {"rollback_of": rec.get("batch"), "actor": actor}
894
- _relocate_layer(cg, nid, e, fm, content, back)
895
- evolution.record(cg, node_id=nid, pattern="提升回滚:长期知识退回情境层",
896
- action=f"层迁移 {rec.get('to')}→{back}",
897
- evidence=f"rollback batch={rec.get('batch')}",
898
- source="consolidate", kind=evolution.KIND_ROLLBACK,
899
- before=before, after=evolution.state_of(cg, nid) or {})
900
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
901
- "t": time.time(), "action": "promote_rollback", "batch": rec.get("batch"),
902
- "id": nid, "to": back, "actor": actor})
903
- reverted += 1
904
- ids.append(nid)
905
- if reverted:
906
- cg.rebuild_index()
907
- return {"ok": True, "reverted": reverted, "ids": ids}
908
-
909
-
910
- # ==========================================================================
911
- # 层归位(consolidate.contextualize)
912
- # ==========================================================================
913
- #
914
- # 场景:批次流水账(note_/milestone_/retest6_)与感知产物(imgpart_/vpipe_)混在
915
- # knowledge 层——它们的语义是**情境**(某次批次的记录 / 某张图的一次观测),不是
916
- # 长期知识;但也不该进 rejected/unresolved(那是「失效 / 未解」,不是「情境」)。
917
- # 故归位到 contextual:
918
- # · 只改 layer 与落点目录;正文 / 密级 / id / tags 一律不动;
919
- # · **保留可召回**(contextual 已在层白名单内,且 predict._SAFE_LAYERS 含之);
920
- # · 可预演(apply=False)/ 可留痕(`_maintain.jsonl`)/ 可追溯(KIND_LAYER_SHIFT)
921
- # / 可回滚(按 batch 或 id 反向迁层)。
922
- #
923
- # 与 promote 的关系:promote 是 contextual→knowledge(升格),本动作是
924
- # knowledge→contextual(归位)。两者共用 `_relocate_layer` 与批次台账,方向相反。
925
-
926
- CONTEXTUALIZE_REASON_DEFAULT = "情境性内容归位(批次流水账 / 感知产物)"
927
-
928
-
929
- # 生效条件:给定 e,返回 e.id 字符串,若缺 id 则回落到 basename(e.path) 去掉 .md。
930
- def _entry_id(e) -> str:
931
- """索引条目取 id:优先 `id` 字段,回落到文件名(索引不保证带 id)。"""
932
- return str(e.get("id") or os.path.basename(e.get("path") or "")[:-3])
933
-
934
-
935
- # 生效条件:给定 root 与 base,若 base 不在维护日志已用批次中则返回 base,否则返回 base.n 且 n 为最小未用序号。
936
- def _unique_batch(root, base) -> str:
937
- """批次号去重:**同一秒内的两次调用不得共用批次号**。
938
-
939
- 否则「按批次回滚」会连带命中上一次的台账记录(回滚必须是精确的、可对账的)。
940
- """
941
- seen = {r.get("batch") for r in _read_maintain(root)}
942
- if base not in seen:
943
- return base
944
- n = 2
945
- while f"{base}.{n}" in seen:
946
- n += 1
947
- return f"{base}.{n}"
948
-
949
-
950
- # 生效条件:给定 root 且 prefixes 或 node_ids 至少一个非空,把 source_layer 中匹配的节点迁到 target_layer,返回统计 rep;两者皆空则抛 ValueError。
951
- def contextualize_prefixes(root, prefixes=None, node_ids=None,
952
- source_layer="knowledge", target_layer="contextual",
953
- reason="", limit=None, apply=False,
954
- actor="maintain") -> dict:
955
- """按 id 前缀(或定向 id 列表)把节点从 source_layer 归位到 target_layer。
956
-
957
- 默认方向 knowledge→contextual。`prefixes` / `node_ids` **至少给一个**:
958
- 宁可少搬,不可全库乱搬——不传白名单直接报错,拒绝「一次误调用把整个知识层改层」
959
- 这种不可归因的批量改写。`node_ids` 用于定向(含「回滚后单独补迁」的对称操作)。
960
- """
961
- pref = tuple(str(p) for p in (prefixes or ()) if str(p))
962
- ids = {str(i) for i in (node_ids or ()) if str(i)} or None
963
- if not pref and not ids:
964
- raise ValueError("contextualize 需要显式 prefixes 或 node_ids"
965
- "(如 ['note_','imgpart_']),拒绝对整层无差别改写")
966
- cg = MdCGOS(root)
967
-
968
- # 生效条件:e 经 _entry_id 得到 nid 后,若闭包 ids 不为 None 则返回 nid in ids 的真假,若 ids 为 None 则返回 nid.startswith(pref) 的真假。
969
- def _hit(e) -> bool:
970
- nid = _entry_id(e)
971
- return nid in ids if ids is not None else nid.startswith(pref)
972
-
973
- entries = [e for e in cg._candidates(layer=source_layer) if _hit(e)]
974
- batch = _unique_batch(root, time.strftime("%Y%m%d-%H%M%S"))
975
- rep = {"root": root, "action": "contextualize", "dry_run": not apply,
976
- "source_layer": source_layer, "target_layer": target_layer,
977
- "prefixes": list(pref), "node_ids": sorted(ids) if ids else [],
978
- "reason": reason or CONTEXTUALIZE_REASON_DEFAULT,
979
- "nodes_scanned": len(entries), "targeted": 0, "skipped_locked": 0,
980
- "skipped_already": 0, "written": 0, "moved": [], "samples": [],
981
- "batch": batch}
982
- for e in entries:
983
- if limit is not None and rep["written"] >= int(limit):
984
- break
985
- nid = _entry_id(e)
986
- if not _hit(e):
987
- continue
988
- fm, content = cg._read(e)
989
- if fm is None:
990
- continue
991
- if crypto.is_encrypted(content):
992
- rep["skipped_locked"] += 1 # 无密钥 → fail-closed,绝不解密回写
993
- continue
994
- if fm.get("layer") != source_layer:
995
- rep["skipped_already"] += 1
996
- continue
997
- rep["targeted"] += 1
998
- if len(rep["samples"]) < 8:
999
- rep["samples"].append({"id": nid, "from": fm.get("layer"),
1000
- "path": e.get("path")})
1001
- if not apply:
1002
- continue
1003
- before = evolution.state_of(cg, nid) or {}
1004
- fm["layer"] = target_layer
1005
- fm["contextualized_from"] = source_layer
1006
- fm["contextualized_at"] = time.time()
1007
- fm["contextualization_basis"] = {"reason": rep["reason"], "batch": batch,
1008
- "actor": actor}
1009
- moved = _relocate_layer(cg, nid, e, fm, content, target_layer)
1010
- evolution.record(
1011
- cg, node_id=nid,
1012
- pattern="情境性内容(批次流水账 / 感知产物)混在知识层 → 归位情境层",
1013
- missing="层归属规则", action=f"层迁移 {source_layer}→{target_layer}",
1014
- evidence=f"prefix={str(nid).split('_')[0]}_ reason={rep['reason']}",
1015
- source="consolidate", kind=evolution.KIND_LAYER_SHIFT,
1016
- before=before, after=evolution.state_of(cg, nid) or {})
1017
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1018
- "t": time.time(), "action": "contextualize", "batch": batch, "id": nid,
1019
- "from": source_layer, "to": target_layer, "path": moved["path"],
1020
- "bucket": moved.get("bucket"), "reason": rep["reason"], "actor": actor})
1021
- rep["moved"].append(nid)
1022
- rep["written"] += 1
1023
- if apply and rep["written"]:
1024
- cg.rebuild_index()
1025
- rep["note"] = ("dry-run:未写盘;apply=True 才归位"
1026
- if not apply else f"已归位 {rep['written']} 个节点到 {target_layer}")
1027
- return rep
1028
-
1029
-
1030
- # 生效条件:给定 root,按 _maintain.jsonl 中 action=contextualize 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
1031
- def rollback_contextualize(root, node_ids=None, batch=None, actor="maintain") -> dict:
1032
- """回滚层归位:按 `_maintain.jsonl` 的 contextualize 记录把节点迁回原层。"""
1033
- cg = MdCGOS(root)
1034
- recs = [r for r in _read_maintain(root)
1035
- if r.get("action") == "contextualize"
1036
- and (not batch or r.get("batch") == batch)
1037
- and (not node_ids or str(r.get("id")) in {str(x) for x in node_ids})]
1038
- if not recs:
1039
- return {"ok": False, "error": "no_records", "reverted": 0}
1040
- reverted, ids = 0, []
1041
- for rec in recs:
1042
- nid = rec["id"]
1043
- e = (cg.index.get("nodes") or {}).get(nid)
1044
- if not e:
1045
- continue
1046
- fm, content = cg._read(e)
1047
- if fm is None or crypto.is_encrypted(content):
1048
- continue
1049
- back = rec.get("from") or "knowledge"
1050
- before = evolution.state_of(cg, nid) or {}
1051
- fm["layer"] = back
1052
- fm["contextualized_from"] = None
1053
- fm["contextualization_basis"] = {"rollback_of": rec.get("batch"),
1054
- "actor": actor}
1055
- _relocate_layer(cg, nid, e, fm, content, back)
1056
- evolution.record(cg, node_id=nid,
1057
- pattern="层归位回滚:情境层迁回原层",
1058
- action=f"层迁移 {rec.get('to')}→{back}",
1059
- evidence=f"rollback batch={rec.get('batch')}",
1060
- source="consolidate", kind=evolution.KIND_ROLLBACK,
1061
- before=before, after=evolution.state_of(cg, nid) or {})
1062
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1063
- "t": time.time(), "action": "contextualize_rollback",
1064
- "batch": rec.get("batch"), "id": nid, "to": back, "actor": actor})
1065
- reverted += 1
1066
- ids.append(nid)
1067
- if reverted:
1068
- cg.rebuild_index()
1069
- return {"ok": True, "reverted": reverted, "ids": ids}
1070
-
1071
-
1072
- # 生效条件:对 _read_maintain(root) 中 action 为 "contextualize" 或 "contextualize_rollback" 的记录,取 recs[-(int(limit) or 50):] 作为 records 返回 {'ok': True, ...}——仅当 int(limit) 成功且为 0 时回落 50,limit 为 None/""/[] 等无法 int() 的值会先抛 TypeError/ValueError。
1073
- def contextualize_history(root, limit=50):
1074
- """层归位的批次记录(只读)。"""
1075
- recs = [r for r in _read_maintain(root)
1076
- if r.get("action") in ("contextualize", "contextualize_rollback")]
1077
- return {"ok": True, "records": recs[-(int(limit) or 50):]}
1078
-
1079
-
1080
- # 生效条件:给定 root,读取 root 下 MAINTAIN_LOG 的 JSONL 并返回记录列表。
1081
- def _read_maintain(root):
1082
- from .fsutil import read_jsonl
1083
- return list(read_jsonl(os.path.join(root, MAINTAIN_LOG)))
1084
-
1085
-
1086
- # ==========================================================================
1087
- # 归纳聚类(consolidate.induce)
1088
- # ==========================================================================
1089
- #
1090
- # 与 promote 的分工:
1091
- # promote —— 把**已经存在**的单条情境记忆升格为长期知识(节点不变,只迁层);
1092
- # induce —— 把**多条**具体记忆归纳为一个**新的概念节点**(新增节点)。
1093
- #
1094
- # 归纳是「由具体到一般」的推理,其输出**不是事实断言**,而是待验证的假设:
1095
- # · 证据基底一律记 inferred(未经验证),不得冒充 verified;
1096
- # · 概念节点必须携带成员清单 + `generalizes`/`instance_of` 对称边,保证可回溯;
1097
- # · 归纳不出「共同条件」时默认**拒绝生成**(没有条件依据的抽象=编造,对齐
1098
- # 纪律 3「不猜测」);确需放宽须显式 require_conditions=False,且概念正文
1099
- # 会写明「未归纳出共同条件」,不掩盖证据缺口。
1100
-
1101
- INDUCE_MIN_CLUSTER = 3
1102
- INDUCE_MIN_JACCARD = 0.30
1103
- INDUCE_MAX_NODES = 400
1104
- INDUCE_MAX_TERMS = 6
1105
- CONCEPT_REL = "generalizes" # concept → member(inferred)
1106
- CONCEPT_MEMBER_REL = "instance_of" # member → concept(inferred)
1107
- CONCEPT_PREFIX = "concept_"
1108
- CONCEPT_IMPORTANCE = 0.5
1109
- CONCEPT_TAGS = ("concept", "induced")
1110
- # 巩固留痕字段(2026-09-19 阶段一):**字段名真源在 md_cg/nodefile.py**,
1111
- # 本处只做短别名引用(非复制),与 `nodefile.VALID_FROM_FIELD` 的登记纪律同构。
1112
- CONSOLIDATED_AT_FIELD = nodefile.CONSOLIDATED_AT_FIELD
1113
- CONSOLIDATED_INTO_FIELD = nodefile.CONSOLIDATED_INTO_FIELD
1114
- # 归纳候选排除:受保护节点,以及洞察/场景/前馈/概念等派生物(避免自我进食)
1115
- INDUCE_SKIP_TAGS = ("insight", "scene", "reconstructed", "gap_hint", "concept")
1116
-
1117
-
1118
- # 生效条件:给定 members,返回 CONCEPT_PREFIX 拼接排序后成员串的 SHA1 前 10 位。
1119
- def _concept_id(members):
1120
- """概念节点 id:由成员清单派生,保证「同成员 ⇒ 同 id」的幂等性。"""
1121
- h = hashlib.sha1("|".join(sorted(str(m) for m in members))
1122
- .encode("utf-8")).hexdigest()
1123
- return CONCEPT_PREFIX + h[:10]
1124
-
1125
-
1126
- # 生效条件:给定 a 与 b,若任一为空集则返回 0.0,否则返回交集大小除以并集大小。
1127
- def _jaccard(a, b):
1128
- if not a or not b:
1129
- return 0.0
1130
- return len(a & b) / float(len(a | b))
1131
-
1132
-
1133
- # 生效条件:给定 term_sets,返回出现次数不小于 max(2, ceil(min_share * len(term_sets))) 的词面排序列表;空输入返回 []。
1134
- def _common_terms(term_sets, min_share=0.6):
1135
- """出现在 ≥ min_share 比例成员中的词面(共同条件);少于 2 个成员共享不算。"""
1136
- if not term_sets:
1137
- return []
1138
- cnt = {}
1139
- for s in term_sets:
1140
- for t in set(s or ()):
1141
- cnt[t] = cnt.get(t, 0) + 1
1142
- need = max(2, int(math.ceil(min_share * len(term_sets))))
1143
- return sorted(t for t, c in cnt.items() if c >= need)
1144
-
1145
-
1146
- # 生效条件:给定 term_sets,按集合排序去重拼接后返回前 limit(默认 INDUCE_MAX_TERMS)个词面。
1147
- def _union_terms(term_sets, limit=INDUCE_MAX_TERMS):
1148
- seen = []
1149
- for s in term_sets:
1150
- for t in sorted(s or ()):
1151
- if t not in seen:
1152
- seen.append(t)
1153
- return seen[:limit]
1154
-
1155
-
1156
- # 生效条件:给定 cg、cid、members、reason、actor、batch,为概念节点与成员节点写对称 inferred 边(已存在则跳过),返回含 concept 与 members 的字典。
1157
- def _link_concept(cg, cid, members, reason, actor, batch):
1158
- """写概念↔成员对称 inferred 边(幂等:已存在则不重复写)。"""
1159
- nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
1160
- out = {"concept": cid, "members": []}
1161
- cnode = cg.get(cid)
1162
- if cnode:
1163
- fm = cnode.get("frontmatter") or {}
1164
- edges = list(fm.get("edges") or [])
1165
- have = {str(e.get("target")) for e in edges if isinstance(e, dict)}
1166
- added = False
1167
- for m in members:
1168
- if m in have:
1169
- continue
1170
- edges.append({"target": m, "relation_type": CONCEPT_REL,
1171
- "reason": reason, "created_at": time.time(),
1172
- "confidence": 0.5, "verified": 0, "evidence": "inferred"})
1173
- added = True
1174
- if added:
1175
- fm["edges"] = edges
1176
- ent = nodes.get(cid) or {}
1177
- cg._write_node(cid, os.path.join(cg.root, ent.get("path") or f"{cid}.md"),
1178
- fm, cnode.get("content") or "")
1179
- if ent:
1180
- ent["edges"] = edges
1181
- for m in members:
1182
- node = cg.get(m)
1183
- if not node:
1184
- continue
1185
- fm = node.get("frontmatter") or {}
1186
- edges = list(fm.get("edges") or [])
1187
- if any(isinstance(e, dict) and str(e.get("target")) == cid for e in edges):
1188
- continue
1189
- edges.append({"target": cid, "relation_type": CONCEPT_MEMBER_REL,
1190
- "reason": reason, "created_at": time.time(),
1191
- "confidence": 0.5, "verified": 0, "evidence": "inferred"})
1192
- fm["edges"] = edges
1193
- # 巩固留痕(2026-09-19 阶段一):`consolidated_into` 为**规范名**,
1194
- # `induced_concept` 保留为历史别名(既有读取面零破坏);`consolidated_at`
1195
- # 补齐**成员侧**巩固时刻——此前只有概念侧 `induced_at`,成员侧无从判定
1196
- # 「何时被并进去」,故「合并后前身可定位」只在概念侧半成立。
1197
- fm[CONSOLIDATED_INTO_FIELD] = cid
1198
- fm[CONSOLIDATED_AT_FIELD] = time.time()
1199
- fm["induced_concept"] = cid
1200
- ent = nodes.get(m) or {}
1201
- cg._write_node(m, os.path.join(cg.root, ent.get("path") or f"{m}.md"),
1202
- fm, node.get("content") or "")
1203
- if ent:
1204
- ent["edges"] = edges
1205
- out["members"].append(m)
1206
- return out
1207
-
1208
-
1209
- # 生效条件:给定 members、common_pos、neg_union,返回标注 inferred 的概念节点正文,含功能名、生效条件、子功能、执行、验证方式、不适用条件。
1210
- def _concept_payload(members, common_pos, neg_union):
1211
- """概念节点正文:把成员的共性条件抽象为可追溯的知识条目(显式标注 inferred)。"""
1212
- label = "、".join(common_pos[:INDUCE_MAX_TERMS])
1213
- pos_txt = ";".join(common_pos[:INDUCE_MAX_TERMS]) or "(未归纳出共同条件)"
1214
- neg_txt = ";".join(neg_union[:INDUCE_MAX_TERMS]) or "(未判定)"
1215
- return (
1216
- "# 功能名:归纳概念:%s\n"
1217
- "# 生效条件:%s\n"
1218
- "# 子功能:%d 条具体记忆的共性(成员:%s)\n"
1219
- "# 执行:由 consolidate.induce 归纳聚合(inferred;未经验证,不得直接当事实使用)\n"
1220
- "# 验证方式:待验证(inferred 假设,需外部证据或实践重复后方可升格)\n"
1221
- "# 不适用条件:%s\n"
1222
- % (label or "共性", pos_txt, len(members), "、".join(members), neg_txt)
1223
- )
1224
-
1225
-
1226
- # 生效条件:给定 cg_or_root,从 source_layer 聚类归纳为 target_layer 概念节点,apply=True 才写盘并返回统计 rep。
1227
- def induce_memories(cg_or_root, source_layer="contextual", target_layer="knowledge",
1228
- min_cluster=INDUCE_MIN_CLUSTER, min_jaccard=INDUCE_MIN_JACCARD,
1229
- max_nodes=INDUCE_MAX_NODES, require_conditions=True,
1230
- limit=None, apply=False, actor="maintain", **extra):
1231
- """归纳聚类:把多条具体记忆归纳为概念层条目(inferred,非事实断言)。
1232
-
1233
- 流程:读取源层 → bigram 相似度贪心聚类 → 提炼共同条件 → 生成概念节点
1234
- (apply=True)→ 写 `generalizes` / `instance_of` 对称 inferred 边 → 写留痕。
1235
-
1236
- apply=False(默认)只出候选报表(可预演);apply=True 才写盘(可留痕、可回溯)。
1237
- 幂等:概念 id 由成员清单派生,同成员重复归纳不新增节点。
1238
- """
1239
- cg = cg_or_root if isinstance(cg_or_root, MdCGOS) else MdCGOS(str(cg_or_root))
1240
- from . import subgraph # 惰性导入:复用统一的条件/词面抽取
1241
-
1242
- # MCP 分发层会把未提供的参数以 None 传入;此处归一化,避免 int(None) 崩溃,
1243
- # 也避免 require_conditions=None 被当成 False 而悄悄关掉「无共同条件即拒绝生成」
1244
- # 这条纪律(默认必须为真,放宽只能显式传 False)。
1245
- min_cluster = INDUCE_MIN_CLUSTER if min_cluster is None else int(min_cluster)
1246
- min_jaccard = INDUCE_MIN_JACCARD if min_jaccard is None else float(min_jaccard)
1247
- max_nodes = INDUCE_MAX_NODES if max_nodes is None else int(max_nodes)
1248
- if require_conditions is None:
1249
- require_conditions = True
1250
-
1251
- nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
1252
- pool = [nid for nid, e in nodes.items()
1253
- if (not source_layer or (e or {}).get("layer") == source_layer)
1254
- and not (e or {}).get("protected")
1255
- and not (set(INDUCE_SKIP_TAGS) & set((e or {}).get("tags") or []))]
1256
- pool.sort()
1257
- truncated = len(pool) > int(max_nodes)
1258
- pool = pool[:int(max_nodes)]
1259
-
1260
- cache = {}
1261
- for nid in pool:
1262
- got = subgraph._node_terms_and_grams(cg, nid)
1263
- if got and got["grams"]:
1264
- cache[nid] = got
1265
- keys = sorted(cache.keys())
1266
-
1267
- rep = {"ok": True, "action": "induce", "op": "consolidate",
1268
- "source_layer": source_layer, "target_layer": target_layer,
1269
- "dry_run": not apply, "nodes_scanned": len(pool), "indexed": len(keys),
1270
- "truncated": truncated, "min_cluster": int(min_cluster),
1271
- "min_jaccard": float(min_jaccard),
1272
- "require_conditions": bool(require_conditions),
1273
- "skipped_small": 0, "skipped_no_condition": 0, "skipped_existing": 0,
1274
- "clusters": 0, "written": 0, "concepts": [], "samples": [],
1275
- "log": MAINTAIN_LOG}
1276
-
1277
- # ---- 贪心聚类(只读) ----
1278
- assigned, proposals = set(), []
1279
- for i, a in enumerate(keys):
1280
- if a in assigned:
1281
- continue
1282
- ga = cache[a]["grams"]
1283
- grp = [b for b in keys[i + 1:]
1284
- if b not in assigned
1285
- and _jaccard(ga, cache[b]["grams"]) >= float(min_jaccard)]
1286
- if len(grp) + 1 < int(min_cluster):
1287
- continue
1288
- members = [a] + grp
1289
- assigned.update(members)
1290
- common_pos = _common_terms([cache[m]["pos"] for m in members])
1291
- if require_conditions and not common_pos:
1292
- rep["skipped_no_condition"] += 1
1293
- continue
1294
- neg_union = _union_terms([cache[m]["neg"] for m in members])
1295
- proposals.append({
1296
- "members": members, "concept_id": _concept_id(members),
1297
- "common_conditions": common_pos, "non_applicable": neg_union,
1298
- "reason": ("%d 条记忆内容相近且共享条件「%s」→ 归纳为概念"
1299
- % (len(members), "、".join(common_pos) or "无")),
1300
- })
1301
- rep["clusters"] = len(proposals)
1302
- for p in proposals[:8]:
1303
- rep["samples"].append(p)
1304
-
1305
- if not apply:
1306
- rep["note"] = ("dry-run:未写盘;apply=True 才生成概念节点与 inferred 边"
1307
- if proposals else "无满足条件的聚类(内容不够相近或缺乏共同条件)")
1308
- rep["concepts"] = [p["concept_id"] for p in proposals]
1309
- return rep
1310
-
1311
- # ---- 落库(可留痕) ----
1312
- batch = time.strftime("%Y%m%d-%H%M%S")
1313
- for p in proposals:
1314
- if limit is not None and rep["written"] >= int(limit):
1315
- break
1316
- cid = p["concept_id"]
1317
- if cid in nodes:
1318
- rep["skipped_existing"] += 1
1319
- continue
1320
- content = _concept_payload(p["members"], p["common_conditions"],
1321
- p["non_applicable"])
1322
- _consolidated_at = time.time() # 概念形成时刻 = 巩固时刻(单一取值,禁两处取时)
1323
- cg.add(cid, content, layer=target_layer, tags=list(CONCEPT_TAGS),
1324
- importance=CONCEPT_IMPORTANCE, verification_basis="other",
1325
- induced_from=list(p["members"]), induced_at=_consolidated_at,
1326
- consolidated_at=_consolidated_at,
1327
- induction={"method": "bigram_jaccard", "min_jaccard": float(min_jaccard),
1328
- "common_conditions": p["common_conditions"], "batch": batch,
1329
- "actor": actor, "evidence": "inferred"},
1330
- actor=actor)
1331
- _link_concept(cg, cid, p["members"], p["reason"], actor, batch)
1332
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1333
- "t": time.time(), "action": "induce", "batch": batch, "concept": cid,
1334
- "members": list(p["members"]), "common_conditions": p["common_conditions"],
1335
- "source_layer": source_layer, "target_layer": target_layer, "actor": actor})
1336
- rep["concepts"].append(cid)
1337
- rep["written"] += 1
1338
- if rep["written"]:
1339
- cg.rebuild_index()
1340
- rep["note"] = (f"已归纳 {rep['written']} 个概念节点(inferred,待验证)"
1341
- if rep["written"] else "无可落库的归纳(均跳过或已达 limit)")
1342
- return rep
1343
-
1344
-
1345
- # ---- CLI ----------------------------------------------------------------
1346
-
1347
- # 生效条件:不适用(无必需形参与模块级常量)
1348
- def _cli(argv=None) -> int:
1349
- ap = argparse.ArgumentParser(
1350
- description="md_cg 离线固化:反思单元(LLM)产出候选 → 白箱闸门 → "
1351
- "验证单元(LLM)否决 → 固化为 md 字段")
1352
- ap.add_argument("--root", required=True, help="md 认知图根目录")
1353
- ap.add_argument("--layer", default=None, help="只处理某层(如 knowledge)")
1354
- ap.add_argument("--limit", type=int, default=None, help="只处理前 N 个待补节点")
1355
- ap.add_argument("--apply", action="store_true", help="真正写盘(默认只验证)")
1356
- ap.add_argument("--dry-run", action="store_true", help="只验证不写盘(默认行为)")
1357
- ap.add_argument("--overwrite", action="store_true",
1358
- help="允许覆盖已有非空字段(默认保护人工既有知识)")
1359
- ap.add_argument("--reflect-model", default=None,
1360
- help=f"反思单元模型(默认 {ROLE_DEFAULT_MODEL[REFLECT_ROLE]})")
1361
- ap.add_argument("--verify-model", default=None,
1362
- help=f"验证单元模型(默认 {ROLE_DEFAULT_MODEL[VERIFY_ROLE]})")
1363
- ap.add_argument("--self-verify", action="store_true",
1364
- help="降级:验证单元复用反思单元模型(非交叉验证,provenance 标记)")
1365
- ap.add_argument("--no-verify", action="store_true",
1366
- help="降级:跳过验证单元,仅靠白箱闸门(不推荐)")
1367
- ap.add_argument("--verification-basis", default=None,
1368
- help="写入 `# 验证方式:` 的声明文本(默认双模型声明)")
1369
- ap.add_argument("--no-basis", action="store_true", help="不写「验证方式」")
1370
- ap.add_argument("--basis-only", action="store_true",
1371
- help="只补「验证方式」(零 LLM 成本),不做四要素反思")
1372
- ap.add_argument("--min-grounding", type=float, default=None,
1373
- help="统一 grounding 阈值(默认按字段 0.5 / 不适用条件 0.34)")
1374
- ap.add_argument("--no-llm", action="store_true",
1375
- help="不调用 LLM,只做四要素完整性普查")
1376
- ap.add_argument("--check", action="store_true",
1377
- help="零 token 探测两个角色网关的可用模型后退出")
1378
- ap.add_argument("--report", default=None, help="把汇总 JSON 另存一份")
1379
- a = ap.parse_args(argv)
1380
-
1381
- r_model, _, r_key = role_config(REFLECT_ROLE, a.reflect_model)
1382
- v_model, _, v_key = role_config(VERIFY_ROLE, a.verify_model)
1383
-
1384
- if a.check:
1385
- out = {"reflect": probe_models(REFLECT_ROLE),
1386
- "verify": probe_models(VERIFY_ROLE)}
1387
- print(json.dumps(out, ensure_ascii=False, indent=2))
1388
- return 0
1389
-
1390
- basis = None if a.no_basis else (
1391
- a.verification_basis or BASIS_TEMPLATE.format(reflect=r_model, verify=v_model))
1392
-
1393
- if a.basis_only:
1394
- rep = fill_verification_basis(a.root, basis, layer=a.layer, limit=a.limit,
1395
- apply=a.apply)
1396
- print(json.dumps(rep, ensure_ascii=False, indent=2))
1397
- if a.report:
1398
- with open(a.report, "w", encoding="utf-8") as f:
1399
- json.dump(rep, f, ensure_ascii=False, indent=2)
1400
- return 0
1401
-
1402
- thresholds = ({f: a.min_grounding for f in CCG_FIELDS}
1403
- if a.min_grounding is not None else None)
1404
-
1405
- reflect_fn = verify_fn = None
1406
- if not a.no_llm:
1407
- if not r_key:
1408
- print(f"[consolidate] 反思单元未配置 key"
1409
- f"({_ROLE_ENV[REFLECT_ROLE][2]} / DEEPSEEK_API_KEY)→ 退化为普查模式",
1410
- file=sys.stderr)
1411
- else:
1412
- reflect_fn = (lambda p: http_llm(p, role=REFLECT_ROLE, # noqa: E731
1413
- model=a.reflect_model))
1414
- if a.self_verify:
1415
- verify_fn = reflect_fn
1416
- elif not a.no_verify:
1417
- if v_key:
1418
- verify_fn = (lambda p: http_llm(p, role=VERIFY_ROLE, # noqa: E731
1419
- model=a.verify_model))
1420
- else:
1421
- print(f"[consolidate] 验证单元未配置 key"
1422
- f"({_ROLE_ENV[VERIFY_ROLE][2]} / ZHIPU_API_KEY / GLM_API_KEY)"
1423
- "→ 待补节点将 DEFER,不写盘(纪律 5:未经验证不固化)",
1424
- file=sys.stderr)
1425
-
1426
- rep = consolidate(a.root, layer=a.layer, limit=a.limit, apply=a.apply,
1427
- overwrite=a.overwrite, reflect_fn=reflect_fn,
1428
- verify_fn=verify_fn, reflect_model=r_model,
1429
- verify_model=v_model if verify_fn else "",
1430
- verification_basis=basis or "",
1431
- require_verify=not a.no_verify, thresholds=thresholds)
1432
- print(json.dumps(rep, ensure_ascii=False, indent=2))
1433
- if a.report:
1434
- with open(a.report, "w", encoding="utf-8") as f:
1435
- json.dump(rep, f, ensure_ascii=False, indent=2)
1436
- return 0
1437
-
1438
-
1439
- if __name__ == "__main__":
1
+ # -*- coding: utf-8 -*-
2
+ """md_cg · 离线固化:LLM 补 CCG 四要素 → 确定性验证 → 固化为 md 字段
3
+
4
+ 为什么是「离线固化」而不是「在线向量」:
5
+ 白箱第 1 篇:相似度可以产生候选,但**不授予执行资格**;资格必须由条件证据裁决。
6
+ LLM 是黑箱,它的输出只能是**候选条件**,不能直接成为检索依据——否则在线检索
7
+ 就被黑箱污染,CCG 28%→88% 的改进会退化回去。故本工具把 LLM 严格限制在
8
+ **离线一次性的固化工序**里:
9
+
10
+ 读节点 → LLM 产出四要素候选 → 确定性验证 → 通过才写进 md 字段
11
+
12
+ 在线检索(search / recall / _path_semantic)仍然只读 md 里已固化的字段,全程白箱。
13
+
14
+ 理论 / 纪律对齐(docs/工作纪律_认知图条目_v1.1.json):
15
+ · 第 3 条 白箱方法:不猜测;**未验证不写入**。
16
+ · 第 5 条 验证纪律:**未经验证不固化**——入库前必须走验证(回放 / 断言 / 回归)。
17
+ · 第 13 条 访谈澄清:节点四要素 = 条件 / 子内容 / 如何执行 / 不适用条件
18
+ ——本工具固化的正是这四个字段(对齐 CCG 的生效条件 / 子功能 / 执行 / 不适用条件)。
19
+ · 《智能的认知过程》:新条件能否**稳定解释误差**?成立 → 纳入知识结构;
20
+ 不成立 → **不固化**,标记为待验证。
21
+ 故 verdict 三态对齐白箱资格判定:ACCEPT(固化)/ REJECT(丢弃)/ DEFER(只存候选)。
22
+
23
+ 验证分三段闸门——前两段确定性零 LLM,第三段是「双模型交叉验证」:
24
+
25
+ 闸门 1 · grounding 支撑度(确定性):候选短语必须能在节点正文里找到字符级依据,
26
+ 否则判为幻觉 → REJECT。(对应「不猜测」)
27
+ 闸门 2 · replay 回放(确定性):把候选条件当作查询,回放生产检索路径的判定:
28
+ · pos_recall 以「生效条件」为查询 → 本节点应被召回,且不被自身负条件挡住;
29
+ · neg_separated 以「不适用条件」为查询 → 应触发条件级负路由,且负条件与正文
30
+ 低相关(负条件必须是「域外」的,不能把知识本身否定掉);
31
+ · no_conflict 生效条件与不适用条件不得互相覆盖。
32
+ 三者同时成立才算「条件稳定」。(对应「回放 / 断言 / 回归」)
33
+ 闸门 3 · 验证单元(GLM,独立模型):逐条核验候选是否有正文依据、负条件是否真域外。
34
+ 硬约束:**验证单元只能否决,不能新增/改写**——它没有产出权,
35
+ 否则验证环节自己就成了新的幻觉源。
36
+
37
+ 回放器复用 _path_semantic 的同一批原语(_declared_conditions / _neg_hit /
38
+ _weighted_coverage / expand_query_terms_weighted),并由 P6 测试与真实
39
+ MdCGOS._path_semantic 做一致性回归,保证不漂移。
40
+
41
+ 双模型角色分工(用户配置,可用环境变量覆盖):
42
+ 反思单元 reflect → 默认 deepseek-flash(实测 2026-09-22 /models 仅
43
+ deepseek-flash / deepseek-v4-pro;原默认
44
+ deepseek-v4.1-flash-expires-on-0910 为限时模型已下架)
45
+ 验证单元 verify → 默认 glm-5.3-flash (IDE 显示名 GLM-5.3-flash)
46
+ 环境变量:MDCG_REFLECT_MODEL/BASE/KEY、MDCG_VERIFY_MODEL/BASE/KEY。
47
+ 注意 IDE 显示名 ≠ API 模型 id;`--check` 可零 token 探测各网关真实 id。
48
+ 两个模型分属不同厂商,避免同源模型的系统性偏见互相印证(交叉验证的本意)。
49
+
50
+ max_tokens 口径(真源:docs/hive/子代理配置标准_v0.5.md §1,2026-09-23 修复 issue #24):
51
+ 思考模型的 reasoning_tokens **计入 max_tokens**(标准 §1 实测:8000 被思考
52
+ 吃满 → content 空 → 假成功;指纹 = usage.completion_tokens≈reasoning_tokens)。
53
+ 故默认 DEFAULT_MAX_TOKENS=200000(标准 §1「思考模型建议 200000」),
54
+ 覆盖链:CLI --max-tokens > env MDCG_LLM_MAX_TOKENS > 常量默认。
55
+ 响应面校验(标准 §1「回收时校验 content 非空而非只看 ok」):finish_reason
56
+ == "length" 或 content 为空 → 明确 RuntimeError(带 max_tokens 与修复指引),
57
+ 绝不静默落成 parse_failed(issue #24 的根因即此静默)。
58
+
59
+ · 验证单元不可用(未配 key)时,默认 **不固化**(DEFER)——纪律 5「未经验证不固化」;
60
+ 确需单模型跑通可显式 --no-verify(provenance 记为 skipped)或 --self-verify
61
+ (同模型自审,provenance 记为 self_verify=true,属于降级模式)。
62
+ · 「验证方式」是 CCG 必需要素,其值 = 声明的验证基底。本工具可写
63
+ `# 验证方式:<声明>`(--verification-basis,默认即上面的双模型声明);
64
+ frontmatter.verification_basis 只能取 nodefile 的枚举
65
+ (compiler/test/measurement/formal_proof/data/other),双 LLM 交叉验证对应 "other"。
66
+ 纯文本声明无需 LLM → --basis-only 可零成本补齐全库(纪律 3:不猜测)。
67
+
68
+ 零第三方依赖(D-005):HTTP 走标准库 urllib.request;reflect_fn / verify_fn 可注入
69
+ (离线可测)。默认 --dry-run,只有 --apply 才写盘(纪律 6:改动可核对)。
70
+ 写盘只动 CCG 字段与 provenance,**不覆盖已有非空字段**(除非 --overwrite)。
71
+ """
72
+ from __future__ import annotations
73
+
74
+ import argparse
75
+ import hashlib
76
+ import json
77
+ import math
78
+ import os
79
+ import re
80
+ import sys
81
+ import time
82
+ import urllib.error
83
+ import urllib.request
84
+
85
+ from . import crypto, evolution, nodefile, routing
86
+ from .fsutil import append_jsonl
87
+ from .mdcg import BUCKETED_LAYERS, bigrams, expand_query_terms_weighted
88
+ from .readcache import direct_read
89
+ from .mdcos import (MdCGOS, _ccg_field, _declared_conditions, _neg_hit, _sig,
90
+ _weighted_coverage)
91
+
92
+ # ---- 常量 ----------------------------------------------------------------
93
+
94
+ # 与工作纪律第 13 条「节点四要素」同构:
95
+ # 生效条件 ↔ conditions(什么时候适用)
96
+ # 子功能 ↔ subgraph / depends_on(子内容)
97
+ # 执行 ↔ execution(如何执行)
98
+ # 不适用条件 ↔ negative(什么时候不适用)
99
+ CCG_FIELDS = ("生效条件", "子功能", "执行", "不适用条件")
100
+ MULTI_FIELDS = ("生效条件", "子功能", "不适用条件") # 列表型
101
+ SINGLE_FIELDS = ("执行",) # 单值型
102
+
103
+ # LLM 常把「子功能」写成「子内容」,别名容错(固化时统一落到标准字段名)
104
+ FIELD_ALIASES = {
105
+ "生效条件": ("生效条件", "适用条件", "conditions", "condition"),
106
+ "子功能": ("子功能", "子内容", "子流程", "subgraph", "sub"),
107
+ "执行": ("执行", "如何执行", "执行方式", "execution", "how"),
108
+ "不适用条件": ("不适用条件", "不适用", "负条件", "negative", "reject"),
109
+ }
110
+
111
+ # 默认 grounding 阈值:不适用条件描述的是「域外」情境,与正文天然低相关,
112
+ # 故阈值放宽;其余三要素必须能在正文里找到实打实的依据。
113
+ DEFAULT_GROUNDING = {"生效条件": 0.5, "子功能": 0.5, "执行": 0.5, "不适用条件": 0.34}
114
+
115
+ MAX_BODY_CHARS = 3000 # 正文截断(控制 token,且条件主要来自开头)
116
+ MAX_TERMS = 8 # 单字段候选条数上限
117
+ MAX_TERM_LEN = 40 # 单条候选长度上限
118
+
119
+ # CCG 声明行(`# 生效条件:…` 等)——它们不是正文,grounding/replay 必须把它们剥掉,
120
+ # 否则已写入的「不适用条件」会在二次运行时被当成正文依据,导致节点自我否定。
121
+ _CCG_LINE_RE = re.compile(
122
+ r"^\s*#\s*(功能名|生效条件|子功能|执行|验证方式|不适用条件)\s*[::]")
123
+
124
+
125
+ # 生效条件:给定 content,返回剔除所有匹配 _CCG_LINE_RE 的行后以换行连接的非声明正文;content 为 None 时按空串处理。
126
+ def body_text(content: str) -> str:
127
+ """剥掉 CCG 声明行后的正文——验证只认正文,不认已写下的声明。"""
128
+ return "\n".join(l for l in (content or "").split("\n")
129
+ if not _CCG_LINE_RE.match(l))
130
+
131
+ # ---- 双模型角色(反思单元 / 验证单元)-------------------------------------
132
+
133
+ REFLECT_ROLE = "reflect" # 反思单元:产出候选
134
+ VERIFY_ROLE = "verify" # 验证单元:否决候选(无产出权)
135
+ ROLES = (REFLECT_ROLE, VERIFY_ROLE)
136
+
137
+ # 推荐模型(真实 API id,经实际调用确认;可用 MDCG_<ROLE>_MODEL 覆盖)
138
+ # 历史:reflect 原默认 deepseek-v4.1-flash-expires-on-0910(限时模型,代码注释曾
139
+ # 预告过期风险)——2026-09-22 实测 /models 仅返回 deepseek-flash 与 deepseek-v4-pro,
140
+ # 限时 id 已下架(issue #24 附带发现)。现行默认 reflect=deepseek-flash(标准 §1
141
+ # 子代理默认档)。
142
+ ROLE_DEFAULT_MODEL = {REFLECT_ROLE: "deepseek-flash",
143
+ VERIFY_ROLE: "glm-5.3-flash"}
144
+
145
+ # max_tokens 默认(真源:docs/hive/子代理配置标准_v0.5.md §1——思考模型建议 200000;
146
+ # reasoning_tokens 计入 max_tokens,预算过小 = content 被思考吃光 = issue #24 根因)。
147
+ # 覆盖链:CLI --max-tokens > env MDCG_LLM_MAX_TOKENS > 本常量。
148
+ DEFAULT_MAX_TOKENS = 200000
149
+ MAX_TOKENS_ENV = "MDCG_LLM_MAX_TOKENS"
150
+ ROLE_DEFAULT_BASE = {REFLECT_ROLE: "https://api.deepseek.com",
151
+ VERIFY_ROLE: "https://open.bigmodel.cn/api/paas/v4"}
152
+ _ROLE_ENV = {REFLECT_ROLE: ("MDCG_REFLECT_MODEL", "MDCG_REFLECT_BASE", "MDCG_REFLECT_KEY"),
153
+ VERIFY_ROLE: ("MDCG_VERIFY_MODEL", "MDCG_VERIFY_BASE", "MDCG_VERIFY_KEY")}
154
+ # 验证单元 key 的常见别名(智谱系)
155
+ VERIFY_KEY_ALIASES = ("ZHIPU_API_KEY", "ZHIPUAI_API_KEY", "GLM_API_KEY", "BIGMODEL_API_KEY")
156
+
157
+ # 「验证方式」的声明文本(可 --verification-basis 覆盖 / --no-basis 关闭)
158
+ BASIS_TEMPLATE = "双模型交叉验证(反思单元={reflect},验证单元={verify})"
159
+ # frontmatter.verification_basis 只能取 nodefile 的枚举;双 LLM 交叉验证 → other
160
+ BASIS_ENUM_DEFAULT = "other"
161
+
162
+ REFLECT_PROMPT = (
163
+ "你是认知图节点的**反思单元**。给定一个知识节点的标题与正文,反思并抽取四要素。\n"
164
+ "只输出一个 JSON 对象,不要任何解释或代码围栏。\n"
165
+ "字段含义:\n"
166
+ ' "生效条件": 什么查询/情境下该知识**适用**(短语数组,2~5 条)\n'
167
+ ' "子功能": 该知识包含的子内容/子步骤(短语数组,2~5 条)\n'
168
+ ' "执行": 如何执行/如何使用该知识(单个字符串)\n'
169
+ ' "不适用条件": 什么查询/情境下该知识**不**适用(短语数组,1~3 条)\n'
170
+ "硬约束:\n"
171
+ " 1. 每条短语必须能在正文中找到依据,禁止编造正文里没有的工具/概念;\n"
172
+ " 2. 不适用条件必须是**正文之外的邻近易混情境**,不得与生效条件语义重叠;\n"
173
+ " 3. 短语要短(不超过 20 字),不要写完整句子。\n"
174
+ "输出格式:"
175
+ '{{"生效条件": ["..."], "子功能": ["..."], "执行": "...", "不适用条件": ["..."]}}\n'
176
+ "标题:{title}\n正文:\n{body}"
177
+ )
178
+
179
+ VERIFY_PROMPT = (
180
+ "你是认知图节点的**验证单元**。你的职责是**否决**,不是补充。\n"
181
+ "只能从候选里删除不成立的条目,**绝不允许新增或改写任何条目**。\n"
182
+ "给定标题、正文与反思单元给出的候选四要素,逐条核验:\n"
183
+ " · 该条目是否真的能在正文中找到依据?找不到依据 → 删除;\n"
184
+ " · 不适用条件是否真的域外?若它其实是该节点的适用情境 → 删除;\n"
185
+ " · 生效条件与不适用条件是否语义重叠?重叠者删除其一(保留更贴合正文的那个)。\n"
186
+ "只输出一个 JSON 对象,键为字段名,值为 "
187
+ '{{"keep": ["保留的条目"], "drop": ["删除的条目"], "reason": "一句话理由"}}。\n'
188
+ "候选:{cand}\n标题:{title}\n正文:\n{body}"
189
+ )
190
+
191
+
192
+ # ---- LLM 侧(黑箱只在离线工序,产出候选)--------------------------------
193
+
194
+ # 生效条件:给定 role,按显式参数、角色环境变量、通用兜底依次解析并返回 (model, base, key);key 不落 DEEPSEEK_API_KEY 除非 role 为 REFLECT_ROLE。
195
+ def role_config(role: str, model: str = None, base: str = None,
196
+ key: str = None) -> tuple:
197
+ """解析某角色的 (model, base, key):显式参数 > 角色环境变量 > 通用兜底。
198
+
199
+ key 刻意**不**让验证单元回落到 DEEPSEEK_API_KEY——跨厂商混用会把一个厂商的
200
+ 凭证发到另一个厂商的网关,既必然失败又构成凭证外泄。
201
+ """
202
+ m_env, b_env, k_env = _ROLE_ENV[role]
203
+ if key is None:
204
+ key = os.environ.get(k_env)
205
+ if key is None and role == VERIFY_ROLE:
206
+ for alias in VERIFY_KEY_ALIASES:
207
+ key = os.environ.get(alias)
208
+ if key:
209
+ break
210
+ if key is None:
211
+ key = os.environ.get("MDCG_LLM_KEY")
212
+ if key is None and role == REFLECT_ROLE:
213
+ key = os.environ.get("DEEPSEEK_API_KEY")
214
+ model = (model or os.environ.get(m_env) or os.environ.get("MDCG_LLM_MODEL")
215
+ or ROLE_DEFAULT_MODEL[role])
216
+ base = (base or os.environ.get(b_env) or os.environ.get("MDCG_LLM_BASE")
217
+ or ROLE_DEFAULT_BASE[role])
218
+ return model, base, key
219
+
220
+
221
+ # 生效条件:给定显式 max_tokens 与 env 值,按 显式参数 > env > DEFAULT_MAX_TOKENS 解析:显式为正整数直接返回;env 可解析且 >0 返回之;否则返回 DEFAULT_MAX_TOKENS;env 存在但非法或非正时忽略(回落默认,不炸批处理)。
222
+ def resolve_max_tokens(explicit: int = None) -> int:
223
+ """max_tokens 三级解析:显式参数 > MDCG_LLM_MAX_TOKENS > DEFAULT_MAX_TOKENS。
224
+
225
+ 为什么默认这么大(200000):思考模型的 reasoning_tokens 计入 max_tokens
226
+ (标准 §1 实测),预算过小 = content 被思考吃光。max_tokens 是预算上限
227
+ 而非计费量——取宽不取窄,实际用量计费不受影响。
228
+ """
229
+ if explicit is not None:
230
+ return max(1, int(explicit))
231
+ raw = (os.environ.get(MAX_TOKENS_ENV) or "").strip()
232
+ if raw:
233
+ try:
234
+ v = int(raw)
235
+ if v > 0:
236
+ return v
237
+ except ValueError:
238
+ pass # 非法 env 不炸批处理,回落默认
239
+ return DEFAULT_MAX_TOKENS
240
+
241
+
242
+ # 生效条件:给定响应 data 与 model,choices 为空、message.content 为空/None、或 finish_reason=="length" 时抛 RuntimeError(含 max_tokens/finish_reason/usage 指纹与修复指引),否则返回 content 字符串。
243
+ def _extract_content(data: dict, model: str, max_tokens: int) -> str:
244
+ """响应面校验(标准 §1:「回收时校验 content 非空而非只看 ok」)。
245
+
246
+ 三种坏形态都不许静默落成 parse_failed(issue #24 根因):
247
+ · choices 空 → 网关异常形态;
248
+ · content 空 → 假成功(指纹:usage.completion_tokens≈reasoning_tokens,
249
+ 思考预算吃光 content——标准 §1 同病实测);
250
+ · finish_reason=="length" → 预算耗尽被截断(截断的 JSON 必然解析失败)。
251
+ """
252
+ choices = data.get("choices") or []
253
+ if not choices:
254
+ raise RuntimeError(
255
+ f"LLM 响应无 choices(model={model}, max_tokens={max_tokens}):"
256
+ "网关异常形态,原样返回体前 300 字符:"
257
+ f"{json.dumps(data, ensure_ascii=False)[:300]}")
258
+ msg = choices[0].get("message") or {}
259
+ finish = choices[0].get("finish_reason")
260
+ usage = data.get("usage") or {}
261
+ if finish == "length":
262
+ raise RuntimeError(
263
+ f"输出预算耗尽(finish_reason=length, model={model}, "
264
+ f"max_tokens={max_tokens}, completion_tokens={usage.get('completion_tokens')})"
265
+ "——思考模型的 reasoning_tokens 计入 max_tokens(子代理配置标准 v0.5 §1)。"
266
+ f"修复:调大 --max-tokens 或 env {MAX_TOKENS_ENV}")
267
+ content = msg.get("content") or ""
268
+ if not content.strip():
269
+ raise RuntimeError(
270
+ f"LLM 返回空 content(model={model}, finish_reason={finish}, "
271
+ f"max_tokens={max_tokens}, "
272
+ f"completion_tokens={usage.get('completion_tokens')})——"
273
+ "假成功指纹:completion_tokens≈reasoning_tokens 表示思考吃满预算;"
274
+ f"修复:调大 --max-tokens 或 env {MAX_TOKENS_ENV}")
275
+ return content
276
+
277
+
278
+ # 生效条件:给定 prompt 且 role 解析或通用兜底得到非空 key 时,向 base 的 /chat/completions 发 POST,max_tokens 按 resolve_max_tokens 解析(显式 > env > 默认 200000),经 _extract_content 校验后返回 content;key 为空则抛 RuntimeError。
279
+ def http_llm(prompt: str, model: str = None, base: str = None, key: str = None,
280
+ role: str = None, timeout: int = 120, max_tokens: int = None) -> str:
281
+ """标准库 HTTP 调 LLM(OpenAI 兼容 /chat/completions)。零第三方依赖。
282
+
283
+ role 给定时按该角色配置解析(reflect / verify),否则走通用配置。
284
+ max_tokens=None 走三级解析(显式 > MDCG_LLM_MAX_TOKENS > DEFAULT_MAX_TOKENS);
285
+ 历史 bug(issue #24):曾硬编码 1200——思考模型 reasoning 吃光预算,
286
+ content 空串静默落成 parse_failed,离线固化 100% DEFER。
287
+ """
288
+ if role:
289
+ model, base, key = role_config(role, model, base, key)
290
+ else:
291
+ model = (model or os.environ.get("MDCG_LLM_MODEL")
292
+ or ROLE_DEFAULT_MODEL[REFLECT_ROLE])
293
+ base = (base or os.environ.get("MDCG_LLM_BASE")
294
+ or ROLE_DEFAULT_BASE[REFLECT_ROLE])
295
+ key = (key or os.environ.get("MDCG_LLM_KEY")
296
+ or os.environ.get("DEEPSEEK_API_KEY"))
297
+ if not key:
298
+ raise RuntimeError(
299
+ f"未配置 {role or 'llm'} 的 API key"
300
+ f"({_ROLE_ENV[role][2] if role in _ROLE_ENV else 'MDCG_LLM_KEY'})")
301
+ max_tokens = resolve_max_tokens(max_tokens)
302
+ payload = json.dumps({
303
+ "model": model,
304
+ "messages": [{"role": "user", "content": prompt}],
305
+ "max_tokens": max_tokens,
306
+ }).encode("utf-8")
307
+ req = urllib.request.Request(
308
+ base.rstrip("/") + "/chat/completions", data=payload,
309
+ headers={"Authorization": f"Bearer {key}",
310
+ "Content-Type": "application/json"})
311
+ try:
312
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
313
+ data = json.loads(resp.read().decode("utf-8"))
314
+ except urllib.error.HTTPError as exc:
315
+ detail = exc.read().decode("utf-8", "replace")[:300]
316
+ raise RuntimeError(f"HTTP {exc.code} model={model} base={base} :: {detail}") from None
317
+ return _extract_content(data, model, max_tokens)
318
+
319
+
320
+ # 生效条件:给定 role,若 role_config 得到非空 key 则 GET base/models 并返回含 ok/model_available/models 的字典;无 key 或请求异常则返回 ok=False 及错误信息。
321
+ def probe_models(role: str, timeout: int = 20) -> dict:
322
+ """零 token 探测:列出该角色网关的可用模型 id(GET /models)。"""
323
+ model, base, key = role_config(role)
324
+ if not key:
325
+ return {"role": role, "model": model, "base": base, "ok": False,
326
+ "error": "no_key", "models": []}
327
+ req = urllib.request.Request(
328
+ base.rstrip("/") + "/models",
329
+ headers={"Authorization": f"Bearer {key}"})
330
+ try:
331
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
332
+ data = json.loads(resp.read().decode("utf-8"))
333
+ ids = [m.get("id") for m in (data.get("data") or []) if m.get("id")]
334
+ except Exception as exc: # noqa: BLE001 —— 探测要抗单点
335
+ return {"role": role, "model": model, "base": base, "ok": False,
336
+ "error": f"{type(exc).__name__}: {exc}"[:200], "models": []}
337
+ res = {"role": role, "model": model, "base": base, "ok": True,
338
+ "model_available": model in ids, "models": ids}
339
+ if not res["model_available"]:
340
+ res["note"] = ("该 id 未出现在 /models 列表:可能是限时/按需模型,或已下架;"
341
+ "以实际 /chat/completions 调用结果为准")
342
+ return res
343
+
344
+
345
+ # 生效条件:raw 为 None 或 strip 后不含 "{"(i<0)、或末个 "}" 的位置 j<=i 时返回 None;否则对 s 从首个 "{" 到末个 "}" 的切片 json.loads,成功则返回解析结果,抛 ValueError 时返回 None。
346
+ def _extract_json_obj(raw: str):
347
+ """从 LLM 输出里抠出第一个 JSON 对象(容忍代码围栏 / 前后废话)。"""
348
+ s = (raw or "").strip()
349
+ i, j = s.find("{"), s.rfind("}")
350
+ if i < 0 or j <= i:
351
+ return None
352
+ try:
353
+ return json.loads(s[i:j + 1])
354
+ except ValueError:
355
+ return None
356
+
357
+
358
+ # 生效条件:v 为 None 返回 [];否则按 v 是 str 取 [v]、是 list/tuple 取逐项、其他取 [str(v)],逐项 strip 并去两端包裹标点后跳过空串及长度超 MAX_TERM_LEN 的项,未出现过的才 append,每次 append 后若 len(out) >= limit 即 break 返回 out(故 limit<=0 且存在有效项时仍返回 1 项)。
359
+ def _as_terms(v, limit: int = MAX_TERMS):
360
+ """把 LLM 给的值规范成去重、限长的短语列表。"""
361
+ if v is None:
362
+ return []
363
+ if isinstance(v, str):
364
+ items = [v]
365
+ elif isinstance(v, (list, tuple)):
366
+ items = list(v)
367
+ else:
368
+ items = [str(v)]
369
+ out = []
370
+ for x in items:
371
+ s = str(x).strip().strip(",。;;、,.;\"'“”")
372
+ if not s or len(s) > MAX_TERM_LEN:
373
+ continue
374
+ if s not in out:
375
+ out.append(s)
376
+ if len(out) >= limit:
377
+ break
378
+ return out
379
+
380
+
381
+ # 生效条件:给定 raw,若 _extract_json_obj 解析出 dict,则按 CCG_FIELDS 与 FIELD_ALIASES 提取非空字段并规范为列表或单值返回字典;否则返回 {}。
382
+ def parse_candidate(raw: str) -> dict:
383
+ """LLM 原始输出 → {字段: 列表/字符串};解析失败返回 {}。"""
384
+ obj = _extract_json_obj(raw)
385
+ if not isinstance(obj, dict):
386
+ return {}
387
+ out = {}
388
+ for field in CCG_FIELDS:
389
+ val = None
390
+ for alias in FIELD_ALIASES[field]:
391
+ if alias in obj and obj[alias] not in (None, "", [], {}):
392
+ val = obj[alias]
393
+ break
394
+ if val is None:
395
+ continue
396
+ if field in SINGLE_FIELDS:
397
+ terms = _as_terms(val, limit=1)
398
+ if terms:
399
+ out[field] = terms[0]
400
+ else:
401
+ terms = _as_terms(val)
402
+ if terms:
403
+ out[field] = terms
404
+ return out
405
+
406
+
407
+ # 生效条件:给定 raw,若解析出 dict,则按 CCG_FIELDS 提取 keep/drop/has_keep/reason 结构返回字典;否则返回 {}。
408
+ def parse_verdict(raw: str) -> dict:
409
+ """验证单元输出 → {字段: {keep, drop, has_keep, reason}};解析失败返回 {}。"""
410
+ obj = _extract_json_obj(raw)
411
+ if not isinstance(obj, dict):
412
+ return {}
413
+ out = {}
414
+ for field in CCG_FIELDS:
415
+ val = None
416
+ for alias in FIELD_ALIASES[field]:
417
+ if alias in obj and obj[alias] not in (None, "", [], {}):
418
+ val = obj[alias]
419
+ break
420
+ if val is None:
421
+ continue
422
+ if isinstance(val, list): # 容忍只给 keep 数组
423
+ out[field] = {"keep": _as_terms(val), "drop": [], "has_keep": True,
424
+ "reason": ""}
425
+ elif isinstance(val, dict):
426
+ out[field] = {"keep": _as_terms(val.get("keep")),
427
+ "drop": _as_terms(val.get("drop")),
428
+ "has_keep": "keep" in val,
429
+ "reason": str(val.get("reason") or "")[:200]}
430
+ return out
431
+
432
+
433
+ # 生效条件:给定 kept 与 verdict,若 verdict 为空则返回 (dict(kept), {});否则按 drop 与 has_keep 收窄候选并返回 (收窄后候选, 被剔除明细)。
434
+ def narrow_by_verdict(kept: dict, verdict: dict):
435
+ """按验证单元裁决收窄候选——**只能否决,不能新增**。
436
+
437
+ · 验证单元未表态的字段 → 保留(沉默不等于否决)
438
+ · has_keep=True → 取「候选 ∩ keep」;否则只按 drop 剔除
439
+ 返回 (收窄后候选, 被剔除明细)。
440
+ """
441
+ if not verdict:
442
+ return dict(kept), {}
443
+ out, dropped = {}, {}
444
+ for field, val in kept.items():
445
+ terms = val if isinstance(val, list) else [val]
446
+ vd = verdict.get(field)
447
+ if vd is None:
448
+ out[field] = val
449
+ continue
450
+ dropset = set(vd.get("drop") or [])
451
+ keepset = set(vd.get("keep") or [])
452
+ surv, gone = [], []
453
+ for t in terms:
454
+ if t in dropset:
455
+ gone.append(t)
456
+ elif vd.get("has_keep") and t not in keepset:
457
+ gone.append(t)
458
+ else:
459
+ surv.append(t)
460
+ if gone:
461
+ dropped[field] = {"terms": gone, "reason": vd.get("reason") or ""}
462
+ if surv:
463
+ out[field] = surv if field in MULTI_FIELDS else surv[0]
464
+ return out, dropped
465
+
466
+
467
+ # ---- 确定性验证(零 LLM)-------------------------------------------------
468
+
469
+ # 生效条件:给定 term 与 body,若 term 的 bigram 序列非空则返回命中 bigram 数除以总 bigram 数,否则返回 0.0。
470
+ def grounding_score(term: str, body: str) -> float:
471
+ """候选短语在正文里的字符级支撑度 = 命中 bigram 数 / 总 bigram 数。"""
472
+ bg = bigrams(term or "")
473
+ if not bg:
474
+ return 0.0
475
+ hit = sum(1 for g in bg if g in (body or ""))
476
+ return hit / len(bg)
477
+
478
+
479
+ # 生效条件:给定 cand 与 body,按 thresholds 更新 DEFAULT_GROUNDING 后逐字段过滤候选,返回 (达标 kept, detail);不达标者丢弃。
480
+ def grounding_filter(cand: dict, body: str, thresholds: dict = None):
481
+ """逐字段过滤候选:返回 (kept, detail)。不达标者丢弃(对应「不猜测」)。"""
482
+ th = dict(DEFAULT_GROUNDING)
483
+ th.update(thresholds or {})
484
+ kept, detail = {}, {}
485
+ for field, val in cand.items():
486
+ terms = val if isinstance(val, list) else [val]
487
+ ok_terms, scores = [], {}
488
+ for t in terms:
489
+ g = grounding_score(t, body)
490
+ scores[t] = round(g, 3)
491
+ if g >= th.get(field, 0.5):
492
+ ok_terms.append(t)
493
+ detail[field] = {"scores": scores, "kept": len(ok_terms)}
494
+ if ok_terms:
495
+ kept[field] = ok_terms if field in MULTI_FIELDS else ok_terms[0]
496
+ return kept, detail
497
+
498
+
499
+ # 生效条件:给定 pos_terms、neg_terms、body,返回含 pos_recall、neg_separated、no_conflict、ok 的回放判定字典。
500
+ def replay_check(pos_terms, neg_terms, body: str) -> dict:
501
+ """回放生产判定:正例召回 + 负例剔除 + 无自相矛盾。
502
+
503
+ 复用 _path_semantic 的同一批原语,保证与生产路同源(P6 与真实路做一致性回归)。
504
+ """
505
+ pos_text = " ".join(pos_terms or [])
506
+ neg_text = " ".join(neg_terms or [])
507
+ tw_pos = expand_query_terms_weighted(pos_text) if pos_text else {}
508
+ tw_neg = expand_query_terms_weighted(neg_text) if neg_text else {}
509
+
510
+ # 1. 正例:以生效条件为查询,本节点正文应被命中,且不被自身负条件挡住
511
+ pos_recall = bool(pos_text) and _weighted_coverage(tw_pos, body) > 0.0 \
512
+ and not _neg_hit(tw_pos, neg_terms)
513
+
514
+ # 2. 负例:以不适用条件为查询,应触发条件级负路由;且负条件与正文低相关
515
+ # (负条件必须是「域外」的,若与正文强相关,等于让知识否定自己)
516
+ if neg_terms:
517
+ neg_separated = _neg_hit(tw_neg, neg_terms) \
518
+ and _weighted_coverage(tw_neg, body) < 0.5
519
+ else:
520
+ neg_separated = True
521
+
522
+ # 3. 生效条件与不适用条件不得互相覆盖
523
+ no_conflict = (not pos_text) or (not neg_text) \
524
+ or _weighted_coverage(tw_pos, neg_text) < 0.5
525
+
526
+ ok = pos_recall and neg_separated and no_conflict
527
+ return {"pos_recall": pos_recall, "neg_separated": neg_separated,
528
+ "no_conflict": no_conflict, "ok": ok}
529
+
530
+
531
+ # ---- 写盘(固化)---------------------------------------------------------
532
+
533
+ # 生效条件:给定 content 与 field,当 content 含 "# field:" 或 "# field:" 时返回 True,否则 False。
534
+ def _has_ccg_line(content: str, field: str) -> bool:
535
+ return f"# {field}:" in (content or "") or f"# {field}:" in (content or "")
536
+
537
+
538
+ # 生效条件:给定 fm 与 content,对每个 CCG_FIELDS,若 frontmatter.comment 值非空或正文含对应 CCG 行则记入,返回已有字段字典。
539
+ def existing_fields(fm: dict, content: str) -> dict:
540
+ """节点当前已有的四要素:正文 CCG 行 或 frontmatter.comment 任一存在即算有。"""
541
+ comment = (fm.get("state_attributes") or {}).get("comment") or {}
542
+ out = {}
543
+ for field in CCG_FIELDS:
544
+ v = comment.get(field)
545
+ if v not in (None, "", [], {}):
546
+ out[field] = v
547
+ elif _has_ccg_line(content, field):
548
+ out[field] = True
549
+ return out
550
+
551
+
552
+ # 生效条件:给定 content、field、value,若已有 "# field:" 行则替换并返回新正文;否则插在 "# 功能名" 之后,若无则该行前置。
553
+ def _upsert_ccg_line(content: str, field: str, value: str) -> str:
554
+ """在正文里写入/替换 `# <字段>:<值>`,优先插在「# 功能名」之后。"""
555
+ lines = (content or "").split("\n")
556
+ for i, ln in enumerate(lines):
557
+ s = ln.strip()
558
+ if not s.startswith("#") or field not in s:
559
+ continue
560
+ name = s.lstrip("#").strip().split(":")[0].split(":")[0].strip()
561
+ if name == field:
562
+ lines[i] = f"# {field}:{value}"
563
+ return "\n".join(lines)
564
+ newline = f"# {field}:{value}"
565
+ for i, ln in enumerate(lines):
566
+ if ln.strip().startswith("# 功能名"):
567
+ lines.insert(i + 1, newline)
568
+ return "\n".join(lines)
569
+ return newline + "\n" + (content or "")
570
+
571
+
572
+ # 生效条件:给定 kept 字段字典,返回一句话规律字符串,列出缺失字段名并声明补齐后可路由。
573
+ def _evo_pattern(kept: dict) -> str:
574
+ """规律(一句话):这一类节点反复缺的正是这批条件。"""
575
+ names = "、".join(kept.keys())
576
+ return f"缺「{names}」的节点条件不可判;补齐后四要素完整、可路由"
577
+
578
+
579
+ # 生效条件:给定 prov 字典,拼接 reflect/verify 模型、grounding、replay、verification_basis 中存在的证据项并返回。
580
+ def _evo_evidence(prov: dict) -> str:
581
+ """证据:本次固化凭什么成立(模型 / 闸门 / 回放)。"""
582
+ parts = []
583
+ rf = (prov.get("reflect") or {}).get("model") or ""
584
+ vf = (prov.get("verify") or {}).get("model") or ""
585
+ if rf:
586
+ parts.append(f"reflect={rf}")
587
+ if vf:
588
+ parts.append(f"verify={vf}")
589
+ if prov.get("grounding"):
590
+ parts.append("grounding通过")
591
+ if prov.get("replay"):
592
+ parts.append("replay通过")
593
+ vb = prov.get("verification_basis") or ""
594
+ if vb:
595
+ parts.append(vb)
596
+ return " · ".join(parts)
597
+
598
+
599
+ # 生效条件:给定 cg、e、fm、content、kept、prov,将 kept 字段写入正文 CCG 行与 frontmatter.comment,不适用条件同步 non_applicable_conditions,并写 llm_consolidation 与演化记录,返回 None。
600
+ def _apply_node(cg, e, fm: dict, content: str, kept: dict, prov: dict,
601
+ basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT):
602
+ """把通过验证的字段固化进 md:正文 CCG 行 + frontmatter.comment + 负条件 + provenance。
603
+
604
+ 固化 = 对一条缺失条件的补充 → 同步落一条演化条目(md 账本,可回滚)。
605
+ """
606
+ nid = e.get("id") or os.path.basename(e["path"])[:-3]
607
+ before = evolution.state_of(cg, nid) or {}
608
+ comment = (fm.get("state_attributes") or {}).get("comment")
609
+ if not isinstance(comment, dict):
610
+ fm["state_attributes"] = dict(fm.get("state_attributes") or {})
611
+ fm["state_attributes"]["comment"] = {}
612
+ comment = fm["state_attributes"]["comment"]
613
+ for field, val in kept.items():
614
+ text = ";".join(val) if isinstance(val, list) else str(val)
615
+ content = _upsert_ccg_line(content, field, text)
616
+ comment[field] = text
617
+ if field == "不适用条件":
618
+ # 同步 frontmatter.non_applicable_conditions(引擎负路由读它)
619
+ cur = [str(x) for x in (fm.get("non_applicable_conditions") or [])]
620
+ for t in (val if isinstance(val, list) else [val]):
621
+ if t not in cur:
622
+ cur.append(t)
623
+ fm["non_applicable_conditions"] = cur
624
+ if basis:
625
+ content = _upsert_ccg_line(content, "验证方式", basis)
626
+ comment["验证方式"] = basis
627
+ if not nodefile.verification_basis_valid(fm):
628
+ # 枚举里没有「LLM 交叉验证」这一档,只能落到 other(声明文本在 CCG 行里)
629
+ fm["verification_basis"] = basis_enum
630
+ fm["llm_consolidation"] = prov
631
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
632
+ durable=True)
633
+ # 每一次修改都是对缺失条件的补充:记录规律 + 状态,不记录实现。
634
+ evolution.record(
635
+ cg, node_id=nid,
636
+ pattern=_evo_pattern(kept),
637
+ missing="、".join(kept.keys()),
638
+ action="补齐 CCG 字段:" + "、".join(kept.keys()),
639
+ evidence=_evo_evidence(prov),
640
+ source="consolidate", kind=evolution.KIND_CONDITION_GAP,
641
+ before=before, after=evolution.state_of(cg, nid) or {})
642
+
643
+
644
+ # 生效条件:给定 root,扫描正排层节点并执行反思→白箱闸门→验证→固化,返回报表 rep;require_verify=True 且无 verify_fn 时全部 DEFER。
645
+ def consolidate(root: str, layer: str = None, limit: int = None, apply: bool = False,
646
+ overwrite: bool = False, llm_fn=None, reflect_fn=None,
647
+ verify_fn=None, reflect_model: str = "", verify_model: str = "",
648
+ verification_basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT,
649
+ require_verify: bool = True, thresholds: dict = None,
650
+ verbose: bool = True) -> dict:
651
+ """对正排层节点做「反思单元产出候选 → 白箱闸门 → 验证单元否决 → 固化」。
652
+
653
+ llm_fn 是 reflect_fn 的旧名(向后兼容,单模型模式)。
654
+ require_verify=True 且无 verify_fn → 一律 DEFER(纪律 5:未经验证不固化)。
655
+ """
656
+ reflect_fn = reflect_fn or llm_fn
657
+ cg = MdCGOS(root)
658
+ entries = cg._candidates(layer=layer)
659
+ t0 = time.time()
660
+ rep = {"root": root, "layer": layer, "dry_run": not apply,
661
+ "reflect_model": reflect_model, "verify_model": verify_model,
662
+ "reflect": bool(reflect_fn), "verify": bool(verify_fn), "llm": bool(reflect_fn),
663
+ "require_verify": require_verify,
664
+ "verification_basis": verification_basis,
665
+ "nodes_scanned": len(entries),
666
+ "targeted": 0, "accepted": 0, "rejected": 0, "deferred": 0,
667
+ "skipped_complete": 0, "written": 0, "reasons": {},
668
+ "per_field": {f: 0 for f in CCG_FIELDS}, "verify_dropped": 0,
669
+ "verification_basis_missing": 0, "samples": []}
670
+
671
+ # 生效条件:以 reason 为键写入闭包 rep["reasons"],计数按 rep["reasons"].get(reason, 0) + 1 递增(键缺失从 0 起算),无返回值。
672
+ def _bump(reason):
673
+ rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
674
+
675
+ for e in entries:
676
+ if limit is not None and rep["targeted"] >= limit:
677
+ break
678
+ nid = os.path.basename(e["path"])[:-3]
679
+ fm, content = direct_read(cg, e) # 写前重查走盘上真值(读缓存口径,见 readcache.direct_read)
680
+ if fm is None:
681
+ _bump("read_failed")
682
+ continue
683
+ if crypto.is_encrypted(content):
684
+ _bump("locked") # 无密钥 → fail-closed:绝不改写密文
685
+ continue
686
+ have = existing_fields(fm, content)
687
+ missing = [f for f in CCG_FIELDS if f not in have]
688
+ if not nodefile.verification_basis_valid(fm):
689
+ rep["verification_basis_missing"] += 1
690
+ if not missing:
691
+ rep["skipped_complete"] += 1
692
+ continue
693
+ rep["targeted"] += 1
694
+
695
+ if not reflect_fn:
696
+ rep["deferred"] += 1
697
+ _bump("no_llm")
698
+ continue
699
+
700
+ # 1) 反思单元:产出候选(黑箱,唯一产出权)
701
+ body = body_text(content)[:MAX_BODY_CHARS]
702
+ title = _ccg_field(content, "功能名") or nid
703
+ prompt = REFLECT_PROMPT.format(title=title, body=body)
704
+ try:
705
+ raw = reflect_fn(prompt)
706
+ except Exception as exc: # noqa: BLE001 —— 离线批处理要抗单点失败
707
+ rep["deferred"] += 1
708
+ _bump(f"reflect_error:{type(exc).__name__}")
709
+ continue
710
+ cand = parse_candidate(raw)
711
+ if not cand:
712
+ rep["deferred"] += 1
713
+ _bump("parse_failed")
714
+ continue
715
+
716
+ # 2) 白箱闸门:grounding + replay(零 LLM,先跑,省调用)
717
+ kept, gdetail = grounding_filter(cand, body, thresholds)
718
+ pos = kept.get("生效条件") or []
719
+ neg = kept.get("不适用条件") or []
720
+ replay = replay_check(pos, neg, body)
721
+ if not kept or not replay["ok"]:
722
+ rep["rejected"] += 1
723
+ _bump("replay_failed" if kept else "grounding_failed")
724
+ if verbose and len(rep["samples"]) < 8:
725
+ rep["samples"].append({"id": nid, "verdict": "REJECT",
726
+ "stage": "whitebox", "grounding": gdetail,
727
+ "replay": replay})
728
+ continue
729
+
730
+ # 3) 验证单元:逐条核验,只能否决、不能新增
731
+ dropped, vprompt, vd = {}, "", None
732
+ if verify_fn:
733
+ vprompt = VERIFY_PROMPT.format(
734
+ cand=json.dumps(kept, ensure_ascii=False), title=title, body=body)
735
+ try:
736
+ vd = parse_verdict(verify_fn(vprompt))
737
+ kept, dropped = narrow_by_verdict(kept, vd)
738
+ except Exception as exc: # noqa: BLE001
739
+ rep["deferred"] += 1
740
+ _bump(f"verify_error:{type(exc).__name__}")
741
+ continue
742
+ if not kept:
743
+ rep["rejected"] += 1
744
+ _bump("verify_rejected")
745
+ if verbose and len(rep["samples"]) < 8:
746
+ rep["samples"].append({"id": nid, "verdict": "REJECT",
747
+ "stage": "verify", "dropped": dropped})
748
+ continue
749
+ rep["verify_dropped"] += sum(len(d["terms"]) for d in dropped.values())
750
+ elif require_verify:
751
+ # 验证单元不可用 → 不固化(纪律 5:未经验证不固化)
752
+ rep["deferred"] += 1
753
+ _bump("verify_unavailable")
754
+ continue
755
+
756
+ # 3) 不覆盖已有非空字段(保护人工既有知识)
757
+ if not overwrite:
758
+ kept = {f: v for f, v in kept.items() if f not in have}
759
+ if not kept:
760
+ rep["skipped_complete"] += 1
761
+ continue
762
+
763
+ prov = {"at": round(time.time(), 3), "verdict": "ACCEPT",
764
+ "source_hash": _sig(content),
765
+ "reflect": {"model": reflect_model, "prompt_hash": _sig(prompt),
766
+ "fields": sorted(kept)},
767
+ "verify": ({"model": verify_model, "prompt_hash": _sig(vprompt),
768
+ "dropped": dropped, "verdict_fields": sorted(vd or {}),
769
+ "self_verify": verify_fn is reflect_fn}
770
+ if verify_fn else {"model": "", "status": "skipped"}),
771
+ "grounding": gdetail, "replay": replay,
772
+ "verification_basis": verification_basis}
773
+ rep["accepted"] += 1
774
+ for f in kept:
775
+ rep["per_field"][f] += 1
776
+ if apply:
777
+ _apply_node(cg, e, fm, content, kept, prov,
778
+ verification_basis, basis_enum)
779
+ append_jsonl(os.path.join(cg.root, "_consolidate.jsonl"),
780
+ {"t": time.time(), "id": nid, "verdict": "ACCEPT",
781
+ "fields": sorted(kept),
782
+ "reflect_model": reflect_model,
783
+ "verify_model": verify_model, "dropped": dropped,
784
+ "source_hash": prov["source_hash"], "replay": replay})
785
+ rep["written"] += 1
786
+ if verbose and len(rep["samples"]) < 8:
787
+ rep["samples"].append({"id": nid, "verdict": "ACCEPT",
788
+ "fields": sorted(kept), "dropped": dropped,
789
+ "replay": replay})
790
+
791
+ if apply and rep["written"]:
792
+ # 正文新增了 `# 不适用条件:` / `# 验证方式:` → 索引字段变了
793
+ cg.rebuild_index()
794
+ rep["elapsed_sec"] = round(time.time() - t0, 3)
795
+ return rep
796
+
797
+
798
+ # 生效条件:给定 root 与 basis,对缺 "# 验证方式" 行的节点补写验证方式并在需要时写入 basis_enum,返回统计 rep。
799
+ def fill_verification_basis(root: str, basis: str, layer: str = None,
800
+ limit: int = None, apply: bool = False,
801
+ basis_enum: str = BASIS_ENUM_DEFAULT) -> dict:
802
+ """只补「验证方式」——声明文本是常量,不需要黑箱生成,零 LLM 成本。
803
+
804
+ 对应纪律 3「不猜测」:验证基底必须由人/流程声明,而不是让模型编出来。
805
+ """
806
+ cg = MdCGOS(root)
807
+ entries = cg._candidates(layer=layer)
808
+ rep = {"root": root, "layer": layer, "dry_run": not apply, "basis": basis,
809
+ "basis_enum": basis_enum, "nodes_scanned": len(entries),
810
+ "targeted": 0, "skipped_present": 0, "skipped_locked": 0,
811
+ "written": 0}
812
+ for e in entries:
813
+ if limit is not None and rep["written"] >= limit:
814
+ break
815
+ fm, content = direct_read(cg, e) # 写前重查走盘上真值(读缓存口径)
816
+ if fm is None:
817
+ continue
818
+ if crypto.is_encrypted(content):
819
+ rep["skipped_locked"] += 1 # 无密钥 → fail-closed:绝不改写密文
820
+ continue
821
+ if _has_ccg_line(content, "验证方式"):
822
+ rep["skipped_present"] += 1
823
+ continue
824
+ rep["targeted"] += 1
825
+ if not apply:
826
+ continue
827
+ comment = (fm.get("state_attributes") or {}).get("comment")
828
+ if not isinstance(comment, dict):
829
+ fm["state_attributes"] = dict(fm.get("state_attributes") or {})
830
+ fm["state_attributes"]["comment"] = {}
831
+ comment = fm["state_attributes"]["comment"]
832
+ content = _upsert_ccg_line(content, "验证方式", basis)
833
+ comment["验证方式"] = basis
834
+ if not nodefile.verification_basis_valid(fm):
835
+ fm["verification_basis"] = basis_enum
836
+ nid = e.get("id") or os.path.basename(e["path"])[:-3]
837
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
838
+ durable=True)
839
+ rep["written"] += 1
840
+ if apply and rep["written"]:
841
+ cg.rebuild_index()
842
+ return rep
843
+
844
+
845
+ # ==========================================================================
846
+ # 情境层批量提升(consolidate.promote)
847
+ # ==========================================================================
848
+ #
849
+ # 场景:情境层(contextual)里有些记忆被反复命中/并入——它们已经不是「一次情境」,
850
+ # 而是稳定的规律。本动作把它们提升为长期知识(knowledge),并保留:
851
+ # · 双向可追溯:promoted_from + 演化账本(KIND_LAYER_SHIFT);
852
+ # · 条件门槛:四要素(CCG)不全者**不提升**(未可判定就不该升格为长期知识);
853
+ # · 可预演:apply=False 只出报表;可留痕:`_maintain.jsonl`。
854
+
855
+ MAINTAIN_LOG = "_maintain.jsonl"
856
+ CCG_REQUIRED = ("生效条件", "子功能", "执行", "不适用条件")
857
+
858
+
859
+ # 生效条件:给定 cg、nid、e、fm、content、target_layer,把节点写入目标层(必要时按 routing 分桶)并删除旧路径,返回新相对路径与 bucket。
860
+ def _relocate_layer(cg, nid, e, fm, content, target_layer):
861
+ """把节点正文迁到目标层的正确目录(含分桶),删除旧文件。返回新相对路径。"""
862
+ d = os.path.join(cg.root, target_layer)
863
+ bucket = None
864
+ if target_layer in BUCKETED_LAYERS:
865
+ bucket = routing.bucket_dir(routing.route_key(fm.get("condition_space"),
866
+ fm.get("tags")))
867
+ d = os.path.join(d, bucket)
868
+ os.makedirs(d, exist_ok=True)
869
+ new_path = os.path.join(d, f"{nid}.md")
870
+ old_path = os.path.join(cg.root, e.get("path") or f"{nid}.md")
871
+ cg._write_node(nid, new_path, fm, content, durable=True)
872
+ if os.path.abspath(old_path) != os.path.abspath(new_path) and os.path.exists(old_path):
873
+ os.remove(old_path)
874
+ return {"path": os.path.relpath(new_path, cg.root).replace("\\", "/"),
875
+ "bucket": bucket}
876
+
877
+
878
+ # 生效条件:给定 root,把 source_layer 中命中次数不小于 min_merge 或 importance 不小于 min_importance 且条件完整的节点提升到 target_layer,返回统计 rep。
879
+ def promote_memories(root, source_layer="contextual", target_layer="knowledge",
880
+ min_merge=2, min_importance=0.6, require_conditions=True,
881
+ limit=None, apply=False, actor="maintain") -> dict:
882
+ """把反复命中的情境记忆批量提升为长期知识(可预演 / 可留痕 / 可追溯)。"""
883
+ cg = MdCGOS(root)
884
+ entries = cg._candidates(layer=source_layer)
885
+ rep = {"root": root, "source_layer": source_layer, "target_layer": target_layer,
886
+ "dry_run": not apply, "nodes_scanned": len(entries), "targeted": 0,
887
+ "skipped_locked": 0, "skipped_incomplete": 0, "skipped_not_hot": 0,
888
+ "written": 0, "promoted": [], "samples": [],
889
+ "min_merge": min_merge, "min_importance": min_importance,
890
+ "require_conditions": bool(require_conditions)}
891
+ batch = time.strftime("%Y%m%d-%H%M%S")
892
+ for e in entries:
893
+ if limit is not None and rep["written"] >= int(limit):
894
+ break
895
+ fm, content = direct_read(cg, e) # 写前重查走盘上真值(读缓存口径)
896
+ if fm is None:
897
+ continue
898
+ if crypto.is_encrypted(content):
899
+ rep["skipped_locked"] += 1 # 无密钥 → fail-closed,绝不解密回写
900
+ continue
901
+ nid = e.get("id") or os.path.basename(e["path"])[:-3]
902
+ hits = max(int(fm.get("merge_count") or 0),
903
+ int(fm.get("access_count") or 0),
904
+ int(fm.get("recall_count") or 0))
905
+ imp = float(fm.get("importance") or e.get("importance") or 0.0)
906
+ complete = all(_has_ccg_line(content, f) for f in CCG_REQUIRED)
907
+ if require_conditions and not complete:
908
+ rep["skipped_incomplete"] += 1 # 四要素不全 → 不可判定,不升格
909
+ continue
910
+ hot = hits >= int(min_merge)
911
+ if not hot and imp < float(min_importance):
912
+ rep["skipped_not_hot"] += 1
913
+ continue
914
+ rep["targeted"] += 1
915
+ item = {"id": nid, "hits": hits, "importance": round(imp, 4),
916
+ "conditions_complete": complete,
917
+ "basis": fm.get("verification_basis")}
918
+ if len(rep["samples"]) < 8:
919
+ rep["samples"].append(item)
920
+ if not apply:
921
+ continue
922
+ before = evolution.state_of(cg, nid) or {}
923
+ fm["layer"] = target_layer
924
+ fm["promoted_from"] = source_layer
925
+ fm["promoted_at"] = time.time()
926
+ fm["promotion_basis"] = {"hits": hits, "importance": round(imp, 4),
927
+ "conditions_complete": complete, "batch": batch,
928
+ "actor": actor}
929
+ moved = _relocate_layer(cg, nid, e, fm, content, target_layer)
930
+ evolution.record(
931
+ cg, node_id=nid,
932
+ pattern="情境记忆反复命中/并入 → 提升为长期知识",
933
+ missing="", action=f"层迁移 {source_layer}→{target_layer}",
934
+ evidence=f"hits={hits} importance={imp:.2f} conditions_complete={complete}",
935
+ source="consolidate", kind=evolution.KIND_LAYER_SHIFT,
936
+ before=before, after=evolution.state_of(cg, nid) or {})
937
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
938
+ "t": time.time(), "action": "promote", "batch": batch, "id": nid,
939
+ "from": source_layer, "to": target_layer, "hits": hits,
940
+ "importance": round(imp, 4), "path": moved["path"], "actor": actor})
941
+ rep["promoted"].append(nid)
942
+ rep["written"] += 1
943
+ if apply and rep["written"]:
944
+ cg.rebuild_index()
945
+ rep["note"] = ("dry-run:未写盘;apply=True 才迁移层"
946
+ if not apply else f"已提升 {rep['written']} 个节点到 {target_layer}")
947
+ return rep
948
+
949
+
950
+ # 生效条件:给定 root,按 _maintain.jsonl 中 action=promote 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
951
+ def rollback_promotion(root, node_ids=None, batch=None, actor="maintain") -> dict:
952
+ """回滚情境提升:把 promoted_from 层迁回,并记一条演化条目。"""
953
+ cg = MdCGOS(root)
954
+ recs = [r for r in _read_maintain(root)
955
+ if r.get("action") == "promote"
956
+ and (not batch or r.get("batch") == batch)
957
+ and (not node_ids or str(r.get("id")) in {str(x) for x in node_ids})]
958
+ if not recs:
959
+ return {"ok": False, "error": "no_records", "reverted": 0}
960
+ reverted, ids = 0, []
961
+ for rec in recs:
962
+ nid = rec["id"]
963
+ e = (cg.index.get("nodes") or {}).get(nid)
964
+ if not e:
965
+ continue
966
+ fm, content = direct_read(cg, e) # 回滚比对走盘上真值(读缓存口径)
967
+ if fm is None or crypto.is_encrypted(content):
968
+ continue
969
+ back = rec.get("from") or "contextual"
970
+ before = evolution.state_of(cg, nid) or {}
971
+ fm["layer"] = back
972
+ fm["promoted_from"] = None
973
+ fm["promotion_basis"] = {"rollback_of": rec.get("batch"), "actor": actor}
974
+ _relocate_layer(cg, nid, e, fm, content, back)
975
+ evolution.record(cg, node_id=nid, pattern="提升回滚:长期知识退回情境层",
976
+ action=f"层迁移 {rec.get('to')}→{back}",
977
+ evidence=f"rollback batch={rec.get('batch')}",
978
+ source="consolidate", kind=evolution.KIND_ROLLBACK,
979
+ before=before, after=evolution.state_of(cg, nid) or {})
980
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
981
+ "t": time.time(), "action": "promote_rollback", "batch": rec.get("batch"),
982
+ "id": nid, "to": back, "actor": actor})
983
+ reverted += 1
984
+ ids.append(nid)
985
+ if reverted:
986
+ cg.rebuild_index()
987
+ return {"ok": True, "reverted": reverted, "ids": ids}
988
+
989
+
990
+ # ==========================================================================
991
+ # 层归位(consolidate.contextualize)
992
+ # ==========================================================================
993
+ #
994
+ # 场景:批次流水账(note_/milestone_/retest6_)与感知产物(imgpart_/vpipe_)混在
995
+ # knowledge 层——它们的语义是**情境**(某次批次的记录 / 某张图的一次观测),不是
996
+ # 长期知识;但也不该进 rejected/unresolved(那是「失效 / 未解」,不是「情境」)。
997
+ # 故归位到 contextual:
998
+ # · 只改 layer 与落点目录;正文 / 密级 / id / tags 一律不动;
999
+ # · **保留可召回**(contextual 已在层白名单内,且 predict._SAFE_LAYERS 含之);
1000
+ # · 可预演(apply=False)/ 可留痕(`_maintain.jsonl`)/ 可追溯(KIND_LAYER_SHIFT)
1001
+ # / 可回滚(按 batch 或 id 反向迁层)。
1002
+ #
1003
+ # 与 promote 的关系:promote 是 contextual→knowledge(升格),本动作是
1004
+ # knowledge→contextual(归位)。两者共用 `_relocate_layer` 与批次台账,方向相反。
1005
+
1006
+ CONTEXTUALIZE_REASON_DEFAULT = "情境性内容归位(批次流水账 / 感知产物)"
1007
+
1008
+
1009
+ # 生效条件:给定 e,返回 e.id 字符串,若缺 id 则回落到 basename(e.path) 去掉 .md。
1010
+ def _entry_id(e) -> str:
1011
+ """索引条目取 id:优先 `id` 字段,回落到文件名(索引不保证带 id)。"""
1012
+ return str(e.get("id") or os.path.basename(e.get("path") or "")[:-3])
1013
+
1014
+
1015
+ # 生效条件:给定 root 与 base,若 base 不在维护日志已用批次中则返回 base,否则返回 base.n 且 n 为最小未用序号。
1016
+ def _unique_batch(root, base) -> str:
1017
+ """批次号去重:**同一秒内的两次调用不得共用批次号**。
1018
+
1019
+ 否则「按批次回滚」会连带命中上一次的台账记录(回滚必须是精确的、可对账的)。
1020
+ """
1021
+ seen = {r.get("batch") for r in _read_maintain(root)}
1022
+ if base not in seen:
1023
+ return base
1024
+ n = 2
1025
+ while f"{base}.{n}" in seen:
1026
+ n += 1
1027
+ return f"{base}.{n}"
1028
+
1029
+
1030
+ # 生效条件:给定 root 且 prefixes 或 node_ids 至少一个非空,把 source_layer 中匹配的节点迁到 target_layer,返回统计 rep;两者皆空则抛 ValueError。
1031
+ def contextualize_prefixes(root, prefixes=None, node_ids=None,
1032
+ source_layer="knowledge", target_layer="contextual",
1033
+ reason="", limit=None, apply=False,
1034
+ actor="maintain") -> dict:
1035
+ """按 id 前缀(或定向 id 列表)把节点从 source_layer 归位到 target_layer。
1036
+
1037
+ 默认方向 knowledge→contextual。`prefixes` / `node_ids` **至少给一个**:
1038
+ 宁可少搬,不可全库乱搬——不传白名单直接报错,拒绝「一次误调用把整个知识层改层」
1039
+ 这种不可归因的批量改写。`node_ids` 用于定向(含「回滚后单独补迁」的对称操作)。
1040
+ """
1041
+ pref = tuple(str(p) for p in (prefixes or ()) if str(p))
1042
+ ids = {str(i) for i in (node_ids or ()) if str(i)} or None
1043
+ if not pref and not ids:
1044
+ raise ValueError("contextualize 需要显式 prefixes 或 node_ids"
1045
+ "(如 ['note_','imgpart_']),拒绝对整层无差别改写")
1046
+ cg = MdCGOS(root)
1047
+
1048
+ # 生效条件:e 经 _entry_id 得到 nid 后,若闭包 ids 不为 None 则返回 nid in ids 的真假,若 ids 为 None 则返回 nid.startswith(pref) 的真假。
1049
+ def _hit(e) -> bool:
1050
+ nid = _entry_id(e)
1051
+ return nid in ids if ids is not None else nid.startswith(pref)
1052
+
1053
+ entries = [e for e in cg._candidates(layer=source_layer) if _hit(e)]
1054
+ batch = _unique_batch(root, time.strftime("%Y%m%d-%H%M%S"))
1055
+ rep = {"root": root, "action": "contextualize", "dry_run": not apply,
1056
+ "source_layer": source_layer, "target_layer": target_layer,
1057
+ "prefixes": list(pref), "node_ids": sorted(ids) if ids else [],
1058
+ "reason": reason or CONTEXTUALIZE_REASON_DEFAULT,
1059
+ "nodes_scanned": len(entries), "targeted": 0, "skipped_locked": 0,
1060
+ "skipped_already": 0, "written": 0, "moved": [], "samples": [],
1061
+ "batch": batch}
1062
+ for e in entries:
1063
+ if limit is not None and rep["written"] >= int(limit):
1064
+ break
1065
+ nid = _entry_id(e)
1066
+ if not _hit(e):
1067
+ continue
1068
+ fm, content = direct_read(cg, e) # 写前重查走盘上真值(读缓存口径)
1069
+ if fm is None:
1070
+ continue
1071
+ if crypto.is_encrypted(content):
1072
+ rep["skipped_locked"] += 1 # 无密钥 → fail-closed,绝不解密回写
1073
+ continue
1074
+ if fm.get("layer") != source_layer:
1075
+ rep["skipped_already"] += 1
1076
+ continue
1077
+ rep["targeted"] += 1
1078
+ if len(rep["samples"]) < 8:
1079
+ rep["samples"].append({"id": nid, "from": fm.get("layer"),
1080
+ "path": e.get("path")})
1081
+ if not apply:
1082
+ continue
1083
+ before = evolution.state_of(cg, nid) or {}
1084
+ fm["layer"] = target_layer
1085
+ fm["contextualized_from"] = source_layer
1086
+ fm["contextualized_at"] = time.time()
1087
+ fm["contextualization_basis"] = {"reason": rep["reason"], "batch": batch,
1088
+ "actor": actor}
1089
+ moved = _relocate_layer(cg, nid, e, fm, content, target_layer)
1090
+ evolution.record(
1091
+ cg, node_id=nid,
1092
+ pattern="情境性内容(批次流水账 / 感知产物)混在知识层 → 归位情境层",
1093
+ missing="层归属规则", action=f"层迁移 {source_layer}→{target_layer}",
1094
+ evidence=f"prefix={str(nid).split('_')[0]}_ reason={rep['reason']}",
1095
+ source="consolidate", kind=evolution.KIND_LAYER_SHIFT,
1096
+ before=before, after=evolution.state_of(cg, nid) or {})
1097
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1098
+ "t": time.time(), "action": "contextualize", "batch": batch, "id": nid,
1099
+ "from": source_layer, "to": target_layer, "path": moved["path"],
1100
+ "bucket": moved.get("bucket"), "reason": rep["reason"], "actor": actor})
1101
+ rep["moved"].append(nid)
1102
+ rep["written"] += 1
1103
+ if apply and rep["written"]:
1104
+ cg.rebuild_index()
1105
+ rep["note"] = ("dry-run:未写盘;apply=True 才归位"
1106
+ if not apply else f"已归位 {rep['written']} 个节点到 {target_layer}")
1107
+ return rep
1108
+
1109
+
1110
+ # 生效条件:给定 root,按 _maintain.jsonl 中 action=contextualize 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
1111
+ def rollback_contextualize(root, node_ids=None, batch=None, actor="maintain") -> dict:
1112
+ """回滚层归位:按 `_maintain.jsonl` 的 contextualize 记录把节点迁回原层。"""
1113
+ cg = MdCGOS(root)
1114
+ recs = [r for r in _read_maintain(root)
1115
+ if r.get("action") == "contextualize"
1116
+ and (not batch or r.get("batch") == batch)
1117
+ and (not node_ids or str(r.get("id")) in {str(x) for x in node_ids})]
1118
+ if not recs:
1119
+ return {"ok": False, "error": "no_records", "reverted": 0}
1120
+ reverted, ids = 0, []
1121
+ for rec in recs:
1122
+ nid = rec["id"]
1123
+ e = (cg.index.get("nodes") or {}).get(nid)
1124
+ if not e:
1125
+ continue
1126
+ fm, content = direct_read(cg, e) # 回滚比对走盘上真值(读缓存口径)
1127
+ if fm is None or crypto.is_encrypted(content):
1128
+ continue
1129
+ back = rec.get("from") or "knowledge"
1130
+ before = evolution.state_of(cg, nid) or {}
1131
+ fm["layer"] = back
1132
+ fm["contextualized_from"] = None
1133
+ fm["contextualization_basis"] = {"rollback_of": rec.get("batch"),
1134
+ "actor": actor}
1135
+ _relocate_layer(cg, nid, e, fm, content, back)
1136
+ evolution.record(cg, node_id=nid,
1137
+ pattern="层归位回滚:情境层迁回原层",
1138
+ action=f"层迁移 {rec.get('to')}→{back}",
1139
+ evidence=f"rollback batch={rec.get('batch')}",
1140
+ source="consolidate", kind=evolution.KIND_ROLLBACK,
1141
+ before=before, after=evolution.state_of(cg, nid) or {})
1142
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1143
+ "t": time.time(), "action": "contextualize_rollback",
1144
+ "batch": rec.get("batch"), "id": nid, "to": back, "actor": actor})
1145
+ reverted += 1
1146
+ ids.append(nid)
1147
+ if reverted:
1148
+ cg.rebuild_index()
1149
+ return {"ok": True, "reverted": reverted, "ids": ids}
1150
+
1151
+
1152
+ # 生效条件:对 _read_maintain(root) 中 action 为 "contextualize" 或 "contextualize_rollback" 的记录,取 recs[-(int(limit) or 50):] 作为 records 返回 {'ok': True, ...}——仅当 int(limit) 成功且为 0 时回落 50,limit 为 None/""/[] 等无法 int() 的值会先抛 TypeError/ValueError。
1153
+ def contextualize_history(root, limit=50):
1154
+ """层归位的批次记录(只读)。"""
1155
+ recs = [r for r in _read_maintain(root)
1156
+ if r.get("action") in ("contextualize", "contextualize_rollback")]
1157
+ return {"ok": True, "records": recs[-(int(limit) or 50):]}
1158
+
1159
+
1160
+ # 生效条件:给定 root,读取 root 下 MAINTAIN_LOG 的 JSONL 并返回记录列表。
1161
+ def _read_maintain(root):
1162
+ from .fsutil import read_jsonl
1163
+ return list(read_jsonl(os.path.join(root, MAINTAIN_LOG)))
1164
+
1165
+
1166
+ # ==========================================================================
1167
+ # 归纳聚类(consolidate.induce)
1168
+ # ==========================================================================
1169
+ #
1170
+ # 与 promote 的分工:
1171
+ # promote —— 把**已经存在**的单条情境记忆升格为长期知识(节点不变,只迁层);
1172
+ # induce —— 把**多条**具体记忆归纳为一个**新的概念节点**(新增节点)。
1173
+ #
1174
+ # 归纳是「由具体到一般」的推理,其输出**不是事实断言**,而是待验证的假设:
1175
+ # · 证据基底一律记 inferred(未经验证),不得冒充 verified;
1176
+ # · 概念节点必须携带成员清单 + `generalizes`/`instance_of` 对称边,保证可回溯;
1177
+ # · 归纳不出「共同条件」时默认**拒绝生成**(没有条件依据的抽象=编造,对齐
1178
+ # 纪律 3「不猜测」);确需放宽须显式 require_conditions=False,且概念正文
1179
+ # 会写明「未归纳出共同条件」,不掩盖证据缺口。
1180
+
1181
+ INDUCE_MIN_CLUSTER = 3
1182
+ INDUCE_MIN_JACCARD = 0.30
1183
+ INDUCE_MAX_NODES = 400
1184
+ INDUCE_MAX_TERMS = 6
1185
+ CONCEPT_REL = "generalizes" # concept → member(inferred)
1186
+ CONCEPT_MEMBER_REL = "instance_of" # member → concept(inferred)
1187
+ CONCEPT_PREFIX = "concept_"
1188
+ CONCEPT_IMPORTANCE = 0.5
1189
+ CONCEPT_TAGS = ("concept", "induced")
1190
+ # 巩固留痕字段(2026-09-19 阶段一):**字段名真源在 md_cg/nodefile.py**,
1191
+ # 本处只做短别名引用(非复制),与 `nodefile.VALID_FROM_FIELD` 的登记纪律同构。
1192
+ CONSOLIDATED_AT_FIELD = nodefile.CONSOLIDATED_AT_FIELD
1193
+ CONSOLIDATED_INTO_FIELD = nodefile.CONSOLIDATED_INTO_FIELD
1194
+ # 归纳候选排除:受保护节点,以及洞察/场景/前馈/概念等派生物(避免自我进食)
1195
+ INDUCE_SKIP_TAGS = ("insight", "scene", "reconstructed", "gap_hint", "concept")
1196
+
1197
+
1198
+ # 生效条件:给定 members,返回 CONCEPT_PREFIX 拼接排序后成员串的 SHA1 前 10 位。
1199
+ def _concept_id(members):
1200
+ """概念节点 id:由成员清单派生,保证「同成员 ⇒ 同 id」的幂等性。"""
1201
+ h = hashlib.sha1("|".join(sorted(str(m) for m in members))
1202
+ .encode("utf-8")).hexdigest()
1203
+ return CONCEPT_PREFIX + h[:10]
1204
+
1205
+
1206
+ # 生效条件:给定 a 与 b,若任一为空集则返回 0.0,否则返回交集大小除以并集大小。
1207
+ def _jaccard(a, b):
1208
+ if not a or not b:
1209
+ return 0.0
1210
+ return len(a & b) / float(len(a | b))
1211
+
1212
+
1213
+ # 生效条件:给定 term_sets,返回出现次数不小于 max(2, ceil(min_share * len(term_sets))) 的词面排序列表;空输入返回 []。
1214
+ def _common_terms(term_sets, min_share=0.6):
1215
+ """出现在 ≥ min_share 比例成员中的词面(共同条件);少于 2 个成员共享不算。"""
1216
+ if not term_sets:
1217
+ return []
1218
+ cnt = {}
1219
+ for s in term_sets:
1220
+ for t in set(s or ()):
1221
+ cnt[t] = cnt.get(t, 0) + 1
1222
+ need = max(2, int(math.ceil(min_share * len(term_sets))))
1223
+ return sorted(t for t, c in cnt.items() if c >= need)
1224
+
1225
+
1226
+ # 生效条件:给定 term_sets,按集合排序去重拼接后返回前 limit(默认 INDUCE_MAX_TERMS)个词面。
1227
+ def _union_terms(term_sets, limit=INDUCE_MAX_TERMS):
1228
+ seen = []
1229
+ for s in term_sets:
1230
+ for t in sorted(s or ()):
1231
+ if t not in seen:
1232
+ seen.append(t)
1233
+ return seen[:limit]
1234
+
1235
+
1236
+ # 生效条件:给定 cg、cid、members、reason、actor、batch,为概念节点与成员节点写对称 inferred 边(已存在则跳过),返回含 concept 与 members 的字典。
1237
+ def _link_concept(cg, cid, members, reason, actor, batch):
1238
+ """写概念↔成员对称 inferred 边(幂等:已存在则不重复写)。"""
1239
+ nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
1240
+ out = {"concept": cid, "members": []}
1241
+ cnode = cg.get(cid)
1242
+ if cnode:
1243
+ fm = cnode.get("frontmatter") or {}
1244
+ edges = list(fm.get("edges") or [])
1245
+ have = {str(e.get("target")) for e in edges if isinstance(e, dict)}
1246
+ added = False
1247
+ for m in members:
1248
+ if m in have:
1249
+ continue
1250
+ edges.append({"target": m, "relation_type": CONCEPT_REL,
1251
+ "reason": reason, "created_at": time.time(),
1252
+ "confidence": 0.5, "verified": 0, "evidence": "inferred"})
1253
+ added = True
1254
+ if added:
1255
+ fm["edges"] = edges
1256
+ ent = nodes.get(cid) or {}
1257
+ cg._write_node(cid, os.path.join(cg.root, ent.get("path") or f"{cid}.md"),
1258
+ fm, cnode.get("content") or "")
1259
+ if ent:
1260
+ ent["edges"] = edges
1261
+ for m in members:
1262
+ node = cg.get(m)
1263
+ if not node:
1264
+ continue
1265
+ fm = node.get("frontmatter") or {}
1266
+ edges = list(fm.get("edges") or [])
1267
+ if any(isinstance(e, dict) and str(e.get("target")) == cid for e in edges):
1268
+ continue
1269
+ edges.append({"target": cid, "relation_type": CONCEPT_MEMBER_REL,
1270
+ "reason": reason, "created_at": time.time(),
1271
+ "confidence": 0.5, "verified": 0, "evidence": "inferred"})
1272
+ fm["edges"] = edges
1273
+ # 巩固留痕(2026-09-19 阶段一):`consolidated_into` 为**规范名**,
1274
+ # `induced_concept` 保留为历史别名(既有读取面零破坏);`consolidated_at`
1275
+ # 补齐**成员侧**巩固时刻——此前只有概念侧 `induced_at`,成员侧无从判定
1276
+ # 「何时被并进去」,故「合并后前身可定位」只在概念侧半成立。
1277
+ fm[CONSOLIDATED_INTO_FIELD] = cid
1278
+ fm[CONSOLIDATED_AT_FIELD] = time.time()
1279
+ fm["induced_concept"] = cid
1280
+ ent = nodes.get(m) or {}
1281
+ cg._write_node(m, os.path.join(cg.root, ent.get("path") or f"{m}.md"),
1282
+ fm, node.get("content") or "")
1283
+ if ent:
1284
+ ent["edges"] = edges
1285
+ out["members"].append(m)
1286
+ return out
1287
+
1288
+
1289
+ # 生效条件:给定 members、common_pos、neg_union,返回标注 inferred 的概念节点正文,含功能名、生效条件、子功能、执行、验证方式、不适用条件。
1290
+ def _concept_payload(members, common_pos, neg_union):
1291
+ """概念节点正文:把成员的共性条件抽象为可追溯的知识条目(显式标注 inferred)。"""
1292
+ label = "、".join(common_pos[:INDUCE_MAX_TERMS])
1293
+ pos_txt = ";".join(common_pos[:INDUCE_MAX_TERMS]) or "(未归纳出共同条件)"
1294
+ neg_txt = ";".join(neg_union[:INDUCE_MAX_TERMS]) or "(未判定)"
1295
+ return (
1296
+ "# 功能名:归纳概念:%s\n"
1297
+ "# 生效条件:%s\n"
1298
+ "# 子功能:%d 条具体记忆的共性(成员:%s)\n"
1299
+ "# 执行:由 consolidate.induce 归纳聚合(inferred;未经验证,不得直接当事实使用)\n"
1300
+ "# 验证方式:待验证(inferred 假设,需外部证据或实践重复后方可升格)\n"
1301
+ "# 不适用条件:%s\n"
1302
+ % (label or "共性", pos_txt, len(members), "、".join(members), neg_txt)
1303
+ )
1304
+
1305
+
1306
+ # 生效条件:给定 cg_or_root,从 source_layer 聚类归纳为 target_layer 概念节点,apply=True 才写盘并返回统计 rep。
1307
+ def induce_memories(cg_or_root, source_layer="contextual", target_layer="knowledge",
1308
+ min_cluster=INDUCE_MIN_CLUSTER, min_jaccard=INDUCE_MIN_JACCARD,
1309
+ max_nodes=INDUCE_MAX_NODES, require_conditions=True,
1310
+ limit=None, apply=False, actor="maintain", **extra):
1311
+ """归纳聚类:把多条具体记忆归纳为概念层条目(inferred,非事实断言)。
1312
+
1313
+ 流程:读取源层 → bigram 相似度贪心聚类 → 提炼共同条件 → 生成概念节点
1314
+ (apply=True)→ 写 `generalizes` / `instance_of` 对称 inferred 边 → 写留痕。
1315
+
1316
+ apply=False(默认)只出候选报表(可预演);apply=True 才写盘(可留痕、可回溯)。
1317
+ 幂等:概念 id 由成员清单派生,同成员重复归纳不新增节点。
1318
+ """
1319
+ cg = cg_or_root if isinstance(cg_or_root, MdCGOS) else MdCGOS(str(cg_or_root))
1320
+ from . import subgraph # 惰性导入:复用统一的条件/词面抽取
1321
+
1322
+ # MCP 分发层会把未提供的参数以 None 传入;此处归一化,避免 int(None) 崩溃,
1323
+ # 也避免 require_conditions=None 被当成 False 而悄悄关掉「无共同条件即拒绝生成」
1324
+ # 这条纪律(默认必须为真,放宽只能显式传 False)。
1325
+ min_cluster = INDUCE_MIN_CLUSTER if min_cluster is None else int(min_cluster)
1326
+ min_jaccard = INDUCE_MIN_JACCARD if min_jaccard is None else float(min_jaccard)
1327
+ max_nodes = INDUCE_MAX_NODES if max_nodes is None else int(max_nodes)
1328
+ if require_conditions is None:
1329
+ require_conditions = True
1330
+
1331
+ nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
1332
+ pool = [nid for nid, e in nodes.items()
1333
+ if (not source_layer or (e or {}).get("layer") == source_layer)
1334
+ and not (e or {}).get("protected")
1335
+ and not (set(INDUCE_SKIP_TAGS) & set((e or {}).get("tags") or []))]
1336
+ pool.sort()
1337
+ truncated = len(pool) > int(max_nodes)
1338
+ pool = pool[:int(max_nodes)]
1339
+
1340
+ cache = {}
1341
+ for nid in pool:
1342
+ got = subgraph._node_terms_and_grams(cg, nid)
1343
+ if got and got["grams"]:
1344
+ cache[nid] = got
1345
+ keys = sorted(cache.keys())
1346
+
1347
+ rep = {"ok": True, "action": "induce", "op": "consolidate",
1348
+ "source_layer": source_layer, "target_layer": target_layer,
1349
+ "dry_run": not apply, "nodes_scanned": len(pool), "indexed": len(keys),
1350
+ "truncated": truncated, "min_cluster": int(min_cluster),
1351
+ "min_jaccard": float(min_jaccard),
1352
+ "require_conditions": bool(require_conditions),
1353
+ "skipped_small": 0, "skipped_no_condition": 0, "skipped_existing": 0,
1354
+ "clusters": 0, "written": 0, "concepts": [], "samples": [],
1355
+ "log": MAINTAIN_LOG}
1356
+
1357
+ # ---- 贪心聚类(只读) ----
1358
+ assigned, proposals = set(), []
1359
+ for i, a in enumerate(keys):
1360
+ if a in assigned:
1361
+ continue
1362
+ ga = cache[a]["grams"]
1363
+ grp = [b for b in keys[i + 1:]
1364
+ if b not in assigned
1365
+ and _jaccard(ga, cache[b]["grams"]) >= float(min_jaccard)]
1366
+ if len(grp) + 1 < int(min_cluster):
1367
+ continue
1368
+ members = [a] + grp
1369
+ assigned.update(members)
1370
+ common_pos = _common_terms([cache[m]["pos"] for m in members])
1371
+ if require_conditions and not common_pos:
1372
+ rep["skipped_no_condition"] += 1
1373
+ continue
1374
+ neg_union = _union_terms([cache[m]["neg"] for m in members])
1375
+ proposals.append({
1376
+ "members": members, "concept_id": _concept_id(members),
1377
+ "common_conditions": common_pos, "non_applicable": neg_union,
1378
+ "reason": ("%d 条记忆内容相近且共享条件「%s」→ 归纳为概念"
1379
+ % (len(members), "、".join(common_pos) or "无")),
1380
+ })
1381
+ rep["clusters"] = len(proposals)
1382
+ for p in proposals[:8]:
1383
+ rep["samples"].append(p)
1384
+
1385
+ if not apply:
1386
+ rep["note"] = ("dry-run:未写盘;apply=True 才生成概念节点与 inferred 边"
1387
+ if proposals else "无满足条件的聚类(内容不够相近或缺乏共同条件)")
1388
+ rep["concepts"] = [p["concept_id"] for p in proposals]
1389
+ return rep
1390
+
1391
+ # ---- 落库(可留痕) ----
1392
+ batch = time.strftime("%Y%m%d-%H%M%S")
1393
+ for p in proposals:
1394
+ if limit is not None and rep["written"] >= int(limit):
1395
+ break
1396
+ cid = p["concept_id"]
1397
+ if cid in nodes:
1398
+ rep["skipped_existing"] += 1
1399
+ continue
1400
+ content = _concept_payload(p["members"], p["common_conditions"],
1401
+ p["non_applicable"])
1402
+ _consolidated_at = time.time() # 概念形成时刻 = 巩固时刻(单一取值,禁两处取时)
1403
+ cg.add(cid, content, layer=target_layer, tags=list(CONCEPT_TAGS),
1404
+ importance=CONCEPT_IMPORTANCE, verification_basis="other",
1405
+ induced_from=list(p["members"]), induced_at=_consolidated_at,
1406
+ consolidated_at=_consolidated_at,
1407
+ induction={"method": "bigram_jaccard", "min_jaccard": float(min_jaccard),
1408
+ "common_conditions": p["common_conditions"], "batch": batch,
1409
+ "actor": actor, "evidence": "inferred"},
1410
+ actor=actor)
1411
+ _link_concept(cg, cid, p["members"], p["reason"], actor, batch)
1412
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1413
+ "t": time.time(), "action": "induce", "batch": batch, "concept": cid,
1414
+ "members": list(p["members"]), "common_conditions": p["common_conditions"],
1415
+ "source_layer": source_layer, "target_layer": target_layer, "actor": actor})
1416
+ rep["concepts"].append(cid)
1417
+ rep["written"] += 1
1418
+ if rep["written"]:
1419
+ cg.rebuild_index()
1420
+ rep["note"] = (f"已归纳 {rep['written']} 个概念节点(inferred,待验证)"
1421
+ if rep["written"] else "无可落库的归纳(均跳过或已达 limit)")
1422
+ return rep
1423
+
1424
+
1425
+ # ---- CLI ----------------------------------------------------------------
1426
+
1427
+ # 生效条件:不适用(无必需形参与模块级常量)
1428
+ def _cli(argv=None) -> int:
1429
+ ap = argparse.ArgumentParser(
1430
+ description="md_cg 离线固化:反思单元(LLM)产出候选 → 白箱闸门 → "
1431
+ "验证单元(LLM)否决 → 固化为 md 字段")
1432
+ ap.add_argument("--root", required=True, help="md 认知图根目录")
1433
+ ap.add_argument("--layer", default=None, help="只处理某层(如 knowledge)")
1434
+ ap.add_argument("--limit", type=int, default=None, help="只处理前 N 个待补节点")
1435
+ ap.add_argument("--apply", action="store_true", help="真正写盘(默认只验证)")
1436
+ ap.add_argument("--dry-run", action="store_true", help="只验证不写盘(默认行为)")
1437
+ ap.add_argument("--overwrite", action="store_true",
1438
+ help="允许覆盖已有非空字段(默认保护人工既有知识)")
1439
+ ap.add_argument("--reflect-model", default=None,
1440
+ help=f"反思单元模型(默认 {ROLE_DEFAULT_MODEL[REFLECT_ROLE]})")
1441
+ ap.add_argument("--verify-model", default=None,
1442
+ help=f"验证单元模型(默认 {ROLE_DEFAULT_MODEL[VERIFY_ROLE]})")
1443
+ ap.add_argument("--self-verify", action="store_true",
1444
+ help="降级:验证单元复用反思单元模型(非交叉验证,provenance 标记)")
1445
+ ap.add_argument("--no-verify", action="store_true",
1446
+ help="降级:跳过验证单元,仅靠白箱闸门(不推荐)")
1447
+ ap.add_argument("--verification-basis", default=None,
1448
+ help="写入 `# 验证方式:` 的声明文本(默认双模型声明)")
1449
+ ap.add_argument("--no-basis", action="store_true", help="不写「验证方式」")
1450
+ ap.add_argument("--basis-only", action="store_true",
1451
+ help="只补「验证方式」(零 LLM 成本),不做四要素反思")
1452
+ ap.add_argument("--min-grounding", type=float, default=None,
1453
+ help="统一 grounding 阈值(默认按字段 0.5 / 不适用条件 0.34)")
1454
+ ap.add_argument("--max-tokens", type=int, default=None,
1455
+ help=f"LLM 输出预算(含思考模型 reasoning_tokens;默认 "
1456
+ f"{DEFAULT_MAX_TOKENS},可 env {MAX_TOKENS_ENV} 覆盖;"
1457
+ "子代理配置标准 v0.5 §1)")
1458
+ ap.add_argument("--no-llm", action="store_true",
1459
+ help="不调用 LLM,只做四要素完整性普查")
1460
+ ap.add_argument("--check", action="store_true",
1461
+ help="零 token 探测两个角色网关的可用模型后退出")
1462
+ ap.add_argument("--report", default=None, help="把汇总 JSON 另存一份")
1463
+ a = ap.parse_args(argv)
1464
+
1465
+ r_model, _, r_key = role_config(REFLECT_ROLE, a.reflect_model)
1466
+ v_model, _, v_key = role_config(VERIFY_ROLE, a.verify_model)
1467
+ if a.self_verify:
1468
+ # 溯源修正(issue #24 附带②):--self-verify 实际调用的是反思单元模型,
1469
+ # verify.model 必须记实际值——此前记 ROLE_DEFAULT_MODEL[verify](glm),
1470
+ # 与真实调用不符,破坏可审计性。basis 声明同步改「同模型自验」,
1471
+ # 不再冒充双模型交叉验证。
1472
+ v_model = r_model
1473
+
1474
+ if a.check:
1475
+ out = {"reflect": probe_models(REFLECT_ROLE),
1476
+ "verify": probe_models(VERIFY_ROLE)}
1477
+ print(json.dumps(out, ensure_ascii=False, indent=2))
1478
+ return 0
1479
+
1480
+ if a.no_basis:
1481
+ basis = None
1482
+ elif a.verification_basis:
1483
+ basis = a.verification_basis
1484
+ elif a.self_verify:
1485
+ basis = f"同模型自验(reflect=verify={r_model},非交叉验证,降级模式)"
1486
+ else:
1487
+ basis = BASIS_TEMPLATE.format(reflect=r_model, verify=v_model)
1488
+
1489
+ if a.basis_only:
1490
+ rep = fill_verification_basis(a.root, basis, layer=a.layer, limit=a.limit,
1491
+ apply=a.apply)
1492
+ print(json.dumps(rep, ensure_ascii=False, indent=2))
1493
+ if a.report:
1494
+ with open(a.report, "w", encoding="utf-8") as f:
1495
+ json.dump(rep, f, ensure_ascii=False, indent=2)
1496
+ return 0
1497
+
1498
+ thresholds = ({f: a.min_grounding for f in CCG_FIELDS}
1499
+ if a.min_grounding is not None else None)
1500
+
1501
+ reflect_fn = verify_fn = None
1502
+ if not a.no_llm:
1503
+ if not r_key:
1504
+ print(f"[consolidate] 反思单元未配置 key"
1505
+ f"({_ROLE_ENV[REFLECT_ROLE][2]} / DEEPSEEK_API_KEY)→ 退化为普查模式",
1506
+ file=sys.stderr)
1507
+ else:
1508
+ reflect_fn = (lambda p: http_llm(p, role=REFLECT_ROLE, # noqa: E731
1509
+ model=a.reflect_model,
1510
+ max_tokens=a.max_tokens))
1511
+ if a.self_verify:
1512
+ verify_fn = reflect_fn
1513
+ elif not a.no_verify:
1514
+ if v_key:
1515
+ verify_fn = (lambda p: http_llm(p, role=VERIFY_ROLE, # noqa: E731
1516
+ model=a.verify_model,
1517
+ max_tokens=a.max_tokens))
1518
+ else:
1519
+ print(f"[consolidate] 验证单元未配置 key"
1520
+ f"({_ROLE_ENV[VERIFY_ROLE][2]} / ZHIPU_API_KEY / GLM_API_KEY)"
1521
+ "→ 待补节点将 DEFER,不写盘(纪律 5:未经验证不固化)",
1522
+ file=sys.stderr)
1523
+
1524
+ rep = consolidate(a.root, layer=a.layer, limit=a.limit, apply=a.apply,
1525
+ overwrite=a.overwrite, reflect_fn=reflect_fn,
1526
+ verify_fn=verify_fn, reflect_model=r_model,
1527
+ verify_model=v_model if verify_fn else "",
1528
+ verification_basis=basis or "",
1529
+ require_verify=not a.no_verify, thresholds=thresholds)
1530
+ print(json.dumps(rep, ensure_ascii=False, indent=2))
1531
+ if a.report:
1532
+ with open(a.report, "w", encoding="utf-8") as f:
1533
+ json.dump(rep, f, ensure_ascii=False, indent=2)
1534
+ return 0
1535
+
1536
+
1537
+ if __name__ == "__main__":
1440
1538
  raise SystemExit(_cli())