@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
@@ -1,1440 +1,1537 @@
1
- # -*- coding: utf-8 -*-
2
- """md_cg · 离线固化:LLM 补 CCG 四要素 → 确定性验证 → 固化为 md 字段
3
-
4
- 为什么是「离线固化」而不是「在线向量」:
5
- 白箱第 1 篇:相似度可以产生候选,但**不授予执行资格**;资格必须由条件证据裁决。
6
- LLM 是黑箱,它的输出只能是**候选条件**,不能直接成为检索依据——否则在线检索
7
- 就被黑箱污染,CCG 28%→88% 的改进会退化回去。故本工具把 LLM 严格限制在
8
- **离线一次性的固化工序**里:
9
-
10
- 读节点 → LLM 产出四要素候选 → 确定性验证 → 通过才写进 md 字段
11
-
12
- 在线检索(search / recall / _path_semantic)仍然只读 md 里已固化的字段,全程白箱。
13
-
14
- 理论 / 纪律对齐(docs/工作纪律_认知图条目_v1.1.json):
15
- · 第 3 条 白箱方法:不猜测;**未验证不写入**。
16
- · 第 5 条 验证纪律:**未经验证不固化**——入库前必须走验证(回放 / 断言 / 回归)。
17
- · 第 13 条 访谈澄清:节点四要素 = 条件 / 子内容 / 如何执行 / 不适用条件
18
- ——本工具固化的正是这四个字段(对齐 CCG 的生效条件 / 子功能 / 执行 / 不适用条件)。
19
- · 《智能的认知过程》:新条件能否**稳定解释误差**?成立 → 纳入知识结构;
20
- 不成立 → **不固化**,标记为待验证。
21
- 故 verdict 三态对齐白箱资格判定:ACCEPT(固化)/ REJECT(丢弃)/ DEFER(只存候选)。
22
-
23
- 验证分三段闸门——前两段确定性零 LLM,第三段是「双模型交叉验证」:
24
-
25
- 闸门 1 · grounding 支撑度(确定性):候选短语必须能在节点正文里找到字符级依据,
26
- 否则判为幻觉 → REJECT。(对应「不猜测」)
27
- 闸门 2 · replay 回放(确定性):把候选条件当作查询,回放生产检索路径的判定:
28
- · pos_recall 以「生效条件」为查询 → 本节点应被召回,且不被自身负条件挡住;
29
- · neg_separated 以「不适用条件」为查询 → 应触发条件级负路由,且负条件与正文
30
- 低相关(负条件必须是「域外」的,不能把知识本身否定掉);
31
- · no_conflict 生效条件与不适用条件不得互相覆盖。
32
- 三者同时成立才算「条件稳定」。(对应「回放 / 断言 / 回归」)
33
- 闸门 3 · 验证单元(GLM,独立模型):逐条核验候选是否有正文依据、负条件是否真域外。
34
- 硬约束:**验证单元只能否决,不能新增/改写**——它没有产出权,
35
- 否则验证环节自己就成了新的幻觉源。
36
-
37
- 回放器复用 _path_semantic 的同一批原语(_declared_conditions / _neg_hit /
38
- _weighted_coverage / expand_query_terms_weighted),并由 P6 测试与真实
39
- MdCGOS._path_semantic 做一致性回归,保证不漂移。
40
-
41
- 双模型角色分工(用户配置,可用环境变量覆盖):
42
- 反思单元 reflect → 默认 deepseek-v4.1-flash-expires-on-0910
43
- (IDE 显示名 DeepSeek-V4.1-Flash;限时模型,见常量注释)
44
- 验证单元 verify → 默认 glm-5.3-flash (IDE 显示名 GLM-5.3-flash)
45
- 环境变量:MDCG_REFLECT_MODEL/BASE/KEY、MDCG_VERIFY_MODEL/BASE/KEY。
46
- 注意 IDE 显示名 ≠ API 模型 id;`--check` 可零 token 探测各网关真实 id。
47
- 两个模型分属不同厂商,避免同源模型的系统性偏见互相印证(交叉验证的本意)。
48
-
49
- · 验证单元不可用(未配 key)时,默认 **不固化**(DEFER)——纪律 5「未经验证不固化」;
50
- 确需单模型跑通可显式 --no-verify(provenance 记为 skipped)或 --self-verify
51
- (同模型自审,provenance 记为 self_verify=true,属于降级模式)。
52
- · 「验证方式」是 CCG 必需要素,其值 = 声明的验证基底。本工具可写
53
- `# 验证方式:<声明>`(--verification-basis,默认即上面的双模型声明);
54
- frontmatter.verification_basis 只能取 nodefile 的枚举
55
- (compiler/test/measurement/formal_proof/data/other),双 LLM 交叉验证对应 "other"。
56
- 纯文本声明无需 LLM → --basis-only 可零成本补齐全库(纪律 3:不猜测)。
57
-
58
- 零第三方依赖(D-005):HTTP 走标准库 urllib.request;reflect_fn / verify_fn 可注入
59
- (离线可测)。默认 --dry-run,只有 --apply 才写盘(纪律 6:改动可核对)。
60
- 写盘只动 CCG 字段与 provenance,**不覆盖已有非空字段**(除非 --overwrite)。
61
- """
62
- from __future__ import annotations
63
-
64
- import argparse
65
- import hashlib
66
- import json
67
- import math
68
- import os
69
- import re
70
- import sys
71
- import time
72
- import urllib.error
73
- import urllib.request
74
-
75
- from . import crypto, evolution, nodefile, routing
76
- from .fsutil import append_jsonl
77
- from .mdcg import BUCKETED_LAYERS, bigrams, expand_query_terms_weighted
78
- from .mdcos import (MdCGOS, _ccg_field, _declared_conditions, _neg_hit, _sig,
79
- _weighted_coverage)
80
-
81
- # ---- 常量 ----------------------------------------------------------------
82
-
83
- # 与工作纪律第 13 条「节点四要素」同构:
84
- # 生效条件 ↔ conditions(什么时候适用)
85
- # 子功能 ↔ subgraph / depends_on(子内容)
86
- # 执行 ↔ execution(如何执行)
87
- # 不适用条件 ↔ negative(什么时候不适用)
88
- CCG_FIELDS = ("生效条件", "子功能", "执行", "不适用条件")
89
- MULTI_FIELDS = ("生效条件", "子功能", "不适用条件") # 列表型
90
- SINGLE_FIELDS = ("执行",) # 单值型
91
-
92
- # LLM 常把「子功能」写成「子内容」,别名容错(固化时统一落到标准字段名)
93
- FIELD_ALIASES = {
94
- "生效条件": ("生效条件", "适用条件", "conditions", "condition"),
95
- "子功能": ("子功能", "子内容", "子流程", "subgraph", "sub"),
96
- "执行": ("执行", "如何执行", "执行方式", "execution", "how"),
97
- "不适用条件": ("不适用条件", "不适用", "负条件", "negative", "reject"),
98
- }
99
-
100
- # 默认 grounding 阈值:不适用条件描述的是「域外」情境,与正文天然低相关,
101
- # 故阈值放宽;其余三要素必须能在正文里找到实打实的依据。
102
- DEFAULT_GROUNDING = {"生效条件": 0.5, "子功能": 0.5, "执行": 0.5, "不适用条件": 0.34}
103
-
104
- MAX_BODY_CHARS = 3000 # 正文截断(控制 token,且条件主要来自开头)
105
- MAX_TERMS = 8 # 单字段候选条数上限
106
- MAX_TERM_LEN = 40 # 单条候选长度上限
107
-
108
- # CCG 声明行(`# 生效条件:…` 等)——它们不是正文,grounding/replay 必须把它们剥掉,
109
- # 否则已写入的「不适用条件」会在二次运行时被当成正文依据,导致节点自我否定。
110
- _CCG_LINE_RE = re.compile(
111
- r"^\s*#\s*(功能名|生效条件|子功能|执行|验证方式|不适用条件)\s*[::]")
112
-
113
-
114
- # 生效条件:给定 content,返回剔除所有匹配 _CCG_LINE_RE 的行后以换行连接的非声明正文;content 为 None 时按空串处理。
115
- def body_text(content: str) -> str:
116
- """剥掉 CCG 声明行后的正文——验证只认正文,不认已写下的声明。"""
117
- return "\n".join(l for l in (content or "").split("\n")
118
- if not _CCG_LINE_RE.match(l))
119
-
120
- # ---- 双模型角色(反思单元 / 验证单元)-------------------------------------
121
-
122
- REFLECT_ROLE = "reflect" # 反思单元:产出候选
123
- VERIFY_ROLE = "verify" # 验证单元:否决候选(无产出权)
124
- ROLES = (REFLECT_ROLE, VERIFY_ROLE)
125
-
126
- # 推荐模型(真实 API id,经实际调用确认;可用 MDCG_<ROLE>_MODEL 覆盖)
127
- # 注意:reflect 的 id 带过期标记(expires-on-0910),属**限时模型**——过期后 /models
128
- # 列表会下架该 id,届时改用 deepseek-v4-flash 或用 MDCG_REFLECT_MODEL 覆盖。
129
- ROLE_DEFAULT_MODEL = {REFLECT_ROLE: "deepseek-v4.1-flash-expires-on-0910",
130
- VERIFY_ROLE: "glm-5.3-flash"}
131
- ROLE_DEFAULT_BASE = {REFLECT_ROLE: "https://api.deepseek.com",
132
- VERIFY_ROLE: "https://open.bigmodel.cn/api/paas/v4"}
133
- _ROLE_ENV = {REFLECT_ROLE: ("MDCG_REFLECT_MODEL", "MDCG_REFLECT_BASE", "MDCG_REFLECT_KEY"),
134
- VERIFY_ROLE: ("MDCG_VERIFY_MODEL", "MDCG_VERIFY_BASE", "MDCG_VERIFY_KEY")}
135
- # 验证单元 key 的常见别名(智谱系)
136
- VERIFY_KEY_ALIASES = ("ZHIPU_API_KEY", "ZHIPUAI_API_KEY", "GLM_API_KEY", "BIGMODEL_API_KEY")
137
-
138
- # 「验证方式」的声明文本(可 --verification-basis 覆盖 / --no-basis 关闭)
139
- BASIS_TEMPLATE = "双模型交叉验证(反思单元={reflect},验证单元={verify})"
140
- # frontmatter.verification_basis 只能取 nodefile 的枚举;双 LLM 交叉验证 → other
141
- BASIS_ENUM_DEFAULT = "other"
142
-
143
- REFLECT_PROMPT = (
144
- "你是认知图节点的**反思单元**。给定一个知识节点的标题与正文,反思并抽取四要素。\n"
145
- "只输出一个 JSON 对象,不要任何解释或代码围栏。\n"
146
- "字段含义:\n"
147
- ' "生效条件": 什么查询/情境下该知识**适用**(短语数组,2~5 条)\n'
148
- ' "子功能": 该知识包含的子内容/子步骤(短语数组,2~5 条)\n'
149
- ' "执行": 如何执行/如何使用该知识(单个字符串)\n'
150
- ' "不适用条件": 什么查询/情境下该知识**不**适用(短语数组,1~3 条)\n'
151
- "硬约束:\n"
152
- " 1. 每条短语必须能在正文中找到依据,禁止编造正文里没有的工具/概念;\n"
153
- " 2. 不适用条件必须是**正文之外的邻近易混情境**,不得与生效条件语义重叠;\n"
154
- " 3. 短语要短(不超过 20 字),不要写完整句子。\n"
155
- "输出格式:"
156
- '{{"生效条件": ["..."], "子功能": ["..."], "执行": "...", "不适用条件": ["..."]}}\n'
157
- "标题:{title}\n正文:\n{body}"
158
- )
159
-
160
- VERIFY_PROMPT = (
161
- "你是认知图节点的**验证单元**。你的职责是**否决**,不是补充。\n"
162
- "只能从候选里删除不成立的条目,**绝不允许新增或改写任何条目**。\n"
163
- "给定标题、正文与反思单元给出的候选四要素,逐条核验:\n"
164
- " · 该条目是否真的能在正文中找到依据?找不到依据 → 删除;\n"
165
- " · 不适用条件是否真的域外?若它其实是该节点的适用情境 → 删除;\n"
166
- " · 生效条件与不适用条件是否语义重叠?重叠者删除其一(保留更贴合正文的那个)。\n"
167
- "只输出一个 JSON 对象,键为字段名,值为 "
168
- '{{"keep": ["保留的条目"], "drop": ["删除的条目"], "reason": "一句话理由"}}。\n'
169
- "候选:{cand}\n标题:{title}\n正文:\n{body}"
170
- )
171
-
172
-
173
- # ---- LLM 侧(黑箱只在离线工序,产出候选)--------------------------------
174
-
175
- # 生效条件:给定 role,按显式参数、角色环境变量、通用兜底依次解析并返回 (model, base, key);key 不落 DEEPSEEK_API_KEY 除非 role 为 REFLECT_ROLE。
176
- def role_config(role: str, model: str = None, base: str = None,
177
- key: str = None) -> tuple:
178
- """解析某角色的 (model, base, key):显式参数 > 角色环境变量 > 通用兜底。
179
-
180
- key 刻意**不**让验证单元回落到 DEEPSEEK_API_KEY——跨厂商混用会把一个厂商的
181
- 凭证发到另一个厂商的网关,既必然失败又构成凭证外泄。
182
- """
183
- m_env, b_env, k_env = _ROLE_ENV[role]
184
- if key is None:
185
- key = os.environ.get(k_env)
186
- if key is None and role == VERIFY_ROLE:
187
- for alias in VERIFY_KEY_ALIASES:
188
- key = os.environ.get(alias)
189
- if key:
190
- break
191
- if key is None:
192
- key = os.environ.get("MDCG_LLM_KEY")
193
- if key is None and role == REFLECT_ROLE:
194
- key = os.environ.get("DEEPSEEK_API_KEY")
195
- model = (model or os.environ.get(m_env) or os.environ.get("MDCG_LLM_MODEL")
196
- or ROLE_DEFAULT_MODEL[role])
197
- base = (base or os.environ.get(b_env) or os.environ.get("MDCG_LLM_BASE")
198
- or ROLE_DEFAULT_BASE[role])
199
- return model, base, key
200
-
201
-
202
- # 生效条件:给定 prompt 且 role 解析或通用兜底得到非空 key 时,向 base 的 /chat/completions 发 POST 并返回首个 choice 的 message.content;key 为空则抛 RuntimeError。
203
- def http_llm(prompt: str, model: str = None, base: str = None, key: str = None,
204
- role: str = None, timeout: int = 120, max_tokens: int = 1200) -> str:
205
- """标准库 HTTP 调 LLM(OpenAI 兼容 /chat/completions)。零第三方依赖。
206
-
207
- role 给定时按该角色配置解析(reflect / verify),否则走通用配置。
208
- """
209
- if role:
210
- model, base, key = role_config(role, model, base, key)
211
- else:
212
- model = (model or os.environ.get("MDCG_LLM_MODEL")
213
- or ROLE_DEFAULT_MODEL[REFLECT_ROLE])
214
- base = (base or os.environ.get("MDCG_LLM_BASE")
215
- or ROLE_DEFAULT_BASE[REFLECT_ROLE])
216
- key = (key or os.environ.get("MDCG_LLM_KEY")
217
- or os.environ.get("DEEPSEEK_API_KEY"))
218
- if not key:
219
- raise RuntimeError(
220
- f"未配置 {role or 'llm'} 的 API key"
221
- f"({_ROLE_ENV[role][2] if role in _ROLE_ENV else 'MDCG_LLM_KEY'})")
222
- payload = json.dumps({
223
- "model": model,
224
- "messages": [{"role": "user", "content": prompt}],
225
- "max_tokens": max_tokens,
226
- }).encode("utf-8")
227
- req = urllib.request.Request(
228
- base.rstrip("/") + "/chat/completions", data=payload,
229
- headers={"Authorization": f"Bearer {key}",
230
- "Content-Type": "application/json"})
231
- try:
232
- with urllib.request.urlopen(req, timeout=timeout) as resp:
233
- data = json.loads(resp.read().decode("utf-8"))
234
- except urllib.error.HTTPError as exc:
235
- detail = exc.read().decode("utf-8", "replace")[:300]
236
- raise RuntimeError(f"HTTP {exc.code} model={model} base={base} :: {detail}") from None
237
- return data["choices"][0]["message"]["content"]
238
-
239
-
240
- # 生效条件:给定 role,若 role_config 得到非空 key 则 GET base/models 并返回含 ok/model_available/models 的字典;无 key 或请求异常则返回 ok=False 及错误信息。
241
- def probe_models(role: str, timeout: int = 20) -> dict:
242
- """零 token 探测:列出该角色网关的可用模型 id(GET /models)。"""
243
- model, base, key = role_config(role)
244
- if not key:
245
- return {"role": role, "model": model, "base": base, "ok": False,
246
- "error": "no_key", "models": []}
247
- req = urllib.request.Request(
248
- base.rstrip("/") + "/models",
249
- headers={"Authorization": f"Bearer {key}"})
250
- try:
251
- with urllib.request.urlopen(req, timeout=timeout) as resp:
252
- data = json.loads(resp.read().decode("utf-8"))
253
- ids = [m.get("id") for m in (data.get("data") or []) if m.get("id")]
254
- except Exception as exc: # noqa: BLE001 —— 探测要抗单点
255
- return {"role": role, "model": model, "base": base, "ok": False,
256
- "error": f"{type(exc).__name__}: {exc}"[:200], "models": []}
257
- res = {"role": role, "model": model, "base": base, "ok": True,
258
- "model_available": model in ids, "models": ids}
259
- if not res["model_available"]:
260
- res["note"] = ("该 id 未出现在 /models 列表:可能是限时/按需模型,或已下架;"
261
- "以实际 /chat/completions 调用结果为准")
262
- return res
263
-
264
-
265
- # 生效条件:raw 为 None 或 strip 后不含 "{"(i<0)、或末个 "}" 的位置 j<=i 时返回 None;否则对 s 从首个 "{" 到末个 "}" 的切片 json.loads,成功则返回解析结果,抛 ValueError 时返回 None。
266
- def _extract_json_obj(raw: str):
267
- """从 LLM 输出里抠出第一个 JSON 对象(容忍代码围栏 / 前后废话)。"""
268
- s = (raw or "").strip()
269
- i, j = s.find("{"), s.rfind("}")
270
- if i < 0 or j <= i:
271
- return None
272
- try:
273
- return json.loads(s[i:j + 1])
274
- except ValueError:
275
- return None
276
-
277
-
278
- # 生效条件:v 为 None 返回 [];否则按 v 是 str 取 [v]、是 list/tuple 取逐项、其他取 [str(v)],逐项 strip 并去两端包裹标点后跳过空串及长度超 MAX_TERM_LEN 的项,未出现过的才 append,每次 append 后若 len(out) >= limit 即 break 返回 out(故 limit<=0 且存在有效项时仍返回 1 项)。
279
- def _as_terms(v, limit: int = MAX_TERMS):
280
- """把 LLM 给的值规范成去重、限长的短语列表。"""
281
- if v is None:
282
- return []
283
- if isinstance(v, str):
284
- items = [v]
285
- elif isinstance(v, (list, tuple)):
286
- items = list(v)
287
- else:
288
- items = [str(v)]
289
- out = []
290
- for x in items:
291
- s = str(x).strip().strip(",。;;、,.;\"'“”")
292
- if not s or len(s) > MAX_TERM_LEN:
293
- continue
294
- if s not in out:
295
- out.append(s)
296
- if len(out) >= limit:
297
- break
298
- return out
299
-
300
-
301
- # 生效条件:给定 raw,若 _extract_json_obj 解析出 dict,则按 CCG_FIELDS 与 FIELD_ALIASES 提取非空字段并规范为列表或单值返回字典;否则返回 {}。
302
- def parse_candidate(raw: str) -> dict:
303
- """LLM 原始输出 → {字段: 列表/字符串};解析失败返回 {}。"""
304
- obj = _extract_json_obj(raw)
305
- if not isinstance(obj, dict):
306
- return {}
307
- out = {}
308
- for field in CCG_FIELDS:
309
- val = None
310
- for alias in FIELD_ALIASES[field]:
311
- if alias in obj and obj[alias] not in (None, "", [], {}):
312
- val = obj[alias]
313
- break
314
- if val is None:
315
- continue
316
- if field in SINGLE_FIELDS:
317
- terms = _as_terms(val, limit=1)
318
- if terms:
319
- out[field] = terms[0]
320
- else:
321
- terms = _as_terms(val)
322
- if terms:
323
- out[field] = terms
324
- return out
325
-
326
-
327
- # 生效条件:给定 raw,若解析出 dict,则按 CCG_FIELDS 提取 keep/drop/has_keep/reason 结构返回字典;否则返回 {}。
328
- def parse_verdict(raw: str) -> dict:
329
- """验证单元输出 → {字段: {keep, drop, has_keep, reason}};解析失败返回 {}。"""
330
- obj = _extract_json_obj(raw)
331
- if not isinstance(obj, dict):
332
- return {}
333
- out = {}
334
- for field in CCG_FIELDS:
335
- val = None
336
- for alias in FIELD_ALIASES[field]:
337
- if alias in obj and obj[alias] not in (None, "", [], {}):
338
- val = obj[alias]
339
- break
340
- if val is None:
341
- continue
342
- if isinstance(val, list): # 容忍只给 keep 数组
343
- out[field] = {"keep": _as_terms(val), "drop": [], "has_keep": True,
344
- "reason": ""}
345
- elif isinstance(val, dict):
346
- out[field] = {"keep": _as_terms(val.get("keep")),
347
- "drop": _as_terms(val.get("drop")),
348
- "has_keep": "keep" in val,
349
- "reason": str(val.get("reason") or "")[:200]}
350
- return out
351
-
352
-
353
- # 生效条件:给定 kept 与 verdict,若 verdict 为空则返回 (dict(kept), {});否则按 drop 与 has_keep 收窄候选并返回 (收窄后候选, 被剔除明细)。
354
- def narrow_by_verdict(kept: dict, verdict: dict):
355
- """按验证单元裁决收窄候选——**只能否决,不能新增**。
356
-
357
- · 验证单元未表态的字段 → 保留(沉默不等于否决)
358
- · has_keep=True → 取「候选 ∩ keep」;否则只按 drop 剔除
359
- 返回 (收窄后候选, 被剔除明细)。
360
- """
361
- if not verdict:
362
- return dict(kept), {}
363
- out, dropped = {}, {}
364
- for field, val in kept.items():
365
- terms = val if isinstance(val, list) else [val]
366
- vd = verdict.get(field)
367
- if vd is None:
368
- out[field] = val
369
- continue
370
- dropset = set(vd.get("drop") or [])
371
- keepset = set(vd.get("keep") or [])
372
- surv, gone = [], []
373
- for t in terms:
374
- if t in dropset:
375
- gone.append(t)
376
- elif vd.get("has_keep") and t not in keepset:
377
- gone.append(t)
378
- else:
379
- surv.append(t)
380
- if gone:
381
- dropped[field] = {"terms": gone, "reason": vd.get("reason") or ""}
382
- if surv:
383
- out[field] = surv if field in MULTI_FIELDS else surv[0]
384
- return out, dropped
385
-
386
-
387
- # ---- 确定性验证(零 LLM)-------------------------------------------------
388
-
389
- # 生效条件:给定 term 与 body,若 term 的 bigram 序列非空则返回命中 bigram 数除以总 bigram 数,否则返回 0.0。
390
- def grounding_score(term: str, body: str) -> float:
391
- """候选短语在正文里的字符级支撑度 = 命中 bigram 数 / 总 bigram 数。"""
392
- bg = bigrams(term or "")
393
- if not bg:
394
- return 0.0
395
- hit = sum(1 for g in bg if g in (body or ""))
396
- return hit / len(bg)
397
-
398
-
399
- # 生效条件:给定 cand 与 body,按 thresholds 更新 DEFAULT_GROUNDING 后逐字段过滤候选,返回 (达标 kept, detail);不达标者丢弃。
400
- def grounding_filter(cand: dict, body: str, thresholds: dict = None):
401
- """逐字段过滤候选:返回 (kept, detail)。不达标者丢弃(对应「不猜测」)。"""
402
- th = dict(DEFAULT_GROUNDING)
403
- th.update(thresholds or {})
404
- kept, detail = {}, {}
405
- for field, val in cand.items():
406
- terms = val if isinstance(val, list) else [val]
407
- ok_terms, scores = [], {}
408
- for t in terms:
409
- g = grounding_score(t, body)
410
- scores[t] = round(g, 3)
411
- if g >= th.get(field, 0.5):
412
- ok_terms.append(t)
413
- detail[field] = {"scores": scores, "kept": len(ok_terms)}
414
- if ok_terms:
415
- kept[field] = ok_terms if field in MULTI_FIELDS else ok_terms[0]
416
- return kept, detail
417
-
418
-
419
- # 生效条件:给定 pos_terms、neg_terms、body,返回含 pos_recall、neg_separated、no_conflict、ok 的回放判定字典。
420
- def replay_check(pos_terms, neg_terms, body: str) -> dict:
421
- """回放生产判定:正例召回 + 负例剔除 + 无自相矛盾。
422
-
423
- 复用 _path_semantic 的同一批原语,保证与生产路同源(P6 与真实路做一致性回归)。
424
- """
425
- pos_text = " ".join(pos_terms or [])
426
- neg_text = " ".join(neg_terms or [])
427
- tw_pos = expand_query_terms_weighted(pos_text) if pos_text else {}
428
- tw_neg = expand_query_terms_weighted(neg_text) if neg_text else {}
429
-
430
- # 1. 正例:以生效条件为查询,本节点正文应被命中,且不被自身负条件挡住
431
- pos_recall = bool(pos_text) and _weighted_coverage(tw_pos, body) > 0.0 \
432
- and not _neg_hit(tw_pos, neg_terms)
433
-
434
- # 2. 负例:以不适用条件为查询,应触发条件级负路由;且负条件与正文低相关
435
- # (负条件必须是「域外」的,若与正文强相关,等于让知识否定自己)
436
- if neg_terms:
437
- neg_separated = _neg_hit(tw_neg, neg_terms) \
438
- and _weighted_coverage(tw_neg, body) < 0.5
439
- else:
440
- neg_separated = True
441
-
442
- # 3. 生效条件与不适用条件不得互相覆盖
443
- no_conflict = (not pos_text) or (not neg_text) \
444
- or _weighted_coverage(tw_pos, neg_text) < 0.5
445
-
446
- ok = pos_recall and neg_separated and no_conflict
447
- return {"pos_recall": pos_recall, "neg_separated": neg_separated,
448
- "no_conflict": no_conflict, "ok": ok}
449
-
450
-
451
- # ---- 写盘(固化)---------------------------------------------------------
452
-
453
- # 生效条件:给定 content 与 field,当 content 含 "# field:" 或 "# field:" 时返回 True,否则 False。
454
- def _has_ccg_line(content: str, field: str) -> bool:
455
- return f"# {field}:" in (content or "") or f"# {field}:" in (content or "")
456
-
457
-
458
- # 生效条件:给定 fm 与 content,对每个 CCG_FIELDS,若 frontmatter.comment 值非空或正文含对应 CCG 行则记入,返回已有字段字典。
459
- def existing_fields(fm: dict, content: str) -> dict:
460
- """节点当前已有的四要素:正文 CCG 行 或 frontmatter.comment 任一存在即算有。"""
461
- comment = (fm.get("state_attributes") or {}).get("comment") or {}
462
- out = {}
463
- for field in CCG_FIELDS:
464
- v = comment.get(field)
465
- if v not in (None, "", [], {}):
466
- out[field] = v
467
- elif _has_ccg_line(content, field):
468
- out[field] = True
469
- return out
470
-
471
-
472
- # 生效条件:给定 content、field、value,若已有 "# field:" 行则替换并返回新正文;否则插在 "# 功能名" 之后,若无则该行前置。
473
- def _upsert_ccg_line(content: str, field: str, value: str) -> str:
474
- """在正文里写入/替换 `# <字段>:<值>`,优先插在「# 功能名」之后。"""
475
- lines = (content or "").split("\n")
476
- for i, ln in enumerate(lines):
477
- s = ln.strip()
478
- if not s.startswith("#") or field not in s:
479
- continue
480
- name = s.lstrip("#").strip().split(":")[0].split(":")[0].strip()
481
- if name == field:
482
- lines[i] = f"# {field}:{value}"
483
- return "\n".join(lines)
484
- newline = f"# {field}:{value}"
485
- for i, ln in enumerate(lines):
486
- if ln.strip().startswith("# 功能名"):
487
- lines.insert(i + 1, newline)
488
- return "\n".join(lines)
489
- return newline + "\n" + (content or "")
490
-
491
-
492
- # 生效条件:给定 kept 字段字典,返回一句话规律字符串,列出缺失字段名并声明补齐后可路由。
493
- def _evo_pattern(kept: dict) -> str:
494
- """规律(一句话):这一类节点反复缺的正是这批条件。"""
495
- names = "、".join(kept.keys())
496
- return f"缺「{names}」的节点条件不可判;补齐后四要素完整、可路由"
497
-
498
-
499
- # 生效条件:给定 prov 字典,拼接 reflect/verify 模型、grounding、replay、verification_basis 中存在的证据项并返回。
500
- def _evo_evidence(prov: dict) -> str:
501
- """证据:本次固化凭什么成立(模型 / 闸门 / 回放)。"""
502
- parts = []
503
- rf = (prov.get("reflect") or {}).get("model") or ""
504
- vf = (prov.get("verify") or {}).get("model") or ""
505
- if rf:
506
- parts.append(f"reflect={rf}")
507
- if vf:
508
- parts.append(f"verify={vf}")
509
- if prov.get("grounding"):
510
- parts.append("grounding通过")
511
- if prov.get("replay"):
512
- parts.append("replay通过")
513
- vb = prov.get("verification_basis") or ""
514
- if vb:
515
- parts.append(vb)
516
- return " · ".join(parts)
517
-
518
-
519
- # 生效条件:给定 cg、e、fm、content、kept、prov,将 kept 字段写入正文 CCG 行与 frontmatter.comment,不适用条件同步 non_applicable_conditions,并写 llm_consolidation 与演化记录,返回 None。
520
- def _apply_node(cg, e, fm: dict, content: str, kept: dict, prov: dict,
521
- basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT):
522
- """把通过验证的字段固化进 md:正文 CCG 行 + frontmatter.comment + 负条件 + provenance。
523
-
524
- 固化 = 对一条缺失条件的补充 → 同步落一条演化条目(md 账本,可回滚)。
525
- """
526
- nid = e.get("id") or os.path.basename(e["path"])[:-3]
527
- before = evolution.state_of(cg, nid) or {}
528
- comment = (fm.get("state_attributes") or {}).get("comment")
529
- if not isinstance(comment, dict):
530
- fm["state_attributes"] = dict(fm.get("state_attributes") or {})
531
- fm["state_attributes"]["comment"] = {}
532
- comment = fm["state_attributes"]["comment"]
533
- for field, val in kept.items():
534
- text = ";".join(val) if isinstance(val, list) else str(val)
535
- content = _upsert_ccg_line(content, field, text)
536
- comment[field] = text
537
- if field == "不适用条件":
538
- # 同步 frontmatter.non_applicable_conditions(引擎负路由读它)
539
- cur = [str(x) for x in (fm.get("non_applicable_conditions") or [])]
540
- for t in (val if isinstance(val, list) else [val]):
541
- if t not in cur:
542
- cur.append(t)
543
- fm["non_applicable_conditions"] = cur
544
- if basis:
545
- content = _upsert_ccg_line(content, "验证方式", basis)
546
- comment["验证方式"] = basis
547
- if not nodefile.verification_basis_valid(fm):
548
- # 枚举里没有「LLM 交叉验证」这一档,只能落到 other(声明文本在 CCG 行里)
549
- fm["verification_basis"] = basis_enum
550
- fm["llm_consolidation"] = prov
551
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
552
- durable=True)
553
- # 每一次修改都是对缺失条件的补充:记录规律 + 状态,不记录实现。
554
- evolution.record(
555
- cg, node_id=nid,
556
- pattern=_evo_pattern(kept),
557
- missing="、".join(kept.keys()),
558
- action="补齐 CCG 字段:" + "、".join(kept.keys()),
559
- evidence=_evo_evidence(prov),
560
- source="consolidate", kind=evolution.KIND_CONDITION_GAP,
561
- before=before, after=evolution.state_of(cg, nid) or {})
562
-
563
-
564
- # 生效条件:给定 root,扫描正排层节点并执行反思→白箱闸门→验证→固化,返回报表 rep;require_verify=True 且无 verify_fn 时全部 DEFER。
565
- def consolidate(root: str, layer: str = None, limit: int = None, apply: bool = False,
566
- overwrite: bool = False, llm_fn=None, reflect_fn=None,
567
- verify_fn=None, reflect_model: str = "", verify_model: str = "",
568
- verification_basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT,
569
- require_verify: bool = True, thresholds: dict = None,
570
- verbose: bool = True) -> dict:
571
- """对正排层节点做「反思单元产出候选 → 白箱闸门 → 验证单元否决 → 固化」。
572
-
573
- llm_fn 是 reflect_fn 的旧名(向后兼容,单模型模式)。
574
- require_verify=True 且无 verify_fn → 一律 DEFER(纪律 5:未经验证不固化)。
575
- """
576
- reflect_fn = reflect_fn or llm_fn
577
- cg = MdCGOS(root)
578
- entries = cg._candidates(layer=layer)
579
- t0 = time.time()
580
- rep = {"root": root, "layer": layer, "dry_run": not apply,
581
- "reflect_model": reflect_model, "verify_model": verify_model,
582
- "reflect": bool(reflect_fn), "verify": bool(verify_fn), "llm": bool(reflect_fn),
583
- "require_verify": require_verify,
584
- "verification_basis": verification_basis,
585
- "nodes_scanned": len(entries),
586
- "targeted": 0, "accepted": 0, "rejected": 0, "deferred": 0,
587
- "skipped_complete": 0, "written": 0, "reasons": {},
588
- "per_field": {f: 0 for f in CCG_FIELDS}, "verify_dropped": 0,
589
- "verification_basis_missing": 0, "samples": []}
590
-
591
- # 生效条件:以 reason 为键写入闭包 rep["reasons"],计数按 rep["reasons"].get(reason, 0) + 1 递增(键缺失从 0 起算),无返回值。
592
- def _bump(reason):
593
- rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
594
-
595
- for e in entries:
596
- if limit is not None and rep["targeted"] >= limit:
597
- break
598
- nid = os.path.basename(e["path"])[:-3]
599
- fm, content = cg._read(e)
600
- if fm is None:
601
- _bump("read_failed")
602
- continue
603
- if crypto.is_encrypted(content):
604
- _bump("locked") # 无密钥 → fail-closed:绝不改写密文
605
- continue
606
- have = existing_fields(fm, content)
607
- missing = [f for f in CCG_FIELDS if f not in have]
608
- if not nodefile.verification_basis_valid(fm):
609
- rep["verification_basis_missing"] += 1
610
- if not missing:
611
- rep["skipped_complete"] += 1
612
- continue
613
- rep["targeted"] += 1
614
-
615
- if not reflect_fn:
616
- rep["deferred"] += 1
617
- _bump("no_llm")
618
- continue
619
-
620
- # 1) 反思单元:产出候选(黑箱,唯一产出权)
621
- body = body_text(content)[:MAX_BODY_CHARS]
622
- title = _ccg_field(content, "功能名") or nid
623
- prompt = REFLECT_PROMPT.format(title=title, body=body)
624
- try:
625
- raw = reflect_fn(prompt)
626
- except Exception as exc: # noqa: BLE001 —— 离线批处理要抗单点失败
627
- rep["deferred"] += 1
628
- _bump(f"reflect_error:{type(exc).__name__}")
629
- continue
630
- cand = parse_candidate(raw)
631
- if not cand:
632
- rep["deferred"] += 1
633
- _bump("parse_failed")
634
- continue
635
-
636
- # 2) 白箱闸门:grounding + replay(零 LLM,先跑,省调用)
637
- kept, gdetail = grounding_filter(cand, body, thresholds)
638
- pos = kept.get("生效条件") or []
639
- neg = kept.get("不适用条件") or []
640
- replay = replay_check(pos, neg, body)
641
- if not kept or not replay["ok"]:
642
- rep["rejected"] += 1
643
- _bump("replay_failed" if kept else "grounding_failed")
644
- if verbose and len(rep["samples"]) < 8:
645
- rep["samples"].append({"id": nid, "verdict": "REJECT",
646
- "stage": "whitebox", "grounding": gdetail,
647
- "replay": replay})
648
- continue
649
-
650
- # 3) 验证单元:逐条核验,只能否决、不能新增
651
- dropped, vprompt, vd = {}, "", None
652
- if verify_fn:
653
- vprompt = VERIFY_PROMPT.format(
654
- cand=json.dumps(kept, ensure_ascii=False), title=title, body=body)
655
- try:
656
- vd = parse_verdict(verify_fn(vprompt))
657
- kept, dropped = narrow_by_verdict(kept, vd)
658
- except Exception as exc: # noqa: BLE001
659
- rep["deferred"] += 1
660
- _bump(f"verify_error:{type(exc).__name__}")
661
- continue
662
- if not kept:
663
- rep["rejected"] += 1
664
- _bump("verify_rejected")
665
- if verbose and len(rep["samples"]) < 8:
666
- rep["samples"].append({"id": nid, "verdict": "REJECT",
667
- "stage": "verify", "dropped": dropped})
668
- continue
669
- rep["verify_dropped"] += sum(len(d["terms"]) for d in dropped.values())
670
- elif require_verify:
671
- # 验证单元不可用 → 不固化(纪律 5:未经验证不固化)
672
- rep["deferred"] += 1
673
- _bump("verify_unavailable")
674
- continue
675
-
676
- # 3) 不覆盖已有非空字段(保护人工既有知识)
677
- if not overwrite:
678
- kept = {f: v for f, v in kept.items() if f not in have}
679
- if not kept:
680
- rep["skipped_complete"] += 1
681
- continue
682
-
683
- prov = {"at": round(time.time(), 3), "verdict": "ACCEPT",
684
- "source_hash": _sig(content),
685
- "reflect": {"model": reflect_model, "prompt_hash": _sig(prompt),
686
- "fields": sorted(kept)},
687
- "verify": ({"model": verify_model, "prompt_hash": _sig(vprompt),
688
- "dropped": dropped, "verdict_fields": sorted(vd or {}),
689
- "self_verify": verify_fn is reflect_fn}
690
- if verify_fn else {"model": "", "status": "skipped"}),
691
- "grounding": gdetail, "replay": replay,
692
- "verification_basis": verification_basis}
693
- rep["accepted"] += 1
694
- for f in kept:
695
- rep["per_field"][f] += 1
696
- if apply:
697
- _apply_node(cg, e, fm, content, kept, prov,
698
- verification_basis, basis_enum)
699
- append_jsonl(os.path.join(cg.root, "_consolidate.jsonl"),
700
- {"t": time.time(), "id": nid, "verdict": "ACCEPT",
701
- "fields": sorted(kept),
702
- "reflect_model": reflect_model,
703
- "verify_model": verify_model, "dropped": dropped,
704
- "source_hash": prov["source_hash"], "replay": replay})
705
- rep["written"] += 1
706
- if verbose and len(rep["samples"]) < 8:
707
- rep["samples"].append({"id": nid, "verdict": "ACCEPT",
708
- "fields": sorted(kept), "dropped": dropped,
709
- "replay": replay})
710
-
711
- if apply and rep["written"]:
712
- # 正文新增了 `# 不适用条件:` / `# 验证方式:` → 索引字段变了
713
- cg.rebuild_index()
714
- rep["elapsed_sec"] = round(time.time() - t0, 3)
715
- return rep
716
-
717
-
718
- # 生效条件:给定 root 与 basis,对缺 "# 验证方式" 行的节点补写验证方式并在需要时写入 basis_enum,返回统计 rep。
719
- def fill_verification_basis(root: str, basis: str, layer: str = None,
720
- limit: int = None, apply: bool = False,
721
- basis_enum: str = BASIS_ENUM_DEFAULT) -> dict:
722
- """只补「验证方式」——声明文本是常量,不需要黑箱生成,零 LLM 成本。
723
-
724
- 对应纪律 3「不猜测」:验证基底必须由人/流程声明,而不是让模型编出来。
725
- """
726
- cg = MdCGOS(root)
727
- entries = cg._candidates(layer=layer)
728
- rep = {"root": root, "layer": layer, "dry_run": not apply, "basis": basis,
729
- "basis_enum": basis_enum, "nodes_scanned": len(entries),
730
- "targeted": 0, "skipped_present": 0, "skipped_locked": 0,
731
- "written": 0}
732
- for e in entries:
733
- if limit is not None and rep["written"] >= limit:
734
- break
735
- fm, content = cg._read(e)
736
- if fm is None:
737
- continue
738
- if crypto.is_encrypted(content):
739
- rep["skipped_locked"] += 1 # 无密钥 → fail-closed:绝不改写密文
740
- continue
741
- if _has_ccg_line(content, "验证方式"):
742
- rep["skipped_present"] += 1
743
- continue
744
- rep["targeted"] += 1
745
- if not apply:
746
- continue
747
- comment = (fm.get("state_attributes") or {}).get("comment")
748
- if not isinstance(comment, dict):
749
- fm["state_attributes"] = dict(fm.get("state_attributes") or {})
750
- fm["state_attributes"]["comment"] = {}
751
- comment = fm["state_attributes"]["comment"]
752
- content = _upsert_ccg_line(content, "验证方式", basis)
753
- comment["验证方式"] = basis
754
- if not nodefile.verification_basis_valid(fm):
755
- fm["verification_basis"] = basis_enum
756
- nid = e.get("id") or os.path.basename(e["path"])[:-3]
757
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
758
- durable=True)
759
- rep["written"] += 1
760
- if apply and rep["written"]:
761
- cg.rebuild_index()
762
- return rep
763
-
764
-
765
- # ==========================================================================
766
- # 情境层批量提升(consolidate.promote)
767
- # ==========================================================================
768
- #
769
- # 场景:情境层(contextual)里有些记忆被反复命中/并入——它们已经不是「一次情境」,
770
- # 而是稳定的规律。本动作把它们提升为长期知识(knowledge),并保留:
771
- # · 双向可追溯:promoted_from + 演化账本(KIND_LAYER_SHIFT);
772
- # · 条件门槛:四要素(CCG)不全者**不提升**(未可判定就不该升格为长期知识);
773
- # · 可预演:apply=False 只出报表;可留痕:`_maintain.jsonl`。
774
-
775
- MAINTAIN_LOG = "_maintain.jsonl"
776
- CCG_REQUIRED = ("生效条件", "子功能", "执行", "不适用条件")
777
-
778
-
779
- # 生效条件:给定 cg、nid、e、fm、content、target_layer,把节点写入目标层(必要时按 routing 分桶)并删除旧路径,返回新相对路径与 bucket。
780
- def _relocate_layer(cg, nid, e, fm, content, target_layer):
781
- """把节点正文迁到目标层的正确目录(含分桶),删除旧文件。返回新相对路径。"""
782
- d = os.path.join(cg.root, target_layer)
783
- bucket = None
784
- if target_layer in BUCKETED_LAYERS:
785
- bucket = routing.bucket_dir(routing.route_key(fm.get("condition_space"),
786
- fm.get("tags")))
787
- d = os.path.join(d, bucket)
788
- os.makedirs(d, exist_ok=True)
789
- new_path = os.path.join(d, f"{nid}.md")
790
- old_path = os.path.join(cg.root, e.get("path") or f"{nid}.md")
791
- cg._write_node(nid, new_path, fm, content, durable=True)
792
- if os.path.abspath(old_path) != os.path.abspath(new_path) and os.path.exists(old_path):
793
- os.remove(old_path)
794
- return {"path": os.path.relpath(new_path, cg.root).replace("\\", "/"),
795
- "bucket": bucket}
796
-
797
-
798
- # 生效条件:给定 root,把 source_layer 中命中次数不小于 min_merge 或 importance 不小于 min_importance 且条件完整的节点提升到 target_layer,返回统计 rep。
799
- def promote_memories(root, source_layer="contextual", target_layer="knowledge",
800
- min_merge=2, min_importance=0.6, require_conditions=True,
801
- limit=None, apply=False, actor="maintain") -> dict:
802
- """把反复命中的情境记忆批量提升为长期知识(可预演 / 可留痕 / 可追溯)。"""
803
- cg = MdCGOS(root)
804
- entries = cg._candidates(layer=source_layer)
805
- rep = {"root": root, "source_layer": source_layer, "target_layer": target_layer,
806
- "dry_run": not apply, "nodes_scanned": len(entries), "targeted": 0,
807
- "skipped_locked": 0, "skipped_incomplete": 0, "skipped_not_hot": 0,
808
- "written": 0, "promoted": [], "samples": [],
809
- "min_merge": min_merge, "min_importance": min_importance,
810
- "require_conditions": bool(require_conditions)}
811
- batch = time.strftime("%Y%m%d-%H%M%S")
812
- for e in entries:
813
- if limit is not None and rep["written"] >= int(limit):
814
- break
815
- fm, content = cg._read(e)
816
- if fm is None:
817
- continue
818
- if crypto.is_encrypted(content):
819
- rep["skipped_locked"] += 1 # 无密钥 → fail-closed,绝不解密回写
820
- continue
821
- nid = e.get("id") or os.path.basename(e["path"])[:-3]
822
- hits = max(int(fm.get("merge_count") or 0),
823
- int(fm.get("access_count") or 0),
824
- int(fm.get("recall_count") or 0))
825
- imp = float(fm.get("importance") or e.get("importance") or 0.0)
826
- complete = all(_has_ccg_line(content, f) for f in CCG_REQUIRED)
827
- if require_conditions and not complete:
828
- rep["skipped_incomplete"] += 1 # 四要素不全 → 不可判定,不升格
829
- continue
830
- hot = hits >= int(min_merge)
831
- if not hot and imp < float(min_importance):
832
- rep["skipped_not_hot"] += 1
833
- continue
834
- rep["targeted"] += 1
835
- item = {"id": nid, "hits": hits, "importance": round(imp, 4),
836
- "conditions_complete": complete,
837
- "basis": fm.get("verification_basis")}
838
- if len(rep["samples"]) < 8:
839
- rep["samples"].append(item)
840
- if not apply:
841
- continue
842
- before = evolution.state_of(cg, nid) or {}
843
- fm["layer"] = target_layer
844
- fm["promoted_from"] = source_layer
845
- fm["promoted_at"] = time.time()
846
- fm["promotion_basis"] = {"hits": hits, "importance": round(imp, 4),
847
- "conditions_complete": complete, "batch": batch,
848
- "actor": actor}
849
- moved = _relocate_layer(cg, nid, e, fm, content, target_layer)
850
- evolution.record(
851
- cg, node_id=nid,
852
- pattern="情境记忆反复命中/并入 → 提升为长期知识",
853
- missing="", action=f"层迁移 {source_layer}→{target_layer}",
854
- evidence=f"hits={hits} importance={imp:.2f} conditions_complete={complete}",
855
- source="consolidate", kind=evolution.KIND_LAYER_SHIFT,
856
- before=before, after=evolution.state_of(cg, nid) or {})
857
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
858
- "t": time.time(), "action": "promote", "batch": batch, "id": nid,
859
- "from": source_layer, "to": target_layer, "hits": hits,
860
- "importance": round(imp, 4), "path": moved["path"], "actor": actor})
861
- rep["promoted"].append(nid)
862
- rep["written"] += 1
863
- if apply and rep["written"]:
864
- cg.rebuild_index()
865
- rep["note"] = ("dry-run:未写盘;apply=True 才迁移层"
866
- if not apply else f"已提升 {rep['written']} 个节点到 {target_layer}")
867
- return rep
868
-
869
-
870
- # 生效条件:给定 root,按 _maintain.jsonl 中 action=promote 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
871
- def rollback_promotion(root, node_ids=None, batch=None, actor="maintain") -> dict:
872
- """回滚情境提升:把 promoted_from 层迁回,并记一条演化条目。"""
873
- cg = MdCGOS(root)
874
- recs = [r for r in _read_maintain(root)
875
- if r.get("action") == "promote"
876
- and (not batch or r.get("batch") == batch)
877
- and (not node_ids or str(r.get("id")) in {str(x) for x in node_ids})]
878
- if not recs:
879
- return {"ok": False, "error": "no_records", "reverted": 0}
880
- reverted, ids = 0, []
881
- for rec in recs:
882
- nid = rec["id"]
883
- e = (cg.index.get("nodes") or {}).get(nid)
884
- if not e:
885
- continue
886
- fm, content = cg._read(e)
887
- if fm is None or crypto.is_encrypted(content):
888
- continue
889
- back = rec.get("from") or "contextual"
890
- before = evolution.state_of(cg, nid) or {}
891
- fm["layer"] = back
892
- fm["promoted_from"] = None
893
- fm["promotion_basis"] = {"rollback_of": rec.get("batch"), "actor": actor}
894
- _relocate_layer(cg, nid, e, fm, content, back)
895
- evolution.record(cg, node_id=nid, pattern="提升回滚:长期知识退回情境层",
896
- action=f"层迁移 {rec.get('to')}→{back}",
897
- evidence=f"rollback batch={rec.get('batch')}",
898
- source="consolidate", kind=evolution.KIND_ROLLBACK,
899
- before=before, after=evolution.state_of(cg, nid) or {})
900
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
901
- "t": time.time(), "action": "promote_rollback", "batch": rec.get("batch"),
902
- "id": nid, "to": back, "actor": actor})
903
- reverted += 1
904
- ids.append(nid)
905
- if reverted:
906
- cg.rebuild_index()
907
- return {"ok": True, "reverted": reverted, "ids": ids}
908
-
909
-
910
- # ==========================================================================
911
- # 层归位(consolidate.contextualize)
912
- # ==========================================================================
913
- #
914
- # 场景:批次流水账(note_/milestone_/retest6_)与感知产物(imgpart_/vpipe_)混在
915
- # knowledge 层——它们的语义是**情境**(某次批次的记录 / 某张图的一次观测),不是
916
- # 长期知识;但也不该进 rejected/unresolved(那是「失效 / 未解」,不是「情境」)。
917
- # 故归位到 contextual:
918
- # · 只改 layer 与落点目录;正文 / 密级 / id / tags 一律不动;
919
- # · **保留可召回**(contextual 已在层白名单内,且 predict._SAFE_LAYERS 含之);
920
- # · 可预演(apply=False)/ 可留痕(`_maintain.jsonl`)/ 可追溯(KIND_LAYER_SHIFT)
921
- # / 可回滚(按 batch 或 id 反向迁层)。
922
- #
923
- # 与 promote 的关系:promote 是 contextual→knowledge(升格),本动作是
924
- # knowledge→contextual(归位)。两者共用 `_relocate_layer` 与批次台账,方向相反。
925
-
926
- CONTEXTUALIZE_REASON_DEFAULT = "情境性内容归位(批次流水账 / 感知产物)"
927
-
928
-
929
- # 生效条件:给定 e,返回 e.id 字符串,若缺 id 则回落到 basename(e.path) 去掉 .md。
930
- def _entry_id(e) -> str:
931
- """索引条目取 id:优先 `id` 字段,回落到文件名(索引不保证带 id)。"""
932
- return str(e.get("id") or os.path.basename(e.get("path") or "")[:-3])
933
-
934
-
935
- # 生效条件:给定 root 与 base,若 base 不在维护日志已用批次中则返回 base,否则返回 base.n 且 n 为最小未用序号。
936
- def _unique_batch(root, base) -> str:
937
- """批次号去重:**同一秒内的两次调用不得共用批次号**。
938
-
939
- 否则「按批次回滚」会连带命中上一次的台账记录(回滚必须是精确的、可对账的)。
940
- """
941
- seen = {r.get("batch") for r in _read_maintain(root)}
942
- if base not in seen:
943
- return base
944
- n = 2
945
- while f"{base}.{n}" in seen:
946
- n += 1
947
- return f"{base}.{n}"
948
-
949
-
950
- # 生效条件:给定 root 且 prefixes 或 node_ids 至少一个非空,把 source_layer 中匹配的节点迁到 target_layer,返回统计 rep;两者皆空则抛 ValueError。
951
- def contextualize_prefixes(root, prefixes=None, node_ids=None,
952
- source_layer="knowledge", target_layer="contextual",
953
- reason="", limit=None, apply=False,
954
- actor="maintain") -> dict:
955
- """按 id 前缀(或定向 id 列表)把节点从 source_layer 归位到 target_layer。
956
-
957
- 默认方向 knowledge→contextual。`prefixes` / `node_ids` **至少给一个**:
958
- 宁可少搬,不可全库乱搬——不传白名单直接报错,拒绝「一次误调用把整个知识层改层」
959
- 这种不可归因的批量改写。`node_ids` 用于定向(含「回滚后单独补迁」的对称操作)。
960
- """
961
- pref = tuple(str(p) for p in (prefixes or ()) if str(p))
962
- ids = {str(i) for i in (node_ids or ()) if str(i)} or None
963
- if not pref and not ids:
964
- raise ValueError("contextualize 需要显式 prefixes 或 node_ids"
965
- "(如 ['note_','imgpart_']),拒绝对整层无差别改写")
966
- cg = MdCGOS(root)
967
-
968
- # 生效条件:e 经 _entry_id 得到 nid 后,若闭包 ids 不为 None 则返回 nid in ids 的真假,若 ids 为 None 则返回 nid.startswith(pref) 的真假。
969
- def _hit(e) -> bool:
970
- nid = _entry_id(e)
971
- return nid in ids if ids is not None else nid.startswith(pref)
972
-
973
- entries = [e for e in cg._candidates(layer=source_layer) if _hit(e)]
974
- batch = _unique_batch(root, time.strftime("%Y%m%d-%H%M%S"))
975
- rep = {"root": root, "action": "contextualize", "dry_run": not apply,
976
- "source_layer": source_layer, "target_layer": target_layer,
977
- "prefixes": list(pref), "node_ids": sorted(ids) if ids else [],
978
- "reason": reason or CONTEXTUALIZE_REASON_DEFAULT,
979
- "nodes_scanned": len(entries), "targeted": 0, "skipped_locked": 0,
980
- "skipped_already": 0, "written": 0, "moved": [], "samples": [],
981
- "batch": batch}
982
- for e in entries:
983
- if limit is not None and rep["written"] >= int(limit):
984
- break
985
- nid = _entry_id(e)
986
- if not _hit(e):
987
- continue
988
- fm, content = cg._read(e)
989
- if fm is None:
990
- continue
991
- if crypto.is_encrypted(content):
992
- rep["skipped_locked"] += 1 # 无密钥 → fail-closed,绝不解密回写
993
- continue
994
- if fm.get("layer") != source_layer:
995
- rep["skipped_already"] += 1
996
- continue
997
- rep["targeted"] += 1
998
- if len(rep["samples"]) < 8:
999
- rep["samples"].append({"id": nid, "from": fm.get("layer"),
1000
- "path": e.get("path")})
1001
- if not apply:
1002
- continue
1003
- before = evolution.state_of(cg, nid) or {}
1004
- fm["layer"] = target_layer
1005
- fm["contextualized_from"] = source_layer
1006
- fm["contextualized_at"] = time.time()
1007
- fm["contextualization_basis"] = {"reason": rep["reason"], "batch": batch,
1008
- "actor": actor}
1009
- moved = _relocate_layer(cg, nid, e, fm, content, target_layer)
1010
- evolution.record(
1011
- cg, node_id=nid,
1012
- pattern="情境性内容(批次流水账 / 感知产物)混在知识层 → 归位情境层",
1013
- missing="层归属规则", action=f"层迁移 {source_layer}→{target_layer}",
1014
- evidence=f"prefix={str(nid).split('_')[0]}_ reason={rep['reason']}",
1015
- source="consolidate", kind=evolution.KIND_LAYER_SHIFT,
1016
- before=before, after=evolution.state_of(cg, nid) or {})
1017
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1018
- "t": time.time(), "action": "contextualize", "batch": batch, "id": nid,
1019
- "from": source_layer, "to": target_layer, "path": moved["path"],
1020
- "bucket": moved.get("bucket"), "reason": rep["reason"], "actor": actor})
1021
- rep["moved"].append(nid)
1022
- rep["written"] += 1
1023
- if apply and rep["written"]:
1024
- cg.rebuild_index()
1025
- rep["note"] = ("dry-run:未写盘;apply=True 才归位"
1026
- if not apply else f"已归位 {rep['written']} 个节点到 {target_layer}")
1027
- return rep
1028
-
1029
-
1030
- # 生效条件:给定 root,按 _maintain.jsonl 中 action=contextualize 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
1031
- def rollback_contextualize(root, node_ids=None, batch=None, actor="maintain") -> dict:
1032
- """回滚层归位:按 `_maintain.jsonl` 的 contextualize 记录把节点迁回原层。"""
1033
- cg = MdCGOS(root)
1034
- recs = [r for r in _read_maintain(root)
1035
- if r.get("action") == "contextualize"
1036
- and (not batch or r.get("batch") == batch)
1037
- and (not node_ids or str(r.get("id")) in {str(x) for x in node_ids})]
1038
- if not recs:
1039
- return {"ok": False, "error": "no_records", "reverted": 0}
1040
- reverted, ids = 0, []
1041
- for rec in recs:
1042
- nid = rec["id"]
1043
- e = (cg.index.get("nodes") or {}).get(nid)
1044
- if not e:
1045
- continue
1046
- fm, content = cg._read(e)
1047
- if fm is None or crypto.is_encrypted(content):
1048
- continue
1049
- back = rec.get("from") or "knowledge"
1050
- before = evolution.state_of(cg, nid) or {}
1051
- fm["layer"] = back
1052
- fm["contextualized_from"] = None
1053
- fm["contextualization_basis"] = {"rollback_of": rec.get("batch"),
1054
- "actor": actor}
1055
- _relocate_layer(cg, nid, e, fm, content, back)
1056
- evolution.record(cg, node_id=nid,
1057
- pattern="层归位回滚:情境层迁回原层",
1058
- action=f"层迁移 {rec.get('to')}→{back}",
1059
- evidence=f"rollback batch={rec.get('batch')}",
1060
- source="consolidate", kind=evolution.KIND_ROLLBACK,
1061
- before=before, after=evolution.state_of(cg, nid) or {})
1062
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1063
- "t": time.time(), "action": "contextualize_rollback",
1064
- "batch": rec.get("batch"), "id": nid, "to": back, "actor": actor})
1065
- reverted += 1
1066
- ids.append(nid)
1067
- if reverted:
1068
- cg.rebuild_index()
1069
- return {"ok": True, "reverted": reverted, "ids": ids}
1070
-
1071
-
1072
- # 生效条件:对 _read_maintain(root) 中 action 为 "contextualize" 或 "contextualize_rollback" 的记录,取 recs[-(int(limit) or 50):] 作为 records 返回 {'ok': True, ...}——仅当 int(limit) 成功且为 0 时回落 50,limit 为 None/""/[] 等无法 int() 的值会先抛 TypeError/ValueError。
1073
- def contextualize_history(root, limit=50):
1074
- """层归位的批次记录(只读)。"""
1075
- recs = [r for r in _read_maintain(root)
1076
- if r.get("action") in ("contextualize", "contextualize_rollback")]
1077
- return {"ok": True, "records": recs[-(int(limit) or 50):]}
1078
-
1079
-
1080
- # 生效条件:给定 root,读取 root 下 MAINTAIN_LOG 的 JSONL 并返回记录列表。
1081
- def _read_maintain(root):
1082
- from .fsutil import read_jsonl
1083
- return list(read_jsonl(os.path.join(root, MAINTAIN_LOG)))
1084
-
1085
-
1086
- # ==========================================================================
1087
- # 归纳聚类(consolidate.induce)
1088
- # ==========================================================================
1089
- #
1090
- # 与 promote 的分工:
1091
- # promote —— 把**已经存在**的单条情境记忆升格为长期知识(节点不变,只迁层);
1092
- # induce —— 把**多条**具体记忆归纳为一个**新的概念节点**(新增节点)。
1093
- #
1094
- # 归纳是「由具体到一般」的推理,其输出**不是事实断言**,而是待验证的假设:
1095
- # · 证据基底一律记 inferred(未经验证),不得冒充 verified;
1096
- # · 概念节点必须携带成员清单 + `generalizes`/`instance_of` 对称边,保证可回溯;
1097
- # · 归纳不出「共同条件」时默认**拒绝生成**(没有条件依据的抽象=编造,对齐
1098
- # 纪律 3「不猜测」);确需放宽须显式 require_conditions=False,且概念正文
1099
- # 会写明「未归纳出共同条件」,不掩盖证据缺口。
1100
-
1101
- INDUCE_MIN_CLUSTER = 3
1102
- INDUCE_MIN_JACCARD = 0.30
1103
- INDUCE_MAX_NODES = 400
1104
- INDUCE_MAX_TERMS = 6
1105
- CONCEPT_REL = "generalizes" # concept → member(inferred)
1106
- CONCEPT_MEMBER_REL = "instance_of" # member → concept(inferred)
1107
- CONCEPT_PREFIX = "concept_"
1108
- CONCEPT_IMPORTANCE = 0.5
1109
- CONCEPT_TAGS = ("concept", "induced")
1110
- # 巩固留痕字段(2026-09-19 阶段一):**字段名真源在 md_cg/nodefile.py**,
1111
- # 本处只做短别名引用(非复制),与 `nodefile.VALID_FROM_FIELD` 的登记纪律同构。
1112
- CONSOLIDATED_AT_FIELD = nodefile.CONSOLIDATED_AT_FIELD
1113
- CONSOLIDATED_INTO_FIELD = nodefile.CONSOLIDATED_INTO_FIELD
1114
- # 归纳候选排除:受保护节点,以及洞察/场景/前馈/概念等派生物(避免自我进食)
1115
- INDUCE_SKIP_TAGS = ("insight", "scene", "reconstructed", "gap_hint", "concept")
1116
-
1117
-
1118
- # 生效条件:给定 members,返回 CONCEPT_PREFIX 拼接排序后成员串的 SHA1 前 10 位。
1119
- def _concept_id(members):
1120
- """概念节点 id:由成员清单派生,保证「同成员 ⇒ 同 id」的幂等性。"""
1121
- h = hashlib.sha1("|".join(sorted(str(m) for m in members))
1122
- .encode("utf-8")).hexdigest()
1123
- return CONCEPT_PREFIX + h[:10]
1124
-
1125
-
1126
- # 生效条件:给定 a 与 b,若任一为空集则返回 0.0,否则返回交集大小除以并集大小。
1127
- def _jaccard(a, b):
1128
- if not a or not b:
1129
- return 0.0
1130
- return len(a & b) / float(len(a | b))
1131
-
1132
-
1133
- # 生效条件:给定 term_sets,返回出现次数不小于 max(2, ceil(min_share * len(term_sets))) 的词面排序列表;空输入返回 []。
1134
- def _common_terms(term_sets, min_share=0.6):
1135
- """出现在 ≥ min_share 比例成员中的词面(共同条件);少于 2 个成员共享不算。"""
1136
- if not term_sets:
1137
- return []
1138
- cnt = {}
1139
- for s in term_sets:
1140
- for t in set(s or ()):
1141
- cnt[t] = cnt.get(t, 0) + 1
1142
- need = max(2, int(math.ceil(min_share * len(term_sets))))
1143
- return sorted(t for t, c in cnt.items() if c >= need)
1144
-
1145
-
1146
- # 生效条件:给定 term_sets,按集合排序去重拼接后返回前 limit(默认 INDUCE_MAX_TERMS)个词面。
1147
- def _union_terms(term_sets, limit=INDUCE_MAX_TERMS):
1148
- seen = []
1149
- for s in term_sets:
1150
- for t in sorted(s or ()):
1151
- if t not in seen:
1152
- seen.append(t)
1153
- return seen[:limit]
1154
-
1155
-
1156
- # 生效条件:给定 cg、cid、members、reason、actor、batch,为概念节点与成员节点写对称 inferred 边(已存在则跳过),返回含 concept 与 members 的字典。
1157
- def _link_concept(cg, cid, members, reason, actor, batch):
1158
- """写概念↔成员对称 inferred 边(幂等:已存在则不重复写)。"""
1159
- nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
1160
- out = {"concept": cid, "members": []}
1161
- cnode = cg.get(cid)
1162
- if cnode:
1163
- fm = cnode.get("frontmatter") or {}
1164
- edges = list(fm.get("edges") or [])
1165
- have = {str(e.get("target")) for e in edges if isinstance(e, dict)}
1166
- added = False
1167
- for m in members:
1168
- if m in have:
1169
- continue
1170
- edges.append({"target": m, "relation_type": CONCEPT_REL,
1171
- "reason": reason, "created_at": time.time(),
1172
- "confidence": 0.5, "verified": 0, "evidence": "inferred"})
1173
- added = True
1174
- if added:
1175
- fm["edges"] = edges
1176
- ent = nodes.get(cid) or {}
1177
- cg._write_node(cid, os.path.join(cg.root, ent.get("path") or f"{cid}.md"),
1178
- fm, cnode.get("content") or "")
1179
- if ent:
1180
- ent["edges"] = edges
1181
- for m in members:
1182
- node = cg.get(m)
1183
- if not node:
1184
- continue
1185
- fm = node.get("frontmatter") or {}
1186
- edges = list(fm.get("edges") or [])
1187
- if any(isinstance(e, dict) and str(e.get("target")) == cid for e in edges):
1188
- continue
1189
- edges.append({"target": cid, "relation_type": CONCEPT_MEMBER_REL,
1190
- "reason": reason, "created_at": time.time(),
1191
- "confidence": 0.5, "verified": 0, "evidence": "inferred"})
1192
- fm["edges"] = edges
1193
- # 巩固留痕(2026-09-19 阶段一):`consolidated_into` 为**规范名**,
1194
- # `induced_concept` 保留为历史别名(既有读取面零破坏);`consolidated_at`
1195
- # 补齐**成员侧**巩固时刻——此前只有概念侧 `induced_at`,成员侧无从判定
1196
- # 「何时被并进去」,故「合并后前身可定位」只在概念侧半成立。
1197
- fm[CONSOLIDATED_INTO_FIELD] = cid
1198
- fm[CONSOLIDATED_AT_FIELD] = time.time()
1199
- fm["induced_concept"] = cid
1200
- ent = nodes.get(m) or {}
1201
- cg._write_node(m, os.path.join(cg.root, ent.get("path") or f"{m}.md"),
1202
- fm, node.get("content") or "")
1203
- if ent:
1204
- ent["edges"] = edges
1205
- out["members"].append(m)
1206
- return out
1207
-
1208
-
1209
- # 生效条件:给定 members、common_pos、neg_union,返回标注 inferred 的概念节点正文,含功能名、生效条件、子功能、执行、验证方式、不适用条件。
1210
- def _concept_payload(members, common_pos, neg_union):
1211
- """概念节点正文:把成员的共性条件抽象为可追溯的知识条目(显式标注 inferred)。"""
1212
- label = "、".join(common_pos[:INDUCE_MAX_TERMS])
1213
- pos_txt = ";".join(common_pos[:INDUCE_MAX_TERMS]) or "(未归纳出共同条件)"
1214
- neg_txt = ";".join(neg_union[:INDUCE_MAX_TERMS]) or "(未判定)"
1215
- return (
1216
- "# 功能名:归纳概念:%s\n"
1217
- "# 生效条件:%s\n"
1218
- "# 子功能:%d 条具体记忆的共性(成员:%s)\n"
1219
- "# 执行:由 consolidate.induce 归纳聚合(inferred;未经验证,不得直接当事实使用)\n"
1220
- "# 验证方式:待验证(inferred 假设,需外部证据或实践重复后方可升格)\n"
1221
- "# 不适用条件:%s\n"
1222
- % (label or "共性", pos_txt, len(members), "、".join(members), neg_txt)
1223
- )
1224
-
1225
-
1226
- # 生效条件:给定 cg_or_root,从 source_layer 聚类归纳为 target_layer 概念节点,apply=True 才写盘并返回统计 rep。
1227
- def induce_memories(cg_or_root, source_layer="contextual", target_layer="knowledge",
1228
- min_cluster=INDUCE_MIN_CLUSTER, min_jaccard=INDUCE_MIN_JACCARD,
1229
- max_nodes=INDUCE_MAX_NODES, require_conditions=True,
1230
- limit=None, apply=False, actor="maintain", **extra):
1231
- """归纳聚类:把多条具体记忆归纳为概念层条目(inferred,非事实断言)。
1232
-
1233
- 流程:读取源层 → bigram 相似度贪心聚类 → 提炼共同条件 → 生成概念节点
1234
- (apply=True)→ 写 `generalizes` / `instance_of` 对称 inferred 边 → 写留痕。
1235
-
1236
- apply=False(默认)只出候选报表(可预演);apply=True 才写盘(可留痕、可回溯)。
1237
- 幂等:概念 id 由成员清单派生,同成员重复归纳不新增节点。
1238
- """
1239
- cg = cg_or_root if isinstance(cg_or_root, MdCGOS) else MdCGOS(str(cg_or_root))
1240
- from . import subgraph # 惰性导入:复用统一的条件/词面抽取
1241
-
1242
- # MCP 分发层会把未提供的参数以 None 传入;此处归一化,避免 int(None) 崩溃,
1243
- # 也避免 require_conditions=None 被当成 False 而悄悄关掉「无共同条件即拒绝生成」
1244
- # 这条纪律(默认必须为真,放宽只能显式传 False)。
1245
- min_cluster = INDUCE_MIN_CLUSTER if min_cluster is None else int(min_cluster)
1246
- min_jaccard = INDUCE_MIN_JACCARD if min_jaccard is None else float(min_jaccard)
1247
- max_nodes = INDUCE_MAX_NODES if max_nodes is None else int(max_nodes)
1248
- if require_conditions is None:
1249
- require_conditions = True
1250
-
1251
- nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
1252
- pool = [nid for nid, e in nodes.items()
1253
- if (not source_layer or (e or {}).get("layer") == source_layer)
1254
- and not (e or {}).get("protected")
1255
- and not (set(INDUCE_SKIP_TAGS) & set((e or {}).get("tags") or []))]
1256
- pool.sort()
1257
- truncated = len(pool) > int(max_nodes)
1258
- pool = pool[:int(max_nodes)]
1259
-
1260
- cache = {}
1261
- for nid in pool:
1262
- got = subgraph._node_terms_and_grams(cg, nid)
1263
- if got and got["grams"]:
1264
- cache[nid] = got
1265
- keys = sorted(cache.keys())
1266
-
1267
- rep = {"ok": True, "action": "induce", "op": "consolidate",
1268
- "source_layer": source_layer, "target_layer": target_layer,
1269
- "dry_run": not apply, "nodes_scanned": len(pool), "indexed": len(keys),
1270
- "truncated": truncated, "min_cluster": int(min_cluster),
1271
- "min_jaccard": float(min_jaccard),
1272
- "require_conditions": bool(require_conditions),
1273
- "skipped_small": 0, "skipped_no_condition": 0, "skipped_existing": 0,
1274
- "clusters": 0, "written": 0, "concepts": [], "samples": [],
1275
- "log": MAINTAIN_LOG}
1276
-
1277
- # ---- 贪心聚类(只读) ----
1278
- assigned, proposals = set(), []
1279
- for i, a in enumerate(keys):
1280
- if a in assigned:
1281
- continue
1282
- ga = cache[a]["grams"]
1283
- grp = [b for b in keys[i + 1:]
1284
- if b not in assigned
1285
- and _jaccard(ga, cache[b]["grams"]) >= float(min_jaccard)]
1286
- if len(grp) + 1 < int(min_cluster):
1287
- continue
1288
- members = [a] + grp
1289
- assigned.update(members)
1290
- common_pos = _common_terms([cache[m]["pos"] for m in members])
1291
- if require_conditions and not common_pos:
1292
- rep["skipped_no_condition"] += 1
1293
- continue
1294
- neg_union = _union_terms([cache[m]["neg"] for m in members])
1295
- proposals.append({
1296
- "members": members, "concept_id": _concept_id(members),
1297
- "common_conditions": common_pos, "non_applicable": neg_union,
1298
- "reason": ("%d 条记忆内容相近且共享条件「%s」→ 归纳为概念"
1299
- % (len(members), "、".join(common_pos) or "无")),
1300
- })
1301
- rep["clusters"] = len(proposals)
1302
- for p in proposals[:8]:
1303
- rep["samples"].append(p)
1304
-
1305
- if not apply:
1306
- rep["note"] = ("dry-run:未写盘;apply=True 才生成概念节点与 inferred 边"
1307
- if proposals else "无满足条件的聚类(内容不够相近或缺乏共同条件)")
1308
- rep["concepts"] = [p["concept_id"] for p in proposals]
1309
- return rep
1310
-
1311
- # ---- 落库(可留痕) ----
1312
- batch = time.strftime("%Y%m%d-%H%M%S")
1313
- for p in proposals:
1314
- if limit is not None and rep["written"] >= int(limit):
1315
- break
1316
- cid = p["concept_id"]
1317
- if cid in nodes:
1318
- rep["skipped_existing"] += 1
1319
- continue
1320
- content = _concept_payload(p["members"], p["common_conditions"],
1321
- p["non_applicable"])
1322
- _consolidated_at = time.time() # 概念形成时刻 = 巩固时刻(单一取值,禁两处取时)
1323
- cg.add(cid, content, layer=target_layer, tags=list(CONCEPT_TAGS),
1324
- importance=CONCEPT_IMPORTANCE, verification_basis="other",
1325
- induced_from=list(p["members"]), induced_at=_consolidated_at,
1326
- consolidated_at=_consolidated_at,
1327
- induction={"method": "bigram_jaccard", "min_jaccard": float(min_jaccard),
1328
- "common_conditions": p["common_conditions"], "batch": batch,
1329
- "actor": actor, "evidence": "inferred"},
1330
- actor=actor)
1331
- _link_concept(cg, cid, p["members"], p["reason"], actor, batch)
1332
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1333
- "t": time.time(), "action": "induce", "batch": batch, "concept": cid,
1334
- "members": list(p["members"]), "common_conditions": p["common_conditions"],
1335
- "source_layer": source_layer, "target_layer": target_layer, "actor": actor})
1336
- rep["concepts"].append(cid)
1337
- rep["written"] += 1
1338
- if rep["written"]:
1339
- cg.rebuild_index()
1340
- rep["note"] = (f"已归纳 {rep['written']} 个概念节点(inferred,待验证)"
1341
- if rep["written"] else "无可落库的归纳(均跳过或已达 limit)")
1342
- return rep
1343
-
1344
-
1345
- # ---- CLI ----------------------------------------------------------------
1346
-
1347
- # 生效条件:不适用(无必需形参与模块级常量)
1348
- def _cli(argv=None) -> int:
1349
- ap = argparse.ArgumentParser(
1350
- description="md_cg 离线固化:反思单元(LLM)产出候选 → 白箱闸门 → "
1351
- "验证单元(LLM)否决 → 固化为 md 字段")
1352
- ap.add_argument("--root", required=True, help="md 认知图根目录")
1353
- ap.add_argument("--layer", default=None, help="只处理某层(如 knowledge)")
1354
- ap.add_argument("--limit", type=int, default=None, help="只处理前 N 个待补节点")
1355
- ap.add_argument("--apply", action="store_true", help="真正写盘(默认只验证)")
1356
- ap.add_argument("--dry-run", action="store_true", help="只验证不写盘(默认行为)")
1357
- ap.add_argument("--overwrite", action="store_true",
1358
- help="允许覆盖已有非空字段(默认保护人工既有知识)")
1359
- ap.add_argument("--reflect-model", default=None,
1360
- help=f"反思单元模型(默认 {ROLE_DEFAULT_MODEL[REFLECT_ROLE]})")
1361
- ap.add_argument("--verify-model", default=None,
1362
- help=f"验证单元模型(默认 {ROLE_DEFAULT_MODEL[VERIFY_ROLE]})")
1363
- ap.add_argument("--self-verify", action="store_true",
1364
- help="降级:验证单元复用反思单元模型(非交叉验证,provenance 标记)")
1365
- ap.add_argument("--no-verify", action="store_true",
1366
- help="降级:跳过验证单元,仅靠白箱闸门(不推荐)")
1367
- ap.add_argument("--verification-basis", default=None,
1368
- help="写入 `# 验证方式:` 的声明文本(默认双模型声明)")
1369
- ap.add_argument("--no-basis", action="store_true", help="不写「验证方式」")
1370
- ap.add_argument("--basis-only", action="store_true",
1371
- help="只补「验证方式」(零 LLM 成本),不做四要素反思")
1372
- ap.add_argument("--min-grounding", type=float, default=None,
1373
- help="统一 grounding 阈值(默认按字段 0.5 / 不适用条件 0.34)")
1374
- ap.add_argument("--no-llm", action="store_true",
1375
- help="不调用 LLM,只做四要素完整性普查")
1376
- ap.add_argument("--check", action="store_true",
1377
- help="零 token 探测两个角色网关的可用模型后退出")
1378
- ap.add_argument("--report", default=None, help="把汇总 JSON 另存一份")
1379
- a = ap.parse_args(argv)
1380
-
1381
- r_model, _, r_key = role_config(REFLECT_ROLE, a.reflect_model)
1382
- v_model, _, v_key = role_config(VERIFY_ROLE, a.verify_model)
1383
-
1384
- if a.check:
1385
- out = {"reflect": probe_models(REFLECT_ROLE),
1386
- "verify": probe_models(VERIFY_ROLE)}
1387
- print(json.dumps(out, ensure_ascii=False, indent=2))
1388
- return 0
1389
-
1390
- basis = None if a.no_basis else (
1391
- a.verification_basis or BASIS_TEMPLATE.format(reflect=r_model, verify=v_model))
1392
-
1393
- if a.basis_only:
1394
- rep = fill_verification_basis(a.root, basis, layer=a.layer, limit=a.limit,
1395
- apply=a.apply)
1396
- print(json.dumps(rep, ensure_ascii=False, indent=2))
1397
- if a.report:
1398
- with open(a.report, "w", encoding="utf-8") as f:
1399
- json.dump(rep, f, ensure_ascii=False, indent=2)
1400
- return 0
1401
-
1402
- thresholds = ({f: a.min_grounding for f in CCG_FIELDS}
1403
- if a.min_grounding is not None else None)
1404
-
1405
- reflect_fn = verify_fn = None
1406
- if not a.no_llm:
1407
- if not r_key:
1408
- print(f"[consolidate] 反思单元未配置 key"
1409
- f"({_ROLE_ENV[REFLECT_ROLE][2]} / DEEPSEEK_API_KEY)→ 退化为普查模式",
1410
- file=sys.stderr)
1411
- else:
1412
- reflect_fn = (lambda p: http_llm(p, role=REFLECT_ROLE, # noqa: E731
1413
- model=a.reflect_model))
1414
- if a.self_verify:
1415
- verify_fn = reflect_fn
1416
- elif not a.no_verify:
1417
- if v_key:
1418
- verify_fn = (lambda p: http_llm(p, role=VERIFY_ROLE, # noqa: E731
1419
- model=a.verify_model))
1420
- else:
1421
- print(f"[consolidate] 验证单元未配置 key"
1422
- f"({_ROLE_ENV[VERIFY_ROLE][2]} / ZHIPU_API_KEY / GLM_API_KEY)"
1423
- "→ 待补节点将 DEFER,不写盘(纪律 5:未经验证不固化)",
1424
- file=sys.stderr)
1425
-
1426
- rep = consolidate(a.root, layer=a.layer, limit=a.limit, apply=a.apply,
1427
- overwrite=a.overwrite, reflect_fn=reflect_fn,
1428
- verify_fn=verify_fn, reflect_model=r_model,
1429
- verify_model=v_model if verify_fn else "",
1430
- verification_basis=basis or "",
1431
- require_verify=not a.no_verify, thresholds=thresholds)
1432
- print(json.dumps(rep, ensure_ascii=False, indent=2))
1433
- if a.report:
1434
- with open(a.report, "w", encoding="utf-8") as f:
1435
- json.dump(rep, f, ensure_ascii=False, indent=2)
1436
- return 0
1437
-
1438
-
1439
- if __name__ == "__main__":
1
+ # -*- coding: utf-8 -*-
2
+ """md_cg · 离线固化:LLM 补 CCG 四要素 → 确定性验证 → 固化为 md 字段
3
+
4
+ 为什么是「离线固化」而不是「在线向量」:
5
+ 白箱第 1 篇:相似度可以产生候选,但**不授予执行资格**;资格必须由条件证据裁决。
6
+ LLM 是黑箱,它的输出只能是**候选条件**,不能直接成为检索依据——否则在线检索
7
+ 就被黑箱污染,CCG 28%→88% 的改进会退化回去。故本工具把 LLM 严格限制在
8
+ **离线一次性的固化工序**里:
9
+
10
+ 读节点 → LLM 产出四要素候选 → 确定性验证 → 通过才写进 md 字段
11
+
12
+ 在线检索(search / recall / _path_semantic)仍然只读 md 里已固化的字段,全程白箱。
13
+
14
+ 理论 / 纪律对齐(docs/工作纪律_认知图条目_v1.1.json):
15
+ · 第 3 条 白箱方法:不猜测;**未验证不写入**。
16
+ · 第 5 条 验证纪律:**未经验证不固化**——入库前必须走验证(回放 / 断言 / 回归)。
17
+ · 第 13 条 访谈澄清:节点四要素 = 条件 / 子内容 / 如何执行 / 不适用条件
18
+ ——本工具固化的正是这四个字段(对齐 CCG 的生效条件 / 子功能 / 执行 / 不适用条件)。
19
+ · 《智能的认知过程》:新条件能否**稳定解释误差**?成立 → 纳入知识结构;
20
+ 不成立 → **不固化**,标记为待验证。
21
+ 故 verdict 三态对齐白箱资格判定:ACCEPT(固化)/ REJECT(丢弃)/ DEFER(只存候选)。
22
+
23
+ 验证分三段闸门——前两段确定性零 LLM,第三段是「双模型交叉验证」:
24
+
25
+ 闸门 1 · grounding 支撑度(确定性):候选短语必须能在节点正文里找到字符级依据,
26
+ 否则判为幻觉 → REJECT。(对应「不猜测」)
27
+ 闸门 2 · replay 回放(确定性):把候选条件当作查询,回放生产检索路径的判定:
28
+ · pos_recall 以「生效条件」为查询 → 本节点应被召回,且不被自身负条件挡住;
29
+ · neg_separated 以「不适用条件」为查询 → 应触发条件级负路由,且负条件与正文
30
+ 低相关(负条件必须是「域外」的,不能把知识本身否定掉);
31
+ · no_conflict 生效条件与不适用条件不得互相覆盖。
32
+ 三者同时成立才算「条件稳定」。(对应「回放 / 断言 / 回归」)
33
+ 闸门 3 · 验证单元(GLM,独立模型):逐条核验候选是否有正文依据、负条件是否真域外。
34
+ 硬约束:**验证单元只能否决,不能新增/改写**——它没有产出权,
35
+ 否则验证环节自己就成了新的幻觉源。
36
+
37
+ 回放器复用 _path_semantic 的同一批原语(_declared_conditions / _neg_hit /
38
+ _weighted_coverage / expand_query_terms_weighted),并由 P6 测试与真实
39
+ MdCGOS._path_semantic 做一致性回归,保证不漂移。
40
+
41
+ 双模型角色分工(用户配置,可用环境变量覆盖):
42
+ 反思单元 reflect → 默认 deepseek-flash(实测 2026-09-22 /models 仅
43
+ deepseek-flash / deepseek-v4-pro;原默认
44
+ deepseek-v4.1-flash-expires-on-0910 为限时模型已下架)
45
+ 验证单元 verify → 默认 glm-5.3-flash (IDE 显示名 GLM-5.3-flash)
46
+ 环境变量:MDCG_REFLECT_MODEL/BASE/KEY、MDCG_VERIFY_MODEL/BASE/KEY。
47
+ 注意 IDE 显示名 ≠ API 模型 id;`--check` 可零 token 探测各网关真实 id。
48
+ 两个模型分属不同厂商,避免同源模型的系统性偏见互相印证(交叉验证的本意)。
49
+
50
+ max_tokens 口径(真源:docs/hive/子代理配置标准_v0.5.md §1,2026-09-23 修复 issue #24):
51
+ 思考模型的 reasoning_tokens **计入 max_tokens**(标准 §1 实测:8000 被思考
52
+ 吃满 → content 空 → 假成功;指纹 = usage.completion_tokens≈reasoning_tokens)。
53
+ 故默认 DEFAULT_MAX_TOKENS=200000(标准 §1「思考模型建议 200000」),
54
+ 覆盖链:CLI --max-tokens > env MDCG_LLM_MAX_TOKENS > 常量默认。
55
+ 响应面校验(标准 §1「回收时校验 content 非空而非只看 ok」):finish_reason
56
+ == "length" 或 content 为空 → 明确 RuntimeError(带 max_tokens 与修复指引),
57
+ 绝不静默落成 parse_failed(issue #24 的根因即此静默)。
58
+
59
+ · 验证单元不可用(未配 key)时,默认 **不固化**(DEFER)——纪律 5「未经验证不固化」;
60
+ 确需单模型跑通可显式 --no-verify(provenance 记为 skipped)或 --self-verify
61
+ (同模型自审,provenance 记为 self_verify=true,属于降级模式)。
62
+ · 「验证方式」是 CCG 必需要素,其值 = 声明的验证基底。本工具可写
63
+ `# 验证方式:<声明>`(--verification-basis,默认即上面的双模型声明);
64
+ frontmatter.verification_basis 只能取 nodefile 的枚举
65
+ (compiler/test/measurement/formal_proof/data/other),双 LLM 交叉验证对应 "other"。
66
+ 纯文本声明无需 LLM → --basis-only 可零成本补齐全库(纪律 3:不猜测)。
67
+
68
+ 零第三方依赖(D-005):HTTP 走标准库 urllib.request;reflect_fn / verify_fn 可注入
69
+ (离线可测)。默认 --dry-run,只有 --apply 才写盘(纪律 6:改动可核对)。
70
+ 写盘只动 CCG 字段与 provenance,**不覆盖已有非空字段**(除非 --overwrite)。
71
+ """
72
+ from __future__ import annotations
73
+
74
+ import argparse
75
+ import hashlib
76
+ import json
77
+ import math
78
+ import os
79
+ import re
80
+ import sys
81
+ import time
82
+ import urllib.error
83
+ import urllib.request
84
+
85
+ from . import crypto, evolution, nodefile, routing
86
+ from .fsutil import append_jsonl
87
+ from .mdcg import BUCKETED_LAYERS, bigrams, expand_query_terms_weighted
88
+ from .mdcos import (MdCGOS, _ccg_field, _declared_conditions, _neg_hit, _sig,
89
+ _weighted_coverage)
90
+
91
+ # ---- 常量 ----------------------------------------------------------------
92
+
93
+ # 与工作纪律第 13 条「节点四要素」同构:
94
+ # 生效条件 ↔ conditions(什么时候适用)
95
+ # 子功能 ↔ subgraph / depends_on(子内容)
96
+ # 执行 ↔ execution(如何执行)
97
+ # 不适用条件 ↔ negative(什么时候不适用)
98
+ CCG_FIELDS = ("生效条件", "子功能", "执行", "不适用条件")
99
+ MULTI_FIELDS = ("生效条件", "子功能", "不适用条件") # 列表型
100
+ SINGLE_FIELDS = ("执行",) # 单值型
101
+
102
+ # LLM 常把「子功能」写成「子内容」,别名容错(固化时统一落到标准字段名)
103
+ FIELD_ALIASES = {
104
+ "生效条件": ("生效条件", "适用条件", "conditions", "condition"),
105
+ "子功能": ("子功能", "子内容", "子流程", "subgraph", "sub"),
106
+ "执行": ("执行", "如何执行", "执行方式", "execution", "how"),
107
+ "不适用条件": ("不适用条件", "不适用", "负条件", "negative", "reject"),
108
+ }
109
+
110
+ # 默认 grounding 阈值:不适用条件描述的是「域外」情境,与正文天然低相关,
111
+ # 故阈值放宽;其余三要素必须能在正文里找到实打实的依据。
112
+ DEFAULT_GROUNDING = {"生效条件": 0.5, "子功能": 0.5, "执行": 0.5, "不适用条件": 0.34}
113
+
114
+ MAX_BODY_CHARS = 3000 # 正文截断(控制 token,且条件主要来自开头)
115
+ MAX_TERMS = 8 # 单字段候选条数上限
116
+ MAX_TERM_LEN = 40 # 单条候选长度上限
117
+
118
+ # CCG 声明行(`# 生效条件:…` 等)——它们不是正文,grounding/replay 必须把它们剥掉,
119
+ # 否则已写入的「不适用条件」会在二次运行时被当成正文依据,导致节点自我否定。
120
+ _CCG_LINE_RE = re.compile(
121
+ r"^\s*#\s*(功能名|生效条件|子功能|执行|验证方式|不适用条件)\s*[::]")
122
+
123
+
124
+ # 生效条件:给定 content,返回剔除所有匹配 _CCG_LINE_RE 的行后以换行连接的非声明正文;content 为 None 时按空串处理。
125
+ def body_text(content: str) -> str:
126
+ """剥掉 CCG 声明行后的正文——验证只认正文,不认已写下的声明。"""
127
+ return "\n".join(l for l in (content or "").split("\n")
128
+ if not _CCG_LINE_RE.match(l))
129
+
130
+ # ---- 双模型角色(反思单元 / 验证单元)-------------------------------------
131
+
132
+ REFLECT_ROLE = "reflect" # 反思单元:产出候选
133
+ VERIFY_ROLE = "verify" # 验证单元:否决候选(无产出权)
134
+ ROLES = (REFLECT_ROLE, VERIFY_ROLE)
135
+
136
+ # 推荐模型(真实 API id,经实际调用确认;可用 MDCG_<ROLE>_MODEL 覆盖)
137
+ # 历史:reflect 原默认 deepseek-v4.1-flash-expires-on-0910(限时模型,代码注释曾
138
+ # 预告过期风险)——2026-09-22 实测 /models 仅返回 deepseek-flash 与 deepseek-v4-pro,
139
+ # 限时 id 已下架(issue #24 附带发现)。现行默认 reflect=deepseek-flash(标准 §1
140
+ # 子代理默认档)。
141
+ ROLE_DEFAULT_MODEL = {REFLECT_ROLE: "deepseek-flash",
142
+ VERIFY_ROLE: "glm-5.3-flash"}
143
+
144
+ # max_tokens 默认(真源:docs/hive/子代理配置标准_v0.5.md §1——思考模型建议 200000;
145
+ # reasoning_tokens 计入 max_tokens,预算过小 = content 被思考吃光 = issue #24 根因)。
146
+ # 覆盖链:CLI --max-tokens > env MDCG_LLM_MAX_TOKENS > 本常量。
147
+ DEFAULT_MAX_TOKENS = 200000
148
+ MAX_TOKENS_ENV = "MDCG_LLM_MAX_TOKENS"
149
+ ROLE_DEFAULT_BASE = {REFLECT_ROLE: "https://api.deepseek.com",
150
+ VERIFY_ROLE: "https://open.bigmodel.cn/api/paas/v4"}
151
+ _ROLE_ENV = {REFLECT_ROLE: ("MDCG_REFLECT_MODEL", "MDCG_REFLECT_BASE", "MDCG_REFLECT_KEY"),
152
+ VERIFY_ROLE: ("MDCG_VERIFY_MODEL", "MDCG_VERIFY_BASE", "MDCG_VERIFY_KEY")}
153
+ # 验证单元 key 的常见别名(智谱系)
154
+ VERIFY_KEY_ALIASES = ("ZHIPU_API_KEY", "ZHIPUAI_API_KEY", "GLM_API_KEY", "BIGMODEL_API_KEY")
155
+
156
+ # 「验证方式」的声明文本(可 --verification-basis 覆盖 / --no-basis 关闭)
157
+ BASIS_TEMPLATE = "双模型交叉验证(反思单元={reflect},验证单元={verify})"
158
+ # frontmatter.verification_basis 只能取 nodefile 的枚举;双 LLM 交叉验证 → other
159
+ BASIS_ENUM_DEFAULT = "other"
160
+
161
+ REFLECT_PROMPT = (
162
+ "你是认知图节点的**反思单元**。给定一个知识节点的标题与正文,反思并抽取四要素。\n"
163
+ "只输出一个 JSON 对象,不要任何解释或代码围栏。\n"
164
+ "字段含义:\n"
165
+ ' "生效条件": 什么查询/情境下该知识**适用**(短语数组,2~5 条)\n'
166
+ ' "子功能": 该知识包含的子内容/子步骤(短语数组,2~5 条)\n'
167
+ ' "执行": 如何执行/如何使用该知识(单个字符串)\n'
168
+ ' "不适用条件": 什么查询/情境下该知识**不**适用(短语数组,1~3 条)\n'
169
+ "硬约束:\n"
170
+ " 1. 每条短语必须能在正文中找到依据,禁止编造正文里没有的工具/概念;\n"
171
+ " 2. 不适用条件必须是**正文之外的邻近易混情境**,不得与生效条件语义重叠;\n"
172
+ " 3. 短语要短(不超过 20 字),不要写完整句子。\n"
173
+ "输出格式:"
174
+ '{{"生效条件": ["..."], "子功能": ["..."], "执行": "...", "不适用条件": ["..."]}}\n'
175
+ "标题:{title}\n正文:\n{body}"
176
+ )
177
+
178
+ VERIFY_PROMPT = (
179
+ "你是认知图节点的**验证单元**。你的职责是**否决**,不是补充。\n"
180
+ "只能从候选里删除不成立的条目,**绝不允许新增或改写任何条目**。\n"
181
+ "给定标题、正文与反思单元给出的候选四要素,逐条核验:\n"
182
+ " · 该条目是否真的能在正文中找到依据?找不到依据 → 删除;\n"
183
+ " · 不适用条件是否真的域外?若它其实是该节点的适用情境 → 删除;\n"
184
+ " · 生效条件与不适用条件是否语义重叠?重叠者删除其一(保留更贴合正文的那个)。\n"
185
+ "只输出一个 JSON 对象,键为字段名,值为 "
186
+ '{{"keep": ["保留的条目"], "drop": ["删除的条目"], "reason": "一句话理由"}}。\n'
187
+ "候选:{cand}\n标题:{title}\n正文:\n{body}"
188
+ )
189
+
190
+
191
+ # ---- LLM 侧(黑箱只在离线工序,产出候选)--------------------------------
192
+
193
+ # 生效条件:给定 role,按显式参数、角色环境变量、通用兜底依次解析并返回 (model, base, key);key 不落 DEEPSEEK_API_KEY 除非 role 为 REFLECT_ROLE。
194
+ def role_config(role: str, model: str = None, base: str = None,
195
+ key: str = None) -> tuple:
196
+ """解析某角色的 (model, base, key):显式参数 > 角色环境变量 > 通用兜底。
197
+
198
+ key 刻意**不**让验证单元回落到 DEEPSEEK_API_KEY——跨厂商混用会把一个厂商的
199
+ 凭证发到另一个厂商的网关,既必然失败又构成凭证外泄。
200
+ """
201
+ m_env, b_env, k_env = _ROLE_ENV[role]
202
+ if key is None:
203
+ key = os.environ.get(k_env)
204
+ if key is None and role == VERIFY_ROLE:
205
+ for alias in VERIFY_KEY_ALIASES:
206
+ key = os.environ.get(alias)
207
+ if key:
208
+ break
209
+ if key is None:
210
+ key = os.environ.get("MDCG_LLM_KEY")
211
+ if key is None and role == REFLECT_ROLE:
212
+ key = os.environ.get("DEEPSEEK_API_KEY")
213
+ model = (model or os.environ.get(m_env) or os.environ.get("MDCG_LLM_MODEL")
214
+ or ROLE_DEFAULT_MODEL[role])
215
+ base = (base or os.environ.get(b_env) or os.environ.get("MDCG_LLM_BASE")
216
+ or ROLE_DEFAULT_BASE[role])
217
+ return model, base, key
218
+
219
+
220
+ # 生效条件:给定显式 max_tokens 与 env 值,按 显式参数 > env > DEFAULT_MAX_TOKENS 解析:显式为正整数直接返回;env 可解析且 >0 返回之;否则返回 DEFAULT_MAX_TOKENS;env 存在但非法或非正时忽略(回落默认,不炸批处理)。
221
+ def resolve_max_tokens(explicit: int = None) -> int:
222
+ """max_tokens 三级解析:显式参数 > MDCG_LLM_MAX_TOKENS > DEFAULT_MAX_TOKENS。
223
+
224
+ 为什么默认这么大(200000):思考模型的 reasoning_tokens 计入 max_tokens
225
+ (标准 §1 实测),预算过小 = content 被思考吃光。max_tokens 是预算上限
226
+ 而非计费量——取宽不取窄,实际用量计费不受影响。
227
+ """
228
+ if explicit is not None:
229
+ return max(1, int(explicit))
230
+ raw = (os.environ.get(MAX_TOKENS_ENV) or "").strip()
231
+ if raw:
232
+ try:
233
+ v = int(raw)
234
+ if v > 0:
235
+ return v
236
+ except ValueError:
237
+ pass # 非法 env 不炸批处理,回落默认
238
+ return DEFAULT_MAX_TOKENS
239
+
240
+
241
+ # 生效条件:给定响应 data 与 model,choices 为空、message.content 为空/None、或 finish_reason=="length" 时抛 RuntimeError(含 max_tokens/finish_reason/usage 指纹与修复指引),否则返回 content 字符串。
242
+ def _extract_content(data: dict, model: str, max_tokens: int) -> str:
243
+ """响应面校验(标准 §1:「回收时校验 content 非空而非只看 ok」)。
244
+
245
+ 三种坏形态都不许静默落成 parse_failed(issue #24 根因):
246
+ · choices 空 → 网关异常形态;
247
+ · content 空 → 假成功(指纹:usage.completion_tokens≈reasoning_tokens,
248
+ 思考预算吃光 content——标准 §1 同病实测);
249
+ · finish_reason=="length" → 预算耗尽被截断(截断的 JSON 必然解析失败)。
250
+ """
251
+ choices = data.get("choices") or []
252
+ if not choices:
253
+ raise RuntimeError(
254
+ f"LLM 响应无 choices(model={model}, max_tokens={max_tokens}):"
255
+ "网关异常形态,原样返回体前 300 字符:"
256
+ f"{json.dumps(data, ensure_ascii=False)[:300]}")
257
+ msg = choices[0].get("message") or {}
258
+ finish = choices[0].get("finish_reason")
259
+ usage = data.get("usage") or {}
260
+ if finish == "length":
261
+ raise RuntimeError(
262
+ f"输出预算耗尽(finish_reason=length, model={model}, "
263
+ f"max_tokens={max_tokens}, completion_tokens={usage.get('completion_tokens')})"
264
+ "——思考模型的 reasoning_tokens 计入 max_tokens(子代理配置标准 v0.5 §1)。"
265
+ f"修复:调大 --max-tokens 或 env {MAX_TOKENS_ENV}")
266
+ content = msg.get("content") or ""
267
+ if not content.strip():
268
+ raise RuntimeError(
269
+ f"LLM 返回空 content(model={model}, finish_reason={finish}, "
270
+ f"max_tokens={max_tokens}, "
271
+ f"completion_tokens={usage.get('completion_tokens')})——"
272
+ "假成功指纹:completion_tokens≈reasoning_tokens 表示思考吃满预算;"
273
+ f"修复:调大 --max-tokens 或 env {MAX_TOKENS_ENV}")
274
+ return content
275
+
276
+
277
+ # 生效条件:给定 prompt 且 role 解析或通用兜底得到非空 key 时,向 base 的 /chat/completions 发 POST,max_tokens 按 resolve_max_tokens 解析(显式 > env > 默认 200000),经 _extract_content 校验后返回 content;key 为空则抛 RuntimeError。
278
+ def http_llm(prompt: str, model: str = None, base: str = None, key: str = None,
279
+ role: str = None, timeout: int = 120, max_tokens: int = None) -> str:
280
+ """标准库 HTTP 调 LLM(OpenAI 兼容 /chat/completions)。零第三方依赖。
281
+
282
+ role 给定时按该角色配置解析(reflect / verify),否则走通用配置。
283
+ max_tokens=None 走三级解析(显式 > MDCG_LLM_MAX_TOKENS > DEFAULT_MAX_TOKENS);
284
+ 历史 bug(issue #24):曾硬编码 1200——思考模型 reasoning 吃光预算,
285
+ content 空串静默落成 parse_failed,离线固化 100% DEFER。
286
+ """
287
+ if role:
288
+ model, base, key = role_config(role, model, base, key)
289
+ else:
290
+ model = (model or os.environ.get("MDCG_LLM_MODEL")
291
+ or ROLE_DEFAULT_MODEL[REFLECT_ROLE])
292
+ base = (base or os.environ.get("MDCG_LLM_BASE")
293
+ or ROLE_DEFAULT_BASE[REFLECT_ROLE])
294
+ key = (key or os.environ.get("MDCG_LLM_KEY")
295
+ or os.environ.get("DEEPSEEK_API_KEY"))
296
+ if not key:
297
+ raise RuntimeError(
298
+ f"未配置 {role or 'llm'} 的 API key"
299
+ f"({_ROLE_ENV[role][2] if role in _ROLE_ENV else 'MDCG_LLM_KEY'})")
300
+ max_tokens = resolve_max_tokens(max_tokens)
301
+ payload = json.dumps({
302
+ "model": model,
303
+ "messages": [{"role": "user", "content": prompt}],
304
+ "max_tokens": max_tokens,
305
+ }).encode("utf-8")
306
+ req = urllib.request.Request(
307
+ base.rstrip("/") + "/chat/completions", data=payload,
308
+ headers={"Authorization": f"Bearer {key}",
309
+ "Content-Type": "application/json"})
310
+ try:
311
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
312
+ data = json.loads(resp.read().decode("utf-8"))
313
+ except urllib.error.HTTPError as exc:
314
+ detail = exc.read().decode("utf-8", "replace")[:300]
315
+ raise RuntimeError(f"HTTP {exc.code} model={model} base={base} :: {detail}") from None
316
+ return _extract_content(data, model, max_tokens)
317
+
318
+
319
+ # 生效条件:给定 role,若 role_config 得到非空 key 则 GET base/models 并返回含 ok/model_available/models 的字典;无 key 或请求异常则返回 ok=False 及错误信息。
320
+ def probe_models(role: str, timeout: int = 20) -> dict:
321
+ """零 token 探测:列出该角色网关的可用模型 id(GET /models)。"""
322
+ model, base, key = role_config(role)
323
+ if not key:
324
+ return {"role": role, "model": model, "base": base, "ok": False,
325
+ "error": "no_key", "models": []}
326
+ req = urllib.request.Request(
327
+ base.rstrip("/") + "/models",
328
+ headers={"Authorization": f"Bearer {key}"})
329
+ try:
330
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
331
+ data = json.loads(resp.read().decode("utf-8"))
332
+ ids = [m.get("id") for m in (data.get("data") or []) if m.get("id")]
333
+ except Exception as exc: # noqa: BLE001 —— 探测要抗单点
334
+ return {"role": role, "model": model, "base": base, "ok": False,
335
+ "error": f"{type(exc).__name__}: {exc}"[:200], "models": []}
336
+ res = {"role": role, "model": model, "base": base, "ok": True,
337
+ "model_available": model in ids, "models": ids}
338
+ if not res["model_available"]:
339
+ res["note"] = ("该 id 未出现在 /models 列表:可能是限时/按需模型,或已下架;"
340
+ "以实际 /chat/completions 调用结果为准")
341
+ return res
342
+
343
+
344
+ # 生效条件:raw 为 None 或 strip 后不含 "{"(i<0)、或末个 "}" 的位置 j<=i 时返回 None;否则对 s 从首个 "{" 到末个 "}" 的切片 json.loads,成功则返回解析结果,抛 ValueError 时返回 None。
345
+ def _extract_json_obj(raw: str):
346
+ """从 LLM 输出里抠出第一个 JSON 对象(容忍代码围栏 / 前后废话)。"""
347
+ s = (raw or "").strip()
348
+ i, j = s.find("{"), s.rfind("}")
349
+ if i < 0 or j <= i:
350
+ return None
351
+ try:
352
+ return json.loads(s[i:j + 1])
353
+ except ValueError:
354
+ return None
355
+
356
+
357
+ # 生效条件:v 为 None 返回 [];否则按 v 是 str 取 [v]、是 list/tuple 取逐项、其他取 [str(v)],逐项 strip 并去两端包裹标点后跳过空串及长度超 MAX_TERM_LEN 的项,未出现过的才 append,每次 append 后若 len(out) >= limit 即 break 返回 out(故 limit<=0 且存在有效项时仍返回 1 项)。
358
+ def _as_terms(v, limit: int = MAX_TERMS):
359
+ """把 LLM 给的值规范成去重、限长的短语列表。"""
360
+ if v is None:
361
+ return []
362
+ if isinstance(v, str):
363
+ items = [v]
364
+ elif isinstance(v, (list, tuple)):
365
+ items = list(v)
366
+ else:
367
+ items = [str(v)]
368
+ out = []
369
+ for x in items:
370
+ s = str(x).strip().strip(",。;;、,.;\"'“”")
371
+ if not s or len(s) > MAX_TERM_LEN:
372
+ continue
373
+ if s not in out:
374
+ out.append(s)
375
+ if len(out) >= limit:
376
+ break
377
+ return out
378
+
379
+
380
+ # 生效条件:给定 raw,若 _extract_json_obj 解析出 dict,则按 CCG_FIELDS 与 FIELD_ALIASES 提取非空字段并规范为列表或单值返回字典;否则返回 {}。
381
+ def parse_candidate(raw: str) -> dict:
382
+ """LLM 原始输出 → {字段: 列表/字符串};解析失败返回 {}。"""
383
+ obj = _extract_json_obj(raw)
384
+ if not isinstance(obj, dict):
385
+ return {}
386
+ out = {}
387
+ for field in CCG_FIELDS:
388
+ val = None
389
+ for alias in FIELD_ALIASES[field]:
390
+ if alias in obj and obj[alias] not in (None, "", [], {}):
391
+ val = obj[alias]
392
+ break
393
+ if val is None:
394
+ continue
395
+ if field in SINGLE_FIELDS:
396
+ terms = _as_terms(val, limit=1)
397
+ if terms:
398
+ out[field] = terms[0]
399
+ else:
400
+ terms = _as_terms(val)
401
+ if terms:
402
+ out[field] = terms
403
+ return out
404
+
405
+
406
+ # 生效条件:给定 raw,若解析出 dict,则按 CCG_FIELDS 提取 keep/drop/has_keep/reason 结构返回字典;否则返回 {}。
407
+ def parse_verdict(raw: str) -> dict:
408
+ """验证单元输出 → {字段: {keep, drop, has_keep, reason}};解析失败返回 {}。"""
409
+ obj = _extract_json_obj(raw)
410
+ if not isinstance(obj, dict):
411
+ return {}
412
+ out = {}
413
+ for field in CCG_FIELDS:
414
+ val = None
415
+ for alias in FIELD_ALIASES[field]:
416
+ if alias in obj and obj[alias] not in (None, "", [], {}):
417
+ val = obj[alias]
418
+ break
419
+ if val is None:
420
+ continue
421
+ if isinstance(val, list): # 容忍只给 keep 数组
422
+ out[field] = {"keep": _as_terms(val), "drop": [], "has_keep": True,
423
+ "reason": ""}
424
+ elif isinstance(val, dict):
425
+ out[field] = {"keep": _as_terms(val.get("keep")),
426
+ "drop": _as_terms(val.get("drop")),
427
+ "has_keep": "keep" in val,
428
+ "reason": str(val.get("reason") or "")[:200]}
429
+ return out
430
+
431
+
432
+ # 生效条件:给定 kept 与 verdict,若 verdict 为空则返回 (dict(kept), {});否则按 drop 与 has_keep 收窄候选并返回 (收窄后候选, 被剔除明细)。
433
+ def narrow_by_verdict(kept: dict, verdict: dict):
434
+ """按验证单元裁决收窄候选——**只能否决,不能新增**。
435
+
436
+ · 验证单元未表态的字段 → 保留(沉默不等于否决)
437
+ · has_keep=True → 取「候选 ∩ keep」;否则只按 drop 剔除
438
+ 返回 (收窄后候选, 被剔除明细)。
439
+ """
440
+ if not verdict:
441
+ return dict(kept), {}
442
+ out, dropped = {}, {}
443
+ for field, val in kept.items():
444
+ terms = val if isinstance(val, list) else [val]
445
+ vd = verdict.get(field)
446
+ if vd is None:
447
+ out[field] = val
448
+ continue
449
+ dropset = set(vd.get("drop") or [])
450
+ keepset = set(vd.get("keep") or [])
451
+ surv, gone = [], []
452
+ for t in terms:
453
+ if t in dropset:
454
+ gone.append(t)
455
+ elif vd.get("has_keep") and t not in keepset:
456
+ gone.append(t)
457
+ else:
458
+ surv.append(t)
459
+ if gone:
460
+ dropped[field] = {"terms": gone, "reason": vd.get("reason") or ""}
461
+ if surv:
462
+ out[field] = surv if field in MULTI_FIELDS else surv[0]
463
+ return out, dropped
464
+
465
+
466
+ # ---- 确定性验证(零 LLM)-------------------------------------------------
467
+
468
+ # 生效条件:给定 term 与 body,若 term 的 bigram 序列非空则返回命中 bigram 数除以总 bigram 数,否则返回 0.0。
469
+ def grounding_score(term: str, body: str) -> float:
470
+ """候选短语在正文里的字符级支撑度 = 命中 bigram 数 / 总 bigram 数。"""
471
+ bg = bigrams(term or "")
472
+ if not bg:
473
+ return 0.0
474
+ hit = sum(1 for g in bg if g in (body or ""))
475
+ return hit / len(bg)
476
+
477
+
478
+ # 生效条件:给定 cand 与 body,按 thresholds 更新 DEFAULT_GROUNDING 后逐字段过滤候选,返回 (达标 kept, detail);不达标者丢弃。
479
+ def grounding_filter(cand: dict, body: str, thresholds: dict = None):
480
+ """逐字段过滤候选:返回 (kept, detail)。不达标者丢弃(对应「不猜测」)。"""
481
+ th = dict(DEFAULT_GROUNDING)
482
+ th.update(thresholds or {})
483
+ kept, detail = {}, {}
484
+ for field, val in cand.items():
485
+ terms = val if isinstance(val, list) else [val]
486
+ ok_terms, scores = [], {}
487
+ for t in terms:
488
+ g = grounding_score(t, body)
489
+ scores[t] = round(g, 3)
490
+ if g >= th.get(field, 0.5):
491
+ ok_terms.append(t)
492
+ detail[field] = {"scores": scores, "kept": len(ok_terms)}
493
+ if ok_terms:
494
+ kept[field] = ok_terms if field in MULTI_FIELDS else ok_terms[0]
495
+ return kept, detail
496
+
497
+
498
+ # 生效条件:给定 pos_terms、neg_terms、body,返回含 pos_recall、neg_separated、no_conflict、ok 的回放判定字典。
499
+ def replay_check(pos_terms, neg_terms, body: str) -> dict:
500
+ """回放生产判定:正例召回 + 负例剔除 + 无自相矛盾。
501
+
502
+ 复用 _path_semantic 的同一批原语,保证与生产路同源(P6 与真实路做一致性回归)。
503
+ """
504
+ pos_text = " ".join(pos_terms or [])
505
+ neg_text = " ".join(neg_terms or [])
506
+ tw_pos = expand_query_terms_weighted(pos_text) if pos_text else {}
507
+ tw_neg = expand_query_terms_weighted(neg_text) if neg_text else {}
508
+
509
+ # 1. 正例:以生效条件为查询,本节点正文应被命中,且不被自身负条件挡住
510
+ pos_recall = bool(pos_text) and _weighted_coverage(tw_pos, body) > 0.0 \
511
+ and not _neg_hit(tw_pos, neg_terms)
512
+
513
+ # 2. 负例:以不适用条件为查询,应触发条件级负路由;且负条件与正文低相关
514
+ # (负条件必须是「域外」的,若与正文强相关,等于让知识否定自己)
515
+ if neg_terms:
516
+ neg_separated = _neg_hit(tw_neg, neg_terms) \
517
+ and _weighted_coverage(tw_neg, body) < 0.5
518
+ else:
519
+ neg_separated = True
520
+
521
+ # 3. 生效条件与不适用条件不得互相覆盖
522
+ no_conflict = (not pos_text) or (not neg_text) \
523
+ or _weighted_coverage(tw_pos, neg_text) < 0.5
524
+
525
+ ok = pos_recall and neg_separated and no_conflict
526
+ return {"pos_recall": pos_recall, "neg_separated": neg_separated,
527
+ "no_conflict": no_conflict, "ok": ok}
528
+
529
+
530
+ # ---- 写盘(固化)---------------------------------------------------------
531
+
532
+ # 生效条件:给定 content 与 field,当 content 含 "# field:" 或 "# field:" 时返回 True,否则 False。
533
+ def _has_ccg_line(content: str, field: str) -> bool:
534
+ return f"# {field}:" in (content or "") or f"# {field}:" in (content or "")
535
+
536
+
537
+ # 生效条件:给定 fm 与 content,对每个 CCG_FIELDS,若 frontmatter.comment 值非空或正文含对应 CCG 行则记入,返回已有字段字典。
538
+ def existing_fields(fm: dict, content: str) -> dict:
539
+ """节点当前已有的四要素:正文 CCG 行 或 frontmatter.comment 任一存在即算有。"""
540
+ comment = (fm.get("state_attributes") or {}).get("comment") or {}
541
+ out = {}
542
+ for field in CCG_FIELDS:
543
+ v = comment.get(field)
544
+ if v not in (None, "", [], {}):
545
+ out[field] = v
546
+ elif _has_ccg_line(content, field):
547
+ out[field] = True
548
+ return out
549
+
550
+
551
+ # 生效条件:给定 content、field、value,若已有 "# field:" 行则替换并返回新正文;否则插在 "# 功能名" 之后,若无则该行前置。
552
+ def _upsert_ccg_line(content: str, field: str, value: str) -> str:
553
+ """在正文里写入/替换 `# <字段>:<值>`,优先插在「# 功能名」之后。"""
554
+ lines = (content or "").split("\n")
555
+ for i, ln in enumerate(lines):
556
+ s = ln.strip()
557
+ if not s.startswith("#") or field not in s:
558
+ continue
559
+ name = s.lstrip("#").strip().split(":")[0].split(":")[0].strip()
560
+ if name == field:
561
+ lines[i] = f"# {field}:{value}"
562
+ return "\n".join(lines)
563
+ newline = f"# {field}:{value}"
564
+ for i, ln in enumerate(lines):
565
+ if ln.strip().startswith("# 功能名"):
566
+ lines.insert(i + 1, newline)
567
+ return "\n".join(lines)
568
+ return newline + "\n" + (content or "")
569
+
570
+
571
+ # 生效条件:给定 kept 字段字典,返回一句话规律字符串,列出缺失字段名并声明补齐后可路由。
572
+ def _evo_pattern(kept: dict) -> str:
573
+ """规律(一句话):这一类节点反复缺的正是这批条件。"""
574
+ names = "、".join(kept.keys())
575
+ return f"缺「{names}」的节点条件不可判;补齐后四要素完整、可路由"
576
+
577
+
578
+ # 生效条件:给定 prov 字典,拼接 reflect/verify 模型、grounding、replay、verification_basis 中存在的证据项并返回。
579
+ def _evo_evidence(prov: dict) -> str:
580
+ """证据:本次固化凭什么成立(模型 / 闸门 / 回放)。"""
581
+ parts = []
582
+ rf = (prov.get("reflect") or {}).get("model") or ""
583
+ vf = (prov.get("verify") or {}).get("model") or ""
584
+ if rf:
585
+ parts.append(f"reflect={rf}")
586
+ if vf:
587
+ parts.append(f"verify={vf}")
588
+ if prov.get("grounding"):
589
+ parts.append("grounding通过")
590
+ if prov.get("replay"):
591
+ parts.append("replay通过")
592
+ vb = prov.get("verification_basis") or ""
593
+ if vb:
594
+ parts.append(vb)
595
+ return " · ".join(parts)
596
+
597
+
598
+ # 生效条件:给定 cg、e、fm、content、kept、prov,将 kept 字段写入正文 CCG 行与 frontmatter.comment,不适用条件同步 non_applicable_conditions,并写 llm_consolidation 与演化记录,返回 None。
599
+ def _apply_node(cg, e, fm: dict, content: str, kept: dict, prov: dict,
600
+ basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT):
601
+ """把通过验证的字段固化进 md:正文 CCG 行 + frontmatter.comment + 负条件 + provenance。
602
+
603
+ 固化 = 对一条缺失条件的补充 → 同步落一条演化条目(md 账本,可回滚)。
604
+ """
605
+ nid = e.get("id") or os.path.basename(e["path"])[:-3]
606
+ before = evolution.state_of(cg, nid) or {}
607
+ comment = (fm.get("state_attributes") or {}).get("comment")
608
+ if not isinstance(comment, dict):
609
+ fm["state_attributes"] = dict(fm.get("state_attributes") or {})
610
+ fm["state_attributes"]["comment"] = {}
611
+ comment = fm["state_attributes"]["comment"]
612
+ for field, val in kept.items():
613
+ text = ";".join(val) if isinstance(val, list) else str(val)
614
+ content = _upsert_ccg_line(content, field, text)
615
+ comment[field] = text
616
+ if field == "不适用条件":
617
+ # 同步 frontmatter.non_applicable_conditions(引擎负路由读它)
618
+ cur = [str(x) for x in (fm.get("non_applicable_conditions") or [])]
619
+ for t in (val if isinstance(val, list) else [val]):
620
+ if t not in cur:
621
+ cur.append(t)
622
+ fm["non_applicable_conditions"] = cur
623
+ if basis:
624
+ content = _upsert_ccg_line(content, "验证方式", basis)
625
+ comment["验证方式"] = basis
626
+ if not nodefile.verification_basis_valid(fm):
627
+ # 枚举里没有「LLM 交叉验证」这一档,只能落到 other(声明文本在 CCG 行里)
628
+ fm["verification_basis"] = basis_enum
629
+ fm["llm_consolidation"] = prov
630
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
631
+ durable=True)
632
+ # 每一次修改都是对缺失条件的补充:记录规律 + 状态,不记录实现。
633
+ evolution.record(
634
+ cg, node_id=nid,
635
+ pattern=_evo_pattern(kept),
636
+ missing="、".join(kept.keys()),
637
+ action="补齐 CCG 字段:" + "、".join(kept.keys()),
638
+ evidence=_evo_evidence(prov),
639
+ source="consolidate", kind=evolution.KIND_CONDITION_GAP,
640
+ before=before, after=evolution.state_of(cg, nid) or {})
641
+
642
+
643
+ # 生效条件:给定 root,扫描正排层节点并执行反思→白箱闸门→验证→固化,返回报表 rep;require_verify=True 且无 verify_fn 时全部 DEFER。
644
+ def consolidate(root: str, layer: str = None, limit: int = None, apply: bool = False,
645
+ overwrite: bool = False, llm_fn=None, reflect_fn=None,
646
+ verify_fn=None, reflect_model: str = "", verify_model: str = "",
647
+ verification_basis: str = "", basis_enum: str = BASIS_ENUM_DEFAULT,
648
+ require_verify: bool = True, thresholds: dict = None,
649
+ verbose: bool = True) -> dict:
650
+ """对正排层节点做「反思单元产出候选 → 白箱闸门 → 验证单元否决 → 固化」。
651
+
652
+ llm_fn 是 reflect_fn 的旧名(向后兼容,单模型模式)。
653
+ require_verify=True 且无 verify_fn → 一律 DEFER(纪律 5:未经验证不固化)。
654
+ """
655
+ reflect_fn = reflect_fn or llm_fn
656
+ cg = MdCGOS(root)
657
+ entries = cg._candidates(layer=layer)
658
+ t0 = time.time()
659
+ rep = {"root": root, "layer": layer, "dry_run": not apply,
660
+ "reflect_model": reflect_model, "verify_model": verify_model,
661
+ "reflect": bool(reflect_fn), "verify": bool(verify_fn), "llm": bool(reflect_fn),
662
+ "require_verify": require_verify,
663
+ "verification_basis": verification_basis,
664
+ "nodes_scanned": len(entries),
665
+ "targeted": 0, "accepted": 0, "rejected": 0, "deferred": 0,
666
+ "skipped_complete": 0, "written": 0, "reasons": {},
667
+ "per_field": {f: 0 for f in CCG_FIELDS}, "verify_dropped": 0,
668
+ "verification_basis_missing": 0, "samples": []}
669
+
670
+ # 生效条件:以 reason 为键写入闭包 rep["reasons"],计数按 rep["reasons"].get(reason, 0) + 1 递增(键缺失从 0 起算),无返回值。
671
+ def _bump(reason):
672
+ rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
673
+
674
+ for e in entries:
675
+ if limit is not None and rep["targeted"] >= limit:
676
+ break
677
+ nid = os.path.basename(e["path"])[:-3]
678
+ fm, content = cg._read(e)
679
+ if fm is None:
680
+ _bump("read_failed")
681
+ continue
682
+ if crypto.is_encrypted(content):
683
+ _bump("locked") # 无密钥 → fail-closed:绝不改写密文
684
+ continue
685
+ have = existing_fields(fm, content)
686
+ missing = [f for f in CCG_FIELDS if f not in have]
687
+ if not nodefile.verification_basis_valid(fm):
688
+ rep["verification_basis_missing"] += 1
689
+ if not missing:
690
+ rep["skipped_complete"] += 1
691
+ continue
692
+ rep["targeted"] += 1
693
+
694
+ if not reflect_fn:
695
+ rep["deferred"] += 1
696
+ _bump("no_llm")
697
+ continue
698
+
699
+ # 1) 反思单元:产出候选(黑箱,唯一产出权)
700
+ body = body_text(content)[:MAX_BODY_CHARS]
701
+ title = _ccg_field(content, "功能名") or nid
702
+ prompt = REFLECT_PROMPT.format(title=title, body=body)
703
+ try:
704
+ raw = reflect_fn(prompt)
705
+ except Exception as exc: # noqa: BLE001 —— 离线批处理要抗单点失败
706
+ rep["deferred"] += 1
707
+ _bump(f"reflect_error:{type(exc).__name__}")
708
+ continue
709
+ cand = parse_candidate(raw)
710
+ if not cand:
711
+ rep["deferred"] += 1
712
+ _bump("parse_failed")
713
+ continue
714
+
715
+ # 2) 白箱闸门:grounding + replay(零 LLM,先跑,省调用)
716
+ kept, gdetail = grounding_filter(cand, body, thresholds)
717
+ pos = kept.get("生效条件") or []
718
+ neg = kept.get("不适用条件") or []
719
+ replay = replay_check(pos, neg, body)
720
+ if not kept or not replay["ok"]:
721
+ rep["rejected"] += 1
722
+ _bump("replay_failed" if kept else "grounding_failed")
723
+ if verbose and len(rep["samples"]) < 8:
724
+ rep["samples"].append({"id": nid, "verdict": "REJECT",
725
+ "stage": "whitebox", "grounding": gdetail,
726
+ "replay": replay})
727
+ continue
728
+
729
+ # 3) 验证单元:逐条核验,只能否决、不能新增
730
+ dropped, vprompt, vd = {}, "", None
731
+ if verify_fn:
732
+ vprompt = VERIFY_PROMPT.format(
733
+ cand=json.dumps(kept, ensure_ascii=False), title=title, body=body)
734
+ try:
735
+ vd = parse_verdict(verify_fn(vprompt))
736
+ kept, dropped = narrow_by_verdict(kept, vd)
737
+ except Exception as exc: # noqa: BLE001
738
+ rep["deferred"] += 1
739
+ _bump(f"verify_error:{type(exc).__name__}")
740
+ continue
741
+ if not kept:
742
+ rep["rejected"] += 1
743
+ _bump("verify_rejected")
744
+ if verbose and len(rep["samples"]) < 8:
745
+ rep["samples"].append({"id": nid, "verdict": "REJECT",
746
+ "stage": "verify", "dropped": dropped})
747
+ continue
748
+ rep["verify_dropped"] += sum(len(d["terms"]) for d in dropped.values())
749
+ elif require_verify:
750
+ # 验证单元不可用 → 不固化(纪律 5:未经验证不固化)
751
+ rep["deferred"] += 1
752
+ _bump("verify_unavailable")
753
+ continue
754
+
755
+ # 3) 不覆盖已有非空字段(保护人工既有知识)
756
+ if not overwrite:
757
+ kept = {f: v for f, v in kept.items() if f not in have}
758
+ if not kept:
759
+ rep["skipped_complete"] += 1
760
+ continue
761
+
762
+ prov = {"at": round(time.time(), 3), "verdict": "ACCEPT",
763
+ "source_hash": _sig(content),
764
+ "reflect": {"model": reflect_model, "prompt_hash": _sig(prompt),
765
+ "fields": sorted(kept)},
766
+ "verify": ({"model": verify_model, "prompt_hash": _sig(vprompt),
767
+ "dropped": dropped, "verdict_fields": sorted(vd or {}),
768
+ "self_verify": verify_fn is reflect_fn}
769
+ if verify_fn else {"model": "", "status": "skipped"}),
770
+ "grounding": gdetail, "replay": replay,
771
+ "verification_basis": verification_basis}
772
+ rep["accepted"] += 1
773
+ for f in kept:
774
+ rep["per_field"][f] += 1
775
+ if apply:
776
+ _apply_node(cg, e, fm, content, kept, prov,
777
+ verification_basis, basis_enum)
778
+ append_jsonl(os.path.join(cg.root, "_consolidate.jsonl"),
779
+ {"t": time.time(), "id": nid, "verdict": "ACCEPT",
780
+ "fields": sorted(kept),
781
+ "reflect_model": reflect_model,
782
+ "verify_model": verify_model, "dropped": dropped,
783
+ "source_hash": prov["source_hash"], "replay": replay})
784
+ rep["written"] += 1
785
+ if verbose and len(rep["samples"]) < 8:
786
+ rep["samples"].append({"id": nid, "verdict": "ACCEPT",
787
+ "fields": sorted(kept), "dropped": dropped,
788
+ "replay": replay})
789
+
790
+ if apply and rep["written"]:
791
+ # 正文新增了 `# 不适用条件:` / `# 验证方式:` → 索引字段变了
792
+ cg.rebuild_index()
793
+ rep["elapsed_sec"] = round(time.time() - t0, 3)
794
+ return rep
795
+
796
+
797
+ # 生效条件:给定 root 与 basis,对缺 "# 验证方式" 行的节点补写验证方式并在需要时写入 basis_enum,返回统计 rep。
798
+ def fill_verification_basis(root: str, basis: str, layer: str = None,
799
+ limit: int = None, apply: bool = False,
800
+ basis_enum: str = BASIS_ENUM_DEFAULT) -> dict:
801
+ """只补「验证方式」——声明文本是常量,不需要黑箱生成,零 LLM 成本。
802
+
803
+ 对应纪律 3「不猜测」:验证基底必须由人/流程声明,而不是让模型编出来。
804
+ """
805
+ cg = MdCGOS(root)
806
+ entries = cg._candidates(layer=layer)
807
+ rep = {"root": root, "layer": layer, "dry_run": not apply, "basis": basis,
808
+ "basis_enum": basis_enum, "nodes_scanned": len(entries),
809
+ "targeted": 0, "skipped_present": 0, "skipped_locked": 0,
810
+ "written": 0}
811
+ for e in entries:
812
+ if limit is not None and rep["written"] >= limit:
813
+ break
814
+ fm, content = cg._read(e)
815
+ if fm is None:
816
+ continue
817
+ if crypto.is_encrypted(content):
818
+ rep["skipped_locked"] += 1 # 无密钥 → fail-closed:绝不改写密文
819
+ continue
820
+ if _has_ccg_line(content, "验证方式"):
821
+ rep["skipped_present"] += 1
822
+ continue
823
+ rep["targeted"] += 1
824
+ if not apply:
825
+ continue
826
+ comment = (fm.get("state_attributes") or {}).get("comment")
827
+ if not isinstance(comment, dict):
828
+ fm["state_attributes"] = dict(fm.get("state_attributes") or {})
829
+ fm["state_attributes"]["comment"] = {}
830
+ comment = fm["state_attributes"]["comment"]
831
+ content = _upsert_ccg_line(content, "验证方式", basis)
832
+ comment["验证方式"] = basis
833
+ if not nodefile.verification_basis_valid(fm):
834
+ fm["verification_basis"] = basis_enum
835
+ nid = e.get("id") or os.path.basename(e["path"])[:-3]
836
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
837
+ durable=True)
838
+ rep["written"] += 1
839
+ if apply and rep["written"]:
840
+ cg.rebuild_index()
841
+ return rep
842
+
843
+
844
+ # ==========================================================================
845
+ # 情境层批量提升(consolidate.promote)
846
+ # ==========================================================================
847
+ #
848
+ # 场景:情境层(contextual)里有些记忆被反复命中/并入——它们已经不是「一次情境」,
849
+ # 而是稳定的规律。本动作把它们提升为长期知识(knowledge),并保留:
850
+ # · 双向可追溯:promoted_from + 演化账本(KIND_LAYER_SHIFT);
851
+ # · 条件门槛:四要素(CCG)不全者**不提升**(未可判定就不该升格为长期知识);
852
+ # · 可预演:apply=False 只出报表;可留痕:`_maintain.jsonl`。
853
+
854
+ MAINTAIN_LOG = "_maintain.jsonl"
855
+ CCG_REQUIRED = ("生效条件", "子功能", "执行", "不适用条件")
856
+
857
+
858
+ # 生效条件:给定 cg、nid、e、fm、content、target_layer,把节点写入目标层(必要时按 routing 分桶)并删除旧路径,返回新相对路径与 bucket。
859
+ def _relocate_layer(cg, nid, e, fm, content, target_layer):
860
+ """把节点正文迁到目标层的正确目录(含分桶),删除旧文件。返回新相对路径。"""
861
+ d = os.path.join(cg.root, target_layer)
862
+ bucket = None
863
+ if target_layer in BUCKETED_LAYERS:
864
+ bucket = routing.bucket_dir(routing.route_key(fm.get("condition_space"),
865
+ fm.get("tags")))
866
+ d = os.path.join(d, bucket)
867
+ os.makedirs(d, exist_ok=True)
868
+ new_path = os.path.join(d, f"{nid}.md")
869
+ old_path = os.path.join(cg.root, e.get("path") or f"{nid}.md")
870
+ cg._write_node(nid, new_path, fm, content, durable=True)
871
+ if os.path.abspath(old_path) != os.path.abspath(new_path) and os.path.exists(old_path):
872
+ os.remove(old_path)
873
+ return {"path": os.path.relpath(new_path, cg.root).replace("\\", "/"),
874
+ "bucket": bucket}
875
+
876
+
877
+ # 生效条件:给定 root,把 source_layer 中命中次数不小于 min_merge 或 importance 不小于 min_importance 且条件完整的节点提升到 target_layer,返回统计 rep。
878
+ def promote_memories(root, source_layer="contextual", target_layer="knowledge",
879
+ min_merge=2, min_importance=0.6, require_conditions=True,
880
+ limit=None, apply=False, actor="maintain") -> dict:
881
+ """把反复命中的情境记忆批量提升为长期知识(可预演 / 可留痕 / 可追溯)。"""
882
+ cg = MdCGOS(root)
883
+ entries = cg._candidates(layer=source_layer)
884
+ rep = {"root": root, "source_layer": source_layer, "target_layer": target_layer,
885
+ "dry_run": not apply, "nodes_scanned": len(entries), "targeted": 0,
886
+ "skipped_locked": 0, "skipped_incomplete": 0, "skipped_not_hot": 0,
887
+ "written": 0, "promoted": [], "samples": [],
888
+ "min_merge": min_merge, "min_importance": min_importance,
889
+ "require_conditions": bool(require_conditions)}
890
+ batch = time.strftime("%Y%m%d-%H%M%S")
891
+ for e in entries:
892
+ if limit is not None and rep["written"] >= int(limit):
893
+ break
894
+ fm, content = cg._read(e)
895
+ if fm is None:
896
+ continue
897
+ if crypto.is_encrypted(content):
898
+ rep["skipped_locked"] += 1 # 无密钥 → fail-closed,绝不解密回写
899
+ continue
900
+ nid = e.get("id") or os.path.basename(e["path"])[:-3]
901
+ hits = max(int(fm.get("merge_count") or 0),
902
+ int(fm.get("access_count") or 0),
903
+ int(fm.get("recall_count") or 0))
904
+ imp = float(fm.get("importance") or e.get("importance") or 0.0)
905
+ complete = all(_has_ccg_line(content, f) for f in CCG_REQUIRED)
906
+ if require_conditions and not complete:
907
+ rep["skipped_incomplete"] += 1 # 四要素不全 → 不可判定,不升格
908
+ continue
909
+ hot = hits >= int(min_merge)
910
+ if not hot and imp < float(min_importance):
911
+ rep["skipped_not_hot"] += 1
912
+ continue
913
+ rep["targeted"] += 1
914
+ item = {"id": nid, "hits": hits, "importance": round(imp, 4),
915
+ "conditions_complete": complete,
916
+ "basis": fm.get("verification_basis")}
917
+ if len(rep["samples"]) < 8:
918
+ rep["samples"].append(item)
919
+ if not apply:
920
+ continue
921
+ before = evolution.state_of(cg, nid) or {}
922
+ fm["layer"] = target_layer
923
+ fm["promoted_from"] = source_layer
924
+ fm["promoted_at"] = time.time()
925
+ fm["promotion_basis"] = {"hits": hits, "importance": round(imp, 4),
926
+ "conditions_complete": complete, "batch": batch,
927
+ "actor": actor}
928
+ moved = _relocate_layer(cg, nid, e, fm, content, target_layer)
929
+ evolution.record(
930
+ cg, node_id=nid,
931
+ pattern="情境记忆反复命中/并入 → 提升为长期知识",
932
+ missing="", action=f"层迁移 {source_layer}→{target_layer}",
933
+ evidence=f"hits={hits} importance={imp:.2f} conditions_complete={complete}",
934
+ source="consolidate", kind=evolution.KIND_LAYER_SHIFT,
935
+ before=before, after=evolution.state_of(cg, nid) or {})
936
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
937
+ "t": time.time(), "action": "promote", "batch": batch, "id": nid,
938
+ "from": source_layer, "to": target_layer, "hits": hits,
939
+ "importance": round(imp, 4), "path": moved["path"], "actor": actor})
940
+ rep["promoted"].append(nid)
941
+ rep["written"] += 1
942
+ if apply and rep["written"]:
943
+ cg.rebuild_index()
944
+ rep["note"] = ("dry-run:未写盘;apply=True 才迁移层"
945
+ if not apply else f"已提升 {rep['written']} 个节点到 {target_layer}")
946
+ return rep
947
+
948
+
949
+ # 生效条件:给定 root,按 _maintain.jsonl 中 action=promote 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
950
+ def rollback_promotion(root, node_ids=None, batch=None, actor="maintain") -> dict:
951
+ """回滚情境提升:把 promoted_from 层迁回,并记一条演化条目。"""
952
+ cg = MdCGOS(root)
953
+ recs = [r for r in _read_maintain(root)
954
+ if r.get("action") == "promote"
955
+ and (not batch or r.get("batch") == batch)
956
+ and (not node_ids or str(r.get("id")) in {str(x) for x in node_ids})]
957
+ if not recs:
958
+ return {"ok": False, "error": "no_records", "reverted": 0}
959
+ reverted, ids = 0, []
960
+ for rec in recs:
961
+ nid = rec["id"]
962
+ e = (cg.index.get("nodes") or {}).get(nid)
963
+ if not e:
964
+ continue
965
+ fm, content = cg._read(e)
966
+ if fm is None or crypto.is_encrypted(content):
967
+ continue
968
+ back = rec.get("from") or "contextual"
969
+ before = evolution.state_of(cg, nid) or {}
970
+ fm["layer"] = back
971
+ fm["promoted_from"] = None
972
+ fm["promotion_basis"] = {"rollback_of": rec.get("batch"), "actor": actor}
973
+ _relocate_layer(cg, nid, e, fm, content, back)
974
+ evolution.record(cg, node_id=nid, pattern="提升回滚:长期知识退回情境层",
975
+ action=f"层迁移 {rec.get('to')}→{back}",
976
+ evidence=f"rollback batch={rec.get('batch')}",
977
+ source="consolidate", kind=evolution.KIND_ROLLBACK,
978
+ before=before, after=evolution.state_of(cg, nid) or {})
979
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
980
+ "t": time.time(), "action": "promote_rollback", "batch": rec.get("batch"),
981
+ "id": nid, "to": back, "actor": actor})
982
+ reverted += 1
983
+ ids.append(nid)
984
+ if reverted:
985
+ cg.rebuild_index()
986
+ return {"ok": True, "reverted": reverted, "ids": ids}
987
+
988
+
989
+ # ==========================================================================
990
+ # 层归位(consolidate.contextualize)
991
+ # ==========================================================================
992
+ #
993
+ # 场景:批次流水账(note_/milestone_/retest6_)与感知产物(imgpart_/vpipe_)混在
994
+ # knowledge 层——它们的语义是**情境**(某次批次的记录 / 某张图的一次观测),不是
995
+ # 长期知识;但也不该进 rejected/unresolved(那是「失效 / 未解」,不是「情境」)。
996
+ # 故归位到 contextual:
997
+ # · 只改 layer 与落点目录;正文 / 密级 / id / tags 一律不动;
998
+ # · **保留可召回**(contextual 已在层白名单内,且 predict._SAFE_LAYERS 含之);
999
+ # · 可预演(apply=False)/ 可留痕(`_maintain.jsonl`)/ 可追溯(KIND_LAYER_SHIFT)
1000
+ # / 可回滚(按 batch 或 id 反向迁层)。
1001
+ #
1002
+ # 与 promote 的关系:promote 是 contextual→knowledge(升格),本动作是
1003
+ # knowledge→contextual(归位)。两者共用 `_relocate_layer` 与批次台账,方向相反。
1004
+
1005
+ CONTEXTUALIZE_REASON_DEFAULT = "情境性内容归位(批次流水账 / 感知产物)"
1006
+
1007
+
1008
+ # 生效条件:给定 e,返回 e.id 字符串,若缺 id 则回落到 basename(e.path) 去掉 .md。
1009
+ def _entry_id(e) -> str:
1010
+ """索引条目取 id:优先 `id` 字段,回落到文件名(索引不保证带 id)。"""
1011
+ return str(e.get("id") or os.path.basename(e.get("path") or "")[:-3])
1012
+
1013
+
1014
+ # 生效条件:给定 root 与 base,若 base 不在维护日志已用批次中则返回 base,否则返回 base.n 且 n 为最小未用序号。
1015
+ def _unique_batch(root, base) -> str:
1016
+ """批次号去重:**同一秒内的两次调用不得共用批次号**。
1017
+
1018
+ 否则「按批次回滚」会连带命中上一次的台账记录(回滚必须是精确的、可对账的)。
1019
+ """
1020
+ seen = {r.get("batch") for r in _read_maintain(root)}
1021
+ if base not in seen:
1022
+ return base
1023
+ n = 2
1024
+ while f"{base}.{n}" in seen:
1025
+ n += 1
1026
+ return f"{base}.{n}"
1027
+
1028
+
1029
+ # 生效条件:给定 root 且 prefixes 或 node_ids 至少一个非空,把 source_layer 中匹配的节点迁到 target_layer,返回统计 rep;两者皆空则抛 ValueError。
1030
+ def contextualize_prefixes(root, prefixes=None, node_ids=None,
1031
+ source_layer="knowledge", target_layer="contextual",
1032
+ reason="", limit=None, apply=False,
1033
+ actor="maintain") -> dict:
1034
+ """按 id 前缀(或定向 id 列表)把节点从 source_layer 归位到 target_layer。
1035
+
1036
+ 默认方向 knowledge→contextual。`prefixes` / `node_ids` **至少给一个**:
1037
+ 宁可少搬,不可全库乱搬——不传白名单直接报错,拒绝「一次误调用把整个知识层改层」
1038
+ 这种不可归因的批量改写。`node_ids` 用于定向(含「回滚后单独补迁」的对称操作)。
1039
+ """
1040
+ pref = tuple(str(p) for p in (prefixes or ()) if str(p))
1041
+ ids = {str(i) for i in (node_ids or ()) if str(i)} or None
1042
+ if not pref and not ids:
1043
+ raise ValueError("contextualize 需要显式 prefixes 或 node_ids"
1044
+ "(如 ['note_','imgpart_']),拒绝对整层无差别改写")
1045
+ cg = MdCGOS(root)
1046
+
1047
+ # 生效条件:e 经 _entry_id 得到 nid 后,若闭包 ids 不为 None 则返回 nid in ids 的真假,若 ids 为 None 则返回 nid.startswith(pref) 的真假。
1048
+ def _hit(e) -> bool:
1049
+ nid = _entry_id(e)
1050
+ return nid in ids if ids is not None else nid.startswith(pref)
1051
+
1052
+ entries = [e for e in cg._candidates(layer=source_layer) if _hit(e)]
1053
+ batch = _unique_batch(root, time.strftime("%Y%m%d-%H%M%S"))
1054
+ rep = {"root": root, "action": "contextualize", "dry_run": not apply,
1055
+ "source_layer": source_layer, "target_layer": target_layer,
1056
+ "prefixes": list(pref), "node_ids": sorted(ids) if ids else [],
1057
+ "reason": reason or CONTEXTUALIZE_REASON_DEFAULT,
1058
+ "nodes_scanned": len(entries), "targeted": 0, "skipped_locked": 0,
1059
+ "skipped_already": 0, "written": 0, "moved": [], "samples": [],
1060
+ "batch": batch}
1061
+ for e in entries:
1062
+ if limit is not None and rep["written"] >= int(limit):
1063
+ break
1064
+ nid = _entry_id(e)
1065
+ if not _hit(e):
1066
+ continue
1067
+ fm, content = cg._read(e)
1068
+ if fm is None:
1069
+ continue
1070
+ if crypto.is_encrypted(content):
1071
+ rep["skipped_locked"] += 1 # 无密钥 → fail-closed,绝不解密回写
1072
+ continue
1073
+ if fm.get("layer") != source_layer:
1074
+ rep["skipped_already"] += 1
1075
+ continue
1076
+ rep["targeted"] += 1
1077
+ if len(rep["samples"]) < 8:
1078
+ rep["samples"].append({"id": nid, "from": fm.get("layer"),
1079
+ "path": e.get("path")})
1080
+ if not apply:
1081
+ continue
1082
+ before = evolution.state_of(cg, nid) or {}
1083
+ fm["layer"] = target_layer
1084
+ fm["contextualized_from"] = source_layer
1085
+ fm["contextualized_at"] = time.time()
1086
+ fm["contextualization_basis"] = {"reason": rep["reason"], "batch": batch,
1087
+ "actor": actor}
1088
+ moved = _relocate_layer(cg, nid, e, fm, content, target_layer)
1089
+ evolution.record(
1090
+ cg, node_id=nid,
1091
+ pattern="情境性内容(批次流水账 / 感知产物)混在知识层 → 归位情境层",
1092
+ missing="层归属规则", action=f"层迁移 {source_layer}→{target_layer}",
1093
+ evidence=f"prefix={str(nid).split('_')[0]}_ reason={rep['reason']}",
1094
+ source="consolidate", kind=evolution.KIND_LAYER_SHIFT,
1095
+ before=before, after=evolution.state_of(cg, nid) or {})
1096
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1097
+ "t": time.time(), "action": "contextualize", "batch": batch, "id": nid,
1098
+ "from": source_layer, "to": target_layer, "path": moved["path"],
1099
+ "bucket": moved.get("bucket"), "reason": rep["reason"], "actor": actor})
1100
+ rep["moved"].append(nid)
1101
+ rep["written"] += 1
1102
+ if apply and rep["written"]:
1103
+ cg.rebuild_index()
1104
+ rep["note"] = ("dry-run:未写盘;apply=True 才归位"
1105
+ if not apply else f"已归位 {rep['written']} 个节点到 {target_layer}")
1106
+ return rep
1107
+
1108
+
1109
+ # 生效条件:给定 root,按 _maintain.jsonl 中 action=contextualize 记录(可再按 node_ids/batch 过滤)把节点迁回原层,成功返回 ok=True/reverted/ids,无记录返回 ok=False/error=no_records。
1110
+ def rollback_contextualize(root, node_ids=None, batch=None, actor="maintain") -> dict:
1111
+ """回滚层归位:按 `_maintain.jsonl` 的 contextualize 记录把节点迁回原层。"""
1112
+ cg = MdCGOS(root)
1113
+ recs = [r for r in _read_maintain(root)
1114
+ if r.get("action") == "contextualize"
1115
+ and (not batch or r.get("batch") == batch)
1116
+ and (not node_ids or str(r.get("id")) in {str(x) for x in node_ids})]
1117
+ if not recs:
1118
+ return {"ok": False, "error": "no_records", "reverted": 0}
1119
+ reverted, ids = 0, []
1120
+ for rec in recs:
1121
+ nid = rec["id"]
1122
+ e = (cg.index.get("nodes") or {}).get(nid)
1123
+ if not e:
1124
+ continue
1125
+ fm, content = cg._read(e)
1126
+ if fm is None or crypto.is_encrypted(content):
1127
+ continue
1128
+ back = rec.get("from") or "knowledge"
1129
+ before = evolution.state_of(cg, nid) or {}
1130
+ fm["layer"] = back
1131
+ fm["contextualized_from"] = None
1132
+ fm["contextualization_basis"] = {"rollback_of": rec.get("batch"),
1133
+ "actor": actor}
1134
+ _relocate_layer(cg, nid, e, fm, content, back)
1135
+ evolution.record(cg, node_id=nid,
1136
+ pattern="层归位回滚:情境层迁回原层",
1137
+ action=f"层迁移 {rec.get('to')}→{back}",
1138
+ evidence=f"rollback batch={rec.get('batch')}",
1139
+ source="consolidate", kind=evolution.KIND_ROLLBACK,
1140
+ before=before, after=evolution.state_of(cg, nid) or {})
1141
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1142
+ "t": time.time(), "action": "contextualize_rollback",
1143
+ "batch": rec.get("batch"), "id": nid, "to": back, "actor": actor})
1144
+ reverted += 1
1145
+ ids.append(nid)
1146
+ if reverted:
1147
+ cg.rebuild_index()
1148
+ return {"ok": True, "reverted": reverted, "ids": ids}
1149
+
1150
+
1151
+ # 生效条件:对 _read_maintain(root) 中 action 为 "contextualize" 或 "contextualize_rollback" 的记录,取 recs[-(int(limit) or 50):] 作为 records 返回 {'ok': True, ...}——仅当 int(limit) 成功且为 0 时回落 50,limit 为 None/""/[] 等无法 int() 的值会先抛 TypeError/ValueError。
1152
+ def contextualize_history(root, limit=50):
1153
+ """层归位的批次记录(只读)。"""
1154
+ recs = [r for r in _read_maintain(root)
1155
+ if r.get("action") in ("contextualize", "contextualize_rollback")]
1156
+ return {"ok": True, "records": recs[-(int(limit) or 50):]}
1157
+
1158
+
1159
+ # 生效条件:给定 root,读取 root 下 MAINTAIN_LOG 的 JSONL 并返回记录列表。
1160
+ def _read_maintain(root):
1161
+ from .fsutil import read_jsonl
1162
+ return list(read_jsonl(os.path.join(root, MAINTAIN_LOG)))
1163
+
1164
+
1165
+ # ==========================================================================
1166
+ # 归纳聚类(consolidate.induce)
1167
+ # ==========================================================================
1168
+ #
1169
+ # 与 promote 的分工:
1170
+ # promote —— 把**已经存在**的单条情境记忆升格为长期知识(节点不变,只迁层);
1171
+ # induce —— 把**多条**具体记忆归纳为一个**新的概念节点**(新增节点)。
1172
+ #
1173
+ # 归纳是「由具体到一般」的推理,其输出**不是事实断言**,而是待验证的假设:
1174
+ # · 证据基底一律记 inferred(未经验证),不得冒充 verified;
1175
+ # · 概念节点必须携带成员清单 + `generalizes`/`instance_of` 对称边,保证可回溯;
1176
+ # · 归纳不出「共同条件」时默认**拒绝生成**(没有条件依据的抽象=编造,对齐
1177
+ # 纪律 3「不猜测」);确需放宽须显式 require_conditions=False,且概念正文
1178
+ # 会写明「未归纳出共同条件」,不掩盖证据缺口。
1179
+
1180
+ INDUCE_MIN_CLUSTER = 3
1181
+ INDUCE_MIN_JACCARD = 0.30
1182
+ INDUCE_MAX_NODES = 400
1183
+ INDUCE_MAX_TERMS = 6
1184
+ CONCEPT_REL = "generalizes" # concept → member(inferred)
1185
+ CONCEPT_MEMBER_REL = "instance_of" # member → concept(inferred)
1186
+ CONCEPT_PREFIX = "concept_"
1187
+ CONCEPT_IMPORTANCE = 0.5
1188
+ CONCEPT_TAGS = ("concept", "induced")
1189
+ # 巩固留痕字段(2026-09-19 阶段一):**字段名真源在 md_cg/nodefile.py**,
1190
+ # 本处只做短别名引用(非复制),与 `nodefile.VALID_FROM_FIELD` 的登记纪律同构。
1191
+ CONSOLIDATED_AT_FIELD = nodefile.CONSOLIDATED_AT_FIELD
1192
+ CONSOLIDATED_INTO_FIELD = nodefile.CONSOLIDATED_INTO_FIELD
1193
+ # 归纳候选排除:受保护节点,以及洞察/场景/前馈/概念等派生物(避免自我进食)
1194
+ INDUCE_SKIP_TAGS = ("insight", "scene", "reconstructed", "gap_hint", "concept")
1195
+
1196
+
1197
+ # 生效条件:给定 members,返回 CONCEPT_PREFIX 拼接排序后成员串的 SHA1 前 10 位。
1198
+ def _concept_id(members):
1199
+ """概念节点 id:由成员清单派生,保证「同成员 ⇒ 同 id」的幂等性。"""
1200
+ h = hashlib.sha1("|".join(sorted(str(m) for m in members))
1201
+ .encode("utf-8")).hexdigest()
1202
+ return CONCEPT_PREFIX + h[:10]
1203
+
1204
+
1205
+ # 生效条件:给定 a 与 b,若任一为空集则返回 0.0,否则返回交集大小除以并集大小。
1206
+ def _jaccard(a, b):
1207
+ if not a or not b:
1208
+ return 0.0
1209
+ return len(a & b) / float(len(a | b))
1210
+
1211
+
1212
+ # 生效条件:给定 term_sets,返回出现次数不小于 max(2, ceil(min_share * len(term_sets))) 的词面排序列表;空输入返回 []。
1213
+ def _common_terms(term_sets, min_share=0.6):
1214
+ """出现在 ≥ min_share 比例成员中的词面(共同条件);少于 2 个成员共享不算。"""
1215
+ if not term_sets:
1216
+ return []
1217
+ cnt = {}
1218
+ for s in term_sets:
1219
+ for t in set(s or ()):
1220
+ cnt[t] = cnt.get(t, 0) + 1
1221
+ need = max(2, int(math.ceil(min_share * len(term_sets))))
1222
+ return sorted(t for t, c in cnt.items() if c >= need)
1223
+
1224
+
1225
+ # 生效条件:给定 term_sets,按集合排序去重拼接后返回前 limit(默认 INDUCE_MAX_TERMS)个词面。
1226
+ def _union_terms(term_sets, limit=INDUCE_MAX_TERMS):
1227
+ seen = []
1228
+ for s in term_sets:
1229
+ for t in sorted(s or ()):
1230
+ if t not in seen:
1231
+ seen.append(t)
1232
+ return seen[:limit]
1233
+
1234
+
1235
+ # 生效条件:给定 cg、cid、members、reason、actor、batch,为概念节点与成员节点写对称 inferred 边(已存在则跳过),返回含 concept 与 members 的字典。
1236
+ def _link_concept(cg, cid, members, reason, actor, batch):
1237
+ """写概念↔成员对称 inferred 边(幂等:已存在则不重复写)。"""
1238
+ nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
1239
+ out = {"concept": cid, "members": []}
1240
+ cnode = cg.get(cid)
1241
+ if cnode:
1242
+ fm = cnode.get("frontmatter") or {}
1243
+ edges = list(fm.get("edges") or [])
1244
+ have = {str(e.get("target")) for e in edges if isinstance(e, dict)}
1245
+ added = False
1246
+ for m in members:
1247
+ if m in have:
1248
+ continue
1249
+ edges.append({"target": m, "relation_type": CONCEPT_REL,
1250
+ "reason": reason, "created_at": time.time(),
1251
+ "confidence": 0.5, "verified": 0, "evidence": "inferred"})
1252
+ added = True
1253
+ if added:
1254
+ fm["edges"] = edges
1255
+ ent = nodes.get(cid) or {}
1256
+ cg._write_node(cid, os.path.join(cg.root, ent.get("path") or f"{cid}.md"),
1257
+ fm, cnode.get("content") or "")
1258
+ if ent:
1259
+ ent["edges"] = edges
1260
+ for m in members:
1261
+ node = cg.get(m)
1262
+ if not node:
1263
+ continue
1264
+ fm = node.get("frontmatter") or {}
1265
+ edges = list(fm.get("edges") or [])
1266
+ if any(isinstance(e, dict) and str(e.get("target")) == cid for e in edges):
1267
+ continue
1268
+ edges.append({"target": cid, "relation_type": CONCEPT_MEMBER_REL,
1269
+ "reason": reason, "created_at": time.time(),
1270
+ "confidence": 0.5, "verified": 0, "evidence": "inferred"})
1271
+ fm["edges"] = edges
1272
+ # 巩固留痕(2026-09-19 阶段一):`consolidated_into` 为**规范名**,
1273
+ # `induced_concept` 保留为历史别名(既有读取面零破坏);`consolidated_at`
1274
+ # 补齐**成员侧**巩固时刻——此前只有概念侧 `induced_at`,成员侧无从判定
1275
+ # 「何时被并进去」,故「合并后前身可定位」只在概念侧半成立。
1276
+ fm[CONSOLIDATED_INTO_FIELD] = cid
1277
+ fm[CONSOLIDATED_AT_FIELD] = time.time()
1278
+ fm["induced_concept"] = cid
1279
+ ent = nodes.get(m) or {}
1280
+ cg._write_node(m, os.path.join(cg.root, ent.get("path") or f"{m}.md"),
1281
+ fm, node.get("content") or "")
1282
+ if ent:
1283
+ ent["edges"] = edges
1284
+ out["members"].append(m)
1285
+ return out
1286
+
1287
+
1288
+ # 生效条件:给定 members、common_pos、neg_union,返回标注 inferred 的概念节点正文,含功能名、生效条件、子功能、执行、验证方式、不适用条件。
1289
+ def _concept_payload(members, common_pos, neg_union):
1290
+ """概念节点正文:把成员的共性条件抽象为可追溯的知识条目(显式标注 inferred)。"""
1291
+ label = "、".join(common_pos[:INDUCE_MAX_TERMS])
1292
+ pos_txt = ";".join(common_pos[:INDUCE_MAX_TERMS]) or "(未归纳出共同条件)"
1293
+ neg_txt = ";".join(neg_union[:INDUCE_MAX_TERMS]) or "(未判定)"
1294
+ return (
1295
+ "# 功能名:归纳概念:%s\n"
1296
+ "# 生效条件:%s\n"
1297
+ "# 子功能:%d 条具体记忆的共性(成员:%s)\n"
1298
+ "# 执行:由 consolidate.induce 归纳聚合(inferred;未经验证,不得直接当事实使用)\n"
1299
+ "# 验证方式:待验证(inferred 假设,需外部证据或实践重复后方可升格)\n"
1300
+ "# 不适用条件:%s\n"
1301
+ % (label or "共性", pos_txt, len(members), "、".join(members), neg_txt)
1302
+ )
1303
+
1304
+
1305
+ # 生效条件:给定 cg_or_root,从 source_layer 聚类归纳为 target_layer 概念节点,apply=True 才写盘并返回统计 rep。
1306
+ def induce_memories(cg_or_root, source_layer="contextual", target_layer="knowledge",
1307
+ min_cluster=INDUCE_MIN_CLUSTER, min_jaccard=INDUCE_MIN_JACCARD,
1308
+ max_nodes=INDUCE_MAX_NODES, require_conditions=True,
1309
+ limit=None, apply=False, actor="maintain", **extra):
1310
+ """归纳聚类:把多条具体记忆归纳为概念层条目(inferred,非事实断言)。
1311
+
1312
+ 流程:读取源层 → bigram 相似度贪心聚类 → 提炼共同条件 → 生成概念节点
1313
+ (apply=True)→ 写 `generalizes` / `instance_of` 对称 inferred 边 → 写留痕。
1314
+
1315
+ apply=False(默认)只出候选报表(可预演);apply=True 才写盘(可留痕、可回溯)。
1316
+ 幂等:概念 id 由成员清单派生,同成员重复归纳不新增节点。
1317
+ """
1318
+ cg = cg_or_root if isinstance(cg_or_root, MdCGOS) else MdCGOS(str(cg_or_root))
1319
+ from . import subgraph # 惰性导入:复用统一的条件/词面抽取
1320
+
1321
+ # MCP 分发层会把未提供的参数以 None 传入;此处归一化,避免 int(None) 崩溃,
1322
+ # 也避免 require_conditions=None 被当成 False 而悄悄关掉「无共同条件即拒绝生成」
1323
+ # 这条纪律(默认必须为真,放宽只能显式传 False)。
1324
+ min_cluster = INDUCE_MIN_CLUSTER if min_cluster is None else int(min_cluster)
1325
+ min_jaccard = INDUCE_MIN_JACCARD if min_jaccard is None else float(min_jaccard)
1326
+ max_nodes = INDUCE_MAX_NODES if max_nodes is None else int(max_nodes)
1327
+ if require_conditions is None:
1328
+ require_conditions = True
1329
+
1330
+ nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
1331
+ pool = [nid for nid, e in nodes.items()
1332
+ if (not source_layer or (e or {}).get("layer") == source_layer)
1333
+ and not (e or {}).get("protected")
1334
+ and not (set(INDUCE_SKIP_TAGS) & set((e or {}).get("tags") or []))]
1335
+ pool.sort()
1336
+ truncated = len(pool) > int(max_nodes)
1337
+ pool = pool[:int(max_nodes)]
1338
+
1339
+ cache = {}
1340
+ for nid in pool:
1341
+ got = subgraph._node_terms_and_grams(cg, nid)
1342
+ if got and got["grams"]:
1343
+ cache[nid] = got
1344
+ keys = sorted(cache.keys())
1345
+
1346
+ rep = {"ok": True, "action": "induce", "op": "consolidate",
1347
+ "source_layer": source_layer, "target_layer": target_layer,
1348
+ "dry_run": not apply, "nodes_scanned": len(pool), "indexed": len(keys),
1349
+ "truncated": truncated, "min_cluster": int(min_cluster),
1350
+ "min_jaccard": float(min_jaccard),
1351
+ "require_conditions": bool(require_conditions),
1352
+ "skipped_small": 0, "skipped_no_condition": 0, "skipped_existing": 0,
1353
+ "clusters": 0, "written": 0, "concepts": [], "samples": [],
1354
+ "log": MAINTAIN_LOG}
1355
+
1356
+ # ---- 贪心聚类(只读) ----
1357
+ assigned, proposals = set(), []
1358
+ for i, a in enumerate(keys):
1359
+ if a in assigned:
1360
+ continue
1361
+ ga = cache[a]["grams"]
1362
+ grp = [b for b in keys[i + 1:]
1363
+ if b not in assigned
1364
+ and _jaccard(ga, cache[b]["grams"]) >= float(min_jaccard)]
1365
+ if len(grp) + 1 < int(min_cluster):
1366
+ continue
1367
+ members = [a] + grp
1368
+ assigned.update(members)
1369
+ common_pos = _common_terms([cache[m]["pos"] for m in members])
1370
+ if require_conditions and not common_pos:
1371
+ rep["skipped_no_condition"] += 1
1372
+ continue
1373
+ neg_union = _union_terms([cache[m]["neg"] for m in members])
1374
+ proposals.append({
1375
+ "members": members, "concept_id": _concept_id(members),
1376
+ "common_conditions": common_pos, "non_applicable": neg_union,
1377
+ "reason": ("%d 条记忆内容相近且共享条件「%s」→ 归纳为概念"
1378
+ % (len(members), "、".join(common_pos) or "无")),
1379
+ })
1380
+ rep["clusters"] = len(proposals)
1381
+ for p in proposals[:8]:
1382
+ rep["samples"].append(p)
1383
+
1384
+ if not apply:
1385
+ rep["note"] = ("dry-run:未写盘;apply=True 才生成概念节点与 inferred 边"
1386
+ if proposals else "无满足条件的聚类(内容不够相近或缺乏共同条件)")
1387
+ rep["concepts"] = [p["concept_id"] for p in proposals]
1388
+ return rep
1389
+
1390
+ # ---- 落库(可留痕) ----
1391
+ batch = time.strftime("%Y%m%d-%H%M%S")
1392
+ for p in proposals:
1393
+ if limit is not None and rep["written"] >= int(limit):
1394
+ break
1395
+ cid = p["concept_id"]
1396
+ if cid in nodes:
1397
+ rep["skipped_existing"] += 1
1398
+ continue
1399
+ content = _concept_payload(p["members"], p["common_conditions"],
1400
+ p["non_applicable"])
1401
+ _consolidated_at = time.time() # 概念形成时刻 = 巩固时刻(单一取值,禁两处取时)
1402
+ cg.add(cid, content, layer=target_layer, tags=list(CONCEPT_TAGS),
1403
+ importance=CONCEPT_IMPORTANCE, verification_basis="other",
1404
+ induced_from=list(p["members"]), induced_at=_consolidated_at,
1405
+ consolidated_at=_consolidated_at,
1406
+ induction={"method": "bigram_jaccard", "min_jaccard": float(min_jaccard),
1407
+ "common_conditions": p["common_conditions"], "batch": batch,
1408
+ "actor": actor, "evidence": "inferred"},
1409
+ actor=actor)
1410
+ _link_concept(cg, cid, p["members"], p["reason"], actor, batch)
1411
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
1412
+ "t": time.time(), "action": "induce", "batch": batch, "concept": cid,
1413
+ "members": list(p["members"]), "common_conditions": p["common_conditions"],
1414
+ "source_layer": source_layer, "target_layer": target_layer, "actor": actor})
1415
+ rep["concepts"].append(cid)
1416
+ rep["written"] += 1
1417
+ if rep["written"]:
1418
+ cg.rebuild_index()
1419
+ rep["note"] = (f"已归纳 {rep['written']} 个概念节点(inferred,待验证)"
1420
+ if rep["written"] else "无可落库的归纳(均跳过或已达 limit)")
1421
+ return rep
1422
+
1423
+
1424
+ # ---- CLI ----------------------------------------------------------------
1425
+
1426
+ # 生效条件:不适用(无必需形参与模块级常量)
1427
+ def _cli(argv=None) -> int:
1428
+ ap = argparse.ArgumentParser(
1429
+ description="md_cg 离线固化:反思单元(LLM)产出候选 → 白箱闸门 → "
1430
+ "验证单元(LLM)否决 → 固化为 md 字段")
1431
+ ap.add_argument("--root", required=True, help="md 认知图根目录")
1432
+ ap.add_argument("--layer", default=None, help="只处理某层(如 knowledge)")
1433
+ ap.add_argument("--limit", type=int, default=None, help="只处理前 N 个待补节点")
1434
+ ap.add_argument("--apply", action="store_true", help="真正写盘(默认只验证)")
1435
+ ap.add_argument("--dry-run", action="store_true", help="只验证不写盘(默认行为)")
1436
+ ap.add_argument("--overwrite", action="store_true",
1437
+ help="允许覆盖已有非空字段(默认保护人工既有知识)")
1438
+ ap.add_argument("--reflect-model", default=None,
1439
+ help=f"反思单元模型(默认 {ROLE_DEFAULT_MODEL[REFLECT_ROLE]})")
1440
+ ap.add_argument("--verify-model", default=None,
1441
+ help=f"验证单元模型(默认 {ROLE_DEFAULT_MODEL[VERIFY_ROLE]})")
1442
+ ap.add_argument("--self-verify", action="store_true",
1443
+ help="降级:验证单元复用反思单元模型(非交叉验证,provenance 标记)")
1444
+ ap.add_argument("--no-verify", action="store_true",
1445
+ help="降级:跳过验证单元,仅靠白箱闸门(不推荐)")
1446
+ ap.add_argument("--verification-basis", default=None,
1447
+ help="写入 `# 验证方式:` 的声明文本(默认双模型声明)")
1448
+ ap.add_argument("--no-basis", action="store_true", help="不写「验证方式」")
1449
+ ap.add_argument("--basis-only", action="store_true",
1450
+ help="只补「验证方式」(零 LLM 成本),不做四要素反思")
1451
+ ap.add_argument("--min-grounding", type=float, default=None,
1452
+ help="统一 grounding 阈值(默认按字段 0.5 / 不适用条件 0.34)")
1453
+ ap.add_argument("--max-tokens", type=int, default=None,
1454
+ help=f"LLM 输出预算(含思考模型 reasoning_tokens;默认 "
1455
+ f"{DEFAULT_MAX_TOKENS},可 env {MAX_TOKENS_ENV} 覆盖;"
1456
+ "子代理配置标准 v0.5 §1)")
1457
+ ap.add_argument("--no-llm", action="store_true",
1458
+ help="不调用 LLM,只做四要素完整性普查")
1459
+ ap.add_argument("--check", action="store_true",
1460
+ help="零 token 探测两个角色网关的可用模型后退出")
1461
+ ap.add_argument("--report", default=None, help="把汇总 JSON 另存一份")
1462
+ a = ap.parse_args(argv)
1463
+
1464
+ r_model, _, r_key = role_config(REFLECT_ROLE, a.reflect_model)
1465
+ v_model, _, v_key = role_config(VERIFY_ROLE, a.verify_model)
1466
+ if a.self_verify:
1467
+ # 溯源修正(issue #24 附带②):--self-verify 实际调用的是反思单元模型,
1468
+ # verify.model 必须记实际值——此前记 ROLE_DEFAULT_MODEL[verify](glm),
1469
+ # 与真实调用不符,破坏可审计性。basis 声明同步改「同模型自验」,
1470
+ # 不再冒充双模型交叉验证。
1471
+ v_model = r_model
1472
+
1473
+ if a.check:
1474
+ out = {"reflect": probe_models(REFLECT_ROLE),
1475
+ "verify": probe_models(VERIFY_ROLE)}
1476
+ print(json.dumps(out, ensure_ascii=False, indent=2))
1477
+ return 0
1478
+
1479
+ if a.no_basis:
1480
+ basis = None
1481
+ elif a.verification_basis:
1482
+ basis = a.verification_basis
1483
+ elif a.self_verify:
1484
+ basis = f"同模型自验(reflect=verify={r_model},非交叉验证,降级模式)"
1485
+ else:
1486
+ basis = BASIS_TEMPLATE.format(reflect=r_model, verify=v_model)
1487
+
1488
+ if a.basis_only:
1489
+ rep = fill_verification_basis(a.root, basis, layer=a.layer, limit=a.limit,
1490
+ apply=a.apply)
1491
+ print(json.dumps(rep, ensure_ascii=False, indent=2))
1492
+ if a.report:
1493
+ with open(a.report, "w", encoding="utf-8") as f:
1494
+ json.dump(rep, f, ensure_ascii=False, indent=2)
1495
+ return 0
1496
+
1497
+ thresholds = ({f: a.min_grounding for f in CCG_FIELDS}
1498
+ if a.min_grounding is not None else None)
1499
+
1500
+ reflect_fn = verify_fn = None
1501
+ if not a.no_llm:
1502
+ if not r_key:
1503
+ print(f"[consolidate] 反思单元未配置 key"
1504
+ f"({_ROLE_ENV[REFLECT_ROLE][2]} / DEEPSEEK_API_KEY)→ 退化为普查模式",
1505
+ file=sys.stderr)
1506
+ else:
1507
+ reflect_fn = (lambda p: http_llm(p, role=REFLECT_ROLE, # noqa: E731
1508
+ model=a.reflect_model,
1509
+ max_tokens=a.max_tokens))
1510
+ if a.self_verify:
1511
+ verify_fn = reflect_fn
1512
+ elif not a.no_verify:
1513
+ if v_key:
1514
+ verify_fn = (lambda p: http_llm(p, role=VERIFY_ROLE, # noqa: E731
1515
+ model=a.verify_model,
1516
+ max_tokens=a.max_tokens))
1517
+ else:
1518
+ print(f"[consolidate] 验证单元未配置 key"
1519
+ f"({_ROLE_ENV[VERIFY_ROLE][2]} / ZHIPU_API_KEY / GLM_API_KEY)"
1520
+ "→ 待补节点将 DEFER,不写盘(纪律 5:未经验证不固化)",
1521
+ file=sys.stderr)
1522
+
1523
+ rep = consolidate(a.root, layer=a.layer, limit=a.limit, apply=a.apply,
1524
+ overwrite=a.overwrite, reflect_fn=reflect_fn,
1525
+ verify_fn=verify_fn, reflect_model=r_model,
1526
+ verify_model=v_model if verify_fn else "",
1527
+ verification_basis=basis or "",
1528
+ require_verify=not a.no_verify, thresholds=thresholds)
1529
+ print(json.dumps(rep, ensure_ascii=False, indent=2))
1530
+ if a.report:
1531
+ with open(a.report, "w", encoding="utf-8") as f:
1532
+ json.dump(rep, f, ensure_ascii=False, indent=2)
1533
+ return 0
1534
+
1535
+
1536
+ if __name__ == "__main__":
1440
1537
  raise SystemExit(_cli())