@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/scrub.py CHANGED
@@ -1,853 +1,853 @@
1
- # -*- coding: utf-8 -*-
2
- """md_cg · 记忆自净(记忆 OS #5):抽查 / 联想 / 去污染 / 校准偏差
3
-
4
- 自维持(`sustain.py`)解决「进程还活着」,本模块解决「记忆还干净」。
5
- 四件事构成一个闭环,挂在常驻循环上周期性跑:
6
-
7
- ① **记忆抽查(sample)**
8
- 分层抽样而非随机抽样:风险层(陈旧 / 低置信 / 有反例 / 从未验证 / 孤立)
9
- + 对照组(高频 / 随机基线)。对照组是刻意的——只查「看起来有问题」的节点
10
- 会形成确认偏差,永远发现不了「看起来没问题其实有问题」的记忆。
11
- `seed` 固定 → 同 seed 同样本(可复现、可审计,对齐本仓确定性白箱取向)。
12
-
13
- ② **联想(associate)**
14
- 从抽样节点出发三路邻域合并:
15
- 关系链(`chain.walk`,带条件序列)· 子图层级(`subgraph.expand`)
16
- · 词法近邻(bigram Jaccard)
17
- 用途是找到「同族节点」——矛盾与重复只能在同族之间判定,孤立地看一个节点
18
- 永远看不出它和谁冲突。
19
-
20
- ③ **去污染(audit → decontaminate)**
21
- 确定性判据(只读、可解释)识别五类污染:
22
- `contradiction` 同族矛盾(同键不同值 / 极性相反且共享词)
23
- `expired` 时效过期(`valid_until` / `expires_at` 已过)
24
- `duplicate` 高冗余(`forgetting.redundancy` ≥ 合并阈值)
25
- `orphan_noise` 孤立噪音(零访问 + 低重要 + 情境层)
26
- `unverified` 长期未验证(knowledge 层且零证据)
27
- 处置**只做可逆动作**且复用既有机制:
28
- · `weaken` → `cg.verify(verdict="weakened")`:反例 +1、置信 -0.15、
29
- 跌破 0.2 自动降级(验证机制天然留痕,不另造一套)
30
- · `demote` → `cg._move_layer("contextual")`:降出可信层(走保护闸门)
31
- · `hint` → 只给建议(合并 / 补证据需语义判断,不代劳)
32
- 受保护节点(self/anchor/importance≥0.7/显式保护)**跳过**;**永不删除**。
33
- 幂等:`_scrub.jsonl` 里已处理过的 (节点, 污染类型) 不重复动手。
34
-
35
- ④ **校准偏差(calibrate)**
36
- 用「自报置信 vs 实测正确率」(`metacognition.calibration`)算全局 gap 与
37
- 分桶偏差,给出偏置建议;`apply=True` 才写回(只动有证据的节点,保护节点跳过)。
38
-
39
- 所有动作留痕 `_scrub.jsonl`(payload-free:只记 id / 判据 / 动作,不记内容)。
40
- 零第三方依赖。
41
- """
42
- from __future__ import annotations
43
-
44
- import os
45
- import random
46
- import time
47
-
48
- from . import forgetting, protect
49
- from .fsutil import append_jsonl, read_jsonl
50
- from .mdcg import bigrams
51
-
52
- SCRUB_LOG = "_scrub.jsonl"
53
-
54
- DEFAULT_SAMPLE = 12
55
- DEFAULT_HOPS = 2
56
- STALE_DAYS = 30.0 # 多久没被访问算「陈旧」
57
- UNVERIFIED_DAYS = 14.0 # knowledge 层多久没验证算「长期未验证」
58
- LOW_CONF = 0.35 # 低于此置信算「低置信」
59
- ORPHAN_IMPORTANCE = 0.30 # 孤立噪音的重要度上限
60
- MAX_ASSOC_SCAN = 600 # 词法近邻比对上限(全库扫描保护)
61
- MAX_BINS_REPORT = 4
62
- MAX_OFFSET = 0.15 # 校准偏置上限(一次最多挪这么多)
63
- SELF_LAYERS = ("self", "anchor")
64
-
65
- STRATA = ("stale", "low_conf", "disputed", "unverified", "orphan", "hot", "random")
66
- STRATUM_WEIGHTS = {"stale": 0.20, "low_conf": 0.20, "disputed": 0.15,
67
- "unverified": 0.15, "orphan": 0.10, "hot": 0.10,
68
- "random": 0.10}
69
-
70
- # 污染类型 → (严重度, 处置动作)
71
- CONTAMINATION = {
72
- "contradiction": ("high", ("weaken", "demote")),
73
- "expired": ("high", ("weaken", "demote")),
74
- # 未生效(双时间轴另一侧,2026-09-19):**不是错误**——只提示「此刻不适用」,
75
- # 故 severity=info、处置仅 hint(不 weaken / 不 demote / 不删)。
76
- "not_yet": ("info", ("hint",)),
77
- "duplicate": ("medium", ("hint",)),
78
- "orphan_noise": ("low", ("weaken", "demote")),
79
- "unverified": ("info", ("hint",)),
80
- }
81
- SEVERITY_ORDER = {"high": 3, "medium": 2, "low": 1, "info": 0}
82
-
83
-
84
- # --------------------------------------------------------------------------
85
- # 工具
86
- # --------------------------------------------------------------------------
87
-
88
- # 生效条件:当 cg 的 index 为真且其 "nodes" 为真时返回该值,否则(cg 无 index、index 为假值、缺 "nodes" 或 "nodes" 为假值)返回 {};
89
- def _nodes(cg) -> dict:
90
- return (getattr(cg, "index", None) or {}).get("nodes") or {}
91
-
92
-
93
- # 生效条件:cg 与 nid 传入后,若 cache 非 None 且 nid 已在 cache 中则直接返回 cache[nid](即使其值为假值);否则尝试 cg.get(nid),异常或返回假值时按空节点处理,再取其中的 "frontmatter" 真值,若为假值则用 {};cache 非 None 时把结果写入 cache[nid] 后返回;
94
- def _fm_of(cg, nid, cache=None) -> dict:
95
- """取节点 frontmatter(带可选缓存)。不可读(缺密钥 / 已删)→ {}。"""
96
- if cache is not None and nid in cache:
97
- return cache[nid]
98
- node = None
99
- try:
100
- node = cg.get(nid)
101
- except Exception:
102
- node = None
103
- fm = (node or {}).get("frontmatter") or {}
104
- if cache is not None:
105
- cache[nid] = fm
106
- return fm
107
-
108
-
109
- # 生效条件:cg.access_counts() 调用成功时返回其结果(访问次数, 最后访问时间);调用抛出任何 Exception 时返回 ({}, {});
110
- def _access(cg):
111
- """(访问次数, 最后访问时间):读访问日志,未 compact 的也算。"""
112
- try:
113
- return cg.access_counts()
114
- except Exception:
115
- return {}, {}
116
-
117
-
118
- # 生效条件:`from . import chain` 成功且 chain.adjacency(cg) 正常返回时返回该 dict,导入或调用抛任何异常时返回 {}。
119
- def _adjacency(cg) -> dict:
120
- try:
121
- from . import chain
122
- return chain.adjacency(cg)
123
- except Exception:
124
- return {}
125
-
126
-
127
- # 生效条件:float(ts or 0) 抛 TypeError/ValueError 时返回 0.0,转换后为 0(含 ts 为 0/空串/None 等假值)时返回 0.0,否则返回 (now - ts)/86400.0。
128
- def _days(ts, now) -> float:
129
- try:
130
- ts = float(ts or 0)
131
- except (TypeError, ValueError):
132
- return 0.0
133
- return (now - ts) / 86400.0 if ts else 0.0
134
-
135
-
136
- # 生效条件:调用即返回带 t 与 op 的 rec;写 append_jsonl(os.path.join(cg.root, SCRUB_LOG), rec) 抛任何异常都被吞掉,不影响返回值。
137
- def _log(cg, op: str, **rec) -> dict:
138
- rec = dict(rec, t=time.time(), op=op)
139
- try:
140
- append_jsonl(os.path.join(cg.root, SCRUB_LOG), rec)
141
- except Exception:
142
- pass
143
- return rec
144
-
145
-
146
- # 生效条件:仅当 read_jsonl(cg.root 下 SCRUB_LOG) 的记录 op=="decontaminate" 且 ok 为真时,把 (rec.get("node_id"), rec.get("kind")) 收进返回集合;无此类记录返回空集。
147
- def _handled(cg) -> set:
148
- """已处置过的 (node_id, kind):保证去污染幂等(审计即状态)。"""
149
- out = set()
150
- for rec in list(read_jsonl(os.path.join(cg.root, SCRUB_LOG))):
151
- if rec.get("op") == "decontaminate" and rec.get("ok"):
152
- out.add((rec.get("node_id"), rec.get("kind")))
153
- return out
154
-
155
-
156
- # --------------------------------------------------------------------------
157
- # ① 记忆抽查
158
- # --------------------------------------------------------------------------
159
-
160
- # 生效条件:当 cg 的节点/访问/邻接数据可取时,对每个「layer 不在 SELF_LAYERS」的节点(源码仅以 `if layer in SELF_LAYERS: continue` 排除,未要求 layer 非空)按 now、stale_days、unverified_days 判定入池——last 访问时间距今≥stale_days 时入 stale,last 为假值且 created_at 距今≥stale_days 时入 stale,访问计数 acc≥3 入 hot,邻接度 deg==0 且 layer=="contextual" 入 orphan,evidence_count≤0 且 layer=="knowledge" 且 age_d≥unverified_days 入 unverified,evidence_count>0 且 reads<int(max_reads) 且 _fm_of 返回非空前台时按 negative_evidence>0 入 disputed、按 confidence<LOW_CONF(confidence 缺键回落 0.6)入 low_conf(max_reads 为 0 时 int(max_reads)=0,reads<0 恒假,故 disputed/low_conf 不产生),每个节点无条件入 random 池,最后按各池 key 排序返回 pools(片段仅见排序段,未展示抽样阶段);多分支无法一句话覆盖全部分支。
161
- def _pool_candidates(cg, *, now, stale_days, unverified_days, max_reads=200):
162
- """构造各层候选池(不读文件的部分先用索引 + 访问日志)。"""
163
- nodes = _nodes(cg)
164
- counts, last = _access(cg)
165
- adj = _adjacency(cg)
166
- indeg = {}
167
- for _src, outs in adj.items():
168
- for tgt, _e in outs:
169
- indeg[tgt] = indeg.get(tgt, 0) + 1
170
-
171
- pools = {s: [] for s in STRATA}
172
- cache, reads = {}, 0
173
- for nid, e in nodes.items():
174
- layer = str(e.get("layer") or "")
175
- if layer in SELF_LAYERS:
176
- continue # 自我认知不抽查、不去污染
177
- created = float(e.get("created_at") or 0)
178
- age_d = _days(created, now)
179
- la = float(last.get(nid) or 0)
180
- acc = int(counts.get(nid) or 0)
181
- ev = int(e.get("evidence_count") or 0)
182
- deg = len(adj.get(nid) or []) + indeg.get(nid, 0)
183
-
184
- if la:
185
- if _days(la, now) >= stale_days:
186
- pools["stale"].append((nid, f"距上次访问 {_days(la, now):.0f} 天"))
187
- elif age_d >= stale_days:
188
- pools["stale"].append((nid, f"从未访问且已存在 {age_d:.0f} 天"))
189
-
190
- if acc >= 3:
191
- pools["hot"].append((nid, f"高频访问 {acc} 次"))
192
-
193
- if deg == 0 and layer == "contextual":
194
- pools["orphan"].append((nid, "无任何关系边(孤立)"))
195
-
196
- if ev <= 0 and layer == "knowledge" and age_d >= unverified_days:
197
- pools["unverified"].append((nid, f"knowledge 层 {age_d:.0f} 天零验证"))
198
-
199
- if ev > 0 and reads < int(max_reads):
200
- fm = _fm_of(cg, nid, cache)
201
- if fm:
202
- reads += 1
203
- neg = int(fm.get("negative_evidence") or 0)
204
- conf = float(fm.get("confidence", 0.6))
205
- if neg > 0:
206
- pools["disputed"].append((nid, f"有 {neg} 条反例"))
207
- if conf < LOW_CONF:
208
- pools["low_conf"].append((nid, f"置信 {conf:.2f}"))
209
-
210
- pools["random"].append((nid, "随机基线"))
211
-
212
- # 池内排序:风险优先(可复现),抽样时再按 seed 洗牌
213
- pools["stale"].sort(key=lambda x: (nodes.get(x[0], {}).get("created_at") or 0))
214
- pools["hot"].sort(key=lambda x: -(int(counts.get(x[0]) or 0)))
215
- pools["unverified"].sort(
216
- key=lambda x: (nodes.get(x[0], {}).get("created_at") or 0))
217
- pools["orphan"].sort(
218
- key=lambda x: float(nodes.get(x[0], {}).get("importance") or 0.5))
219
- pools["disputed"].sort(key=lambda x: -len(x[1]))
220
- pools["low_conf"].sort(key=lambda x: x[1])
221
- return pools
222
-
223
-
224
- # 生效条件:strategy=="random" 时直接返回 {"random": int(n)};strategy=="risk" 时按剔除 hot/random 后的 STRATUM_WEIGHTS 权重分配;其它策略名走全权重分配,两者都把 int(n) 余量补进已存在的 random 键否则补进 stale。
225
- def _quota(n: int, strategy: str) -> dict:
226
- if strategy == "random":
227
- return {"random": int(n)}
228
- weights = dict(STRATUM_WEIGHTS)
229
- if strategy == "risk":
230
- weights.pop("hot", None)
231
- weights.pop("random", None)
232
- total = sum(weights.values()) or 1.0
233
- out, used = {}, 0
234
- for s in STRATA:
235
- if s in weights:
236
- k = int(round(int(n) * weights[s] / total))
237
- out[s] = k
238
- used += k
239
- rest = int(n) - used
240
- if rest:
241
- key = "random" if "random" in out else "stale"
242
- out[key] = out.get(key, 0) + rest
243
- return out
244
-
245
-
246
- # 生效条件:k<=0 或 pool 为假值(空池)时返回 [],否则取池前 k*3 项后用 random.Random(f"{seed}:{stratum}") 稳定洗牌并返回前 k 项(池长不足 k*3 时对全池洗牌)。
247
- def _pick(pool, k: int, seed, stratum: str):
248
- """从池中取 k 个:风险最高的 3k 个入池,再按 seed 稳定洗牌。"""
249
- if k <= 0 or not pool:
250
- return []
251
- cand = list(pool)
252
- if len(cand) > k * 3:
253
- cand = cand[:k * 3]
254
- random.Random(f"{seed}:{stratum}").shuffle(cand)
255
- return cand[:k]
256
-
257
-
258
- # 生效条件:strategy 经 str(strategy or "stratified").lower() 后属于 stratified/risk/random 时返回带分层标签的样本(seed 为 None 时取 0,n 为 0 时 picked 为空),否则抛 ValueError。
259
- def sample(cg, n: int = DEFAULT_SAMPLE, *, strategy: str = "stratified",
260
- seed=None, stale_days: float = STALE_DAYS,
261
- unverified_days: float = UNVERIFIED_DAYS,
262
- max_reads: int = 200) -> dict:
263
- """分层抽查记忆,返回带分层标签的样本。
264
-
265
- strategy:`stratified`(默认,风险层 + 对照组)· `risk`(只查风险层)
266
- · `random`(纯随机基线)。seed 固定 → 同样本可复现。
267
- """
268
- strategy = str(strategy or "stratified").lower()
269
- if strategy not in ("stratified", "risk", "random"):
270
- raise ValueError(f"未知抽样策略:{strategy}(stratified|risk|random)")
271
- now = time.time()
272
- seed = 0 if seed is None else seed
273
- pools = _pool_candidates(cg, now=now, stale_days=stale_days,
274
- unverified_days=unverified_days,
275
- max_reads=max_reads)
276
- quota = _quota(n, strategy)
277
- picked, seen = [], set()
278
- for s in STRATA:
279
- for nid, reason in _pick(pools.get(s), quota.get(s, 0), seed, s):
280
- if nid in seen:
281
- continue
282
- seen.add(nid)
283
- picked.append({"node_id": nid, "stratum": s, "reason": reason})
284
- # 风险层不够时用随机池补足(先剔除已选,避免跳过造成不满额)
285
- if len(picked) < int(n):
286
- rest = [x for x in pools["random"] if x[0] not in seen]
287
- for nid, reason in _pick(rest, int(n) - len(picked), seed,
288
- "random-fill"):
289
- seen.add(nid)
290
- picked.append({"node_id": nid, "stratum": "random",
291
- "reason": reason})
292
- return {"ok": True, "strategy": strategy, "seed": seed, "n": len(picked),
293
- "strata": {s: len(pools.get(s) or []) for s in STRATA},
294
- "sample": picked, "t": now}
295
-
296
-
297
- # --------------------------------------------------------------------------
298
- # ② 联想
299
- # --------------------------------------------------------------------------
300
-
301
- # 生效条件:a 或 b 为假值(空集)时返回 0.0,否则返回 len(a & b)/len(a | b)。
302
- def _jaccard(a: set, b: set) -> float:
303
- if not a or not b:
304
- return 0.0
305
- return len(a & b) / len(a | b)
306
-
307
-
308
- # 生效条件:对 node_id 汇总关系链(chain.walk 出边与入边,max_depth=int(hops))、子图层级(subgraph.expand 命中项与父索引逐级上溯 int(hops) 层)以及 lexical 为真时的词法近邻(只在 _nodes 前 int(max_scan) 个节点内比 bigram、sim≥min_sim 者),返回按 weight 降序的 items[:int(limit)]。
309
- def associate(cg, node_id: str, *, hops: int = DEFAULT_HOPS, limit: int = 30,
310
- lexical: bool = True, min_sim: float = 0.25,
311
- max_scan: int = MAX_ASSOC_SCAN) -> dict:
312
- """三路邻域联想:关系链 + 子图层级 + 词法近邻。
313
-
314
- 返回按权重降序的 `related`;每条带 `via`(哪一路发现)与 `weight`,
315
- 便于审计「为什么把这两个节点算作同族」。
316
- """
317
- related = {}
318
-
319
- # 生效条件:nid 为真且 nid != node_id 时,构造 rec={'node_id':nid,'via':via,'weight':round(float(weight),4),'depth':int(depth)} 并并入 extra;仅当 related 中 nid 不存在或 rec['weight'] 大于已有 weight 时更新 related[nid];nid 为假或等于 node_id 时直接返回;
320
- def put(nid, via, weight, depth=1, **extra):
321
- if not nid or nid == node_id:
322
- return
323
- rec = {"node_id": nid, "via": via, "weight": round(float(weight), 4),
324
- "depth": int(depth)}
325
- rec.update(extra)
326
- cur = related.get(nid)
327
- if cur is None or rec["weight"] > cur["weight"]:
328
- related[nid] = rec
329
-
330
- # 联想是双向的:同族既可能是「我指向的」也可能是「指向我的」,
331
- # 故关系链同时沿出边与入边展开(`chain_t --causal--> a` 也要能从 a 找到)。
332
- try:
333
- from . import chain
334
- for direction in ("out", "in"):
335
- for c in chain.walk(cg, node_id,
336
- relation_types=chain.CHAIN_TYPES_DEFAULT,
337
- max_depth=int(hops), max_chains=200,
338
- include_hierarchy=True, direction=direction):
339
- for i, h in enumerate(c.get("hops") or [], 1):
340
- put(h["to"], "chain",
341
- float(h.get("weight") or 1.0) * (0.8 ** (i - 1)),
342
- depth=i, relation_type=h.get("relation_type"),
343
- condition=h.get("condition") or "")
344
- except Exception:
345
- pass
346
-
347
- # 子图层级:向下展开子树 + 向上追溯祖先(父边唯一,逐级上溯)。
348
- try:
349
- from . import subgraph
350
- ex = subgraph.expand(cg, node_id, max_depth=int(hops))
351
- for nid, path in (ex.get("paths") or {}).items():
352
- d = str(path).count("/")
353
- put(nid, "subgraph", 1.0 / (1 + d), depth=d)
354
- pmap = subgraph.parents_index(cg)
355
- cur, d = node_id, 0
356
- while d < int(hops):
357
- ps = pmap.get(cur) or []
358
- if not ps:
359
- break
360
- d += 1
361
- for pid in ps:
362
- put(pid, "subgraph", 1.0 / (1 + d), depth=d)
363
- cur = ps[0]
364
- except Exception:
365
- pass
366
-
367
- if lexical:
368
- base = None
369
- try:
370
- base = cg.get(node_id)
371
- except Exception:
372
- base = None
373
- bg = bigrams((base or {}).get("content") or "")
374
- if bg:
375
- cands = []
376
- for i, nid in enumerate(_nodes(cg)):
377
- if i >= int(max_scan):
378
- break
379
- if nid == node_id or nid in related:
380
- continue
381
- try:
382
- node = cg.get(nid)
383
- except Exception:
384
- node = None
385
- b2 = bigrams((node or {}).get("content") or "")
386
- sim = _jaccard(bg, b2)
387
- if sim >= min_sim:
388
- cands.append((sim, nid))
389
- cands.sort(key=lambda x: (-x[0], x[1]))
390
- for sim, nid in cands[:int(limit)]:
391
- put(nid, "lexical", sim, depth=1, similarity=round(sim, 4))
392
-
393
- items = sorted(related.values(), key=lambda r: (-r["weight"], r["node_id"]))
394
- by_via = {}
395
- for r in items:
396
- by_via[r["via"]] = by_via.get(r["via"], 0) + 1
397
- return {"ok": True, "node_id": node_id, "n": len(items),
398
- "related": items[:int(limit)], "by_via": by_via}
399
-
400
-
401
- # --------------------------------------------------------------------------
402
- # ③ 去污染:确定性判据
403
- # --------------------------------------------------------------------------
404
-
405
- # 已结束键族(2026-09-19 阶段一:补规范名 `effective_until` 与冗余时刻 `expired_at`)。
406
- # 纪律:`believed_at`(信念时间)**两族都不入**——它既不是「已结束」也不是「尚未开始」,
407
- # 而是「体系何时确认此条」的取代/审核锚。键族真源见 md_cg/trust.py(FROM_ALIASES/UNTIL_ALIASES),
408
- # 两侧新增键须同步(交叉守卫 test_validity_filter)。
409
- _EXPIRY_KEYS = ("effective_until", "valid_until", "expires_at", "expire_at",
410
- "expiry", "deadline", "expired_at")
411
- _SKIP_KEYS = ("功能名", "执行", "条件", "来源", "标签", "状态", "备注", "标题",
412
- "描述", "name", "id", "title", "layer", "tags", "说明")
413
- _NEG_WORDS = ("禁止", "不得", "不要", "不能", "切勿", "避免", "不应", "不可")
414
- _POS_WORDS = ("应当", "建议", "必须", "需要", "可以", "允许", "推荐", "宜")
415
-
416
-
417
- # 生效条件:v 为 bool 返回 None;v 为 int/float 时仅 f>1e8 返回 f 否则 None;str(v or "").strip() 为空返回 None;否则按 "%Y-%m-%dT%H:%M:%S"/"%Y-%m-%d %H:%M:%S"/"%Y-%m-%d" 依次取前 19/19/10 字符尝试解析,全失败后再试 float(s),>1e8 返回否则 None,float 亦失败返回 None。
418
- def _to_ts(v):
419
- """宽松时间解析:秒级时间戳 / ISO / `YYYY-MM-DD`。无法识别 → None。"""
420
- if isinstance(v, bool):
421
- return None
422
- if isinstance(v, (int, float)):
423
- f = float(v)
424
- return f if f > 1e8 else None
425
- s = str(v or "").strip()
426
- if not s:
427
- return None
428
- for fmt, n in (("%Y-%m-%dT%H:%M:%S", 19), ("%Y-%m-%d %H:%M:%S", 19),
429
- ("%Y-%m-%d", 10)):
430
- try:
431
- return time.mktime(time.strptime(s[:n], fmt))
432
- except ValueError:
433
- continue
434
- try:
435
- f = float(s)
436
- return f if f > 1e8 else None
437
- except ValueError:
438
- return None
439
-
440
-
441
- # 生效条件:对 (content or "").splitlines() 的每行去 # 后,先试中文全角 ":"、无则试 ":",切出的键长度在 2–12、值非空且键不在 _SKIP_KEYS 时以 setdefault 记录(每键只留首次出现),无合格行返回 {}。
442
- def _kv_pairs(content) -> dict:
443
- """抽 `键:值` 对(中文/英文冒号),跳过结构字段。"""
444
- out = {}
445
- for raw in (content or "").splitlines():
446
- line = raw.strip().lstrip("#").strip()
447
- if not line:
448
- continue
449
- for sep in (":", ":"):
450
- if sep in line:
451
- k, _, v = line.partition(sep)
452
- k, v = k.strip(), v.strip()
453
- if 2 <= len(k) <= 12 and v and k not in _SKIP_KEYS:
454
- out.setdefault(k, v)
455
- break
456
- return out
457
-
458
-
459
- # 生效条件:c 取 content or "",含 _NEG_WORDS 任一返回 -1 分量、含 _POS_WORDS 任一返回 +1 分量,结果为两者之和(都不含时为 0)。
460
- def _polarity(content) -> int:
461
- c = content or ""
462
- return (-1 if any(w in c for w in _NEG_WORDS) else 0) + \
463
- (1 if any(w in c for w in _POS_WORDS) else 0)
464
-
465
-
466
- # 生效条件:按 _EXPIRY_KEYS 顺序遍历 fm,返回首个满足「键在 fm 中且 _to_ts 非 None 且 ts<now」的 (k, fm.get(k));该键解析为 None 或 ts≥now 时继续检查后续键,全不满足返回 None。
467
- def _expired(fm, now):
468
- for k in _EXPIRY_KEYS:
469
- if k in fm:
470
- ts = _to_ts(fm.get(k))
471
- if ts is not None and ts < now:
472
- return k, fm.get(k)
473
- return None
474
-
475
-
476
- # 未生效键(双时间轴的起点,2026-09-19 · 真源 md_cg/trust.py)。
477
- # 纪律一:`valid_from` **绝不并入 `_EXPIRY_KEYS`**——两者语义相反(「尚未开始」vs
478
- # 「已经结束」),并入会让「未来才生效」被误判为「已失效」并触发 weaken/demote。
479
- # 纪律二:`believed_at`(信念时间)同上,**两族都不入**——第三类语义(体系何时确认)。
480
- _NOT_YET_KEYS = ("valid_from", "valid_since", "effective_from", "starts_at")
481
-
482
-
483
- # 生效条件:按 _NOT_YET_KEYS 顺序遍历 fm,返回首个满足「键在 fm 中且 _to_ts 非 None 且 ts>now」的 (k, fm.get(k));全不满足返回 None。
484
- def _not_yet(fm, now):
485
- """宽松判定「尚未生效」(与 `_expired` 同口径,方向相反)。"""
486
- for k in _NOT_YET_KEYS:
487
- if k in fm:
488
- ts = _to_ts(fm.get(k))
489
- if ts is not None and ts > now:
490
- return k, fm.get(k)
491
- return None
492
-
493
-
494
- # 生效条件:对 content 取 kv_pairs、bigrams 和 polarity,遍历 related(若 related 为假值则视为空)的前 int(max_compare) 个 r,以 r["node_id"] 调 cg.get;若某 oid 节点可读非空,先在其 content 与 content 的共同键中找到值不同者并返回 {'with':oid,'via':r.get('via'),'why':'同键不同值:...'};否则若极性乘积 <0 且双方 bigram 非空,且共享 bigram 数 >=3 且 ratio>=0.15,返回 {'with':oid,'via':r.get('via'),'why':'极性相反且共享内容:...'};全部遍历完无命中则返回 None;
495
- def _contradiction(cg, content, related, max_compare=10):
496
- """与同族节点比对:同键不同值 / 极性相反且共享 bigram。"""
497
- kv_a = _kv_pairs(content)
498
- bg_a, pol_a = bigrams(content or ""), _polarity(content)
499
- for r in (related or [])[:int(max_compare)]:
500
- oid = r["node_id"]
501
- try:
502
- onode = cg.get(oid)
503
- except Exception:
504
- onode = None
505
- if not onode:
506
- continue
507
- oc = onode.get("content") or ""
508
- kv_b = _kv_pairs(oc)
509
- for k in sorted(set(kv_a) & set(kv_b)):
510
- if kv_a[k] != kv_b[k]:
511
- return {"with": oid, "via": r.get("via"),
512
- "why": f"同键不同值:{k}={kv_a[k]} vs {kv_b[k]}"}
513
- bg_b = bigrams(oc)
514
- if pol_a * _polarity(oc) < 0 and bg_a and bg_b:
515
- shared = bg_a & bg_b
516
- ratio = len(shared) / max(1, min(len(bg_a), len(bg_b)))
517
- if len(shared) >= 3 and ratio >= 0.15:
518
- return {"with": oid, "via": r.get("via"),
519
- "why": "极性相反且共享内容:" + "、".join(
520
- sorted(shared)[:5])}
521
- return None
522
-
523
-
524
- # 生效条件:ids 在 node_ids 为 None 时取索引全部键、为 str 时取单元素列表、否则取 list(node_ids),limit 为真值时截断为前 int(limit) 个;跳过 layer 在 SELF_LAYERS 的节点和 cg.get 取不到正文的节点,对剩余每个节点按 min_severity 门限累加 expired/unverified/orphan_noise/duplicate/contradiction,返回 ok=无 high 且无 medium 的结果。
525
- def audit(cg, node_ids=None, *, hops: int = 1, min_severity: str = "info",
526
- limit=None, unverified_days: float = UNVERIFIED_DAYS,
527
- lexical: bool = True, min_sim: float = 0.15) -> dict:
528
- """对指定节点(默认全库)做污染体检。**只读**,不改动任何节点。"""
529
- nodes = _nodes(cg)
530
- ids = list(nodes.keys()) if node_ids is None else (
531
- [node_ids] if isinstance(node_ids, str) else list(node_ids))
532
- if limit:
533
- ids = ids[:int(limit)]
534
- counts, _last = _access(cg)
535
- adj = _adjacency(cg)
536
- indeg = {}
537
- for _src, outs in adj.items():
538
- for tgt, _e in outs:
539
- indeg[tgt] = indeg.get(tgt, 0) + 1
540
- now = time.time()
541
- issues, checked = [], 0
542
-
543
- for nid in ids:
544
- e = nodes.get(nid)
545
- if not e:
546
- continue
547
- layer = str(e.get("layer") or "")
548
- if layer in SELF_LAYERS:
549
- continue
550
- try:
551
- node = cg.get(nid)
552
- except Exception:
553
- node = None
554
- if not node:
555
- continue
556
- checked += 1
557
- fm = node.get("frontmatter") or {}
558
- content = node.get("content") or ""
559
- age_d = _days(fm.get("created_at") or e.get("created_at"), now)
560
-
561
- # 生效条件:kind 是 CONTAMINATION 的键(否则 KeyError)且 CONTAMINATION[kind] 取出的 sev 在 SEVERITY_ORDER 中的值(缺键按 0)不小于闭包变量 min_severity 在 SEVERITY_ORDER 中的值(缺键按 0)时,把 {node_id, layer, kind, severity, detail, fix} 用 **extra 覆盖更新后追加到闭包 issues;sev 的值更小则直接 return 不追加(该函数无返回值)。
562
- def add(kind, detail, **extra):
563
- sev, actions = CONTAMINATION[kind]
564
- if SEVERITY_ORDER.get(sev, 0) < SEVERITY_ORDER.get(min_severity, 0):
565
- return
566
- rec = {"node_id": nid, "layer": layer, "kind": kind,
567
- "severity": sev, "detail": detail, "fix": list(actions)}
568
- rec.update(extra)
569
- issues.append(rec)
570
-
571
- hit = _expired(fm, now)
572
- if hit:
573
- add("expired", f"时效已过:{hit[0]}={hit[1]}", evidence=hit[0])
574
-
575
- # 双时间轴另一侧(2026-09-19):尚未生效**不是错误**——只提示「此刻不适用」,
576
- # 处置由 CONTAMINATION["not_yet"] 定为 info 级 + 仅 hint(不 weaken / 不 demote)。
577
- ny = _not_yet(fm, now)
578
- if ny:
579
- add("not_yet", f"尚未生效:{ny[0]}={ny[1]}", evidence=ny[0])
580
-
581
- if (layer == "knowledge" and age_d >= unverified_days
582
- and int(e.get("evidence_count") or 0) <= 0):
583
- add("unverified", f"knowledge 层 {age_d:.0f} 天零验证")
584
-
585
- if ((len(adj.get(nid) or []) + indeg.get(nid, 0)) == 0
586
- and layer == "contextual"
587
- and int(counts.get(nid) or 0) == 0
588
- and float(fm.get("importance") or e.get("importance") or 0.5)
589
- < ORPHAN_IMPORTANCE):
590
- add("orphan_noise", "孤立且零访问、低重要(情境噪音)")
591
-
592
- try:
593
- red = forgetting.redundancy(cg, content, layer=layer, exclude=nid)
594
- except Exception:
595
- red = None
596
- if red and float(red.get("max") or 0) >= float(forgetting.DUP_MERGE):
597
- add("duplicate", f"冗余 {red['max']:.2f}(与 {red.get('with')})",
598
- duplicate_with=red.get("with"))
599
-
600
- rel = associate(cg, nid, hops=max(1, int(hops)), limit=10,
601
- lexical=lexical, min_sim=min_sim)["related"]
602
- bad = _contradiction(cg, content, rel)
603
- if bad:
604
- add("contradiction", bad["why"], conflict_with=bad.get("with"))
605
-
606
- by_kind = {}
607
- for i in issues:
608
- by_kind[i["kind"]] = by_kind.get(i["kind"], 0) + 1
609
- high = sum(1 for i in issues if i["severity"] == "high")
610
- med = sum(1 for i in issues if i["severity"] == "medium")
611
- return {"ok": not (high or med), "n_checked": checked,
612
- "n_issues": len(issues), "issues": issues, "by_kind": by_kind,
613
- "stats": {"checked": checked, "high": high, "medium": med,
614
- "low": sum(1 for i in issues if i["severity"] == "low")},
615
- "t": now}
616
-
617
-
618
- # 生效条件:循环 max(1, int(times)) 次调用 cg.verify(nid, "scrub:"+kind+":"+str(detail)[:120], "weakened"),返回最后一次调用结果(times≤0 时按 1 次执行)。
619
- def _weaken(cg, nid, kind, detail, times=1):
620
- res = None
621
- for _ in range(max(1, int(times))):
622
- res = cg.verify(nid, f"scrub:{kind}:{str(detail)[:120]}", "weakened")
623
- return res
624
-
625
-
626
- # 生效条件:kinds 为真且 issue 的 kind 不在其中则跳过;kind 为 duplicate/unverified 记 hint,已出现在 _handled 记 skip,protect.is_protected 为真且 override 为假记 skip_protected,dry_run 为真记 planned;仅 dry_run 为假时对余下 issue 执行 _weaken,并在 demote 为真、severity 为 high 或 low 且当前层非 contextual 时降级到 contextual,返回计数与 actions。
627
- def decontaminate(cg, node_ids=None, *, kinds=None, dry_run: bool = True,
628
- min_severity: str = "medium", hops: int = 1, actor=None,
629
- override: bool = False, weaken_times: int = 1,
630
- demote: bool = True) -> dict:
631
- """按体检结果处置污染。**默认 dry_run**;实修只做可逆动作。
632
-
633
- 动作:`weaken`(verify weakened → 反例+1、置信-0.15、跌破 0.2 自动降级)、
634
- `demote`(降级到情境层)、`hint`(只建议,不改)。保护节点跳过;永不删除。
635
- """
636
- rep = audit(cg, node_ids, hops=hops, min_severity=min_severity)
637
- handled = _handled(cg)
638
- actions = []
639
- n_applied = n_prot = n_done = n_hint = 0
640
-
641
- for issue in rep["issues"]:
642
- nid, kind = issue["node_id"], issue["kind"]
643
- if kinds and kind not in kinds:
644
- continue
645
- if kind in ("duplicate", "unverified"):
646
- n_hint += 1
647
- actions.append({"node_id": nid, "kind": kind, "action": "hint",
648
- "detail": issue["detail"], "applied": False})
649
- continue
650
- if (nid, kind) in handled:
651
- n_done += 1
652
- actions.append({"node_id": nid, "kind": kind, "action": "skip",
653
- "detail": "已处置过(幂等)", "applied": False})
654
- continue
655
- prot, why = protect.is_protected(cg, nid)
656
- if prot and not override:
657
- n_prot += 1
658
- actions.append({"node_id": nid, "kind": kind,
659
- "action": "skip_protected", "detail": why,
660
- "applied": False})
661
- continue
662
- if dry_run:
663
- actions.append({"node_id": nid, "kind": kind,
664
- "action": "planned",
665
- "planned": list(CONTAMINATION[kind][1]),
666
- "detail": issue["detail"], "applied": False})
667
- continue
668
-
669
- done = []
670
- try:
671
- _weaken(cg, nid, kind, issue["detail"], weaken_times)
672
- done.append("weaken")
673
- except Exception as exc: # noqa: BLE001
674
- done.append(f"weaken_failed:{type(exc).__name__}")
675
- if demote and issue["severity"] in ("high", "low"):
676
- try:
677
- fm = _fm_of(cg, nid, None)
678
- if str(fm.get("layer") or "") != "contextual":
679
- cg._move_layer(nid, "contextual", reason=f"scrub:{kind}")
680
- # 层降级 = 生命周期降级(②):状态同批推进(负路由,失败不抛)
681
- _st = cg.set_state(nid, "demoted", reason=f"scrub:{kind}",
682
- actor=actor or "scrub")
683
- done.append("demote" if _st.get("ok") else
684
- f"demote_state:{_st.get('error')}")
685
- except Exception as exc: # noqa: BLE001
686
- done.append(f"demote_failed:{type(exc).__name__}")
687
- n_applied += 1
688
- _log(cg, "decontaminate", node_id=nid, kind=kind, ok=True,
689
- actions=done, severity=issue["severity"], actor=actor)
690
- actions.append({"node_id": nid, "kind": kind, "action": "applied",
691
- "done": done, "detail": issue["detail"],
692
- "applied": True})
693
-
694
- if not dry_run:
695
- _log(cg, "decontaminate_batch", ok=True, applied=n_applied,
696
- skipped_protected=n_prot, skipped_done=n_done, hints=n_hint,
697
- actor=actor)
698
- return {"ok": True, "dry_run": dry_run, "n_issues": rep["n_issues"],
699
- "applied": n_applied, "skipped_protected": n_prot,
700
- "skipped_done": n_done, "hints": n_hint, "actions": actions,
701
- "audit": rep, "t": time.time()}
702
-
703
-
704
- # --------------------------------------------------------------------------
705
- # ④ 校准偏差
706
- # --------------------------------------------------------------------------
707
-
708
- # 生效条件:对 cg 中每个「layer 不在 SELF_LAYERS」且 evidence_count≥int(min_evidence)(min_evidence=0 时该比较恒假而不早退)的节点,若 protect.is_protected 为假或 override 为真,且 cg.get(nid) 未抛异常并返回真值节点,则取 fm.get("confidence", 0.6)(缺键才回落 0.6,键存在为 None/假值不回落)为 old,算出 round(max(0.0, min(0.99, old+float(offset))),4),与 old 差<1e-9 时跳过,否则写 fm["confidence"] 与 fm["calibration"] 并调用 cg._write_node 成功时 adjusted+1(写回异常被吞掉不计数),返回 (adjusted, skipped),其中 skipped 只累计「layer 属 SELF_LAYERS」或被 protect 拦下且非 override 的节点。
709
- def _apply_offset(cg, offset, *, override=False, min_evidence=1):
710
- nodes = _nodes(cg)
711
- adjusted = skipped = 0
712
- for nid, e in nodes.items():
713
- if str(e.get("layer") or "") in SELF_LAYERS:
714
- skipped += 1
715
- continue
716
- if int(e.get("evidence_count") or 0) < int(min_evidence):
717
- continue
718
- prot, _why = protect.is_protected(cg, nid)
719
- if prot and not override:
720
- skipped += 1
721
- continue
722
- try:
723
- node = cg.get(nid)
724
- except Exception:
725
- node = None
726
- if not node:
727
- continue
728
- fm = node.get("frontmatter") or {}
729
- old = float(fm.get("confidence", 0.6))
730
- new = round(max(0.0, min(0.99, old + float(offset))), 4)
731
- if abs(new - old) < 1e-9:
732
- continue
733
- fm["confidence"] = new
734
- fm["calibration"] = {"t": time.time(), "offset": round(float(offset), 4),
735
- "from": old, "to": new}
736
- try:
737
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm,
738
- node.get("content") or "")
739
- adjusted += 1
740
- except Exception: # noqa: BLE001
741
- pass
742
- return adjusted, skipped
743
-
744
-
745
- # 生效条件:cg 和 apply/override/actor/max_offset/min_evidence 传入后,若 metacognition.calibration(cg) 返回 ok 假,则返回 {'ok':False,'reason':cal.get('reason') or 'insufficient_data',...};若 ok 真,则用 gap=float(cal.get('gap') or 0.0) 和 max_offset 计算 offset=round(max(-max_offset,min(max_offset,-gap)),4),对 cal.get('bins') or [] 中 accuracy 非 None 的 bin 生成 bins_bias 并排序;仅当 apply 为真且 abs(offset)>1e-9 时调用 _apply_offset(cg,offset,override=override,min_evidence=min_evidence) 并写日志,最后返回 ok True 及 verdict/gap/建议 offset 等字段;
746
- def calibrate(cg, *, apply: bool = False, override: bool = False, actor=None,
747
- max_offset: float = MAX_OFFSET, min_evidence: int = 1) -> dict:
748
- """校准偏差:自报置信 vs 实测正确率 → 偏置建议(`apply=True` 才写回)。"""
749
- from . import metacognition
750
- cal = metacognition.calibration(cg)
751
- if not cal.get("ok"):
752
- return {"ok": False, "reason": cal.get("reason") or "insufficient_data",
753
- "calibration": cal, "applied": False, "n_adjusted": 0,
754
- "hint": "先 verify() 积累证据,校准才有样本"}
755
- gap = float(cal.get("gap") or 0.0)
756
- offset = round(max(-float(max_offset), min(float(max_offset), -gap)), 4)
757
- bins_bias = []
758
- for b in cal.get("bins") or []:
759
- if b.get("accuracy") is None:
760
- continue
761
- bins_bias.append({"bin": b["bin"], "n_nodes": b["n_nodes"],
762
- "confidence": b["confidence"],
763
- "accuracy": b["accuracy"],
764
- "bias": round(float(b["confidence"])
765
- - float(b["accuracy"]), 4)})
766
- bins_bias.sort(key=lambda x: -abs(x["bias"]))
767
- adjusted = skipped = 0
768
- if apply and abs(offset) > 1e-9:
769
- adjusted, skipped = _apply_offset(cg, offset, override=override,
770
- min_evidence=min_evidence)
771
- _log(cg, "calibrate", ok=True, offset=offset, gap=gap,
772
- adjusted=adjusted, actor=actor)
773
- blind = 0
774
- try:
775
- blind = int(metacognition.blindspots(cg, limit=1)
776
- .get("unresolved_count") or 0)
777
- except Exception:
778
- pass
779
- return {"ok": True, "verdict": cal.get("verdict"), "gap": gap,
780
- "expected_accuracy": cal.get("expected_accuracy"),
781
- "actual_accuracy": cal.get("actual_accuracy"),
782
- "ece": cal.get("ece"), "n_nodes": cal.get("n_nodes"),
783
- "suggested_offset": offset, "applied": bool(apply),
784
- "n_adjusted": adjusted, "skipped_protected": skipped,
785
- "bins_bias": bins_bias[:MAX_BINS_REPORT], "blindspots": blind,
786
- "hint": ("整体偏过度自信" if gap > 0.05 else
787
- "整体偏保守" if gap < -0.05 else "校准良好")}
788
-
789
-
790
- # --------------------------------------------------------------------------
791
- # 一轮完整自净 + 审计
792
- # --------------------------------------------------------------------------
793
-
794
- # 生效条件:cg 与 n/seed/dry_run/hops/strategy/apply_calibration/actor 传入后,按 n 与 strategy、seed 调 sample;对 sample 结果前 8 个 node_id 按 hops 调 associate;按 dry_run/hops/actor 调 decontaminate;按 apply_calibration/actor 调 calibrate;若 decontaminate 的 audit.issues 中存在 severity 为 high 或 medium 的项则 out.ok 为 False,否则为 True,并返回含 sample/associate/audit/decontaminate/calibration/t 的 out;
795
- def sweep(cg, *, n: int = DEFAULT_SAMPLE, seed=None, dry_run: bool = True,
796
- hops: int = DEFAULT_HOPS, strategy: str = "stratified",
797
- apply_calibration: bool = False, actor=None) -> dict:
798
- """一轮自净闭环:抽查 → 联想 → 体检 → 去污染 → 校准偏差。"""
799
- smp = sample(cg, n, strategy=strategy, seed=seed)
800
- ids = [s["node_id"] for s in smp["sample"]]
801
- assoc = {}
802
- for nid in ids[:8]:
803
- assoc[nid] = [r["node_id"] for r in associate(
804
- cg, nid, hops=hops, lexical=False, limit=5)["related"]]
805
- dec = decontaminate(cg, ids, dry_run=dry_run, hops=hops, actor=actor)
806
- rep = dec["audit"]
807
- cal = calibrate(cg, apply=apply_calibration, actor=actor)
808
- hi = [i for i in rep["issues"] if i["severity"] in ("high", "medium")]
809
- out = {"ok": not hi, "dry_run": dry_run, "n_high_medium": len(hi),
810
- "sample": smp, "associate": assoc, "audit": rep,
811
- "decontaminate": dec, "calibration": cal, "t": time.time()}
812
- _log(cg, "sweep", n_sample=smp["n"], n_issues=rep["n_issues"],
813
- n_high_medium=len(hi), applied=dec["applied"], dry_run=dry_run,
814
- calibration=(cal.get("verdict") if cal.get("ok")
815
- else cal.get("reason")))
816
- return out
817
-
818
-
819
- # 生效条件:返回 cg.root 下 SCRUB_LOG 的全部记录条数 n 与 recs[-int(limit):](limit=0 时切片为 recs[0:] 即返回全部记录)。
820
- def history(cg, limit: int = 100) -> dict:
821
- recs = list(read_jsonl(os.path.join(cg.root, SCRUB_LOG)))
822
- return {"n": len(recs), "records": recs[-int(limit):]}
823
-
824
-
825
- # 生效条件:读取 cg.root 下 SCRUB_LOG 的 JSONL 记录,过滤 op=="sweep" 得 sweeps、op=="decontaminate" 且 ok 为真得 decs;last 为 sweeps 最后一项或 None;返回 {'sweeps':len(sweeps),'decontaminated':len(decs),'last_sweep':last 的 t/n_issues/n_high_medium/applied/dry_run/calibration 或 None};
826
- def summary(cg) -> dict:
827
- """给 health_os / 自维持循环用的只读摘要。"""
828
- recs = list(read_jsonl(os.path.join(cg.root, SCRUB_LOG)))
829
- sweeps = [r for r in recs if r.get("op") == "sweep"]
830
- decs = [r for r in recs if r.get("op") == "decontaminate" and r.get("ok")]
831
- last = sweeps[-1] if sweeps else None
832
- return {"sweeps": len(sweeps), "decontaminated": len(decs),
833
- "last_sweep": ({"t": last.get("t"),
834
- "n_issues": last.get("n_issues"),
835
- "n_high_medium": last.get("n_high_medium"),
836
- "applied": last.get("applied"),
837
- "dry_run": last.get("dry_run"),
838
- "calibration": last.get("calibration")}
839
- if last else None)}
840
-
841
-
842
- # 生效条件:无入参调用即返回固定的 actions 列表、STRATA 列表、CONTAMINATION 映射(每项取 severity 与 actions)与 STALE_DAYS/UNVERIFIED_DAYS/LOW_CONF/ORPHAN_IMPORTANCE/MAX_OFFSET 阈值字典。
843
- def catalog() -> dict:
844
- return {"actions": ["sample", "associate", "audit", "decontaminate",
845
- "calibrate", "sweep", "history", "summary", "catalog"],
846
- "strata": list(STRATA),
847
- "contamination": {k: {"severity": v[0], "actions": list(v[1])}
848
- for k, v in CONTAMINATION.items()},
849
- "thresholds": {"stale_days": STALE_DAYS,
850
- "unverified_days": UNVERIFIED_DAYS,
851
- "low_conf": LOW_CONF,
852
- "orphan_importance": ORPHAN_IMPORTANCE,
1
+ # -*- coding: utf-8 -*-
2
+ """md_cg · 记忆自净(记忆 OS #5):抽查 / 联想 / 去污染 / 校准偏差
3
+
4
+ 自维持(`sustain.py`)解决「进程还活着」,本模块解决「记忆还干净」。
5
+ 四件事构成一个闭环,挂在常驻循环上周期性跑:
6
+
7
+ ① **记忆抽查(sample)**
8
+ 分层抽样而非随机抽样:风险层(陈旧 / 低置信 / 有反例 / 从未验证 / 孤立)
9
+ + 对照组(高频 / 随机基线)。对照组是刻意的——只查「看起来有问题」的节点
10
+ 会形成确认偏差,永远发现不了「看起来没问题其实有问题」的记忆。
11
+ `seed` 固定 → 同 seed 同样本(可复现、可审计,对齐本仓确定性白箱取向)。
12
+
13
+ ② **联想(associate)**
14
+ 从抽样节点出发三路邻域合并:
15
+ 关系链(`chain.walk`,带条件序列)· 子图层级(`subgraph.expand`)
16
+ · 词法近邻(bigram Jaccard)
17
+ 用途是找到「同族节点」——矛盾与重复只能在同族之间判定,孤立地看一个节点
18
+ 永远看不出它和谁冲突。
19
+
20
+ ③ **去污染(audit → decontaminate)**
21
+ 确定性判据(只读、可解释)识别五类污染:
22
+ `contradiction` 同族矛盾(同键不同值 / 极性相反且共享词)
23
+ `expired` 时效过期(`valid_until` / `expires_at` 已过)
24
+ `duplicate` 高冗余(`forgetting.redundancy` ≥ 合并阈值)
25
+ `orphan_noise` 孤立噪音(零访问 + 低重要 + 情境层)
26
+ `unverified` 长期未验证(knowledge 层且零证据)
27
+ 处置**只做可逆动作**且复用既有机制:
28
+ · `weaken` → `cg.verify(verdict="weakened")`:反例 +1、置信 -0.15、
29
+ 跌破 0.2 自动降级(验证机制天然留痕,不另造一套)
30
+ · `demote` → `cg._move_layer("contextual")`:降出可信层(走保护闸门)
31
+ · `hint` → 只给建议(合并 / 补证据需语义判断,不代劳)
32
+ 受保护节点(self/anchor/importance≥0.7/显式保护)**跳过**;**永不删除**。
33
+ 幂等:`_scrub.jsonl` 里已处理过的 (节点, 污染类型) 不重复动手。
34
+
35
+ ④ **校准偏差(calibrate)**
36
+ 用「自报置信 vs 实测正确率」(`metacognition.calibration`)算全局 gap 与
37
+ 分桶偏差,给出偏置建议;`apply=True` 才写回(只动有证据的节点,保护节点跳过)。
38
+
39
+ 所有动作留痕 `_scrub.jsonl`(payload-free:只记 id / 判据 / 动作,不记内容)。
40
+ 零第三方依赖。
41
+ """
42
+ from __future__ import annotations
43
+
44
+ import os
45
+ import random
46
+ import time
47
+
48
+ from . import forgetting, protect
49
+ from .fsutil import append_jsonl, read_jsonl
50
+ from .mdcg import bigrams
51
+
52
+ SCRUB_LOG = "_scrub.jsonl"
53
+
54
+ DEFAULT_SAMPLE = 12
55
+ DEFAULT_HOPS = 2
56
+ STALE_DAYS = 30.0 # 多久没被访问算「陈旧」
57
+ UNVERIFIED_DAYS = 14.0 # knowledge 层多久没验证算「长期未验证」
58
+ LOW_CONF = 0.35 # 低于此置信算「低置信」
59
+ ORPHAN_IMPORTANCE = 0.30 # 孤立噪音的重要度上限
60
+ MAX_ASSOC_SCAN = 600 # 词法近邻比对上限(全库扫描保护)
61
+ MAX_BINS_REPORT = 4
62
+ MAX_OFFSET = 0.15 # 校准偏置上限(一次最多挪这么多)
63
+ SELF_LAYERS = ("self", "anchor")
64
+
65
+ STRATA = ("stale", "low_conf", "disputed", "unverified", "orphan", "hot", "random")
66
+ STRATUM_WEIGHTS = {"stale": 0.20, "low_conf": 0.20, "disputed": 0.15,
67
+ "unverified": 0.15, "orphan": 0.10, "hot": 0.10,
68
+ "random": 0.10}
69
+
70
+ # 污染类型 → (严重度, 处置动作)
71
+ CONTAMINATION = {
72
+ "contradiction": ("high", ("weaken", "demote")),
73
+ "expired": ("high", ("weaken", "demote")),
74
+ # 未生效(双时间轴另一侧,2026-09-19):**不是错误**——只提示「此刻不适用」,
75
+ # 故 severity=info、处置仅 hint(不 weaken / 不 demote / 不删)。
76
+ "not_yet": ("info", ("hint",)),
77
+ "duplicate": ("medium", ("hint",)),
78
+ "orphan_noise": ("low", ("weaken", "demote")),
79
+ "unverified": ("info", ("hint",)),
80
+ }
81
+ SEVERITY_ORDER = {"high": 3, "medium": 2, "low": 1, "info": 0}
82
+
83
+
84
+ # --------------------------------------------------------------------------
85
+ # 工具
86
+ # --------------------------------------------------------------------------
87
+
88
+ # 生效条件:当 cg 的 index 为真且其 "nodes" 为真时返回该值,否则(cg 无 index、index 为假值、缺 "nodes" 或 "nodes" 为假值)返回 {};
89
+ def _nodes(cg) -> dict:
90
+ return (getattr(cg, "index", None) or {}).get("nodes") or {}
91
+
92
+
93
+ # 生效条件:cg 与 nid 传入后,若 cache 非 None 且 nid 已在 cache 中则直接返回 cache[nid](即使其值为假值);否则尝试 cg.get(nid),异常或返回假值时按空节点处理,再取其中的 "frontmatter" 真值,若为假值则用 {};cache 非 None 时把结果写入 cache[nid] 后返回;
94
+ def _fm_of(cg, nid, cache=None) -> dict:
95
+ """取节点 frontmatter(带可选缓存)。不可读(缺密钥 / 已删)→ {}。"""
96
+ if cache is not None and nid in cache:
97
+ return cache[nid]
98
+ node = None
99
+ try:
100
+ node = cg.get(nid)
101
+ except Exception:
102
+ node = None
103
+ fm = (node or {}).get("frontmatter") or {}
104
+ if cache is not None:
105
+ cache[nid] = fm
106
+ return fm
107
+
108
+
109
+ # 生效条件:cg.access_counts() 调用成功时返回其结果(访问次数, 最后访问时间);调用抛出任何 Exception 时返回 ({}, {});
110
+ def _access(cg):
111
+ """(访问次数, 最后访问时间):读访问日志,未 compact 的也算。"""
112
+ try:
113
+ return cg.access_counts()
114
+ except Exception:
115
+ return {}, {}
116
+
117
+
118
+ # 生效条件:`from . import chain` 成功且 chain.adjacency(cg) 正常返回时返回该 dict,导入或调用抛任何异常时返回 {}。
119
+ def _adjacency(cg) -> dict:
120
+ try:
121
+ from . import chain
122
+ return chain.adjacency(cg)
123
+ except Exception:
124
+ return {}
125
+
126
+
127
+ # 生效条件:float(ts or 0) 抛 TypeError/ValueError 时返回 0.0,转换后为 0(含 ts 为 0/空串/None 等假值)时返回 0.0,否则返回 (now - ts)/86400.0。
128
+ def _days(ts, now) -> float:
129
+ try:
130
+ ts = float(ts or 0)
131
+ except (TypeError, ValueError):
132
+ return 0.0
133
+ return (now - ts) / 86400.0 if ts else 0.0
134
+
135
+
136
+ # 生效条件:调用即返回带 t 与 op 的 rec;写 append_jsonl(os.path.join(cg.root, SCRUB_LOG), rec) 抛任何异常都被吞掉,不影响返回值。
137
+ def _log(cg, op: str, **rec) -> dict:
138
+ rec = dict(rec, t=time.time(), op=op)
139
+ try:
140
+ append_jsonl(os.path.join(cg.root, SCRUB_LOG), rec)
141
+ except Exception:
142
+ pass
143
+ return rec
144
+
145
+
146
+ # 生效条件:仅当 read_jsonl(cg.root 下 SCRUB_LOG) 的记录 op=="decontaminate" 且 ok 为真时,把 (rec.get("node_id"), rec.get("kind")) 收进返回集合;无此类记录返回空集。
147
+ def _handled(cg) -> set:
148
+ """已处置过的 (node_id, kind):保证去污染幂等(审计即状态)。"""
149
+ out = set()
150
+ for rec in list(read_jsonl(os.path.join(cg.root, SCRUB_LOG))):
151
+ if rec.get("op") == "decontaminate" and rec.get("ok"):
152
+ out.add((rec.get("node_id"), rec.get("kind")))
153
+ return out
154
+
155
+
156
+ # --------------------------------------------------------------------------
157
+ # ① 记忆抽查
158
+ # --------------------------------------------------------------------------
159
+
160
+ # 生效条件:当 cg 的节点/访问/邻接数据可取时,对每个「layer 不在 SELF_LAYERS」的节点(源码仅以 `if layer in SELF_LAYERS: continue` 排除,未要求 layer 非空)按 now、stale_days、unverified_days 判定入池——last 访问时间距今≥stale_days 时入 stale,last 为假值且 created_at 距今≥stale_days 时入 stale,访问计数 acc≥3 入 hot,邻接度 deg==0 且 layer=="contextual" 入 orphan,evidence_count≤0 且 layer=="knowledge" 且 age_d≥unverified_days 入 unverified,evidence_count>0 且 reads<int(max_reads) 且 _fm_of 返回非空前台时按 negative_evidence>0 入 disputed、按 confidence<LOW_CONF(confidence 缺键回落 0.6)入 low_conf(max_reads 为 0 时 int(max_reads)=0,reads<0 恒假,故 disputed/low_conf 不产生),每个节点无条件入 random 池,最后按各池 key 排序返回 pools(片段仅见排序段,未展示抽样阶段);多分支无法一句话覆盖全部分支。
161
+ def _pool_candidates(cg, *, now, stale_days, unverified_days, max_reads=200):
162
+ """构造各层候选池(不读文件的部分先用索引 + 访问日志)。"""
163
+ nodes = _nodes(cg)
164
+ counts, last = _access(cg)
165
+ adj = _adjacency(cg)
166
+ indeg = {}
167
+ for _src, outs in adj.items():
168
+ for tgt, _e in outs:
169
+ indeg[tgt] = indeg.get(tgt, 0) + 1
170
+
171
+ pools = {s: [] for s in STRATA}
172
+ cache, reads = {}, 0
173
+ for nid, e in nodes.items():
174
+ layer = str(e.get("layer") or "")
175
+ if layer in SELF_LAYERS:
176
+ continue # 自我认知不抽查、不去污染
177
+ created = float(e.get("created_at") or 0)
178
+ age_d = _days(created, now)
179
+ la = float(last.get(nid) or 0)
180
+ acc = int(counts.get(nid) or 0)
181
+ ev = int(e.get("evidence_count") or 0)
182
+ deg = len(adj.get(nid) or []) + indeg.get(nid, 0)
183
+
184
+ if la:
185
+ if _days(la, now) >= stale_days:
186
+ pools["stale"].append((nid, f"距上次访问 {_days(la, now):.0f} 天"))
187
+ elif age_d >= stale_days:
188
+ pools["stale"].append((nid, f"从未访问且已存在 {age_d:.0f} 天"))
189
+
190
+ if acc >= 3:
191
+ pools["hot"].append((nid, f"高频访问 {acc} 次"))
192
+
193
+ if deg == 0 and layer == "contextual":
194
+ pools["orphan"].append((nid, "无任何关系边(孤立)"))
195
+
196
+ if ev <= 0 and layer == "knowledge" and age_d >= unverified_days:
197
+ pools["unverified"].append((nid, f"knowledge 层 {age_d:.0f} 天零验证"))
198
+
199
+ if ev > 0 and reads < int(max_reads):
200
+ fm = _fm_of(cg, nid, cache)
201
+ if fm:
202
+ reads += 1
203
+ neg = int(fm.get("negative_evidence") or 0)
204
+ conf = float(fm.get("confidence", 0.6))
205
+ if neg > 0:
206
+ pools["disputed"].append((nid, f"有 {neg} 条反例"))
207
+ if conf < LOW_CONF:
208
+ pools["low_conf"].append((nid, f"置信 {conf:.2f}"))
209
+
210
+ pools["random"].append((nid, "随机基线"))
211
+
212
+ # 池内排序:风险优先(可复现),抽样时再按 seed 洗牌
213
+ pools["stale"].sort(key=lambda x: (nodes.get(x[0], {}).get("created_at") or 0))
214
+ pools["hot"].sort(key=lambda x: -(int(counts.get(x[0]) or 0)))
215
+ pools["unverified"].sort(
216
+ key=lambda x: (nodes.get(x[0], {}).get("created_at") or 0))
217
+ pools["orphan"].sort(
218
+ key=lambda x: float(nodes.get(x[0], {}).get("importance") or 0.5))
219
+ pools["disputed"].sort(key=lambda x: -len(x[1]))
220
+ pools["low_conf"].sort(key=lambda x: x[1])
221
+ return pools
222
+
223
+
224
+ # 生效条件:strategy=="random" 时直接返回 {"random": int(n)};strategy=="risk" 时按剔除 hot/random 后的 STRATUM_WEIGHTS 权重分配;其它策略名走全权重分配,两者都把 int(n) 余量补进已存在的 random 键否则补进 stale。
225
+ def _quota(n: int, strategy: str) -> dict:
226
+ if strategy == "random":
227
+ return {"random": int(n)}
228
+ weights = dict(STRATUM_WEIGHTS)
229
+ if strategy == "risk":
230
+ weights.pop("hot", None)
231
+ weights.pop("random", None)
232
+ total = sum(weights.values()) or 1.0
233
+ out, used = {}, 0
234
+ for s in STRATA:
235
+ if s in weights:
236
+ k = int(round(int(n) * weights[s] / total))
237
+ out[s] = k
238
+ used += k
239
+ rest = int(n) - used
240
+ if rest:
241
+ key = "random" if "random" in out else "stale"
242
+ out[key] = out.get(key, 0) + rest
243
+ return out
244
+
245
+
246
+ # 生效条件:k<=0 或 pool 为假值(空池)时返回 [],否则取池前 k*3 项后用 random.Random(f"{seed}:{stratum}") 稳定洗牌并返回前 k 项(池长不足 k*3 时对全池洗牌)。
247
+ def _pick(pool, k: int, seed, stratum: str):
248
+ """从池中取 k 个:风险最高的 3k 个入池,再按 seed 稳定洗牌。"""
249
+ if k <= 0 or not pool:
250
+ return []
251
+ cand = list(pool)
252
+ if len(cand) > k * 3:
253
+ cand = cand[:k * 3]
254
+ random.Random(f"{seed}:{stratum}").shuffle(cand)
255
+ return cand[:k]
256
+
257
+
258
+ # 生效条件:strategy 经 str(strategy or "stratified").lower() 后属于 stratified/risk/random 时返回带分层标签的样本(seed 为 None 时取 0,n 为 0 时 picked 为空),否则抛 ValueError。
259
+ def sample(cg, n: int = DEFAULT_SAMPLE, *, strategy: str = "stratified",
260
+ seed=None, stale_days: float = STALE_DAYS,
261
+ unverified_days: float = UNVERIFIED_DAYS,
262
+ max_reads: int = 200) -> dict:
263
+ """分层抽查记忆,返回带分层标签的样本。
264
+
265
+ strategy:`stratified`(默认,风险层 + 对照组)· `risk`(只查风险层)
266
+ · `random`(纯随机基线)。seed 固定 → 同样本可复现。
267
+ """
268
+ strategy = str(strategy or "stratified").lower()
269
+ if strategy not in ("stratified", "risk", "random"):
270
+ raise ValueError(f"未知抽样策略:{strategy}(stratified|risk|random)")
271
+ now = time.time()
272
+ seed = 0 if seed is None else seed
273
+ pools = _pool_candidates(cg, now=now, stale_days=stale_days,
274
+ unverified_days=unverified_days,
275
+ max_reads=max_reads)
276
+ quota = _quota(n, strategy)
277
+ picked, seen = [], set()
278
+ for s in STRATA:
279
+ for nid, reason in _pick(pools.get(s), quota.get(s, 0), seed, s):
280
+ if nid in seen:
281
+ continue
282
+ seen.add(nid)
283
+ picked.append({"node_id": nid, "stratum": s, "reason": reason})
284
+ # 风险层不够时用随机池补足(先剔除已选,避免跳过造成不满额)
285
+ if len(picked) < int(n):
286
+ rest = [x for x in pools["random"] if x[0] not in seen]
287
+ for nid, reason in _pick(rest, int(n) - len(picked), seed,
288
+ "random-fill"):
289
+ seen.add(nid)
290
+ picked.append({"node_id": nid, "stratum": "random",
291
+ "reason": reason})
292
+ return {"ok": True, "strategy": strategy, "seed": seed, "n": len(picked),
293
+ "strata": {s: len(pools.get(s) or []) for s in STRATA},
294
+ "sample": picked, "t": now}
295
+
296
+
297
+ # --------------------------------------------------------------------------
298
+ # ② 联想
299
+ # --------------------------------------------------------------------------
300
+
301
+ # 生效条件:a 或 b 为假值(空集)时返回 0.0,否则返回 len(a & b)/len(a | b)。
302
+ def _jaccard(a: set, b: set) -> float:
303
+ if not a or not b:
304
+ return 0.0
305
+ return len(a & b) / len(a | b)
306
+
307
+
308
+ # 生效条件:对 node_id 汇总关系链(chain.walk 出边与入边,max_depth=int(hops))、子图层级(subgraph.expand 命中项与父索引逐级上溯 int(hops) 层)以及 lexical 为真时的词法近邻(只在 _nodes 前 int(max_scan) 个节点内比 bigram、sim≥min_sim 者),返回按 weight 降序的 items[:int(limit)]。
309
+ def associate(cg, node_id: str, *, hops: int = DEFAULT_HOPS, limit: int = 30,
310
+ lexical: bool = True, min_sim: float = 0.25,
311
+ max_scan: int = MAX_ASSOC_SCAN) -> dict:
312
+ """三路邻域联想:关系链 + 子图层级 + 词法近邻。
313
+
314
+ 返回按权重降序的 `related`;每条带 `via`(哪一路发现)与 `weight`,
315
+ 便于审计「为什么把这两个节点算作同族」。
316
+ """
317
+ related = {}
318
+
319
+ # 生效条件:nid 为真且 nid != node_id 时,构造 rec={'node_id':nid,'via':via,'weight':round(float(weight),4),'depth':int(depth)} 并并入 extra;仅当 related 中 nid 不存在或 rec['weight'] 大于已有 weight 时更新 related[nid];nid 为假或等于 node_id 时直接返回;
320
+ def put(nid, via, weight, depth=1, **extra):
321
+ if not nid or nid == node_id:
322
+ return
323
+ rec = {"node_id": nid, "via": via, "weight": round(float(weight), 4),
324
+ "depth": int(depth)}
325
+ rec.update(extra)
326
+ cur = related.get(nid)
327
+ if cur is None or rec["weight"] > cur["weight"]:
328
+ related[nid] = rec
329
+
330
+ # 联想是双向的:同族既可能是「我指向的」也可能是「指向我的」,
331
+ # 故关系链同时沿出边与入边展开(`chain_t --causal--> a` 也要能从 a 找到)。
332
+ try:
333
+ from . import chain
334
+ for direction in ("out", "in"):
335
+ for c in chain.walk(cg, node_id,
336
+ relation_types=chain.CHAIN_TYPES_DEFAULT,
337
+ max_depth=int(hops), max_chains=200,
338
+ include_hierarchy=True, direction=direction):
339
+ for i, h in enumerate(c.get("hops") or [], 1):
340
+ put(h["to"], "chain",
341
+ float(h.get("weight") or 1.0) * (0.8 ** (i - 1)),
342
+ depth=i, relation_type=h.get("relation_type"),
343
+ condition=h.get("condition") or "")
344
+ except Exception:
345
+ pass
346
+
347
+ # 子图层级:向下展开子树 + 向上追溯祖先(父边唯一,逐级上溯)。
348
+ try:
349
+ from . import subgraph
350
+ ex = subgraph.expand(cg, node_id, max_depth=int(hops))
351
+ for nid, path in (ex.get("paths") or {}).items():
352
+ d = str(path).count("/")
353
+ put(nid, "subgraph", 1.0 / (1 + d), depth=d)
354
+ pmap = subgraph.parents_index(cg)
355
+ cur, d = node_id, 0
356
+ while d < int(hops):
357
+ ps = pmap.get(cur) or []
358
+ if not ps:
359
+ break
360
+ d += 1
361
+ for pid in ps:
362
+ put(pid, "subgraph", 1.0 / (1 + d), depth=d)
363
+ cur = ps[0]
364
+ except Exception:
365
+ pass
366
+
367
+ if lexical:
368
+ base = None
369
+ try:
370
+ base = cg.get(node_id)
371
+ except Exception:
372
+ base = None
373
+ bg = bigrams((base or {}).get("content") or "")
374
+ if bg:
375
+ cands = []
376
+ for i, nid in enumerate(_nodes(cg)):
377
+ if i >= int(max_scan):
378
+ break
379
+ if nid == node_id or nid in related:
380
+ continue
381
+ try:
382
+ node = cg.get(nid)
383
+ except Exception:
384
+ node = None
385
+ b2 = bigrams((node or {}).get("content") or "")
386
+ sim = _jaccard(bg, b2)
387
+ if sim >= min_sim:
388
+ cands.append((sim, nid))
389
+ cands.sort(key=lambda x: (-x[0], x[1]))
390
+ for sim, nid in cands[:int(limit)]:
391
+ put(nid, "lexical", sim, depth=1, similarity=round(sim, 4))
392
+
393
+ items = sorted(related.values(), key=lambda r: (-r["weight"], r["node_id"]))
394
+ by_via = {}
395
+ for r in items:
396
+ by_via[r["via"]] = by_via.get(r["via"], 0) + 1
397
+ return {"ok": True, "node_id": node_id, "n": len(items),
398
+ "related": items[:int(limit)], "by_via": by_via}
399
+
400
+
401
+ # --------------------------------------------------------------------------
402
+ # ③ 去污染:确定性判据
403
+ # --------------------------------------------------------------------------
404
+
405
+ # 已结束键族(2026-09-19 阶段一:补规范名 `effective_until` 与冗余时刻 `expired_at`)。
406
+ # 纪律:`believed_at`(信念时间)**两族都不入**——它既不是「已结束」也不是「尚未开始」,
407
+ # 而是「体系何时确认此条」的取代/审核锚。键族真源见 md_cg/trust.py(FROM_ALIASES/UNTIL_ALIASES),
408
+ # 两侧新增键须同步(交叉守卫 test_validity_filter)。
409
+ _EXPIRY_KEYS = ("effective_until", "valid_until", "expires_at", "expire_at",
410
+ "expiry", "deadline", "expired_at")
411
+ _SKIP_KEYS = ("功能名", "执行", "条件", "来源", "标签", "状态", "备注", "标题",
412
+ "描述", "name", "id", "title", "layer", "tags", "说明")
413
+ _NEG_WORDS = ("禁止", "不得", "不要", "不能", "切勿", "避免", "不应", "不可")
414
+ _POS_WORDS = ("应当", "建议", "必须", "需要", "可以", "允许", "推荐", "宜")
415
+
416
+
417
+ # 生效条件:v 为 bool 返回 None;v 为 int/float 时仅 f>1e8 返回 f 否则 None;str(v or "").strip() 为空返回 None;否则按 "%Y-%m-%dT%H:%M:%S"/"%Y-%m-%d %H:%M:%S"/"%Y-%m-%d" 依次取前 19/19/10 字符尝试解析,全失败后再试 float(s),>1e8 返回否则 None,float 亦失败返回 None。
418
+ def _to_ts(v):
419
+ """宽松时间解析:秒级时间戳 / ISO / `YYYY-MM-DD`。无法识别 → None。"""
420
+ if isinstance(v, bool):
421
+ return None
422
+ if isinstance(v, (int, float)):
423
+ f = float(v)
424
+ return f if f > 1e8 else None
425
+ s = str(v or "").strip()
426
+ if not s:
427
+ return None
428
+ for fmt, n in (("%Y-%m-%dT%H:%M:%S", 19), ("%Y-%m-%d %H:%M:%S", 19),
429
+ ("%Y-%m-%d", 10)):
430
+ try:
431
+ return time.mktime(time.strptime(s[:n], fmt))
432
+ except ValueError:
433
+ continue
434
+ try:
435
+ f = float(s)
436
+ return f if f > 1e8 else None
437
+ except ValueError:
438
+ return None
439
+
440
+
441
+ # 生效条件:对 (content or "").splitlines() 的每行去 # 后,先试中文全角 ":"、无则试 ":",切出的键长度在 2–12、值非空且键不在 _SKIP_KEYS 时以 setdefault 记录(每键只留首次出现),无合格行返回 {}。
442
+ def _kv_pairs(content) -> dict:
443
+ """抽 `键:值` 对(中文/英文冒号),跳过结构字段。"""
444
+ out = {}
445
+ for raw in (content or "").splitlines():
446
+ line = raw.strip().lstrip("#").strip()
447
+ if not line:
448
+ continue
449
+ for sep in (":", ":"):
450
+ if sep in line:
451
+ k, _, v = line.partition(sep)
452
+ k, v = k.strip(), v.strip()
453
+ if 2 <= len(k) <= 12 and v and k not in _SKIP_KEYS:
454
+ out.setdefault(k, v)
455
+ break
456
+ return out
457
+
458
+
459
+ # 生效条件:c 取 content or "",含 _NEG_WORDS 任一返回 -1 分量、含 _POS_WORDS 任一返回 +1 分量,结果为两者之和(都不含时为 0)。
460
+ def _polarity(content) -> int:
461
+ c = content or ""
462
+ return (-1 if any(w in c for w in _NEG_WORDS) else 0) + \
463
+ (1 if any(w in c for w in _POS_WORDS) else 0)
464
+
465
+
466
+ # 生效条件:按 _EXPIRY_KEYS 顺序遍历 fm,返回首个满足「键在 fm 中且 _to_ts 非 None 且 ts<now」的 (k, fm.get(k));该键解析为 None 或 ts≥now 时继续检查后续键,全不满足返回 None。
467
+ def _expired(fm, now):
468
+ for k in _EXPIRY_KEYS:
469
+ if k in fm:
470
+ ts = _to_ts(fm.get(k))
471
+ if ts is not None and ts < now:
472
+ return k, fm.get(k)
473
+ return None
474
+
475
+
476
+ # 未生效键(双时间轴的起点,2026-09-19 · 真源 md_cg/trust.py)。
477
+ # 纪律一:`valid_from` **绝不并入 `_EXPIRY_KEYS`**——两者语义相反(「尚未开始」vs
478
+ # 「已经结束」),并入会让「未来才生效」被误判为「已失效」并触发 weaken/demote。
479
+ # 纪律二:`believed_at`(信念时间)同上,**两族都不入**——第三类语义(体系何时确认)。
480
+ _NOT_YET_KEYS = ("valid_from", "valid_since", "effective_from", "starts_at")
481
+
482
+
483
+ # 生效条件:按 _NOT_YET_KEYS 顺序遍历 fm,返回首个满足「键在 fm 中且 _to_ts 非 None 且 ts>now」的 (k, fm.get(k));全不满足返回 None。
484
+ def _not_yet(fm, now):
485
+ """宽松判定「尚未生效」(与 `_expired` 同口径,方向相反)。"""
486
+ for k in _NOT_YET_KEYS:
487
+ if k in fm:
488
+ ts = _to_ts(fm.get(k))
489
+ if ts is not None and ts > now:
490
+ return k, fm.get(k)
491
+ return None
492
+
493
+
494
+ # 生效条件:对 content 取 kv_pairs、bigrams 和 polarity,遍历 related(若 related 为假值则视为空)的前 int(max_compare) 个 r,以 r["node_id"] 调 cg.get;若某 oid 节点可读非空,先在其 content 与 content 的共同键中找到值不同者并返回 {'with':oid,'via':r.get('via'),'why':'同键不同值:...'};否则若极性乘积 <0 且双方 bigram 非空,且共享 bigram 数 >=3 且 ratio>=0.15,返回 {'with':oid,'via':r.get('via'),'why':'极性相反且共享内容:...'};全部遍历完无命中则返回 None;
495
+ def _contradiction(cg, content, related, max_compare=10):
496
+ """与同族节点比对:同键不同值 / 极性相反且共享 bigram。"""
497
+ kv_a = _kv_pairs(content)
498
+ bg_a, pol_a = bigrams(content or ""), _polarity(content)
499
+ for r in (related or [])[:int(max_compare)]:
500
+ oid = r["node_id"]
501
+ try:
502
+ onode = cg.get(oid)
503
+ except Exception:
504
+ onode = None
505
+ if not onode:
506
+ continue
507
+ oc = onode.get("content") or ""
508
+ kv_b = _kv_pairs(oc)
509
+ for k in sorted(set(kv_a) & set(kv_b)):
510
+ if kv_a[k] != kv_b[k]:
511
+ return {"with": oid, "via": r.get("via"),
512
+ "why": f"同键不同值:{k}={kv_a[k]} vs {kv_b[k]}"}
513
+ bg_b = bigrams(oc)
514
+ if pol_a * _polarity(oc) < 0 and bg_a and bg_b:
515
+ shared = bg_a & bg_b
516
+ ratio = len(shared) / max(1, min(len(bg_a), len(bg_b)))
517
+ if len(shared) >= 3 and ratio >= 0.15:
518
+ return {"with": oid, "via": r.get("via"),
519
+ "why": "极性相反且共享内容:" + "、".join(
520
+ sorted(shared)[:5])}
521
+ return None
522
+
523
+
524
+ # 生效条件:ids 在 node_ids 为 None 时取索引全部键、为 str 时取单元素列表、否则取 list(node_ids),limit 为真值时截断为前 int(limit) 个;跳过 layer 在 SELF_LAYERS 的节点和 cg.get 取不到正文的节点,对剩余每个节点按 min_severity 门限累加 expired/unverified/orphan_noise/duplicate/contradiction,返回 ok=无 high 且无 medium 的结果。
525
+ def audit(cg, node_ids=None, *, hops: int = 1, min_severity: str = "info",
526
+ limit=None, unverified_days: float = UNVERIFIED_DAYS,
527
+ lexical: bool = True, min_sim: float = 0.15) -> dict:
528
+ """对指定节点(默认全库)做污染体检。**只读**,不改动任何节点。"""
529
+ nodes = _nodes(cg)
530
+ ids = list(nodes.keys()) if node_ids is None else (
531
+ [node_ids] if isinstance(node_ids, str) else list(node_ids))
532
+ if limit:
533
+ ids = ids[:int(limit)]
534
+ counts, _last = _access(cg)
535
+ adj = _adjacency(cg)
536
+ indeg = {}
537
+ for _src, outs in adj.items():
538
+ for tgt, _e in outs:
539
+ indeg[tgt] = indeg.get(tgt, 0) + 1
540
+ now = time.time()
541
+ issues, checked = [], 0
542
+
543
+ for nid in ids:
544
+ e = nodes.get(nid)
545
+ if not e:
546
+ continue
547
+ layer = str(e.get("layer") or "")
548
+ if layer in SELF_LAYERS:
549
+ continue
550
+ try:
551
+ node = cg.get(nid)
552
+ except Exception:
553
+ node = None
554
+ if not node:
555
+ continue
556
+ checked += 1
557
+ fm = node.get("frontmatter") or {}
558
+ content = node.get("content") or ""
559
+ age_d = _days(fm.get("created_at") or e.get("created_at"), now)
560
+
561
+ # 生效条件:kind 是 CONTAMINATION 的键(否则 KeyError)且 CONTAMINATION[kind] 取出的 sev 在 SEVERITY_ORDER 中的值(缺键按 0)不小于闭包变量 min_severity 在 SEVERITY_ORDER 中的值(缺键按 0)时,把 {node_id, layer, kind, severity, detail, fix} 用 **extra 覆盖更新后追加到闭包 issues;sev 的值更小则直接 return 不追加(该函数无返回值)。
562
+ def add(kind, detail, **extra):
563
+ sev, actions = CONTAMINATION[kind]
564
+ if SEVERITY_ORDER.get(sev, 0) < SEVERITY_ORDER.get(min_severity, 0):
565
+ return
566
+ rec = {"node_id": nid, "layer": layer, "kind": kind,
567
+ "severity": sev, "detail": detail, "fix": list(actions)}
568
+ rec.update(extra)
569
+ issues.append(rec)
570
+
571
+ hit = _expired(fm, now)
572
+ if hit:
573
+ add("expired", f"时效已过:{hit[0]}={hit[1]}", evidence=hit[0])
574
+
575
+ # 双时间轴另一侧(2026-09-19):尚未生效**不是错误**——只提示「此刻不适用」,
576
+ # 处置由 CONTAMINATION["not_yet"] 定为 info 级 + 仅 hint(不 weaken / 不 demote)。
577
+ ny = _not_yet(fm, now)
578
+ if ny:
579
+ add("not_yet", f"尚未生效:{ny[0]}={ny[1]}", evidence=ny[0])
580
+
581
+ if (layer == "knowledge" and age_d >= unverified_days
582
+ and int(e.get("evidence_count") or 0) <= 0):
583
+ add("unverified", f"knowledge 层 {age_d:.0f} 天零验证")
584
+
585
+ if ((len(adj.get(nid) or []) + indeg.get(nid, 0)) == 0
586
+ and layer == "contextual"
587
+ and int(counts.get(nid) or 0) == 0
588
+ and float(fm.get("importance") or e.get("importance") or 0.5)
589
+ < ORPHAN_IMPORTANCE):
590
+ add("orphan_noise", "孤立且零访问、低重要(情境噪音)")
591
+
592
+ try:
593
+ red = forgetting.redundancy(cg, content, layer=layer, exclude=nid)
594
+ except Exception:
595
+ red = None
596
+ if red and float(red.get("max") or 0) >= float(forgetting.DUP_MERGE):
597
+ add("duplicate", f"冗余 {red['max']:.2f}(与 {red.get('with')})",
598
+ duplicate_with=red.get("with"))
599
+
600
+ rel = associate(cg, nid, hops=max(1, int(hops)), limit=10,
601
+ lexical=lexical, min_sim=min_sim)["related"]
602
+ bad = _contradiction(cg, content, rel)
603
+ if bad:
604
+ add("contradiction", bad["why"], conflict_with=bad.get("with"))
605
+
606
+ by_kind = {}
607
+ for i in issues:
608
+ by_kind[i["kind"]] = by_kind.get(i["kind"], 0) + 1
609
+ high = sum(1 for i in issues if i["severity"] == "high")
610
+ med = sum(1 for i in issues if i["severity"] == "medium")
611
+ return {"ok": not (high or med), "n_checked": checked,
612
+ "n_issues": len(issues), "issues": issues, "by_kind": by_kind,
613
+ "stats": {"checked": checked, "high": high, "medium": med,
614
+ "low": sum(1 for i in issues if i["severity"] == "low")},
615
+ "t": now}
616
+
617
+
618
+ # 生效条件:循环 max(1, int(times)) 次调用 cg.verify(nid, "scrub:"+kind+":"+str(detail)[:120], "weakened"),返回最后一次调用结果(times≤0 时按 1 次执行)。
619
+ def _weaken(cg, nid, kind, detail, times=1):
620
+ res = None
621
+ for _ in range(max(1, int(times))):
622
+ res = cg.verify(nid, f"scrub:{kind}:{str(detail)[:120]}", "weakened")
623
+ return res
624
+
625
+
626
+ # 生效条件:kinds 为真且 issue 的 kind 不在其中则跳过;kind 为 duplicate/unverified 记 hint,已出现在 _handled 记 skip,protect.is_protected 为真且 override 为假记 skip_protected,dry_run 为真记 planned;仅 dry_run 为假时对余下 issue 执行 _weaken,并在 demote 为真、severity 为 high 或 low 且当前层非 contextual 时降级到 contextual,返回计数与 actions。
627
+ def decontaminate(cg, node_ids=None, *, kinds=None, dry_run: bool = True,
628
+ min_severity: str = "medium", hops: int = 1, actor=None,
629
+ override: bool = False, weaken_times: int = 1,
630
+ demote: bool = True) -> dict:
631
+ """按体检结果处置污染。**默认 dry_run**;实修只做可逆动作。
632
+
633
+ 动作:`weaken`(verify weakened → 反例+1、置信-0.15、跌破 0.2 自动降级)、
634
+ `demote`(降级到情境层)、`hint`(只建议,不改)。保护节点跳过;永不删除。
635
+ """
636
+ rep = audit(cg, node_ids, hops=hops, min_severity=min_severity)
637
+ handled = _handled(cg)
638
+ actions = []
639
+ n_applied = n_prot = n_done = n_hint = 0
640
+
641
+ for issue in rep["issues"]:
642
+ nid, kind = issue["node_id"], issue["kind"]
643
+ if kinds and kind not in kinds:
644
+ continue
645
+ if kind in ("duplicate", "unverified"):
646
+ n_hint += 1
647
+ actions.append({"node_id": nid, "kind": kind, "action": "hint",
648
+ "detail": issue["detail"], "applied": False})
649
+ continue
650
+ if (nid, kind) in handled:
651
+ n_done += 1
652
+ actions.append({"node_id": nid, "kind": kind, "action": "skip",
653
+ "detail": "已处置过(幂等)", "applied": False})
654
+ continue
655
+ prot, why = protect.is_protected(cg, nid)
656
+ if prot and not override:
657
+ n_prot += 1
658
+ actions.append({"node_id": nid, "kind": kind,
659
+ "action": "skip_protected", "detail": why,
660
+ "applied": False})
661
+ continue
662
+ if dry_run:
663
+ actions.append({"node_id": nid, "kind": kind,
664
+ "action": "planned",
665
+ "planned": list(CONTAMINATION[kind][1]),
666
+ "detail": issue["detail"], "applied": False})
667
+ continue
668
+
669
+ done = []
670
+ try:
671
+ _weaken(cg, nid, kind, issue["detail"], weaken_times)
672
+ done.append("weaken")
673
+ except Exception as exc: # noqa: BLE001
674
+ done.append(f"weaken_failed:{type(exc).__name__}")
675
+ if demote and issue["severity"] in ("high", "low"):
676
+ try:
677
+ fm = _fm_of(cg, nid, None)
678
+ if str(fm.get("layer") or "") != "contextual":
679
+ cg._move_layer(nid, "contextual", reason=f"scrub:{kind}")
680
+ # 层降级 = 生命周期降级(②):状态同批推进(负路由,失败不抛)
681
+ _st = cg.set_state(nid, "demoted", reason=f"scrub:{kind}",
682
+ actor=actor or "scrub")
683
+ done.append("demote" if _st.get("ok") else
684
+ f"demote_state:{_st.get('error')}")
685
+ except Exception as exc: # noqa: BLE001
686
+ done.append(f"demote_failed:{type(exc).__name__}")
687
+ n_applied += 1
688
+ _log(cg, "decontaminate", node_id=nid, kind=kind, ok=True,
689
+ actions=done, severity=issue["severity"], actor=actor)
690
+ actions.append({"node_id": nid, "kind": kind, "action": "applied",
691
+ "done": done, "detail": issue["detail"],
692
+ "applied": True})
693
+
694
+ if not dry_run:
695
+ _log(cg, "decontaminate_batch", ok=True, applied=n_applied,
696
+ skipped_protected=n_prot, skipped_done=n_done, hints=n_hint,
697
+ actor=actor)
698
+ return {"ok": True, "dry_run": dry_run, "n_issues": rep["n_issues"],
699
+ "applied": n_applied, "skipped_protected": n_prot,
700
+ "skipped_done": n_done, "hints": n_hint, "actions": actions,
701
+ "audit": rep, "t": time.time()}
702
+
703
+
704
+ # --------------------------------------------------------------------------
705
+ # ④ 校准偏差
706
+ # --------------------------------------------------------------------------
707
+
708
+ # 生效条件:对 cg 中每个「layer 不在 SELF_LAYERS」且 evidence_count≥int(min_evidence)(min_evidence=0 时该比较恒假而不早退)的节点,若 protect.is_protected 为假或 override 为真,且 cg.get(nid) 未抛异常并返回真值节点,则取 fm.get("confidence", 0.6)(缺键才回落 0.6,键存在为 None/假值不回落)为 old,算出 round(max(0.0, min(0.99, old+float(offset))),4),与 old 差<1e-9 时跳过,否则写 fm["confidence"] 与 fm["calibration"] 并调用 cg._write_node 成功时 adjusted+1(写回异常被吞掉不计数),返回 (adjusted, skipped),其中 skipped 只累计「layer 属 SELF_LAYERS」或被 protect 拦下且非 override 的节点。
709
+ def _apply_offset(cg, offset, *, override=False, min_evidence=1):
710
+ nodes = _nodes(cg)
711
+ adjusted = skipped = 0
712
+ for nid, e in nodes.items():
713
+ if str(e.get("layer") or "") in SELF_LAYERS:
714
+ skipped += 1
715
+ continue
716
+ if int(e.get("evidence_count") or 0) < int(min_evidence):
717
+ continue
718
+ prot, _why = protect.is_protected(cg, nid)
719
+ if prot and not override:
720
+ skipped += 1
721
+ continue
722
+ try:
723
+ node = cg.get(nid)
724
+ except Exception:
725
+ node = None
726
+ if not node:
727
+ continue
728
+ fm = node.get("frontmatter") or {}
729
+ old = float(fm.get("confidence", 0.6))
730
+ new = round(max(0.0, min(0.99, old + float(offset))), 4)
731
+ if abs(new - old) < 1e-9:
732
+ continue
733
+ fm["confidence"] = new
734
+ fm["calibration"] = {"t": time.time(), "offset": round(float(offset), 4),
735
+ "from": old, "to": new}
736
+ try:
737
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm,
738
+ node.get("content") or "")
739
+ adjusted += 1
740
+ except Exception: # noqa: BLE001
741
+ pass
742
+ return adjusted, skipped
743
+
744
+
745
+ # 生效条件:cg 和 apply/override/actor/max_offset/min_evidence 传入后,若 metacognition.calibration(cg) 返回 ok 假,则返回 {'ok':False,'reason':cal.get('reason') or 'insufficient_data',...};若 ok 真,则用 gap=float(cal.get('gap') or 0.0) 和 max_offset 计算 offset=round(max(-max_offset,min(max_offset,-gap)),4),对 cal.get('bins') or [] 中 accuracy 非 None 的 bin 生成 bins_bias 并排序;仅当 apply 为真且 abs(offset)>1e-9 时调用 _apply_offset(cg,offset,override=override,min_evidence=min_evidence) 并写日志,最后返回 ok True 及 verdict/gap/建议 offset 等字段;
746
+ def calibrate(cg, *, apply: bool = False, override: bool = False, actor=None,
747
+ max_offset: float = MAX_OFFSET, min_evidence: int = 1) -> dict:
748
+ """校准偏差:自报置信 vs 实测正确率 → 偏置建议(`apply=True` 才写回)。"""
749
+ from . import metacognition
750
+ cal = metacognition.calibration(cg)
751
+ if not cal.get("ok"):
752
+ return {"ok": False, "reason": cal.get("reason") or "insufficient_data",
753
+ "calibration": cal, "applied": False, "n_adjusted": 0,
754
+ "hint": "先 verify() 积累证据,校准才有样本"}
755
+ gap = float(cal.get("gap") or 0.0)
756
+ offset = round(max(-float(max_offset), min(float(max_offset), -gap)), 4)
757
+ bins_bias = []
758
+ for b in cal.get("bins") or []:
759
+ if b.get("accuracy") is None:
760
+ continue
761
+ bins_bias.append({"bin": b["bin"], "n_nodes": b["n_nodes"],
762
+ "confidence": b["confidence"],
763
+ "accuracy": b["accuracy"],
764
+ "bias": round(float(b["confidence"])
765
+ - float(b["accuracy"]), 4)})
766
+ bins_bias.sort(key=lambda x: -abs(x["bias"]))
767
+ adjusted = skipped = 0
768
+ if apply and abs(offset) > 1e-9:
769
+ adjusted, skipped = _apply_offset(cg, offset, override=override,
770
+ min_evidence=min_evidence)
771
+ _log(cg, "calibrate", ok=True, offset=offset, gap=gap,
772
+ adjusted=adjusted, actor=actor)
773
+ blind = 0
774
+ try:
775
+ blind = int(metacognition.blindspots(cg, limit=1)
776
+ .get("unresolved_count") or 0)
777
+ except Exception:
778
+ pass
779
+ return {"ok": True, "verdict": cal.get("verdict"), "gap": gap,
780
+ "expected_accuracy": cal.get("expected_accuracy"),
781
+ "actual_accuracy": cal.get("actual_accuracy"),
782
+ "ece": cal.get("ece"), "n_nodes": cal.get("n_nodes"),
783
+ "suggested_offset": offset, "applied": bool(apply),
784
+ "n_adjusted": adjusted, "skipped_protected": skipped,
785
+ "bins_bias": bins_bias[:MAX_BINS_REPORT], "blindspots": blind,
786
+ "hint": ("整体偏过度自信" if gap > 0.05 else
787
+ "整体偏保守" if gap < -0.05 else "校准良好")}
788
+
789
+
790
+ # --------------------------------------------------------------------------
791
+ # 一轮完整自净 + 审计
792
+ # --------------------------------------------------------------------------
793
+
794
+ # 生效条件:cg 与 n/seed/dry_run/hops/strategy/apply_calibration/actor 传入后,按 n 与 strategy、seed 调 sample;对 sample 结果前 8 个 node_id 按 hops 调 associate;按 dry_run/hops/actor 调 decontaminate;按 apply_calibration/actor 调 calibrate;若 decontaminate 的 audit.issues 中存在 severity 为 high 或 medium 的项则 out.ok 为 False,否则为 True,并返回含 sample/associate/audit/decontaminate/calibration/t 的 out;
795
+ def sweep(cg, *, n: int = DEFAULT_SAMPLE, seed=None, dry_run: bool = True,
796
+ hops: int = DEFAULT_HOPS, strategy: str = "stratified",
797
+ apply_calibration: bool = False, actor=None) -> dict:
798
+ """一轮自净闭环:抽查 → 联想 → 体检 → 去污染 → 校准偏差。"""
799
+ smp = sample(cg, n, strategy=strategy, seed=seed)
800
+ ids = [s["node_id"] for s in smp["sample"]]
801
+ assoc = {}
802
+ for nid in ids[:8]:
803
+ assoc[nid] = [r["node_id"] for r in associate(
804
+ cg, nid, hops=hops, lexical=False, limit=5)["related"]]
805
+ dec = decontaminate(cg, ids, dry_run=dry_run, hops=hops, actor=actor)
806
+ rep = dec["audit"]
807
+ cal = calibrate(cg, apply=apply_calibration, actor=actor)
808
+ hi = [i for i in rep["issues"] if i["severity"] in ("high", "medium")]
809
+ out = {"ok": not hi, "dry_run": dry_run, "n_high_medium": len(hi),
810
+ "sample": smp, "associate": assoc, "audit": rep,
811
+ "decontaminate": dec, "calibration": cal, "t": time.time()}
812
+ _log(cg, "sweep", n_sample=smp["n"], n_issues=rep["n_issues"],
813
+ n_high_medium=len(hi), applied=dec["applied"], dry_run=dry_run,
814
+ calibration=(cal.get("verdict") if cal.get("ok")
815
+ else cal.get("reason")))
816
+ return out
817
+
818
+
819
+ # 生效条件:返回 cg.root 下 SCRUB_LOG 的全部记录条数 n 与 recs[-int(limit):](limit=0 时切片为 recs[0:] 即返回全部记录)。
820
+ def history(cg, limit: int = 100) -> dict:
821
+ recs = list(read_jsonl(os.path.join(cg.root, SCRUB_LOG)))
822
+ return {"n": len(recs), "records": recs[-int(limit):]}
823
+
824
+
825
+ # 生效条件:读取 cg.root 下 SCRUB_LOG 的 JSONL 记录,过滤 op=="sweep" 得 sweeps、op=="decontaminate" 且 ok 为真得 decs;last 为 sweeps 最后一项或 None;返回 {'sweeps':len(sweeps),'decontaminated':len(decs),'last_sweep':last 的 t/n_issues/n_high_medium/applied/dry_run/calibration 或 None};
826
+ def summary(cg) -> dict:
827
+ """给 health_os / 自维持循环用的只读摘要。"""
828
+ recs = list(read_jsonl(os.path.join(cg.root, SCRUB_LOG)))
829
+ sweeps = [r for r in recs if r.get("op") == "sweep"]
830
+ decs = [r for r in recs if r.get("op") == "decontaminate" and r.get("ok")]
831
+ last = sweeps[-1] if sweeps else None
832
+ return {"sweeps": len(sweeps), "decontaminated": len(decs),
833
+ "last_sweep": ({"t": last.get("t"),
834
+ "n_issues": last.get("n_issues"),
835
+ "n_high_medium": last.get("n_high_medium"),
836
+ "applied": last.get("applied"),
837
+ "dry_run": last.get("dry_run"),
838
+ "calibration": last.get("calibration")}
839
+ if last else None)}
840
+
841
+
842
+ # 生效条件:无入参调用即返回固定的 actions 列表、STRATA 列表、CONTAMINATION 映射(每项取 severity 与 actions)与 STALE_DAYS/UNVERIFIED_DAYS/LOW_CONF/ORPHAN_IMPORTANCE/MAX_OFFSET 阈值字典。
843
+ def catalog() -> dict:
844
+ return {"actions": ["sample", "associate", "audit", "decontaminate",
845
+ "calibrate", "sweep", "history", "summary", "catalog"],
846
+ "strata": list(STRATA),
847
+ "contamination": {k: {"severity": v[0], "actions": list(v[1])}
848
+ for k, v in CONTAMINATION.items()},
849
+ "thresholds": {"stale_days": STALE_DAYS,
850
+ "unverified_days": UNVERIFIED_DAYS,
851
+ "low_conf": LOW_CONF,
852
+ "orphan_importance": ORPHAN_IMPORTANCE,
853
853
  "max_offset": MAX_OFFSET}}