@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
@@ -1,582 +1,582 @@
1
- # -*- coding: utf-8 -*-
2
- """md_cg · 主动遗忘闸门(写入情景层前的三问筛选)
3
-
4
- 理论出处(全部来自本仓已有文档):
5
-
6
- · `memory_score.md:12`
7
- J 判断引擎 9-10 档 = 「独立元认知 + **主动遗忘**」;灵枢正因「无主动遗忘」
8
- 停在 8.0。→ 主动遗忘是 J 维上 9 分的门槛项,不是可选优化。
9
- · `docs/白箱智能系列·第五篇:174-180`
10
- 「把经历兑换成结构…整理完之后记忆库变小了,但信息量反而更可用——
11
- 噪音被扔掉了,骨架被留下。」
12
- · AEIS 工具表 `docs/mdcg/tool_table_v0.3.0.md:15-19`
13
- `prefeed`(H1 新奇检测 → 高新奇输入当场强化编码)、
14
- `pattern_separation`(H3 扫描相似节点对)、
15
- `nightly_cleanup`(知识层夜间整理、无边孤岛降级)。
16
- 本模块 = 这三件事的**写入侧前置版**:不等夜间整理,写之前就裁决。
17
- · `docs/theory/智能的公理化基石.md:758-763` —— **诚实边界**
18
- 「信息差与热力学熵之间只能进行结构类比,不应宣称数学同构」。
19
- 故本模块一律称「自信息代理 / 惊奇度」,**不称香农熵**,也不做熵的物理断言。
20
-
21
- 三问 → 四态裁决(对齐白箱四态,落库动作分四种):
22
-
23
- Q1 重复? redundancy = 新内容被既有同层节点覆盖的最大比例(bigram 覆盖率)
24
- Q2 重要? importance = 显式 hint 优先,否则启发式(新奇/来源/长度)
25
- Q3 惊奇? self_info = -log2(dup + ε)(bit,**代理量**,非香农熵)
26
-
27
- ACCEPT 写入 / MERGE 并入既有(不新增,强化既有节点)
28
- DROP 丢弃 / DEFER 待定(不写,留痕待复核)
29
-
30
- 裁决顺序(**顺序即语义**):
31
- 1) 重要度 ≥0.7 → ACCEPT(保护优先)
32
- 2) 确定性内部产生 且 冗余 → DROP ← 先于 MERGE:机器例行输出再"重复"也只是
33
- 噪音,不该去强化既有记忆(否则例行日志
34
- 会把普通记忆刷成高重要性)
35
- 3) 冗余 ≥0.85 → MERGE ← 外部/未知来源的重复 = 又一次确认,强化
36
- 4) 半重复 且 不重要 → DEFER
37
- 5) 重要度 ≥0.30 → ACCEPT
38
- 6) 新信息 ≥0.15 → ACCEPT
39
- 7) 其余 → DEFER
40
-
41
- 一切裁决都写进 `_forgetting.jsonl`(append-only),可审计:
42
- 「这条为什么没被记住」和「为什么被记住」同样有据可查。
43
- """
44
- import hashlib
45
- import json
46
- import math
47
- import os
48
- import time
49
-
50
- from . import lifecycle, nodefile
51
- from .fsutil import append_jsonl, atomic_write, read_jsonl
52
- from .mdcg import bigrams
53
-
54
- # ---------------------------------------------------------------- 判据常量
55
-
56
- DUP_MERGE = 0.85 # 重复度 ≥ 此值 → MERGE
57
- DUP_DROP = 0.60 # 重复度 ≥ 此值 → 进入 DROP / DEFER 判据
58
- NOVELTY_MIN = 0.15 # 新信息 < 此值 → 视为无新信息
59
- IMPORTANCE_MIN = 0.30 # 重要度 < 此值 → 不予写入
60
- PROTECT_IMPORTANCE = 0.70 # 对齐 tool_table:≥0.7 触发不可遗忘保护
61
- MAX_BITS = 4.0 # 自信息归一化上限(dup=0 时 4.0 bit)
62
- EPS = 0.0625 # 自信息平滑(避免 dup=0 时取 log(0))
63
- MAX_COMPARE = 240 # 单次重复检测最多比对的同层节点数(写入非热路径)
64
-
65
- # 来源类型 → 权重(确定性内部产生 = 低权;外部惊奇 = 高权)
66
- SOURCE_WEIGHT = {
67
- "external_surprising": 1.00,
68
- "unknown": 0.60,
69
- "self_generated": 0.50,
70
- "internal_deterministic": 0.25,
71
- }
72
- EXTERNAL_ROLES = ("user",)
73
- INTERNAL_ROLES = ("command", "tool-output", "edit", "system")
74
- # 注意:文科的 textbook/public_kb **不在此列**——它们是「权威来源表述一致」,
75
- # 不是「内部确定性产生」,故仍按外部来源计权(见 source_kind)。
76
- DETERMINISTIC_BASIS = ("data", "measurement", "compiler", "test", "formal_proof")
77
-
78
- LOG_FILE = "_forgetting.jsonl"
79
-
80
-
81
- # ---------------------------------------------------------------- 三问
82
-
83
- # 生效条件:role 与 verification_basis 各自经 str(x or "").strip().lower() 后按序判——role 命中模块常量 EXTERNAL_ROLES 返回 "external_surprising";否则 role 命中 INTERNAL_ROLES、或两者都不命中前者时 verification_basis 命中 DETERMINISTIC_BASIS,返回 "internal_deterministic";否则 role 为 "assistant"/"agent" 返回 "self_generated";全不命中返回 "unknown"。
84
- def source_kind(role=None, verification_basis=None):
85
- """Q3 的来源面:内部确定性产生 vs 外部惊奇来源。"""
86
- r = str(role or "").strip().lower()
87
- vb = str(verification_basis or "").strip().lower()
88
- if r in EXTERNAL_ROLES:
89
- return "external_surprising"
90
- if r in INTERNAL_ROLES:
91
- return "internal_deterministic"
92
- if vb in DETERMINISTIC_BASIS:
93
- return "internal_deterministic"
94
- if r in ("assistant", "agent"):
95
- return "self_generated"
96
- return "unknown"
97
-
98
-
99
- # 生效条件:new_grams 为空集(假值)时返回 0.0;非空时返回 len(new_grams & body_grams)/len(new_grams)。
100
- def _coverage(new_grams, body_grams):
101
- if not new_grams:
102
- return 0.0
103
- return len(new_grams & body_grams) / float(len(new_grams))
104
-
105
-
106
- # CCG 五要素的固定标签:所有节点都一样,属**模板骨架而非内容**。
107
- # 不剥离它们,任何两条记忆都会因共享 `# 功能名:`/`# 生效条件:` 而虚高重复度
108
- # (实测:两条毫不相关的记忆 dup≈0.33,全部来自模板)。故重复检测只看"值"。
109
- _TEMPLATE_LABELS = ("功能名", "生效条件", "子功能", "执行", "验证方式", "不适用条件")
110
-
111
-
112
- # 生效条件:content 为 None 或假值时按 "" 处理,结果为空串;否则逐行剥离 "#" 与 _TEMPLATE_LABELS 标签后以 "" 直接拼接。
113
- def payload(content):
114
- """剥离 CCG 固定标签后的**内容骨架**(保留字段值,丢弃字段名与标记)。"""
115
- out = []
116
- for line in (content or "").splitlines():
117
- s = line.strip()
118
- if s.startswith("#"):
119
- s = s.lstrip("#").strip()
120
- for lab in _TEMPLATE_LABELS:
121
- if s.startswith(lab):
122
- s = s[len(lab):].lstrip(":: ").strip()
123
- break
124
- if s:
125
- out.append(s)
126
- return "".join(out)
127
-
128
-
129
- # 生效条件:content 经 payload/bigrams 得空集合时直接返回零值 best(max=0.0、with=None、compared=0);否则遍历 cg.index 的 nodes,跳过 nid==exclude,layer 为真值时只比较 str(layer 字段 or "")==layer 的节点,cg.get(nid) 抛异常/返回假值、或该节点 content 的 bigrams 为空则跳过,每计入一个节点后若 n>=limit 立即 break(故 limit 为 0 或负数时只比较首项即停),返回覆盖度最大者 best(无覆盖度提升时不更新 with/jaccard,compared 为实际计入数)。
130
- def redundancy(cg, content, layer="contextual", exclude=None, limit=MAX_COMPARE):
131
- """Q1 重复?——新内容被既有同层节点覆盖的最大比例。"""
132
- new = bigrams(payload(content))
133
- best = {"max": 0.0, "with": None, "jaccard": 0.0, "compared": 0}
134
- if not new:
135
- return best
136
- nodes = ((getattr(cg, "index", None) or {}).get("nodes") or {})
137
- n = 0
138
- for nid in list(nodes.keys()):
139
- if nid == exclude:
140
- continue
141
- if layer and str(nodes[nid].get("layer") or "") != layer:
142
- continue
143
- try:
144
- node = cg.get(nid)
145
- except Exception:
146
- node = None
147
- if not node:
148
- continue
149
- body = bigrams(payload(node.get("content") or ""))
150
- if not body:
151
- continue
152
- n += 1
153
- cov = _coverage(new, body)
154
- if cov > best["max"]:
155
- best = {"max": cov, "with": nid,
156
- "jaccard": len(new & body) / float(len(new | body) or 1),
157
- "compared": n}
158
- if n >= limit:
159
- break
160
- best["compared"] = n
161
- return best
162
-
163
-
164
- # 生效条件:dup 必填并转 float;dup=0 时 p 取 EPS,返回 -log2(EPS) 这一有限大值;dup>=1 时返回 0.0。
165
- def self_information(dup):
166
- """Q3 的自信息代理:I = -log2(min(1, dup + ε)),单位 bit。
167
-
168
- 注意:dup 是「被既有记忆覆盖率」的估计,不是概率模型的真实 P(x),
169
- 因此这是**结构类比的代理量**(见模块 docstring 的诚实边界)。
170
- """
171
- p = min(1.0, max(0.0, float(dup)) + EPS)
172
- return -math.log(p, 2.0)
173
-
174
-
175
- # 生效条件:hint 非 None 且可转 float(含 hint=0)时返回 from="hint" 的裁剪分数;否则用 novelty、SOURCE_WEIGHT.get(kind, SOURCE_WEIGHT["unknown"])、len(content)/200 三因子启发式。
176
- def importance_score(hint, novelty, kind, content):
177
- """Q2 重要?——显式 hint 优先,否则启发式(对齐 longterm_snapshot 四因子简化版)。"""
178
- if hint is not None:
179
- try:
180
- return {"score": round(max(0.0, min(1.0, float(hint))), 4),
181
- "from": "hint"}
182
- except (TypeError, ValueError):
183
- pass
184
- lf = min(1.0, len(content or "") / 200.0)
185
- s = (0.5 * novelty
186
- + 0.3 * SOURCE_WEIGHT.get(kind, SOURCE_WEIGHT["unknown"])
187
- + 0.2 * lf)
188
- return {"score": round(max(0.0, min(1.0, s)), 4), "from": "heuristic"}
189
-
190
-
191
- # 生效条件:以 source_kind(role,verification_basis) 的 kind 与 redundancy(cg,content,layer=layer,exclude=node_id) 的 red["max"] 为输入,按 if/elif 顺序取首个命中分支——imp["score"]≥PROTECT_IMPORTANCE→"ACCEPT";否则 kind=="internal_deterministic" 且 red["max"]≥DUP_DROP→"DROP";否则 red["max"]≥DUP_MERGE→"MERGE";否则 red["max"]≥DUP_DROP 且 imp["score"]<IMPORTANCE_MIN→"DEFER";否则 imp["score"]≥IMPORTANCE_MIN→"ACCEPT";否则 novelty≥NOVELTY_MIN→"ACCEPT";否则→"DEFER"。
192
- def assess(cg, content, layer="contextual", role=None, verification_basis=None,
193
- importance_hint=None, node_id=None):
194
- """三问 → 四态裁决。返回完整判据(可审计,不只给结论)。"""
195
- kind = source_kind(role, verification_basis)
196
- red = redundancy(cg, content, layer=layer, exclude=node_id)
197
- novelty = round(1.0 - red["max"], 4)
198
- bits = round(self_information(red["max"]), 4)
199
- imp = importance_score(importance_hint, novelty, kind, content)
200
- entropy = {
201
- "source_kind": kind,
202
- "novelty": novelty,
203
- "self_information_bits": bits,
204
- "normalized": round(min(1.0, bits / MAX_BITS), 4),
205
- "duplicate_with": red["with"],
206
- "duplicate_ratio": round(red["max"], 4),
207
- "compared": red["compared"],
208
- }
209
-
210
- if imp["score"] >= PROTECT_IMPORTANCE:
211
- verdict, why = "ACCEPT", (f"重要度 {imp['score']:.2f}≥{PROTECT_IMPORTANCE}"
212
- f"(触发不可遗忘保护)")
213
- elif kind == "internal_deterministic" and red["max"] >= DUP_DROP:
214
- verdict, why = "DROP", (f"确定性内部产生且冗余 {red['max']:.2f}≥{DUP_DROP}"
215
- f"(低熵噪音,不编码)")
216
- elif red["max"] >= DUP_MERGE:
217
- verdict, why = "MERGE", (f"重复度 {red['max']:.2f}≥{DUP_MERGE}"
218
- f"(并入 {red['with']},强化既有)")
219
- elif red["max"] >= DUP_DROP and imp["score"] < IMPORTANCE_MIN:
220
- verdict, why = "DEFER", (f"半重复 {red['max']:.2f}∈[{DUP_DROP},{DUP_MERGE})"
221
- f" 且重要度 {imp['score']:.2f}<{IMPORTANCE_MIN}"
222
- f"(待定复核)")
223
- elif imp["score"] >= IMPORTANCE_MIN:
224
- verdict, why = "ACCEPT", f"重要度 {imp['score']:.2f}≥{IMPORTANCE_MIN}"
225
- elif novelty >= NOVELTY_MIN:
226
- verdict, why = "ACCEPT", f"新信息 {novelty:.2f}≥{NOVELTY_MIN}"
227
- else:
228
- verdict, why = "DEFER", "重要度与新信息均不足判据(待定)"
229
-
230
- return {"verdict": verdict, "reason": why, "redundancy": red,
231
- "importance": imp, "entropy": entropy}
232
-
233
-
234
- # ---------------------------------------------------------------- 落库动作
235
-
236
- # 生效条件:cg 与 rec 必填;append_jsonl 写 cg.root/LOG_FILE 抛任意异常时被吞掉,仍返回 rec。
237
- def log(cg, rec):
238
- """裁决留痕(append-only)。DROP/DEFER 也留痕——否则遗忘变黑箱。"""
239
- try:
240
- append_jsonl(os.path.join(cg.root, LOG_FILE), rec)
241
- except Exception:
242
- pass
243
- return rec
244
-
245
-
246
- # 生效条件:cg 与 node_id 必填,delta 默认 0.05;cg.get(node_id) 抛异常或返回假值时返回 None;imp 跨过 PROTECT_IMPORTANCE 即写 protected。
247
- def reinforce(cg, node_id, delta=0.05):
248
- """MERGE 的落库动作:不新增节点,把「又一次见到」折算成既有节点的强化。
249
-
250
- 重要性 +delta,merge_count +1;一旦跨过 0.7 自动打上保护标记
251
- (对齐「importance 提升(保护:不可遗忘…且受保护标记)」)。
252
- """
253
- try:
254
- node = cg.get(node_id)
255
- except Exception:
256
- node = None
257
- if not node:
258
- return None
259
- fm = node.get("frontmatter") or {}
260
- imp = min(1.0, float(fm.get("importance") or 0.5) + delta)
261
- fm["importance"] = imp
262
- fm["merge_count"] = int(fm.get("merge_count") or 0) + 1
263
- fm["last_merge_at"] = time.time()
264
- if imp >= PROTECT_IMPORTANCE:
265
- fm["protected"] = True
266
- fm["protection_reason"] = (f"importance={imp:.2f}≥{PROTECT_IMPORTANCE}"
267
- f"(重复强化)")
268
- # ② 显式状态机收口(2026-09-16):MERGE 的语义是「又一次见到」= **重新激活**
269
- # 信号——已降权(demoted)/已定型(converged)的节点经状态机**逐级回升**到
270
- # active(archived→active 亦合法,归档节点被再次见到即恢复参与);active 为
271
- # 幂等 no-op(不写字段、不留痕)。protected 只豁免**降级**,回升不受限。
272
- lifecycle.stamp(fm, "active", reason="MERGE 重复强化(回升)",
273
- actor="forgetting:reinforce")
274
- cg._write_node(node_id, os.path.join(cg.root, node["path"]),
275
- fm, node.get("content") or "")
276
- e = ((getattr(cg, "index", None) or {}).get("nodes") or {}).get(node_id)
277
- if e is not None:
278
- e["importance"] = imp
279
- if fm.get(lifecycle.STATE_FIELD):
280
- e[lifecycle.STATE_FIELD] = fm[lifecycle.STATE_FIELD]
281
- if fm.get("protected"):
282
- e["protected"] = True
283
- e["protection_reason"] = fm["protection_reason"]
284
- return {"node_id": node_id, "importance": imp,
285
- "merge_count": fm["merge_count"], "protected": bool(fm.get("protected"))}
286
-
287
-
288
- # 生效条件:cg 必填,limit 默认 100;日志路径不存在时返回 [];否则返回 out[-limit:],limit=0 时 -0 退化为 out[0:] 即全量。
289
- def history(cg, limit=100):
290
- """读取遗忘留痕(最近 limit 条)。"""
291
- p = os.path.join(cg.root, LOG_FILE)
292
- if not os.path.exists(p):
293
- return []
294
- out = []
295
- try:
296
- with open(p, "r", encoding="utf-8") as f:
297
- for line in f:
298
- line = line.strip()
299
- if line:
300
- try:
301
- out.append(__import__("json").loads(line))
302
- except Exception:
303
- continue
304
- except Exception:
305
- return []
306
- return out[-limit:]
307
-
308
-
309
- # 生效条件:cg 必填;日志路径不存在返回 {"total": 0, "by_verdict": {}};否则流式累计行数与 verdict 分布。
310
- def summary(cg):
311
- """遗忘留痕聚合(流式,不把全量日志读进内存):总数 + 四态分布。"""
312
- p = os.path.join(cg.root, LOG_FILE)
313
- counts, total = {}, 0
314
- if not os.path.exists(p):
315
- return {"total": 0, "by_verdict": {}}
316
- try:
317
- with open(p, "r", encoding="utf-8") as f:
318
- for line in f:
319
- line = line.strip()
320
- if not line:
321
- continue
322
- try:
323
- v = json.loads(line).get("verdict") or "?"
324
- except Exception:
325
- continue
326
- counts[v] = counts.get(v, 0) + 1
327
- total += 1
328
- except Exception:
329
- return {"total": total, "by_verdict": counts}
330
- return {"total": total, "by_verdict": counts}
331
-
332
-
333
- # ==========================================================================
334
- # 长期记忆快照(maintain.longterm)
335
- # ==========================================================================
336
- #
337
- # 目标:评估后**分层落盘**,形成可回溯的历史断面(哪一刻哪些记忆处于长期态)。
338
- # 与上面写入侧闸门的分工:闸门管「这条要不要记」,快照管「记住的现在稳稳站在哪一层」。
339
- #
340
- # 性能纪律:全部判据来自**索引快照**(免读节点文件),流式写 JSONL,不全量载入内存。
341
- # 4500+ 节点下 dry-run 为 O(N) 纯内存计算;apply 为顺序写文件。
342
-
343
- MAINTAIN_LOG = "_maintain.jsonl"
344
- LONGTERM_DIR = "_longterm"
345
- LONGTERM_KEEP = 10 # 保留最近 N 个断面(多了自动清理)
346
- TIERS = ("longterm", "working", "candidate")
347
- # 白箱可 ACCEPT 的基底档位;文科来源一致性档同样认账(否则「白箱判定已通过、
348
- # 长期分层却视作未验证」自相矛盾)。
349
- VERIFIED_BASES = ("formal_proof", "compiler", "test", "textbook", "public_kb")
350
- TIER_WORKING = 0.40
351
-
352
-
353
- # 生效条件:e 必填;protected 为真、importance>=PROTECT_IMPORTANCE、或 evidence_count>=3 且 verification_basis 在 VERIFIED_BASES → "longterm";importance>=TIER_WORKING 或 vb 在 VERIFIED_BASES → "working";否则 "candidate"。
354
- def _tier_of(e: dict) -> str:
355
- """索引快照 → 分层:longterm(长期)/ working(工作)/ candidate(候选待评估)。"""
356
- imp = float(e.get("importance", 0.5) or 0.5)
357
- vb = e.get("verification_basis")
358
- ev = int(e.get("evidence_count", 0) or 0)
359
- if e.get("protected") or imp >= PROTECT_IMPORTANCE or (ev >= 3 and vb in VERIFIED_BASES):
360
- return "longterm"
361
- if imp >= TIER_WORKING or vb in VERIFIED_BASES:
362
- return "working"
363
- return "candidate"
364
-
365
-
366
- # 生效条件:e 必填;e["edges"] 为假值(缺失/空列表)且 e["subgraph"] 为假值时返回 True,否则 False。
367
- def _is_island(e: dict) -> bool:
368
- """无边孤岛:既无出边也无子图声明(夜间整理的首要候选)。"""
369
- return (not (e.get("edges") or [])) and (not e.get("subgraph"))
370
-
371
-
372
- # 生效条件:cg 必填且提供 cg.root,恒返回 os.path.join(cg.root, LONGTERM_DIR)。
373
- def longterm_dir(cg) -> str:
374
- return os.path.join(cg.root, LONGTERM_DIR)
375
-
376
-
377
- # 生效条件:cg 必填,恒返回 longterm_dir(cg) 下的 "current.json" 路径。
378
- def current_path(cg) -> str:
379
- return os.path.join(longterm_dir(cg), "current.json")
380
-
381
-
382
- # 生效条件:apply 为真且由 cg.index 的 nodes(layer 为假值时不过滤、为真时仅取 layer 字段相等者,max_rows 为真值时先取 ids[:int(max_rows)])算出的 snapshot_id 与 current.json 所记 snapshot_id 不同或其记录的 path 文件不存在(same 为假)时,才写断面文件、原子更新 current 指针、执行 _prune 并追加维护日志;apply 为假时只返回 dry_run=True 的统计(out 与 force 在源码中未被引用)。
383
- def longterm_assess(cg, apply=False, out=None, layer=None, keep=LONGTERM_KEEP,
384
- max_rows=None, force=False, actor="maintain"):
385
- """评估后分层落盘:生成一个可回溯的长期记忆断面。
386
-
387
- apply=False(默认)只出报表;apply=True 写 `_longterm/<ts>.jsonl` 并更新
388
- `current.json` 指针。幂等:断面内容相同则跳过重写(除非 force=True)。
389
- """
390
- nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
391
- ids = sorted(nid for nid, e in nodes.items()
392
- if not layer or e.get("layer") == layer)
393
- if max_rows:
394
- ids = ids[:int(max_rows)]
395
- tiers, by_layer, islands, digest = {}, {}, 0, hashlib.sha1()
396
- t0 = time.time()
397
- total = len(ids)
398
- for nid in ids:
399
- e = nodes.get(nid) or {}
400
- t = _tier_of(e)
401
- tiers[t] = tiers.get(t, 0) + 1
402
- lay = e.get("layer") or "?"
403
- bl = by_layer.setdefault(lay, {k: 0 for k in TIERS})
404
- bl[t] += 1
405
- if _is_island(e):
406
- islands += 1
407
- digest.update(f"{nid}:{e.get('importance')}:{t};".encode("utf-8"))
408
- snapshot_id = digest.hexdigest()[:12]
409
- ts = time.strftime("%Y%m%d-%H%M%S")
410
- rel = f"{LONGTERM_DIR}/{ts}-{snapshot_id}.jsonl"
411
- path = os.path.join(cg.root, rel)
412
- cur = None
413
- try:
414
- with open(current_path(cg), encoding="utf-8") as f:
415
- cur = json.load(f)
416
- except (OSError, ValueError):
417
- cur = None
418
- same = bool(cur and cur.get("snapshot_id") == snapshot_id
419
- and os.path.exists(os.path.join(cg.root, cur.get("path") or "")))
420
- written, pruned = 0, []
421
- if apply and not same:
422
- d = longterm_dir(cg)
423
- os.makedirs(d, exist_ok=True)
424
- tmp = path + ".tmp"
425
- with open(tmp, "w", encoding="utf-8") as f:
426
- for nid in ids:
427
- e = nodes.get(nid) or {}
428
- row = {"t": time.time(), "id": nid, "layer": e.get("layer"),
429
- "importance": e.get("importance"),
430
- "tier": _tier_of(e),
431
- "verification_basis": e.get("verification_basis"),
432
- "evidence_count": e.get("evidence_count", 0),
433
- "edges": len(e.get("edges") or []),
434
- "island": _is_island(e),
435
- "content_hash": e.get("content_hash")}
436
- f.write(json.dumps(row, ensure_ascii=False) + "\n")
437
- written += 1
438
- os.replace(tmp, path)
439
- atomic_write(current_path(cg), json.dumps(
440
- {"snapshot_id": snapshot_id, "ts": time.time(), "path": rel,
441
- "total": total, "tiers": tiers, "by_layer": by_layer,
442
- "islands": islands}, ensure_ascii=False))
443
- pruned = _prune(cg, keep)
444
- append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
445
- "t": time.time(), "action": "longterm", "snapshot_id": snapshot_id,
446
- "path": rel, "total": total, "tiers": tiers, "actor": actor})
447
- return {
448
- "ok": True, "action": "longterm", "dry_run": not apply,
449
- "snapshot_id": snapshot_id, "path": rel, "same_as_current": same,
450
- "total": total, "tiers": tiers, "by_layer": by_layer,
451
- "islands": islands, "written": written, "pruned": pruned,
452
- "elapsed_ms": int((time.time() - t0) * 1000), "log": MAINTAIN_LOG,
453
- "note": ("dry-run:未写盘" if not apply else
454
- (f"断面与 current 相同,跳过重写(id={snapshot_id})" if same
455
- else f"已写断面 {rel}({written} 行)")),
456
- }
457
-
458
-
459
- # 生效条件:cg 与 keep 必填;keep<=0 时不删除任何断面返回 [];否则删除除最近 keep 个 .jsonl 外的旧断面。
460
- def _prune(cg, keep):
461
- """只保留最近 keep 个断面文件(按文件名时间前缀排序)。"""
462
- d = longterm_dir(cg)
463
- try:
464
- files = sorted(x for x in os.listdir(d) if x.endswith(".jsonl"))
465
- except OSError:
466
- return []
467
- removed = []
468
- for x in files[:-int(keep)] if int(keep) > 0 else []:
469
- try:
470
- os.remove(os.path.join(d, x))
471
- removed.append(x)
472
- except OSError:
473
- pass
474
- return removed
475
-
476
-
477
- # 生效条件:cg 必填,limit 默认 20;目录不可读返回 [];否则新的在前逐个 append,因先 append 后判 len(out)>=limit,limit=0 时仍返回 1 条快照。
478
- def longterm_list(cg, limit=20):
479
- """列出历史断面(新的在前):{snapshot_id, path, ts, total, tiers}。"""
480
- d = longterm_dir(cg)
481
- out = []
482
- try:
483
- for x in sorted(os.listdir(d), reverse=True):
484
- if not x.endswith(".jsonl"):
485
- continue
486
- p = os.path.join(d, x)
487
- out.append({"file": x, "path": f"{LONGTERM_DIR}/{x}",
488
- "bytes": os.path.getsize(p)})
489
- if len(out) >= int(limit):
490
- break
491
- except OSError:
492
- return []
493
- cur = None
494
- try:
495
- with open(current_path(cg), encoding="utf-8") as f:
496
- cur = json.load(f)
497
- except (OSError, ValueError):
498
- cur = None
499
- return {"current": cur, "snapshots": out}
500
-
501
-
502
- # 生效条件:longterm_dir(cg) 不可列出(OSError)时返回 {"ok":False,"error":"no_snapshot"};否则在倒序文件名中取首个满足 snapshot_id 为 None 或为其子串的 .jsonl(snapshot_id="" 与任意文件名匹配),无匹配返回 {"ok":False,"error":"snapshot_not_found"};命中则逐行聚合该文件(空行与 json.loads 抛 ValueError 的行跳过),返回 file/total/tiers/by_layer/islands。
503
- def longterm_show(cg, snapshot_id=None):
504
- """读取某个断面的分层统计(不载全量行,只聚合)。"""
505
- d = longterm_dir(cg)
506
- target = None
507
- try:
508
- files = sorted(x for x in os.listdir(d) if x.endswith(".jsonl"))
509
- except OSError:
510
- return {"ok": False, "error": "no_snapshot"}
511
- for x in reversed(files):
512
- if snapshot_id is None or snapshot_id in x:
513
- target = x
514
- break
515
- if not target:
516
- return {"ok": False, "error": "snapshot_not_found", "snapshot_id": snapshot_id}
517
- tiers, by_layer, islands, n = {}, {}, 0, 0
518
- with open(os.path.join(d, target), encoding="utf-8") as f:
519
- for line in f:
520
- line = line.strip()
521
- if not line:
522
- continue
523
- try:
524
- r = json.loads(line)
525
- except ValueError:
526
- continue
527
- n += 1
528
- t = r.get("tier") or "?"
529
- tiers[t] = tiers.get(t, 0) + 1
530
- lay = r.get("layer") or "?"
531
- by_layer.setdefault(lay, {k: 0 for k in TIERS})
532
- by_layer[lay][t] = by_layer[lay].get(t, 0) + 1
533
- if r.get("island"):
534
- islands += 1
535
- return {"ok": True, "file": target, "total": n, "tiers": tiers,
536
- "by_layer": by_layer, "islands": islands}
537
-
538
-
539
- # ==========================================================================
540
- # 海马体前馈(maintain.prefeed)
541
- # ==========================================================================
542
- #
543
- # 写入**之前**的新奇检测:重复项并入既有(MERGE),而非新增;无关噪音丢弃;
544
- # 有歧义的半重复留痕待复核。这是「写入侧前置」的落库动作,比夜间整理更早一步。
545
-
546
- # 生效条件:cg 与 content 必填;恒经 assess 得四态并映射 decision(ACCEPT→write 等),落留痕后返回 ok=True,不写任何节点。
547
- def prefeed(cg, content, layer="contextual", role=None, verification_basis=None,
548
- importance_hint=None, node_id=None):
549
- """前馈裁决(不写盘):返回四态 + 判据,并留痕 `_forgetting.jsonl`。
550
-
551
- 落库由调用方按 verdict 执行(ACCEPT 新增 / MERGE 强化 / DROP|DEFER 不写),
552
- 使「裁决」与「落库」解耦——便于 dry-run 预演与单测。
553
- """
554
- vd = assess(cg, content, layer=layer, role=role,
555
- verification_basis=verification_basis,
556
- importance_hint=importance_hint, node_id=node_id)
557
- decision = {"ACCEPT": "write", "MERGE": "reinforce",
558
- "DROP": "discard", "DEFER": "defer"}.get(vd["verdict"], "defer")
559
- rec = {"kind": "prefeed", "layer": layer,
560
- "node_id": node_id or _prefeed_id(content),
561
- "verdict": vd["verdict"], "decision": decision,
562
- "reason": vd["reason"], "duplicate_with": vd["redundancy"]["with"],
563
- "duplicate_ratio": vd["redundancy"]["max"],
564
- "novelty": vd["entropy"]["novelty"],
565
- "self_information_bits": vd["entropy"]["self_information_bits"],
566
- "actor": getattr(cg, "actor", "unknown"), "t": time.time()}
567
- log(cg, rec)
568
- return {"ok": True, "action": "prefeed", **rec}
569
-
570
-
571
- # 生效条件:content 为 None 或假值时按 "" 计算,恒返回 "pre_"+sha1(content).hexdigest()[:12]。
572
- def _prefeed_id(content):
573
- return "pre_" + hashlib.sha1((content or "").encode("utf-8")).hexdigest()[:12]
574
-
575
-
576
- # 生效条件:cg 必填,limit 默认 100;action 为假值(None/空串)时不过滤,真值只留该 action;返回 recs[-int(limit):],limit=0 时退化为全量。
577
- def maintain_history(cg, limit=100, action=None):
578
- """维护留痕(`_maintain.jsonl` 最近 limit 条),可按 action 过滤。"""
579
- recs = list(read_jsonl(os.path.join(cg.root, MAINTAIN_LOG)))
580
- if action:
581
- recs = [r for r in recs if r.get("action") == action]
1
+ # -*- coding: utf-8 -*-
2
+ """md_cg · 主动遗忘闸门(写入情景层前的三问筛选)
3
+
4
+ 理论出处(全部来自本仓已有文档):
5
+
6
+ · `memory_score.md:12`
7
+ J 判断引擎 9-10 档 = 「独立元认知 + **主动遗忘**」;灵枢正因「无主动遗忘」
8
+ 停在 8.0。→ 主动遗忘是 J 维上 9 分的门槛项,不是可选优化。
9
+ · `docs/白箱智能系列·第五篇:174-180`
10
+ 「把经历兑换成结构…整理完之后记忆库变小了,但信息量反而更可用——
11
+ 噪音被扔掉了,骨架被留下。」
12
+ · AEIS 工具表 `docs/mdcg/tool_table_v0.3.0.md:15-19`
13
+ `prefeed`(H1 新奇检测 → 高新奇输入当场强化编码)、
14
+ `pattern_separation`(H3 扫描相似节点对)、
15
+ `nightly_cleanup`(知识层夜间整理、无边孤岛降级)。
16
+ 本模块 = 这三件事的**写入侧前置版**:不等夜间整理,写之前就裁决。
17
+ · `docs/theory/智能的公理化基石.md:758-763` —— **诚实边界**
18
+ 「信息差与热力学熵之间只能进行结构类比,不应宣称数学同构」。
19
+ 故本模块一律称「自信息代理 / 惊奇度」,**不称香农熵**,也不做熵的物理断言。
20
+
21
+ 三问 → 四态裁决(对齐白箱四态,落库动作分四种):
22
+
23
+ Q1 重复? redundancy = 新内容被既有同层节点覆盖的最大比例(bigram 覆盖率)
24
+ Q2 重要? importance = 显式 hint 优先,否则启发式(新奇/来源/长度)
25
+ Q3 惊奇? self_info = -log2(dup + ε)(bit,**代理量**,非香农熵)
26
+
27
+ ACCEPT 写入 / MERGE 并入既有(不新增,强化既有节点)
28
+ DROP 丢弃 / DEFER 待定(不写,留痕待复核)
29
+
30
+ 裁决顺序(**顺序即语义**):
31
+ 1) 重要度 ≥0.7 → ACCEPT(保护优先)
32
+ 2) 确定性内部产生 且 冗余 → DROP ← 先于 MERGE:机器例行输出再"重复"也只是
33
+ 噪音,不该去强化既有记忆(否则例行日志
34
+ 会把普通记忆刷成高重要性)
35
+ 3) 冗余 ≥0.85 → MERGE ← 外部/未知来源的重复 = 又一次确认,强化
36
+ 4) 半重复 且 不重要 → DEFER
37
+ 5) 重要度 ≥0.30 → ACCEPT
38
+ 6) 新信息 ≥0.15 → ACCEPT
39
+ 7) 其余 → DEFER
40
+
41
+ 一切裁决都写进 `_forgetting.jsonl`(append-only),可审计:
42
+ 「这条为什么没被记住」和「为什么被记住」同样有据可查。
43
+ """
44
+ import hashlib
45
+ import json
46
+ import math
47
+ import os
48
+ import time
49
+
50
+ from . import lifecycle, nodefile
51
+ from .fsutil import append_jsonl, atomic_write, read_jsonl
52
+ from .mdcg import bigrams
53
+
54
+ # ---------------------------------------------------------------- 判据常量
55
+
56
+ DUP_MERGE = 0.85 # 重复度 ≥ 此值 → MERGE
57
+ DUP_DROP = 0.60 # 重复度 ≥ 此值 → 进入 DROP / DEFER 判据
58
+ NOVELTY_MIN = 0.15 # 新信息 < 此值 → 视为无新信息
59
+ IMPORTANCE_MIN = 0.30 # 重要度 < 此值 → 不予写入
60
+ PROTECT_IMPORTANCE = 0.70 # 对齐 tool_table:≥0.7 触发不可遗忘保护
61
+ MAX_BITS = 4.0 # 自信息归一化上限(dup=0 时 4.0 bit)
62
+ EPS = 0.0625 # 自信息平滑(避免 dup=0 时取 log(0))
63
+ MAX_COMPARE = 240 # 单次重复检测最多比对的同层节点数(写入非热路径)
64
+
65
+ # 来源类型 → 权重(确定性内部产生 = 低权;外部惊奇 = 高权)
66
+ SOURCE_WEIGHT = {
67
+ "external_surprising": 1.00,
68
+ "unknown": 0.60,
69
+ "self_generated": 0.50,
70
+ "internal_deterministic": 0.25,
71
+ }
72
+ EXTERNAL_ROLES = ("user",)
73
+ INTERNAL_ROLES = ("command", "tool-output", "edit", "system")
74
+ # 注意:文科的 textbook/public_kb **不在此列**——它们是「权威来源表述一致」,
75
+ # 不是「内部确定性产生」,故仍按外部来源计权(见 source_kind)。
76
+ DETERMINISTIC_BASIS = ("data", "measurement", "compiler", "test", "formal_proof")
77
+
78
+ LOG_FILE = "_forgetting.jsonl"
79
+
80
+
81
+ # ---------------------------------------------------------------- 三问
82
+
83
+ # 生效条件:role 与 verification_basis 各自经 str(x or "").strip().lower() 后按序判——role 命中模块常量 EXTERNAL_ROLES 返回 "external_surprising";否则 role 命中 INTERNAL_ROLES、或两者都不命中前者时 verification_basis 命中 DETERMINISTIC_BASIS,返回 "internal_deterministic";否则 role 为 "assistant"/"agent" 返回 "self_generated";全不命中返回 "unknown"。
84
+ def source_kind(role=None, verification_basis=None):
85
+ """Q3 的来源面:内部确定性产生 vs 外部惊奇来源。"""
86
+ r = str(role or "").strip().lower()
87
+ vb = str(verification_basis or "").strip().lower()
88
+ if r in EXTERNAL_ROLES:
89
+ return "external_surprising"
90
+ if r in INTERNAL_ROLES:
91
+ return "internal_deterministic"
92
+ if vb in DETERMINISTIC_BASIS:
93
+ return "internal_deterministic"
94
+ if r in ("assistant", "agent"):
95
+ return "self_generated"
96
+ return "unknown"
97
+
98
+
99
+ # 生效条件:new_grams 为空集(假值)时返回 0.0;非空时返回 len(new_grams & body_grams)/len(new_grams)。
100
+ def _coverage(new_grams, body_grams):
101
+ if not new_grams:
102
+ return 0.0
103
+ return len(new_grams & body_grams) / float(len(new_grams))
104
+
105
+
106
+ # CCG 五要素的固定标签:所有节点都一样,属**模板骨架而非内容**。
107
+ # 不剥离它们,任何两条记忆都会因共享 `# 功能名:`/`# 生效条件:` 而虚高重复度
108
+ # (实测:两条毫不相关的记忆 dup≈0.33,全部来自模板)。故重复检测只看"值"。
109
+ _TEMPLATE_LABELS = ("功能名", "生效条件", "子功能", "执行", "验证方式", "不适用条件")
110
+
111
+
112
+ # 生效条件:content 为 None 或假值时按 "" 处理,结果为空串;否则逐行剥离 "#" 与 _TEMPLATE_LABELS 标签后以 "" 直接拼接。
113
+ def payload(content):
114
+ """剥离 CCG 固定标签后的**内容骨架**(保留字段值,丢弃字段名与标记)。"""
115
+ out = []
116
+ for line in (content or "").splitlines():
117
+ s = line.strip()
118
+ if s.startswith("#"):
119
+ s = s.lstrip("#").strip()
120
+ for lab in _TEMPLATE_LABELS:
121
+ if s.startswith(lab):
122
+ s = s[len(lab):].lstrip(":: ").strip()
123
+ break
124
+ if s:
125
+ out.append(s)
126
+ return "".join(out)
127
+
128
+
129
+ # 生效条件:content 经 payload/bigrams 得空集合时直接返回零值 best(max=0.0、with=None、compared=0);否则遍历 cg.index 的 nodes,跳过 nid==exclude,layer 为真值时只比较 str(layer 字段 or "")==layer 的节点,cg.get(nid) 抛异常/返回假值、或该节点 content 的 bigrams 为空则跳过,每计入一个节点后若 n>=limit 立即 break(故 limit 为 0 或负数时只比较首项即停),返回覆盖度最大者 best(无覆盖度提升时不更新 with/jaccard,compared 为实际计入数)。
130
+ def redundancy(cg, content, layer="contextual", exclude=None, limit=MAX_COMPARE):
131
+ """Q1 重复?——新内容被既有同层节点覆盖的最大比例。"""
132
+ new = bigrams(payload(content))
133
+ best = {"max": 0.0, "with": None, "jaccard": 0.0, "compared": 0}
134
+ if not new:
135
+ return best
136
+ nodes = ((getattr(cg, "index", None) or {}).get("nodes") or {})
137
+ n = 0
138
+ for nid in list(nodes.keys()):
139
+ if nid == exclude:
140
+ continue
141
+ if layer and str(nodes[nid].get("layer") or "") != layer:
142
+ continue
143
+ try:
144
+ node = cg.get(nid)
145
+ except Exception:
146
+ node = None
147
+ if not node:
148
+ continue
149
+ body = bigrams(payload(node.get("content") or ""))
150
+ if not body:
151
+ continue
152
+ n += 1
153
+ cov = _coverage(new, body)
154
+ if cov > best["max"]:
155
+ best = {"max": cov, "with": nid,
156
+ "jaccard": len(new & body) / float(len(new | body) or 1),
157
+ "compared": n}
158
+ if n >= limit:
159
+ break
160
+ best["compared"] = n
161
+ return best
162
+
163
+
164
+ # 生效条件:dup 必填并转 float;dup=0 时 p 取 EPS,返回 -log2(EPS) 这一有限大值;dup>=1 时返回 0.0。
165
+ def self_information(dup):
166
+ """Q3 的自信息代理:I = -log2(min(1, dup + ε)),单位 bit。
167
+
168
+ 注意:dup 是「被既有记忆覆盖率」的估计,不是概率模型的真实 P(x),
169
+ 因此这是**结构类比的代理量**(见模块 docstring 的诚实边界)。
170
+ """
171
+ p = min(1.0, max(0.0, float(dup)) + EPS)
172
+ return -math.log(p, 2.0)
173
+
174
+
175
+ # 生效条件:hint 非 None 且可转 float(含 hint=0)时返回 from="hint" 的裁剪分数;否则用 novelty、SOURCE_WEIGHT.get(kind, SOURCE_WEIGHT["unknown"])、len(content)/200 三因子启发式。
176
+ def importance_score(hint, novelty, kind, content):
177
+ """Q2 重要?——显式 hint 优先,否则启发式(对齐 longterm_snapshot 四因子简化版)。"""
178
+ if hint is not None:
179
+ try:
180
+ return {"score": round(max(0.0, min(1.0, float(hint))), 4),
181
+ "from": "hint"}
182
+ except (TypeError, ValueError):
183
+ pass
184
+ lf = min(1.0, len(content or "") / 200.0)
185
+ s = (0.5 * novelty
186
+ + 0.3 * SOURCE_WEIGHT.get(kind, SOURCE_WEIGHT["unknown"])
187
+ + 0.2 * lf)
188
+ return {"score": round(max(0.0, min(1.0, s)), 4), "from": "heuristic"}
189
+
190
+
191
+ # 生效条件:以 source_kind(role,verification_basis) 的 kind 与 redundancy(cg,content,layer=layer,exclude=node_id) 的 red["max"] 为输入,按 if/elif 顺序取首个命中分支——imp["score"]≥PROTECT_IMPORTANCE→"ACCEPT";否则 kind=="internal_deterministic" 且 red["max"]≥DUP_DROP→"DROP";否则 red["max"]≥DUP_MERGE→"MERGE";否则 red["max"]≥DUP_DROP 且 imp["score"]<IMPORTANCE_MIN→"DEFER";否则 imp["score"]≥IMPORTANCE_MIN→"ACCEPT";否则 novelty≥NOVELTY_MIN→"ACCEPT";否则→"DEFER"。
192
+ def assess(cg, content, layer="contextual", role=None, verification_basis=None,
193
+ importance_hint=None, node_id=None):
194
+ """三问 → 四态裁决。返回完整判据(可审计,不只给结论)。"""
195
+ kind = source_kind(role, verification_basis)
196
+ red = redundancy(cg, content, layer=layer, exclude=node_id)
197
+ novelty = round(1.0 - red["max"], 4)
198
+ bits = round(self_information(red["max"]), 4)
199
+ imp = importance_score(importance_hint, novelty, kind, content)
200
+ entropy = {
201
+ "source_kind": kind,
202
+ "novelty": novelty,
203
+ "self_information_bits": bits,
204
+ "normalized": round(min(1.0, bits / MAX_BITS), 4),
205
+ "duplicate_with": red["with"],
206
+ "duplicate_ratio": round(red["max"], 4),
207
+ "compared": red["compared"],
208
+ }
209
+
210
+ if imp["score"] >= PROTECT_IMPORTANCE:
211
+ verdict, why = "ACCEPT", (f"重要度 {imp['score']:.2f}≥{PROTECT_IMPORTANCE}"
212
+ f"(触发不可遗忘保护)")
213
+ elif kind == "internal_deterministic" and red["max"] >= DUP_DROP:
214
+ verdict, why = "DROP", (f"确定性内部产生且冗余 {red['max']:.2f}≥{DUP_DROP}"
215
+ f"(低熵噪音,不编码)")
216
+ elif red["max"] >= DUP_MERGE:
217
+ verdict, why = "MERGE", (f"重复度 {red['max']:.2f}≥{DUP_MERGE}"
218
+ f"(并入 {red['with']},强化既有)")
219
+ elif red["max"] >= DUP_DROP and imp["score"] < IMPORTANCE_MIN:
220
+ verdict, why = "DEFER", (f"半重复 {red['max']:.2f}∈[{DUP_DROP},{DUP_MERGE})"
221
+ f" 且重要度 {imp['score']:.2f}<{IMPORTANCE_MIN}"
222
+ f"(待定复核)")
223
+ elif imp["score"] >= IMPORTANCE_MIN:
224
+ verdict, why = "ACCEPT", f"重要度 {imp['score']:.2f}≥{IMPORTANCE_MIN}"
225
+ elif novelty >= NOVELTY_MIN:
226
+ verdict, why = "ACCEPT", f"新信息 {novelty:.2f}≥{NOVELTY_MIN}"
227
+ else:
228
+ verdict, why = "DEFER", "重要度与新信息均不足判据(待定)"
229
+
230
+ return {"verdict": verdict, "reason": why, "redundancy": red,
231
+ "importance": imp, "entropy": entropy}
232
+
233
+
234
+ # ---------------------------------------------------------------- 落库动作
235
+
236
+ # 生效条件:cg 与 rec 必填;append_jsonl 写 cg.root/LOG_FILE 抛任意异常时被吞掉,仍返回 rec。
237
+ def log(cg, rec):
238
+ """裁决留痕(append-only)。DROP/DEFER 也留痕——否则遗忘变黑箱。"""
239
+ try:
240
+ append_jsonl(os.path.join(cg.root, LOG_FILE), rec)
241
+ except Exception:
242
+ pass
243
+ return rec
244
+
245
+
246
+ # 生效条件:cg 与 node_id 必填,delta 默认 0.05;cg.get(node_id) 抛异常或返回假值时返回 None;imp 跨过 PROTECT_IMPORTANCE 即写 protected。
247
+ def reinforce(cg, node_id, delta=0.05):
248
+ """MERGE 的落库动作:不新增节点,把「又一次见到」折算成既有节点的强化。
249
+
250
+ 重要性 +delta,merge_count +1;一旦跨过 0.7 自动打上保护标记
251
+ (对齐「importance 提升(保护:不可遗忘…且受保护标记)」)。
252
+ """
253
+ try:
254
+ node = cg.get(node_id)
255
+ except Exception:
256
+ node = None
257
+ if not node:
258
+ return None
259
+ fm = node.get("frontmatter") or {}
260
+ imp = min(1.0, float(fm.get("importance") or 0.5) + delta)
261
+ fm["importance"] = imp
262
+ fm["merge_count"] = int(fm.get("merge_count") or 0) + 1
263
+ fm["last_merge_at"] = time.time()
264
+ if imp >= PROTECT_IMPORTANCE:
265
+ fm["protected"] = True
266
+ fm["protection_reason"] = (f"importance={imp:.2f}≥{PROTECT_IMPORTANCE}"
267
+ f"(重复强化)")
268
+ # ② 显式状态机收口(2026-09-16):MERGE 的语义是「又一次见到」= **重新激活**
269
+ # 信号——已降权(demoted)/已定型(converged)的节点经状态机**逐级回升**到
270
+ # active(archived→active 亦合法,归档节点被再次见到即恢复参与);active 为
271
+ # 幂等 no-op(不写字段、不留痕)。protected 只豁免**降级**,回升不受限。
272
+ lifecycle.stamp(fm, "active", reason="MERGE 重复强化(回升)",
273
+ actor="forgetting:reinforce")
274
+ cg._write_node(node_id, os.path.join(cg.root, node["path"]),
275
+ fm, node.get("content") or "")
276
+ e = ((getattr(cg, "index", None) or {}).get("nodes") or {}).get(node_id)
277
+ if e is not None:
278
+ e["importance"] = imp
279
+ if fm.get(lifecycle.STATE_FIELD):
280
+ e[lifecycle.STATE_FIELD] = fm[lifecycle.STATE_FIELD]
281
+ if fm.get("protected"):
282
+ e["protected"] = True
283
+ e["protection_reason"] = fm["protection_reason"]
284
+ return {"node_id": node_id, "importance": imp,
285
+ "merge_count": fm["merge_count"], "protected": bool(fm.get("protected"))}
286
+
287
+
288
+ # 生效条件:cg 必填,limit 默认 100;日志路径不存在时返回 [];否则返回 out[-limit:],limit=0 时 -0 退化为 out[0:] 即全量。
289
+ def history(cg, limit=100):
290
+ """读取遗忘留痕(最近 limit 条)。"""
291
+ p = os.path.join(cg.root, LOG_FILE)
292
+ if not os.path.exists(p):
293
+ return []
294
+ out = []
295
+ try:
296
+ with open(p, "r", encoding="utf-8") as f:
297
+ for line in f:
298
+ line = line.strip()
299
+ if line:
300
+ try:
301
+ out.append(__import__("json").loads(line))
302
+ except Exception:
303
+ continue
304
+ except Exception:
305
+ return []
306
+ return out[-limit:]
307
+
308
+
309
+ # 生效条件:cg 必填;日志路径不存在返回 {"total": 0, "by_verdict": {}};否则流式累计行数与 verdict 分布。
310
+ def summary(cg):
311
+ """遗忘留痕聚合(流式,不把全量日志读进内存):总数 + 四态分布。"""
312
+ p = os.path.join(cg.root, LOG_FILE)
313
+ counts, total = {}, 0
314
+ if not os.path.exists(p):
315
+ return {"total": 0, "by_verdict": {}}
316
+ try:
317
+ with open(p, "r", encoding="utf-8") as f:
318
+ for line in f:
319
+ line = line.strip()
320
+ if not line:
321
+ continue
322
+ try:
323
+ v = json.loads(line).get("verdict") or "?"
324
+ except Exception:
325
+ continue
326
+ counts[v] = counts.get(v, 0) + 1
327
+ total += 1
328
+ except Exception:
329
+ return {"total": total, "by_verdict": counts}
330
+ return {"total": total, "by_verdict": counts}
331
+
332
+
333
+ # ==========================================================================
334
+ # 长期记忆快照(maintain.longterm)
335
+ # ==========================================================================
336
+ #
337
+ # 目标:评估后**分层落盘**,形成可回溯的历史断面(哪一刻哪些记忆处于长期态)。
338
+ # 与上面写入侧闸门的分工:闸门管「这条要不要记」,快照管「记住的现在稳稳站在哪一层」。
339
+ #
340
+ # 性能纪律:全部判据来自**索引快照**(免读节点文件),流式写 JSONL,不全量载入内存。
341
+ # 4500+ 节点下 dry-run 为 O(N) 纯内存计算;apply 为顺序写文件。
342
+
343
+ MAINTAIN_LOG = "_maintain.jsonl"
344
+ LONGTERM_DIR = "_longterm"
345
+ LONGTERM_KEEP = 10 # 保留最近 N 个断面(多了自动清理)
346
+ TIERS = ("longterm", "working", "candidate")
347
+ # 白箱可 ACCEPT 的基底档位;文科来源一致性档同样认账(否则「白箱判定已通过、
348
+ # 长期分层却视作未验证」自相矛盾)。
349
+ VERIFIED_BASES = ("formal_proof", "compiler", "test", "textbook", "public_kb")
350
+ TIER_WORKING = 0.40
351
+
352
+
353
+ # 生效条件:e 必填;protected 为真、importance>=PROTECT_IMPORTANCE、或 evidence_count>=3 且 verification_basis 在 VERIFIED_BASES → "longterm";importance>=TIER_WORKING 或 vb 在 VERIFIED_BASES → "working";否则 "candidate"。
354
+ def _tier_of(e: dict) -> str:
355
+ """索引快照 → 分层:longterm(长期)/ working(工作)/ candidate(候选待评估)。"""
356
+ imp = float(e.get("importance", 0.5) or 0.5)
357
+ vb = e.get("verification_basis")
358
+ ev = int(e.get("evidence_count", 0) or 0)
359
+ if e.get("protected") or imp >= PROTECT_IMPORTANCE or (ev >= 3 and vb in VERIFIED_BASES):
360
+ return "longterm"
361
+ if imp >= TIER_WORKING or vb in VERIFIED_BASES:
362
+ return "working"
363
+ return "candidate"
364
+
365
+
366
+ # 生效条件:e 必填;e["edges"] 为假值(缺失/空列表)且 e["subgraph"] 为假值时返回 True,否则 False。
367
+ def _is_island(e: dict) -> bool:
368
+ """无边孤岛:既无出边也无子图声明(夜间整理的首要候选)。"""
369
+ return (not (e.get("edges") or [])) and (not e.get("subgraph"))
370
+
371
+
372
+ # 生效条件:cg 必填且提供 cg.root,恒返回 os.path.join(cg.root, LONGTERM_DIR)。
373
+ def longterm_dir(cg) -> str:
374
+ return os.path.join(cg.root, LONGTERM_DIR)
375
+
376
+
377
+ # 生效条件:cg 必填,恒返回 longterm_dir(cg) 下的 "current.json" 路径。
378
+ def current_path(cg) -> str:
379
+ return os.path.join(longterm_dir(cg), "current.json")
380
+
381
+
382
+ # 生效条件:apply 为真且由 cg.index 的 nodes(layer 为假值时不过滤、为真时仅取 layer 字段相等者,max_rows 为真值时先取 ids[:int(max_rows)])算出的 snapshot_id 与 current.json 所记 snapshot_id 不同或其记录的 path 文件不存在(same 为假)时,才写断面文件、原子更新 current 指针、执行 _prune 并追加维护日志;apply 为假时只返回 dry_run=True 的统计(out 与 force 在源码中未被引用)。
383
+ def longterm_assess(cg, apply=False, out=None, layer=None, keep=LONGTERM_KEEP,
384
+ max_rows=None, force=False, actor="maintain"):
385
+ """评估后分层落盘:生成一个可回溯的长期记忆断面。
386
+
387
+ apply=False(默认)只出报表;apply=True 写 `_longterm/<ts>.jsonl` 并更新
388
+ `current.json` 指针。幂等:断面内容相同则跳过重写(除非 force=True)。
389
+ """
390
+ nodes = (getattr(cg, "index", None) or {}).get("nodes") or {}
391
+ ids = sorted(nid for nid, e in nodes.items()
392
+ if not layer or e.get("layer") == layer)
393
+ if max_rows:
394
+ ids = ids[:int(max_rows)]
395
+ tiers, by_layer, islands, digest = {}, {}, 0, hashlib.sha1()
396
+ t0 = time.time()
397
+ total = len(ids)
398
+ for nid in ids:
399
+ e = nodes.get(nid) or {}
400
+ t = _tier_of(e)
401
+ tiers[t] = tiers.get(t, 0) + 1
402
+ lay = e.get("layer") or "?"
403
+ bl = by_layer.setdefault(lay, {k: 0 for k in TIERS})
404
+ bl[t] += 1
405
+ if _is_island(e):
406
+ islands += 1
407
+ digest.update(f"{nid}:{e.get('importance')}:{t};".encode("utf-8"))
408
+ snapshot_id = digest.hexdigest()[:12]
409
+ ts = time.strftime("%Y%m%d-%H%M%S")
410
+ rel = f"{LONGTERM_DIR}/{ts}-{snapshot_id}.jsonl"
411
+ path = os.path.join(cg.root, rel)
412
+ cur = None
413
+ try:
414
+ with open(current_path(cg), encoding="utf-8") as f:
415
+ cur = json.load(f)
416
+ except (OSError, ValueError):
417
+ cur = None
418
+ same = bool(cur and cur.get("snapshot_id") == snapshot_id
419
+ and os.path.exists(os.path.join(cg.root, cur.get("path") or "")))
420
+ written, pruned = 0, []
421
+ if apply and not same:
422
+ d = longterm_dir(cg)
423
+ os.makedirs(d, exist_ok=True)
424
+ tmp = path + ".tmp"
425
+ with open(tmp, "w", encoding="utf-8") as f:
426
+ for nid in ids:
427
+ e = nodes.get(nid) or {}
428
+ row = {"t": time.time(), "id": nid, "layer": e.get("layer"),
429
+ "importance": e.get("importance"),
430
+ "tier": _tier_of(e),
431
+ "verification_basis": e.get("verification_basis"),
432
+ "evidence_count": e.get("evidence_count", 0),
433
+ "edges": len(e.get("edges") or []),
434
+ "island": _is_island(e),
435
+ "content_hash": e.get("content_hash")}
436
+ f.write(json.dumps(row, ensure_ascii=False) + "\n")
437
+ written += 1
438
+ os.replace(tmp, path)
439
+ atomic_write(current_path(cg), json.dumps(
440
+ {"snapshot_id": snapshot_id, "ts": time.time(), "path": rel,
441
+ "total": total, "tiers": tiers, "by_layer": by_layer,
442
+ "islands": islands}, ensure_ascii=False))
443
+ pruned = _prune(cg, keep)
444
+ append_jsonl(os.path.join(cg.root, MAINTAIN_LOG), {
445
+ "t": time.time(), "action": "longterm", "snapshot_id": snapshot_id,
446
+ "path": rel, "total": total, "tiers": tiers, "actor": actor})
447
+ return {
448
+ "ok": True, "action": "longterm", "dry_run": not apply,
449
+ "snapshot_id": snapshot_id, "path": rel, "same_as_current": same,
450
+ "total": total, "tiers": tiers, "by_layer": by_layer,
451
+ "islands": islands, "written": written, "pruned": pruned,
452
+ "elapsed_ms": int((time.time() - t0) * 1000), "log": MAINTAIN_LOG,
453
+ "note": ("dry-run:未写盘" if not apply else
454
+ (f"断面与 current 相同,跳过重写(id={snapshot_id})" if same
455
+ else f"已写断面 {rel}({written} 行)")),
456
+ }
457
+
458
+
459
+ # 生效条件:cg 与 keep 必填;keep<=0 时不删除任何断面返回 [];否则删除除最近 keep 个 .jsonl 外的旧断面。
460
+ def _prune(cg, keep):
461
+ """只保留最近 keep 个断面文件(按文件名时间前缀排序)。"""
462
+ d = longterm_dir(cg)
463
+ try:
464
+ files = sorted(x for x in os.listdir(d) if x.endswith(".jsonl"))
465
+ except OSError:
466
+ return []
467
+ removed = []
468
+ for x in files[:-int(keep)] if int(keep) > 0 else []:
469
+ try:
470
+ os.remove(os.path.join(d, x))
471
+ removed.append(x)
472
+ except OSError:
473
+ pass
474
+ return removed
475
+
476
+
477
+ # 生效条件:cg 必填,limit 默认 20;目录不可读返回 [];否则新的在前逐个 append,因先 append 后判 len(out)>=limit,limit=0 时仍返回 1 条快照。
478
+ def longterm_list(cg, limit=20):
479
+ """列出历史断面(新的在前):{snapshot_id, path, ts, total, tiers}。"""
480
+ d = longterm_dir(cg)
481
+ out = []
482
+ try:
483
+ for x in sorted(os.listdir(d), reverse=True):
484
+ if not x.endswith(".jsonl"):
485
+ continue
486
+ p = os.path.join(d, x)
487
+ out.append({"file": x, "path": f"{LONGTERM_DIR}/{x}",
488
+ "bytes": os.path.getsize(p)})
489
+ if len(out) >= int(limit):
490
+ break
491
+ except OSError:
492
+ return []
493
+ cur = None
494
+ try:
495
+ with open(current_path(cg), encoding="utf-8") as f:
496
+ cur = json.load(f)
497
+ except (OSError, ValueError):
498
+ cur = None
499
+ return {"current": cur, "snapshots": out}
500
+
501
+
502
+ # 生效条件:longterm_dir(cg) 不可列出(OSError)时返回 {"ok":False,"error":"no_snapshot"};否则在倒序文件名中取首个满足 snapshot_id 为 None 或为其子串的 .jsonl(snapshot_id="" 与任意文件名匹配),无匹配返回 {"ok":False,"error":"snapshot_not_found"};命中则逐行聚合该文件(空行与 json.loads 抛 ValueError 的行跳过),返回 file/total/tiers/by_layer/islands。
503
+ def longterm_show(cg, snapshot_id=None):
504
+ """读取某个断面的分层统计(不载全量行,只聚合)。"""
505
+ d = longterm_dir(cg)
506
+ target = None
507
+ try:
508
+ files = sorted(x for x in os.listdir(d) if x.endswith(".jsonl"))
509
+ except OSError:
510
+ return {"ok": False, "error": "no_snapshot"}
511
+ for x in reversed(files):
512
+ if snapshot_id is None or snapshot_id in x:
513
+ target = x
514
+ break
515
+ if not target:
516
+ return {"ok": False, "error": "snapshot_not_found", "snapshot_id": snapshot_id}
517
+ tiers, by_layer, islands, n = {}, {}, 0, 0
518
+ with open(os.path.join(d, target), encoding="utf-8") as f:
519
+ for line in f:
520
+ line = line.strip()
521
+ if not line:
522
+ continue
523
+ try:
524
+ r = json.loads(line)
525
+ except ValueError:
526
+ continue
527
+ n += 1
528
+ t = r.get("tier") or "?"
529
+ tiers[t] = tiers.get(t, 0) + 1
530
+ lay = r.get("layer") or "?"
531
+ by_layer.setdefault(lay, {k: 0 for k in TIERS})
532
+ by_layer[lay][t] = by_layer[lay].get(t, 0) + 1
533
+ if r.get("island"):
534
+ islands += 1
535
+ return {"ok": True, "file": target, "total": n, "tiers": tiers,
536
+ "by_layer": by_layer, "islands": islands}
537
+
538
+
539
+ # ==========================================================================
540
+ # 海马体前馈(maintain.prefeed)
541
+ # ==========================================================================
542
+ #
543
+ # 写入**之前**的新奇检测:重复项并入既有(MERGE),而非新增;无关噪音丢弃;
544
+ # 有歧义的半重复留痕待复核。这是「写入侧前置」的落库动作,比夜间整理更早一步。
545
+
546
+ # 生效条件:cg 与 content 必填;恒经 assess 得四态并映射 decision(ACCEPT→write 等),落留痕后返回 ok=True,不写任何节点。
547
+ def prefeed(cg, content, layer="contextual", role=None, verification_basis=None,
548
+ importance_hint=None, node_id=None):
549
+ """前馈裁决(不写盘):返回四态 + 判据,并留痕 `_forgetting.jsonl`。
550
+
551
+ 落库由调用方按 verdict 执行(ACCEPT 新增 / MERGE 强化 / DROP|DEFER 不写),
552
+ 使「裁决」与「落库」解耦——便于 dry-run 预演与单测。
553
+ """
554
+ vd = assess(cg, content, layer=layer, role=role,
555
+ verification_basis=verification_basis,
556
+ importance_hint=importance_hint, node_id=node_id)
557
+ decision = {"ACCEPT": "write", "MERGE": "reinforce",
558
+ "DROP": "discard", "DEFER": "defer"}.get(vd["verdict"], "defer")
559
+ rec = {"kind": "prefeed", "layer": layer,
560
+ "node_id": node_id or _prefeed_id(content),
561
+ "verdict": vd["verdict"], "decision": decision,
562
+ "reason": vd["reason"], "duplicate_with": vd["redundancy"]["with"],
563
+ "duplicate_ratio": vd["redundancy"]["max"],
564
+ "novelty": vd["entropy"]["novelty"],
565
+ "self_information_bits": vd["entropy"]["self_information_bits"],
566
+ "actor": getattr(cg, "actor", "unknown"), "t": time.time()}
567
+ log(cg, rec)
568
+ return {"ok": True, "action": "prefeed", **rec}
569
+
570
+
571
+ # 生效条件:content 为 None 或假值时按 "" 计算,恒返回 "pre_"+sha1(content).hexdigest()[:12]。
572
+ def _prefeed_id(content):
573
+ return "pre_" + hashlib.sha1((content or "").encode("utf-8")).hexdigest()[:12]
574
+
575
+
576
+ # 生效条件:cg 必填,limit 默认 100;action 为假值(None/空串)时不过滤,真值只留该 action;返回 recs[-int(limit):],limit=0 时退化为全量。
577
+ def maintain_history(cg, limit=100, action=None):
578
+ """维护留痕(`_maintain.jsonl` 最近 limit 条),可按 action 过滤。"""
579
+ recs = list(read_jsonl(os.path.join(cg.root, MAINTAIN_LOG)))
580
+ if action:
581
+ recs = [r for r in recs if r.get("action") == action]
582
582
  return recs[-int(limit):]