@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/sources.py CHANGED
@@ -1,583 +1,816 @@
1
- # -*- coding: utf-8 -*-
2
- """md_cg · 设备驱动(记忆 OS #3):把外部会话流接进认知图
3
-
4
- 设计:
5
- Source(事件源)—— 把某种外部存储解析成统一事件流:
6
- {"t": 毫秒时间戳, "seq": 序号, "role": ..., "text": ..., "session": ..., "cwd": ...}
7
- Ingestor(摄取器)—— 增量 watermark + 去重 + 写节点 + 自动 fix-pair 挖掘。
8
-
9
- 内置源:
10
- · JsonlSource —— 通用 JSONL(每行一个事件对象)
11
- · DSHSessionSource —— DeepSeek Harness 会话(~/.dsh/sessions/**/session.jsonl[.zstd])
12
- zstd 为**可选**依赖:缺失时优雅降级(跳过 .zstd 文件并告警)
13
-
14
- 默认敏感度:会话内容是私有记忆 → sensitivity="private"(避免落入公开根)。
15
-
16
- 零第三方依赖(zstd 为可选增强)。
17
- """
18
- from __future__ import annotations
19
-
20
- import glob
21
- import json
22
- import os
23
- import time
24
-
25
- from .security import DEFAULT_SENSITIVITY
26
-
27
- # 会话事件的默认落层与敏感度
28
- SESSION_LAYER = "contextual"
29
- SESSION_SENSITIVITY = "private"
30
-
31
-
32
- # --------------------------------------------------------------------------
33
- # 事件源
34
- # --------------------------------------------------------------------------
35
-
36
- # 生效条件:无必填构造形参,类常量 name="source" 即实例默认;仅当子类覆写 events() 时才产出事件,基类 events() 恒抛 NotImplementedError。
37
- class Source:
38
- """事件源基类。"""
39
-
40
- name = "source"
41
-
42
- # 生效条件:任何调用都直接 raise NotImplementedError(基类占位,无其它分支)。
43
- def events(self):
44
- raise NotImplementedError
45
-
46
- # 生效条件:无前置;返回类常量 self.name(基类为 "source"),作为 watermark 的稳定标识,不含路径与运行期状态;
47
- def key(self):
48
- """源的稳定标识(用于 watermark)。"""
49
- return self.name
50
-
51
-
52
- # 生效条件:required 形参 path 总被存入 self.path;可选形参 name 为假值(None/空串)时 self.name 回落到 "jsonl:"+os.path.basename(path),为真值时用 name 本身,t_key/role_key/text_key/default_role 原样存入属性。
53
- class JsonlSource(Source):
54
- """通用 JSONL 会话源。
55
-
56
- 每行是事件对象;字段映射可配置:
57
- t_key —— 时间戳字段(默认 "time",也接受 ISO 字符串)
58
- role_key —— 角色字段(默认 "role")
59
- text_key —— 文本字段(默认 "text")
60
- """
61
-
62
- # 生效条件:传入 path;name 为假值(None/空串)时回落为 "jsonl:"+os.path.basename(path),t_key/role_key/text_key/default_role 原样存为属性(默认值 "time"/"role"/"text"/"user")。
63
- def __init__(self, path: str, name: str = None, t_key="time", role_key="role",
64
- text_key="text", default_role="user"):
65
- self.path = path
66
- self.name = name or ("jsonl:" + os.path.basename(path))
67
- self.t_key, self.role_key, self.text_key = t_key, role_key, text_key
68
- self.default_role = default_role
69
-
70
- # 生效条件:无前置;返回 self.name(构造时已回落为 "jsonl:"+basename(path)),只随构造参数变化、不随文件内容变化;
71
- def key(self):
72
- return self.name
73
-
74
- # 生效条件:o.get(self.t_key) 为 int/float 时返回 float(v) 乘 1000(v<1e12)或乘 1(否则);为字符串且 v[:19] 按 "%Y-%m-%dT%H:%M:%S" 解析成功时返回 mktime*1000,抛 ValueError 时返回 0.0;键缺失或其它类型返回 0.0。
75
- def _ts(self, o):
76
- v = o.get(self.t_key)
77
- if isinstance(v, (int, float)):
78
- return float(v) * (1000.0 if v < 1e12 else 1.0)
79
- if isinstance(v, str):
80
- try:
81
- return time.mktime(time.strptime(v[:19], "%Y-%m-%dT%H:%M:%S")) * 1000
82
- except ValueError:
83
- return 0.0
84
- return 0.0
85
-
86
- # 生效条件:os.path.exists(self.path) 为真时逐行产出,空行、json.loads 抛 ValueError、o.get(self.text_key) 为假的行被跳过,产出项为 t=self._ts(o)、seq=o.get("seq", i)、role=o.get(self.role_key) or self.default_role、text=str(text)、session/cwd 取 o 同名键;path 不存在时直接 return 不产出。
87
- def events(self):
88
- if not os.path.exists(self.path):
89
- return
90
- with open(self.path, encoding="utf-8", errors="replace") as f:
91
- for i, line in enumerate(f):
92
- line = line.strip()
93
- if not line:
94
- continue
95
- try:
96
- o = json.loads(line)
97
- except ValueError:
98
- continue
99
- text = o.get(self.text_key)
100
- if not text:
101
- continue
102
- yield {"t": self._ts(o), "seq": o.get("seq", i),
103
- "role": o.get(self.role_key) or self.default_role,
104
- "text": str(text), "session": o.get("session"),
105
- "cwd": o.get("cwd")}
106
-
107
-
108
- # 生效条件:path 指向的内容可读且 zstandard 可导入时返回 StringIO(raw.decode("utf-8", errors="replace"));ImportError 时返回 None。
109
- def _zstd_reader(path):
110
- """返回可读的文本迭代器;zstd 不可用返回 None。"""
111
- import io
112
- try:
113
- import zstandard as zstd
114
- except ImportError:
115
- return None
116
- with open(path, "rb") as f:
117
- raw = zstd.ZstdDecompressor().stream_reader(f).read()
118
- return io.StringIO(raw.decode("utf-8", errors="replace"))
119
-
120
-
121
- # 生效条件:required 形参 path 总被存入 self.path,可选形参 include_reasoning 原样存入,self.name 恒为 "dsh:"+os.path.basename(os.path.dirname(path))(与 include_reasoning 取值无关)。
122
- class DSHSessionSource(Source):
123
- """DeepSeek Harness 会话源。
124
-
125
- 事件映射:
126
- user/message → role=user
127
- assistant/message → role=assistant(只取 content[].text,reasoning 默认丢弃)
128
- tool/call → role=command("name(args)" 形式)
129
- tool/result → role=tool-output
130
- """
131
-
132
- # 生效条件:传入 path 即成立,include_reasoning 原样存为属性(默认 False),name 固定为 "dsh:"+os.path.basename(os.path.dirname(path))。
133
- def __init__(self, path: str, include_reasoning: bool = False):
134
- self.path = path
135
- self.include_reasoning = include_reasoning
136
- self.name = "dsh:" + os.path.basename(os.path.dirname(path))
137
-
138
- # 生效条件:无前置;返回 self.name(构造时固定为 "dsh:"+basename(dirname(path))),与 include_reasoning 取值无关;
139
- def key(self):
140
- return self.name
141
-
142
- @staticmethod
143
- # 生效条件:可选形参 root 为假值(None/空串)时改用默认目录 os.path.join(expanduser("~"),".dsh","sessions"),否则用传入 root;对 root 下递归 glob 到的 session.jsonl 与 session.jsonl.zstd 按 -os.path.getsize 降序排序(两处 glob 均无命中时为空列表),可选形参 limit 为假值(None/0)时返回全部 files,否则返回 files[:limit]。
144
- def discover(root: str = None, limit: int = None):
145
- root = root or os.path.join(os.path.expanduser("~"), ".dsh", "sessions")
146
- files = glob.glob(os.path.join(root, "**", "session.jsonl"), recursive=True)
147
- files += glob.glob(os.path.join(root, "**", "session.jsonl.zstd"), recursive=True)
148
- files.sort(key=lambda p: (-os.path.getsize(p), p))
149
- return files[:limit] if limit else files
150
-
151
- # 生效条件:self.path 以 ".zstd" 结尾时经 _zstd_reader 逐行产出(其返回 None 时 raise RuntimeError),否则以 utf-8/errors=replace 打开 self.path 逐行产出。
152
- def _lines(self):
153
- if self.path.endswith(".zstd"):
154
- fh = _zstd_reader(self.path)
155
- if fh is None:
156
- raise RuntimeError("zstd 不可用(pip install zstandard),跳过该会话")
157
- try:
158
- yield from fh
159
- finally:
160
- fh.close()
161
- else:
162
- with open(self.path, encoding="utf-8", errors="replace") as f:
163
- yield from f
164
-
165
- @staticmethod
166
- # 生效条件:required 形参 content 为 str 时原样返回 content;为 list 时收集其中 str 元素及 type 属于 ("text","input-text") 且 text 为真值的 dict 元素(取 str(c["text"])),以 "\n" 连接返回(无可收集元素时为空串 "");既非 str 也非 list 时返回 ""。
167
- def _text_of(content):
168
- """content 可能是 [{type,text}] 或字符串。"""
169
- if isinstance(content, str):
170
- return content
171
- if isinstance(content, list):
172
- parts = []
173
- for c in content:
174
- if isinstance(c, dict):
175
- if c.get("type") in ("text", "input-text") and c.get("text"):
176
- parts.append(str(c["text"]))
177
- elif isinstance(c, str):
178
- parts.append(c)
179
- return "\n".join(parts)
180
- return ""
181
-
182
- # 生效条件:逐行解析后按 o.get("type") 分派——"session" 只更新 sess/cwd 不产出;"user/message" 产出 role=user 与 _text_of(data.get("content"));"assistant/message" 产出 role=assistant 与 _text_of(msg.get("content")),include_reasoning 为真时再把 content 中 type=="reasoning" 且有 text 的项追加 "\n[reasoning] "+str(c["text"]);"tool/call" 产出 role=command 与 f"{name}({arguments or ''})";"tool/result" 产出 role=tool-output,仅当 inner[0] 为 dict 时 text=_text_of(inner[0].get("content"));其它 type 或 ev 中 text 为空的事件不产出。
183
- def events(self):
184
- sess = None
185
- cwd = None
186
- for line in self._lines():
187
- line = line.strip()
188
- if not line:
189
- continue
190
- try:
191
- o = json.loads(line)
192
- except ValueError:
193
- continue
194
- t = o.get("type")
195
- data = o.get("data") or {}
196
- if t == "session":
197
- sess, cwd = o.get("id"), o.get("cwd")
198
- continue
199
- ev = {"t": float(o.get("time") or 0), "seq": o.get("seq"),
200
- "session": sess, "cwd": cwd}
201
- if t == "user/message":
202
- ev["role"], ev["text"] = "user", self._text_of(data.get("content"))
203
- elif t == "assistant/message":
204
- msg = data.get("message") or {}
205
- ev["role"] = "assistant"
206
- ev["text"] = self._text_of(msg.get("content"))
207
- if self.include_reasoning:
208
- for c in (msg.get("content") or []):
209
- if isinstance(c, dict) and c.get("type") == "reasoning" \
210
- and c.get("text"):
211
- ev["text"] = (ev["text"] or "") + "\n[reasoning] " + str(c["text"])
212
- elif t == "tool/call":
213
- ev["role"] = "command"
214
- ev["text"] = f"{data.get('name')}({data.get('arguments') or ''})"
215
- elif t == "tool/result":
216
- msg = data.get("message") or {}
217
- inner = msg.get("content") or []
218
- txt = ""
219
- if inner and isinstance(inner[0], dict):
220
- txt = self._text_of(inner[0].get("content"))
221
- ev["role"], ev["text"] = "tool-output", txt
222
- else:
223
- continue
224
- if ev.get("text"):
225
- yield ev
226
-
227
-
228
- # --------------------------------------------------------------------------
229
- # 摄取器
230
- # --------------------------------------------------------------------------
231
-
232
- # 生效条件:required 形参 cg 总被存入并以其 cg.root 拼出 self.path=os.path.join(cg.root,"_sources.json");layer/sensitivity 原样存入,未传时取模块级常量 SESSION_LAYER、SESSION_SENSITIVITY 作为默认值。
233
- class Ingestor:
234
- """增量摄取:watermark + 去重 + 写节点 + 自动 fix-pair 挖掘。
235
-
236
- watermark 文件:<root>/_sources.json
237
- {source_key: {"t": 最后时间戳(ms), "seq": 最后序号, "count": 已摄取条数}}
238
- """
239
-
240
- # 生效条件:传入带 root 的 cg 即成立,layer/sensitivity 默认 SESSION_LAYER/SESSION_SENSITIVITY 并原样存为属性,路径为 os.path.join(cg.root, "_sources.json")。
241
- def __init__(self, cg, layer: str = SESSION_LAYER,
242
- sensitivity: str = SESSION_SENSITIVITY):
243
- self.cg = cg
244
- self.layer = layer
245
- self.sensitivity = sensitivity
246
- self.path = os.path.join(cg.root, "_sources.json")
247
-
248
- # ---- watermark ----
249
-
250
- # 生效条件:self.path 存在、json.load 成功且结果为 dict 时返回该 dict;path 不存在、抛 ValueError/OSError 或结果非 dict 时返回 {"schema": 1, "sources": {}}。
251
- def _load(self):
252
- if os.path.exists(self.path):
253
- try:
254
- with open(self.path, encoding="utf-8") as f:
255
- d = json.load(f)
256
- if isinstance(d, dict):
257
- return d
258
- except (ValueError, OSError):
259
- pass
260
- return {"schema": 1, "sources": {}}
261
-
262
- # 生效条件:传入 d 时以 ensure_ascii=False/indent=1 写入 self.path+".tmp",再 os.replace 覆盖 self.path。
263
- def _save(self, d):
264
- tmp = self.path + ".tmp"
265
- with open(tmp, "w", encoding="utf-8") as f:
266
- json.dump(d, f, ensure_ascii=False, indent=1)
267
- os.replace(tmp, self.path)
268
-
269
- # 生效条件:key 命中 self._load()["sources"] 时返回其值,缺 key 时返回 {}(_load 结果缺 "sources" 键则抛 KeyError)。
270
- def watermark(self, key: str):
271
- return self._load()["sources"].get(key, {})
272
-
273
- # 生效条件:调用即返回 dict(self._load()["sources"]) 的浅拷贝(_load 结果缺 "sources" 键则抛 KeyError)。
274
- def watermarks(self):
275
- return dict(self._load()["sources"])
276
-
277
- # ---- 摄取 ----
278
-
279
- # 生效条件:对必需形参 source,先按 watermark(source.key()) 跳过 t<last_t(wm 的 "t" 为假值时视作 0)及 t==last_t 且 last_seq 非 None 且 ev["seq"] 为 int 且 <= last_seq 的事件、并按 (session, seq) 去重,max_events 为真值(非 0/None)时取满即停;dry_run 为假时逐条 cg.add(nid 已在 cg.index["nodes"] 中则跳过,cg.add 抛异常则 denied+=1、记 last_error 并继续),denied 与 last_error 同时成立时补 last_error/hint,mine_fix_pairs 为真且 new_events 非空且非 dry_run 时结果附 fix_pairs,new_events 非空且非 dry_run 时以末事件写回水位(count 累加 written),最后返回 result。
280
- def ingest(self, source, mine_fix_pairs: bool = True, max_events: int = None,
281
- dry_run: bool = False):
282
- """摄取一个源的新事件。返回统计。
283
-
284
- 去重键:(session, seq) —— 同一事件不重复入库。
285
- 增量:只处理 (t, seq) 大于 watermark 的事件。
286
- """
287
- key = source.key()
288
- wm = self.watermark(key)
289
- last_t, last_seq = float(wm.get("t") or 0), wm.get("seq")
290
- new_events, seen = [], set()
291
- for ev in source.events():
292
- if ev.get("t", 0) < last_t:
293
- continue
294
- if ev.get("t", 0) == last_t and last_seq is not None \
295
- and isinstance(ev.get("seq"), int) and ev["seq"] <= last_seq:
296
- continue
297
- dedup = (ev.get("session"), ev.get("seq"))
298
- if dedup in seen:
299
- continue
300
- seen.add(dedup)
301
- new_events.append(ev)
302
- if max_events and len(new_events) >= max_events:
303
- break
304
-
305
- written, ids, denied = 0, [], 0
306
- if not dry_run:
307
- for ev in new_events:
308
- nid = "src_%s_%s" % (_sig(key)[:6], _sig(
309
- f"{ev.get('session')}:{ev.get('seq')}")[:10])
310
- if nid in self.cg.index["nodes"]:
311
- continue # 幂等
312
- body = ("# 功能名:会话事件\n"
313
- f"# 生效条件:检索「{str(ev.get('text'))[:20]}」\n"
314
- "# 子功能:记录会话事件\n"
315
- f"# 执行:{str(ev.get('text'))[:80]}\n"
316
- "# 验证方式:data(会话原始记录)\n"
317
- "# 不适用条件:其它会话\n\n"
318
- f"{ev.get('text')}\n")
319
- try:
320
- self.cg.add(nid, body, layer=self.layer, role=ev.get("role"),
321
- tags=["session", key], sensitivity=self.sensitivity,
322
- verification_basis="data",
323
- condition_space={"observation_position": key,
324
- "observation_tool": "会话流",
325
- "time_window": [ev.get("t") or 0,
326
- ev.get("t") or 0]})
327
- except Exception as exc: # noqa: BLE001 —— 权限/层错误不中断整批
328
- denied += 1
329
- self.last_error = f"{type(exc).__name__}: {exc}"
330
- continue
331
- ids.append(nid)
332
- written += 1
333
-
334
- result = {"source": key, "new_events": len(new_events), "written": written,
335
- "denied": denied, "ids": ids, "dry_run": dry_run,
336
- "sensitivity": self.sensitivity}
337
- if denied and getattr(self, "last_error", None):
338
- result["last_error"] = self.last_error
339
- result["hint"] = ("会话内容默认 sensitivity=private;"
340
- "调用方需 MDCG_CLEARANCE=private 才能写入")
341
- # 自动 fix-pair 挖掘(对标 deja-vu:错误→修复)
342
- if mine_fix_pairs and new_events and not dry_run:
343
- result["fix_pairs"] = self.cg.mine_fix_pairs(
344
- [{"role": e.get("role"), "text": e.get("text")} for e in new_events])
345
-
346
- if new_events and not dry_run:
347
- d = self._load()
348
- last = new_events[-1]
349
- d["sources"][key] = {
350
- "t": last.get("t") or last_t,
351
- "seq": last.get("seq"),
352
- "count": (wm.get("count") or 0) + written,
353
- "updated_at": time.time(),
354
- "path": getattr(source, "path", None),
355
- }
356
- self._save(d)
357
- return result
358
-
359
-
360
- # 生效条件:text 为 None 或假值时按 "" 参与 sha1;n 默认 12,返回 hexdigest 前 n 位(n=0 得空串)。
361
- def _sig(text: str, n: int = 12) -> str:
362
- import hashlib
363
- return hashlib.sha1((text or "").encode("utf-8")).hexdigest()[:n]
364
-
365
-
366
- # --------------------------------------------------------------------------
367
- # 摄取分派(P0 · ingest op):按扩展名选摄取方式,单一入口吃多种文件
368
- # --------------------------------------------------------------------------
369
- #
370
- # 三条摄取链:
371
- # session —— 会话流(.jsonl):走 Ingestor(watermark + 去重 + fix-pair)
372
- # doc —— 文档(.md/.txt/...):走 docindex.extract + refindex.add_items
373
- # code —— 代码(.py/.ts/...):走 codeindex.extract + refindex.add_items
374
- #
375
- # 设计要点:
376
- # - 注册表 INGEST_REGISTRY 是唯一真源:新增后缀只改这里。
377
- # - dir 动作分链处理:目录里 doc / code / jsonl 混放时各链互不干扰。
378
- # - 幂等:沿用 refindex.Ledger(size+mtime 水位)与 Ingestor watermark。
379
- # - 预演:dry_run=True 只统计、不写入(对应计划「可预演」要求)。
380
-
381
- INGEST_ACTIONS = ("file", "dir", "jsonl", "stat")
382
-
383
- INGEST_REGISTRY = {
384
- # 会话流
385
- ".jsonl": "session", ".ndjson": "session",
386
- # 文档
387
- ".md": "doc", ".markdown": "doc", ".txt": "doc", ".rst": "doc",
388
- ".html": "doc", ".htm": "doc",
389
- # 代码
390
- ".py": "code", ".js": "code", ".mjs": "code", ".ts": "code",
391
- ".tsx": "code", ".jsx": "code", ".go": "code", ".rs": "code",
392
- ".java": "code", ".kt": "code", ".swift": "code", ".rb": "code",
393
- ".php": "code", ".cs": "code", ".cpp": "code", ".cc": "code",
394
- ".c": "code", ".h": "code", ".hpp": "code",
395
- }
396
-
397
-
398
- # 生效条件:path 为 None 或空串时 splitext 得 "" 且未登记 → 返回 None;扩展名(小写)存在于 INGEST_REGISTRY 时返回其 kind。
399
- def dispatch_of(path: str):
400
- """按扩展名返回摄取方式(session / doc / code);未登记返回 None。"""
401
- return INGEST_REGISTRY.get(os.path.splitext(path or "")[1].lower())
402
-
403
-
404
- # 生效条件:required 形参 cg 总被存入;可选形参 sensitivity 为假值(None/空串)时内部 Ingestor 的 sensitivity 回落模块级常量 SESSION_SENSITIVITY,为真值时用传入的 sensitivity。
405
- class FileDispatcher:
406
- """单一入口吃多种文件:按扩展名分派到会话流 / 文档 / 代码三条摄取链。"""
407
-
408
- # 生效条件:传入 cg 即成立,sensitivity 为假值(None/空串)时所用 Ingestor 回落 SESSION_SENSITIVITY,否则用传入值。
409
- def __init__(self, cg, sensitivity=None):
410
- self.cg = cg
411
- # 会话链默认 sensitivity=private;调用方可显式覆盖(测试/受限环境)
412
- self.ingestor = Ingestor(cg, sensitivity=sensitivity or SESSION_SENSITIVITY)
413
-
414
- # ---- stat:看水位与支持面 ----
415
-
416
- # 生效条件:from . import refindex 与 refindex.Ledger(self.cg.root).stat() 均不抛异常时返回该 stat 结果,抛任何异常时返回 {}。
417
- def _ledger_stat(self):
418
- try:
419
- from . import refindex
420
- return refindex.Ledger(self.cg.root).stat()
421
- except Exception: # noqa: BLE001
422
- return {}
423
-
424
- # 生效条件:调用即返回由 INGEST_REGISTRY 按 kind 分组并排序后的 extensions、list(INGEST_ACTIONS)、self.ingestor.watermarks()、self._ledger_stat() 与固定 note。
425
- def stat(self):
426
- kinds = {}
427
- for ext, kind in INGEST_REGISTRY.items():
428
- kinds.setdefault(kind, []).append(ext)
429
- return {"ok": True, "kind": "stat", "actions": list(INGEST_ACTIONS),
430
- "extensions": {k: sorted(v) for k, v in sorted(kinds.items())},
431
- "watermarks": self.ingestor.watermarks(),
432
- "ledger": self._ledger_stat(),
433
- "note": ("注册表是唯一真源:新增后缀只改 INGEST_REGISTRY。"
434
- "watermarks=会话流水位;ledger=文档/代码文件水位。")}
435
-
436
- # ---- 单文件 ----
437
-
438
- # 生效条件:dispatch_of(path) 为 None 时返回 ok=False 的「不支持的后缀」结果;path 不是文件时返回 ok=False 的「文件不存在」结果;kind=="session" 时转 ingest_jsonl(path, dry_run=dry_run)(layer/sensitivity 不参与);其它 kind 转 _ingest_doc_or_code(path, kind, layer=layer, sensitivity=sensitivity, dry_run=dry_run)。
439
- def ingest_file(self, path, layer=None, sensitivity=None, dry_run=False):
440
- kind = dispatch_of(path)
441
- if kind is None:
442
- ext = os.path.splitext(path or "")[1] or "(无后缀)"
443
- return {"ok": False, "path": path, "error": f"不支持的后缀:{ext}",
444
- "supported": sorted(INGEST_REGISTRY)}
445
- if not os.path.isfile(path):
446
- return {"ok": False, "path": path, "error": "文件不存在"}
447
- if kind == "session":
448
- return self.ingest_jsonl(path, dry_run=dry_run)
449
- return self._ingest_doc_or_code(path, kind, layer=layer,
450
- sensitivity=sensitivity, dry_run=dry_run)
451
-
452
- # 生效条件:kind=="code" 用 codeindex 否则用 docindex;mod.extract 抛 ValueError 时返回 ok=False 的「抽取失败」;dry_run 为真时只返回 items 计数与前 20 个 node_id 不写盘;否则经 refindex.add_items(kind 为 code_ref/doc_ref)写入并返回 indexed、ids[:20] 与 sensitivity。
453
- def _ingest_doc_or_code(self, path, kind, layer=None, sensitivity=None,
454
- dry_run=False):
455
- from . import codeindex, docindex, refindex
456
- mod = codeindex if kind == "code" else docindex
457
- rel = os.path.basename(path)
458
- with open(path, encoding="utf-8", errors="replace") as f:
459
- src = f.read()
460
- try:
461
- items = mod.extract(src, path=rel,
462
- suffix=os.path.splitext(path)[1].lower())
463
- except ValueError as exc:
464
- return {"ok": False, "path": path, "error": f"抽取失败:{exc}"}
465
- items = items or []
466
- if dry_run:
467
- return {"ok": True, "dry_run": True, "kind": kind, "path": path,
468
- "items": len(items),
469
- "ids": [mod.node_id(i) for i in items[:20]],
470
- "note": "预演:只抽取计数,未写盘"}
471
- ref_kind = "code_ref" if kind == "code" else "doc_ref"
472
- ids, sens = refindex.add_items(
473
- self.cg, items, kind=ref_kind,
474
- root=os.path.dirname(path) or ".", layer=layer,
475
- sensitivity=sensitivity)
476
- return {"ok": True, "kind": kind, "path": path, "items": len(items),
477
- "indexed": len(ids), "ids": ids[:20], "sensitivity": sens}
478
-
479
- # ---- 目录 ----
480
-
481
- # 生效条件:调用即 os.walk(root) 统计每个文件名经 dispatch_of 得到的 kind(无匹配记 "unsupported")并返回 dry_run 预演计数结果。
482
- def _dry_dir(self, root):
483
- counts = {}
484
- for _dp, _dn, fns in os.walk(root):
485
- for name in fns:
486
- k = dispatch_of(name) or "unsupported"
487
- counts[k] = counts.get(k, 0) + 1
488
- return {"ok": True, "dry_run": True, "kind": "dir", "root": root,
489
- "counts": counts,
490
- "note": "预演:仅统计各链文件数,未做任何写入"}
491
-
492
- # 生效条件:root 非目录时返回 ok=False 的「目录不存在」;dry_run 为真时返回 _dry_dir(root);否则对 doc_ref/code_ref 两链各以 patterns/max_files/max_items/incremental/ledger 调 refindex.index_dir 与 add_items,并把 root 下 **/*.jsonl 前 max_files 个逐个 ingest_jsonl 后返回 out。
493
- def ingest_dir(self, root, layer=None, sensitivity=None, patterns=None,
494
- max_files=500, max_items=2000, incremental=False,
495
- dry_run=False):
496
- from . import refindex
497
- if not os.path.isdir(root):
498
- return {"ok": False, "error": f"目录不存在:{root}"}
499
- if dry_run:
500
- return self._dry_dir(root)
501
- ledger = refindex.Ledger(self.cg.root)
502
- out = {"ok": True, "kind": "dir", "root": root, "chains": {}}
503
- for ref_kind, key in (("doc_ref", "doc"), ("code_ref", "code")):
504
- items, errors, stats = refindex.index_dir(
505
- root, kind=ref_kind, patterns=patterns, max_files=max_files,
506
- max_items=max_items, incremental=incremental, ledger=ledger)
507
- ids, sens = refindex.add_items(self.cg, items, kind=ref_kind,
508
- root=root, layer=layer,
509
- sensitivity=sensitivity)
510
- out["chains"][key] = {
511
- "indexed": len(ids), "errors": len(errors),
512
- "files": stats.get("files"), "truncated": stats.get("truncated"),
513
- "skipped_unchanged": stats.get("skipped_unchanged", 0),
514
- "skipped_suffixes": stats.get("skipped_suffixes", []),
515
- "sensitivity": sens}
516
- # 会话流(.jsonl)逐个增量摄取
517
- jsons = sorted(glob.glob(os.path.join(root, "**", "*.jsonl"),
518
- recursive=True))[:max_files]
519
- ses = []
520
- for p in jsons:
521
- r = self.ingest_jsonl(p)
522
- ses.append({"path": p, "written": r.get("written", 0),
523
- "new_events": r.get("new_events", 0)})
524
- out["chains"]["session"] = {"files": len(jsons), "results": ses}
525
- return out
526
-
527
- # ---- 会话流 ----
528
-
529
- @staticmethod
530
- # 生效条件:required 形参 path 能被 open(...,encoding="utf-8",errors="replace") 打开时读前 50000 字符,head 含 '"user/message"'、'"assistant/message"'、'"tool/call"' 任一标记则返回 DSHSessionSource(path),否则返回 JsonlSource(path);打开抛 OSError 时直接返回 JsonlSource(path)。
531
- def _auto_source(path):
532
- """通用 JSONL vs DSH 会话:按内容探测,避免调用方选错源类型。"""
533
- try:
534
- with open(path, encoding="utf-8", errors="replace") as f:
535
- head = f.read(50000)
536
- except OSError:
537
- return JsonlSource(path)
538
- for marker in ('"user/message"', '"assistant/message"', '"tool/call"'):
539
- if marker in head:
540
- return DSHSessionSource(path)
541
- return JsonlSource(path)
542
-
543
- # 生效条件:path 不是文件时返回 ok=False 的「文件不存在」;否则经 _auto_source(path) 选源后调 self.ingestor.ingest(src, dry_run=dry_run, max_events=max_events),并补上 ok=True/kind/path/source_class 后返回。
544
- def ingest_jsonl(self, path, dry_run=False, max_events=None):
545
- if not os.path.isfile(path):
546
- return {"ok": False, "error": f"文件不存在:{path}"}
547
- src = self._auto_source(path)
548
- res = self.ingestor.ingest(src, dry_run=dry_run, max_events=max_events)
549
- res.update({"ok": True, "kind": "session", "path": path,
550
- "source_class": type(src).__name__})
551
- return res
552
-
553
-
554
- # 生效条件:cg 必填;action 为 None/空串时 (action or "stat") 归为 stat;file/jsonl 缺 path 返回 ok=False;dir 的 max_files/max_items 走 int(x or 500)/int(x or 2000),传 0 也变 500/2000;未知 action 抛 ValueError。
555
- def run(cg, action: str = "stat", **kw):
556
- """ingest op 唯一入口。"""
557
- act = (action or "stat").strip().lower()
558
- d = FileDispatcher(cg, sensitivity=kw.get("sensitivity"))
559
- if act == "stat":
560
- return d.stat()
561
- if act == "file":
562
- p = kw.get("path")
563
- if not p:
564
- return {"ok": False, "error": "file 动作需要 path"}
565
- return d.ingest_file(p, layer=kw.get("layer"),
566
- sensitivity=kw.get("sensitivity"),
567
- dry_run=bool(kw.get("dry_run")))
568
- if act == "dir":
569
- p = kw.get("path") or cg.root
570
- return d.ingest_dir(p, layer=kw.get("layer"),
571
- sensitivity=kw.get("sensitivity"),
572
- patterns=kw.get("patterns"),
573
- max_files=int(kw.get("max_files") or 500),
574
- max_items=int(kw.get("max_items") or 2000),
575
- incremental=bool(kw.get("incremental")),
576
- dry_run=bool(kw.get("dry_run")))
577
- if act == "jsonl":
578
- p = kw.get("path")
579
- if not p:
580
- return {"ok": False, "error": "jsonl 动作需要 path"}
581
- return d.ingest_jsonl(p, dry_run=bool(kw.get("dry_run")),
582
- max_events=kw.get("max_events"))
1
+ # -*- coding: utf-8 -*-
2
+ """md_cg · 设备驱动(记忆 OS #3):把外部会话流接进认知图
3
+
4
+ 设计:
5
+ Source(事件源)—— 把某种外部存储解析成统一事件流:
6
+ {"t": epoch 秒, "seq": 序号, "role": ..., "text": ..., "session": ..., "cwd": ...}
7
+
8
+ ⚠ **单位纪律(issue #23)**:`t` 一律 **epoch 秒**,与图内 `created_at` /
9
+ `condition_space.time_window` / `trust.parse_time` 同口径。外部源里的 13 位毫秒
10
+ (DSH 的 `time` 字段、通用 JSONL 的毫秒戳)在**源适配层**经 `trust.epoch_seconds`
11
+ 归一后即不外溢——否则 `condition_space.time_window` 会落毫秒,把
12
+ `stg(op=timeline)` 的倒序头部整片占满并顶掉 auto-recall。
13
+
14
+ Ingestor(摄取器)—— 增量 watermark + 去重 + 写节点 + 自动 fix-pair 挖掘。
15
+
16
+ 内置源:
17
+ · JsonlSource —— 通用 JSONL(每行一个事件对象)
18
+ · DSHSessionSource —— DeepSeek Harness 会话(~/.dsh/sessions/**/session.jsonl[.zstd])
19
+ zstd 为**可选**依赖:缺失时优雅降级(跳过 .zstd 文件并告警)
20
+
21
+ 默认敏感度:会话内容是私有记忆 → sensitivity="private"(避免落入公开根)。
22
+
23
+ 零第三方依赖(zstd 为可选增强)。
24
+ """
25
+ from __future__ import annotations
26
+
27
+ import glob
28
+ import json
29
+ import os
30
+ import time
31
+
32
+ from . import trust
33
+ from .security import DEFAULT_SENSITIVITY
34
+
35
+ # 会话事件的默认落层与敏感度
36
+ SESSION_LAYER = "contextual"
37
+ SESSION_SENSITIVITY = "private"
38
+
39
+
40
+ # --------------------------------------------------------------------------
41
+ # 事件源
42
+ # --------------------------------------------------------------------------
43
+
44
+ # 生效条件:无必填构造形参,类常量 name="source" 即实例默认;仅当子类覆写 events() 时才产出事件,基类 events() 恒抛 NotImplementedError。
45
+ class Source:
46
+ """事件源基类。"""
47
+
48
+ name = "source"
49
+
50
+ # 生效条件:任何调用都直接 raise NotImplementedError(基类占位,无其它分支)。
51
+ def events(self):
52
+ raise NotImplementedError
53
+
54
+ # 生效条件:无前置;返回类常量 self.name(基类为 "source"),作为 watermark 的稳定标识,不含路径与运行期状态;
55
+ def key(self):
56
+ """源的稳定标识(用于 watermark)。"""
57
+ return self.name
58
+
59
+
60
+ # 生效条件:required 形参 path 总被存入 self.path;可选形参 name 为假值(None/空串)时 self.name 回落到 "jsonl:"+os.path.basename(path),为真值时用 name 本身,t_key/role_key/text_key/default_role 原样存入属性。
61
+ class JsonlSource(Source):
62
+ """通用 JSONL 会话源。
63
+
64
+ 每行是事件对象;字段映射可配置:
65
+ t_key —— 时间戳字段(默认 "time",也接受 ISO 字符串)
66
+ role_key —— 角色字段(默认 "role")
67
+ text_key —— 文本字段(默认 "text")
68
+ """
69
+
70
+ # 生效条件:传入 path;name 为假值(None/空串)时回落为 "jsonl:"+os.path.basename(path),t_key/role_key/text_key/default_role 原样存为属性(默认值 "time"/"role"/"text"/"user")。
71
+ def __init__(self, path: str, name: str = None, t_key="time", role_key="role",
72
+ text_key="text", default_role="user"):
73
+ self.path = path
74
+ self.name = name or ("jsonl:" + os.path.basename(path))
75
+ self.t_key, self.role_key, self.text_key = t_key, role_key, text_key
76
+ self.default_role = default_role
77
+
78
+ # 生效条件:无前置;返回 self.name(构造时已回落为 "jsonl:"+basename(path)),只随构造参数变化、不随文件内容变化;
79
+ def key(self):
80
+ return self.name
81
+
82
+ # 生效条件:o.get(self.t_key) 经 float 可转数值时返回 trust.epoch_seconds(该值)(秒值原样、13 位毫秒 /1000 归一到**秒**);float 抛 TypeError/ValueError 且 v 为字符串时按 v[:19] 与 "%Y-%m-%dT%H:%M:%S" 解析,成功返回 mktime(**秒**)、抛 ValueError 返回 0.0;键缺失/None/其它不可转类型返回 0.0。
83
+ def _ts(self, o):
84
+ v = o.get(self.t_key)
85
+ try:
86
+ return trust.epoch_seconds(float(v)) or 0.0
87
+ except (TypeError, ValueError):
88
+ pass
89
+ if isinstance(v, str):
90
+ try:
91
+ return time.mktime(time.strptime(v[:19], "%Y-%m-%dT%H:%M:%S"))
92
+ except ValueError:
93
+ return 0.0
94
+ return 0.0
95
+
96
+ # 生效条件:os.path.exists(self.path) 为真时逐行产出,空行、json.loads 抛 ValueError、o.get(self.text_key) 为假的行被跳过,产出项为 t=self._ts(o)、seq=o.get("seq", i)、role=o.get(self.role_key) or self.default_role、text=str(text)、session/cwd 取 o 同名键;path 不存在时直接 return 不产出。
97
+ def events(self):
98
+ if not os.path.exists(self.path):
99
+ return
100
+ with open(self.path, encoding="utf-8", errors="replace") as f:
101
+ for i, line in enumerate(f):
102
+ line = line.strip()
103
+ if not line:
104
+ continue
105
+ try:
106
+ o = json.loads(line)
107
+ except ValueError:
108
+ continue
109
+ text = o.get(self.text_key)
110
+ if not text:
111
+ continue
112
+ yield {"t": self._ts(o), "seq": o.get("seq", i),
113
+ "role": o.get(self.role_key) or self.default_role,
114
+ "text": str(text), "session": o.get("session"),
115
+ "cwd": o.get("cwd")}
116
+
117
+
118
+ # 生效条件:path 指向的内容可读且 zstandard 可导入时返回 StringIO(raw.decode("utf-8", errors="replace"));ImportError 时返回 None。
119
+ def _zstd_reader(path):
120
+ """返回可读的文本迭代器;zstd 不可用返回 None。"""
121
+ import io
122
+ try:
123
+ import zstandard as zstd
124
+ except ImportError:
125
+ return None
126
+ with open(path, "rb") as f:
127
+ raw = zstd.ZstdDecompressor().stream_reader(f).read()
128
+ return io.StringIO(raw.decode("utf-8", errors="replace"))
129
+
130
+
131
+ # 生效条件:required 形参 path 总被存入 self.path,可选形参 include_reasoning 原样存入,self.name 恒为 "dsh:"+os.path.basename(os.path.dirname(path))(与 include_reasoning 取值无关)。
132
+ class DSHSessionSource(Source):
133
+ """DeepSeek Harness 会话源。
134
+
135
+ 事件映射:
136
+ user/message → role=user
137
+ assistant/message → role=assistant(只取 content[].text,reasoning 默认丢弃)
138
+ tool/call → role=command("name(args)" 形式)
139
+ tool/result → role=tool-output
140
+ """
141
+
142
+ # 生效条件:传入 path 即成立,include_reasoning 原样存为属性(默认 False),name 固定为 "dsh:"+os.path.basename(os.path.dirname(path))。
143
+ def __init__(self, path: str, include_reasoning: bool = False):
144
+ self.path = path
145
+ self.include_reasoning = include_reasoning
146
+ self.name = "dsh:" + os.path.basename(os.path.dirname(path))
147
+
148
+ # 生效条件:无前置;返回 self.name(构造时固定为 "dsh:"+basename(dirname(path))),与 include_reasoning 取值无关;
149
+ def key(self):
150
+ return self.name
151
+
152
+ @staticmethod
153
+ # 生效条件:可选形参 root 为假值(None/空串)时改用默认目录 os.path.join(expanduser("~"),".dsh","sessions"),否则用传入 root;对 root 下递归 glob 到的 session.jsonl 与 session.jsonl.zstd 逐条 os.stat(抛 OSError 的条目跳过),按 (-st_size, path) 升序键排序(两处 glob 均无命中时为空列表),可选形参 limit 为假值(None/0)时返回全部 rows,否则返回 rows[:limit];每行为 (path, size, mtime) 三元组。
154
+ def discover_detailed(root: str = None, limit: int = None):
155
+ """候选会话 → `[(path, size, mtime), ...]`(**排序唯一真源**,`discover` 复用)。
156
+
157
+ 排序:会话**文件体积**降序 → `path` 升序(确定性终键)。
158
+ **刻意不按最近活动排序**——这正是 `source='auto'` 可能选中很久以前会话的
159
+ 原因;把 size/mtime 一并透出,让「为什么选它」在返回体里可辨
160
+ (issue #23 附带建议:先让依据可见,是否换策略另议)。
161
+ """
162
+ root = root or os.path.join(os.path.expanduser("~"), ".dsh", "sessions")
163
+ files = glob.glob(os.path.join(root, "**", "session.jsonl"), recursive=True)
164
+ files += glob.glob(os.path.join(root, "**", "session.jsonl.zstd"), recursive=True)
165
+ rows = []
166
+ for p in files:
167
+ try:
168
+ st = os.stat(p)
169
+ except OSError:
170
+ continue # 竞态删除:跳过而非整体失败
171
+ rows.append((p, st.st_size, st.st_mtime))
172
+ rows.sort(key=lambda r: (-r[1], r[0]))
173
+ return rows[:limit] if limit else rows
174
+
175
+ @staticmethod
176
+ # 生效条件:可选形参 root/limit 原样转交 discover_detailed,返回其结果的 path 列(同为体积降序、同确定性终键),limit 为假值时返回全部。
177
+ def discover(root: str = None, limit: int = None):
178
+ """候选会话路径(体积降序)——`discover_detailed` 的路径投影。"""
179
+ return [p for p, _s, _m in DSHSessionSource.discover_detailed(root, limit)]
180
+
181
+ # 生效条件:self.path 以 ".zstd" 结尾时经 _zstd_reader 逐行产出(其返回 None 时 raise RuntimeError),否则以 utf-8/errors=replace 打开 self.path 逐行产出。
182
+ def _lines(self):
183
+ if self.path.endswith(".zstd"):
184
+ fh = _zstd_reader(self.path)
185
+ if fh is None:
186
+ raise RuntimeError("zstd 不可用(pip install zstandard),跳过该会话")
187
+ try:
188
+ yield from fh
189
+ finally:
190
+ fh.close()
191
+ else:
192
+ with open(self.path, encoding="utf-8", errors="replace") as f:
193
+ yield from f
194
+
195
+ @staticmethod
196
+ # 生效条件:required 形参 content 为 str 时原样返回 content;为 list 时收集其中 str 元素及 type 属于 ("text","input-text") 且 text 为真值的 dict 元素(取 str(c["text"])),以 "\n" 连接返回(无可收集元素时为空串 "");既非 str 也非 list 时返回 ""。
197
+ def _text_of(content):
198
+ """content 可能是 [{type,text}] 或字符串。"""
199
+ if isinstance(content, str):
200
+ return content
201
+ if isinstance(content, list):
202
+ parts = []
203
+ for c in content:
204
+ if isinstance(c, dict):
205
+ if c.get("type") in ("text", "input-text") and c.get("text"):
206
+ parts.append(str(c["text"]))
207
+ elif isinstance(c, str):
208
+ parts.append(c)
209
+ return "\n".join(parts)
210
+ return ""
211
+
212
+ # 生效条件:逐行解析后按 o.get("type") 分派——"session" 只更新 sess/cwd 不产出;"user/message" 产出 role=user 与 _text_of(data.get("content"));"assistant/message" 产出 role=assistant 与 _text_of(msg.get("content")),include_reasoning 为真时再把 content 中 type=="reasoning" 且有 text 的项追加 "\n[reasoning] "+str(c["text"]);"tool/call" 产出 role=command 与 f"{name}({arguments or ''})";"tool/result" 产出 role=tool-output,仅当 inner[0] 为 dict 时 text=_text_of(inner[0].get("content"));其它 type 或 ev 中 text 为空的事件不产出。
213
+ def events(self):
214
+ sess = None
215
+ cwd = None
216
+ for line in self._lines():
217
+ line = line.strip()
218
+ if not line:
219
+ continue
220
+ try:
221
+ o = json.loads(line)
222
+ except ValueError:
223
+ continue
224
+ t = o.get("type")
225
+ data = o.get("data") or {}
226
+ if t == "session":
227
+ sess, cwd = o.get("id"), o.get("cwd")
228
+ continue
229
+ ev = {"t": trust.epoch_seconds(float(o.get("time") or 0)) or 0.0,
230
+ "seq": o.get("seq"),
231
+ "session": sess, "cwd": cwd}
232
+ if t == "user/message":
233
+ ev["role"], ev["text"] = "user", self._text_of(data.get("content"))
234
+ elif t == "assistant/message":
235
+ msg = data.get("message") or {}
236
+ ev["role"] = "assistant"
237
+ ev["text"] = self._text_of(msg.get("content"))
238
+ if self.include_reasoning:
239
+ for c in (msg.get("content") or []):
240
+ if isinstance(c, dict) and c.get("type") == "reasoning" \
241
+ and c.get("text"):
242
+ ev["text"] = (ev["text"] or "") + "\n[reasoning] " + str(c["text"])
243
+ elif t == "tool/call":
244
+ ev["role"] = "command"
245
+ ev["text"] = f"{data.get('name')}({data.get('arguments') or ''})"
246
+ elif t == "tool/result":
247
+ msg = data.get("message") or {}
248
+ inner = msg.get("content") or []
249
+ txt = ""
250
+ if inner and isinstance(inner[0], dict):
251
+ txt = self._text_of(inner[0].get("content"))
252
+ ev["role"], ev["text"] = "tool-output", txt
253
+ else:
254
+ continue
255
+ if ev.get("text"):
256
+ yield ev
257
+
258
+
259
+ # --------------------------------------------------------------------------
260
+ # 蜂巢任务事件源(M6:hive/jobs → contextual,D:\2_ai 蜂巢记忆架构设计.md §5)
261
+ # --------------------------------------------------------------------------
262
+
263
+ class HiveJobsSource(Source):
264
+ """蜂巢任务事件源。
265
+
266
+ 事件映射(设计稿 §5.3):
267
+ start → role=user,任务开始(task 摘要)
268
+ tool → **跳过**(防流水账爆炸;tool_trace 已在 result 有摘要)
269
+ handoff → role=assistant,续跑卡
270
+ error → role=assistant,「任务失败(job=…):…」(fix-pair 前件)
271
+ final → role=assistant,content 头
272
+ result.json 终态 → **权威确认事件**(progress 可能因强杀缺失 final):
273
+ done → 「任务完成(job=…)」;error/timeout/killed → 「任务失败(job=…)」
274
+
275
+ 排序:job_id 名升序 = 时间升序(job.rs 命名保证 h<unix_ms>_<pid>),
276
+ 跨 job 全序成立;seq 由本源按枚举顺序递增(ingest 去重键 = (session, seq))。
277
+ 解析失败的行/条目计入 self.skipped,不终杀批次。
278
+ """
279
+
280
+ name = "hive_jobs"
281
+
282
+ def __init__(self, root: str, error_sensitivity: str = "private"):
283
+ self.root = os.path.abspath(root)
284
+ self.skipped = 0
285
+ # §5.5「error:true 建议 private」——默认 private;调用方 clearance 不足时
286
+ # 可显式降级 error_sensitivity="internal"(显式权衡:接受失败细节入 internal,
287
+ # 换取误差归因原料不丢)。能否落盘由写入者 clearance 裁决(写隔离)。
288
+ self.error_sensitivity = error_sensitivity
289
+
290
+ def key(self):
291
+ """多仓/多池隔离:root 参与键(watermark 按源路径各自推进)。"""
292
+ return f"hive_jobs:{self.root}"
293
+
294
+ def _jobs(self):
295
+ try:
296
+ return sorted(d for d in os.listdir(self.root)
297
+ if d.startswith("h")
298
+ and os.path.isdir(os.path.join(self.root, d)))
299
+ except OSError:
300
+ return []
301
+
302
+ def _read_json(self, path):
303
+ try:
304
+ with open(path, encoding="utf-8") as f:
305
+ return json.load(f)
306
+ except (OSError, ValueError):
307
+ self.skipped += 1
308
+ return None
309
+
310
+ def events(self):
311
+ seq = 0
312
+ for job_id in self._jobs():
313
+ jdir = os.path.join(self.root, job_id)
314
+ sess = f"hive:{job_id}"
315
+ spec = self._read_json(os.path.join(jdir, "spec.json")) or {}
316
+ task = str(spec.get("user_prompt") or "")[:200]
317
+
318
+ has_final = False
319
+ prog = os.path.join(jdir, "progress.jsonl")
320
+ if os.path.isfile(prog):
321
+ try:
322
+ fh = open(prog, encoding="utf-8", errors="replace")
323
+ except OSError:
324
+ fh = None
325
+ if fh is not None:
326
+ with fh:
327
+ for line in fh:
328
+ line = line.strip()
329
+ if not line:
330
+ continue
331
+ try:
332
+ o = json.loads(line)
333
+ except ValueError:
334
+ self.skipped += 1
335
+ continue
336
+ kind = o.get("kind")
337
+ t = trust.epoch_seconds(float(o.get("ts") or 0)) or 0.0
338
+ if kind == "start":
339
+ seq += 1
340
+ yield {"t": t, "seq": seq, "session": sess,
341
+ "role": "user",
342
+ "text": f"任务开始(job={job_id}):{task}"}
343
+ elif kind == "handoff":
344
+ seq += 1
345
+ has_final = True
346
+ yield {"t": t, "seq": seq, "session": sess,
347
+ "role": "assistant",
348
+ "text": f"续跑卡(job={job_id}):"
349
+ f"{str(o.get('summary') or '')[:400]}"}
350
+ elif kind == "error":
351
+ seq += 1
352
+ ev = {"t": t, "seq": seq, "session": sess,
353
+ "role": "assistant",
354
+ "text": f"任务失败(job={job_id}):"
355
+ f"{str(o.get('error') or '')[:300]}"}
356
+ if self.error_sensitivity:
357
+ ev["sensitivity"] = self.error_sensitivity
358
+ yield ev
359
+ elif kind == "final":
360
+ seq += 1
361
+ has_final = True
362
+ yield {"t": t, "seq": seq, "session": sess,
363
+ "role": "assistant",
364
+ "text": str(o.get("content_head") or "")[:400]}
365
+ # tool / budget_stop / force_final / 其它 → 纯跳过
366
+ # (§5.3 原设计为聚合计数,实现从简——终态由 result.json
367
+ # 权威确认承载,无需逐条累计;防流水账爆炸目标不变。
368
+ # 注释曾照抄设计稿口径称「仅聚合计数」,与实现漂移,
369
+ # zcode 外评抓出,2026-09-23 修正)
370
+
371
+ # result.json 终态权威确认(progress 可能因强杀缺失 final)
372
+ res = self._read_json(os.path.join(jdir, "result.json")) or {}
373
+ state = "done" if res.get("ok") is True else \
374
+ ("error" if res.get("ok") is False else None)
375
+ if state is None:
376
+ continue
377
+ t = trust.epoch_seconds(float(res.get("finished_ts") or 0))
378
+ if not t:
379
+ # 兜底(批次8 实测缺陷):exec_cmd 旧版 result 无 finished_ts,
380
+ # t=0 会被水位整片过滤(确定性任务事件永远摄不进来)——
381
+ # 回落 result.json 的 mtime(产物诞生时间,同「产物说了算」族)
382
+ rpath = os.path.join(jdir, "result.json")
383
+ t = trust.epoch_seconds(os.path.getmtime(rpath)) if \
384
+ os.path.isfile(rpath) else 0.0
385
+ t = t or 0.0
386
+ head = str(res.get("content") or "")[:300]
387
+ err = str(res.get("error") or "")[:300]
388
+ seq += 1
389
+ if state == "done":
390
+ # 前缀独立成行:content 的行首结构(命令行等)不被破坏,
391
+ # 保证 mine_fix_pairs 的 _FIX_RE 行首启发式仍能命中修复证据
392
+ text = f"任务完成(job={job_id}):\n{head}"
393
+ # 重放命令行(批次8):cmd 任务(exec_cmd)的 result.steps 携带
394
+ # 原始 argv——附到事件正文,事件自身携带「做了什么」的可复放
395
+ # 信息,同时命令形态可被 _FIX_RE 行首启发式命中(fix-pair 原料)
396
+ steps = res.get("steps")
397
+ if isinstance(steps, list) and steps:
398
+ argv = (steps[-1] or {}).get("command") or []
399
+ if argv:
400
+ text += "\n" + " ".join(str(a) for a in argv)
401
+ yield {"t": t, "seq": seq, "session": sess, "role": "assistant",
402
+ "text": text}
403
+ else:
404
+ tag = "超时强杀" if st_err_is_timeout(err) else "失败"
405
+ ev = {"t": t, "seq": seq, "session": sess, "role": "assistant",
406
+ "text": f"任务{tag}(job={job_id}):{err or head}"
407
+ + ("" if has_final else "(progress 缺 final,"
408
+ "本条为 result 终态补位)")}
409
+ if self.error_sensitivity:
410
+ ev["sensitivity"] = self.error_sensitivity
411
+ yield ev
412
+
413
+
414
+ def st_err_is_timeout(err: str) -> bool:
415
+ return "超时" in (err or "")
416
+
417
+
418
+ # --------------------------------------------------------------------------
419
+ # 摄取器
420
+ # --------------------------------------------------------------------------
421
+
422
+ # 生效条件:required 形参 cg 总被存入并以其 cg.root 拼出 self.path=os.path.join(cg.root,"_sources.json");layer/sensitivity 原样存入,未传时取模块级常量 SESSION_LAYER、SESSION_SENSITIVITY 作为默认值。
423
+ class Ingestor:
424
+ """增量摄取:watermark + 去重 + 写节点 + 自动 fix-pair 挖掘。
425
+
426
+ watermark 文件:<root>/_sources.json
427
+ {source_key: {"t": 最后时间戳(epoch 秒), "seq": 最后序号, "count": 已摄取条数}}
428
+
429
+ 存量水位兼容:**旧版写入的是毫秒**,读回时经 `trust.epoch_seconds` 归一到秒,
430
+ 故升级后无需清 `_sources.json` 重建(否则 `ev["t"] < last_t` 恒真 → 新事件全被跳过)。
431
+ """
432
+
433
+ # 生效条件:传入带 root 的 cg 即成立,layer/sensitivity 默认 SESSION_LAYER/SESSION_SENSITIVITY 并原样存为属性,路径为 os.path.join(cg.root, "_sources.json")。
434
+ def __init__(self, cg, layer: str = SESSION_LAYER,
435
+ sensitivity: str = SESSION_SENSITIVITY):
436
+ self.cg = cg
437
+ self.layer = layer
438
+ self.sensitivity = sensitivity
439
+ self.path = os.path.join(cg.root, "_sources.json")
440
+
441
+ # ---- watermark ----
442
+
443
+ # 生效条件:self.path 存在、json.load 成功且结果为 dict 时返回该 dict;path 不存在、抛 ValueError/OSError 或结果非 dict 时返回 {"schema": 1, "sources": {}}。
444
+ def _load(self):
445
+ if os.path.exists(self.path):
446
+ try:
447
+ with open(self.path, encoding="utf-8") as f:
448
+ d = json.load(f)
449
+ if isinstance(d, dict):
450
+ return d
451
+ except (ValueError, OSError):
452
+ pass
453
+ return {"schema": 1, "sources": {}}
454
+
455
+ # 生效条件:传入 d 时以 ensure_ascii=False/indent=1 写入 self.path+".tmp",再 os.replace 覆盖 self.path。
456
+ def _save(self, d):
457
+ tmp = self.path + ".tmp"
458
+ with open(tmp, "w", encoding="utf-8") as f:
459
+ json.dump(d, f, ensure_ascii=False, indent=1)
460
+ os.replace(tmp, self.path)
461
+
462
+ # 生效条件:key 命中 self._load()["sources"] 时返回其值,缺 key 时返回 {}(_load 结果缺 "sources" 键则抛 KeyError)。
463
+ def watermark(self, key: str):
464
+ return self._load()["sources"].get(key, {})
465
+
466
+ # 生效条件:调用即返回 dict(self._load()["sources"]) 的浅拷贝(_load 结果缺 "sources" 键则抛 KeyError)。
467
+ def watermarks(self):
468
+ return dict(self._load()["sources"])
469
+
470
+ # ---- 摄取 ----
471
+
472
+ # 生效条件:对必需形参 source,先按 watermark(source.key()) 跳过 t<last_t(wm 的 "t" 为假值时视作 0)及 t==last_t 且 last_seq 非 None 且 ev["seq"] 为 int 且 <= last_seq 的事件、并按 (session, seq) 去重,max_events 为真值(非 0/None)时取满即停;dry_run 为假时逐条 cg.add(nid 已在 cg.index["nodes"] 中则跳过,cg.add 抛异常则 denied+=1、记 last_error 并继续),denied 与 last_error 同时成立时补 last_error/hint 与 denied_events 明细,mine_fix_pairs 为真且 new_events 非空且非 dry_run 时以 as_proposals=True 调 mine_fix_pairs(自动产物走审核队列)并结果附 fix_pairs,new_events 非空且非 dry_run 时以末事件写回水位(count 累加 written),最后返回 result。
473
+ def ingest(self, source, mine_fix_pairs: bool = False, max_events: int = None,
474
+ dry_run: bool = False):
475
+ """摄取一个源的新事件。返回统计。
476
+
477
+ 去重键:(session, seq) —— 同一事件不重复入库。
478
+ 增量:只处理 (t, seq) 大于 watermark 的事件。
479
+ mine_fix_pairs 默认 False(先落账后挖矿——§5.5 系统纪律化,
480
+ zcode 外评 break#2:默认 True 曾使 file/jsonl/mdcg_ingest 三入口
481
+ 绕过审核队列直写 knowledge 层);显式开启时产物走 propose 队列。
482
+ """
483
+ key = source.key()
484
+ wm = self.watermark(key)
485
+ # 水位单位归一:**存量水位是毫秒**(旧版 `_ts` 产出),不归一会让
486
+ # `ev["t"] < last_t` 恒真 → 升级后新事件被整片误判为「已处理」而跳过。
487
+ last_t, last_seq = trust.epoch_seconds(float(wm.get("t") or 0)) or 0.0, \
488
+ wm.get("seq")
489
+ new_events, seen = [], set()
490
+ denied_events = [] # 被拒事件明细(session/seq/原因)——可观测可重放
491
+ for ev in source.events():
492
+ if ev.get("t", 0) < last_t:
493
+ continue
494
+ if ev.get("t", 0) == last_t and last_seq is not None \
495
+ and isinstance(ev.get("seq"), int) and ev["seq"] <= last_seq:
496
+ continue
497
+ dedup = (ev.get("session"), ev.get("seq"))
498
+ if dedup in seen:
499
+ continue
500
+ seen.add(dedup)
501
+ new_events.append(ev)
502
+ if max_events and len(new_events) >= max_events:
503
+ break
504
+
505
+ written, ids, denied = 0, [], 0
506
+ if not dry_run:
507
+ for ev in new_events:
508
+ nid = "src_%s_%s" % (_sig(key)[:6], _sig(
509
+ f"{ev.get('session')}:{ev.get('seq')}")[:10])
510
+ if nid in self.cg.index["nodes"]:
511
+ continue # 幂等
512
+ body = ("# 功能名:会话事件\n"
513
+ f"# 生效条件:检索「{str(ev.get('text'))[:20]}」\n"
514
+ "# 子功能:记录会话事件\n"
515
+ f"# 执行:{str(ev.get('text'))[:80]}\n"
516
+ "# 验证方式:data(会话原始记录)\n"
517
+ "# 不适用条件:其它会话\n\n"
518
+ f"{ev.get('text')}\n")
519
+ try:
520
+ self.cg.add(nid, body, layer=self.layer, role=ev.get("role"),
521
+ tags=["session", key],
522
+ # 事件级密级覆盖(M6 §5.5):error 事件建议 private
523
+ # (失败细节可能含路径/配置),缺省回落源级默认
524
+ sensitivity=ev.get("sensitivity") or self.sensitivity,
525
+ verification_basis="data",
526
+ condition_space={"observation_position": key,
527
+ "observation_tool": "会话流",
528
+ "time_window": [ev.get("t") or 0,
529
+ ev.get("t") or 0]})
530
+ except Exception as exc: # noqa: BLE001 —— 权限/层错误不中断整批
531
+ denied += 1
532
+ # 诚实化(M6):denied 事件虽被水位跳过,但明细必须可见——
533
+ # 静默丢失会吞掉误差归因闭环的原料(error 事件恰是原料)
534
+ denied_events.append({
535
+ "session": ev.get("session"), "seq": ev.get("seq"),
536
+ "error": str(exc)[:150]})
537
+ self.last_error = f"{type(exc).__name__}: {exc}"
538
+ continue
539
+ ids.append(nid)
540
+ written += 1
541
+
542
+ result = {"source": key, "new_events": len(new_events), "written": written,
543
+ "denied": denied, "ids": ids, "dry_run": dry_run,
544
+ "sensitivity": self.sensitivity}
545
+ if denied_events:
546
+ result["denied_events"] = denied_events
547
+ if denied and getattr(self, "last_error", None):
548
+ result["last_error"] = self.last_error
549
+ result["hint"] = ("会话内容默认 sensitivity=private;"
550
+ "调用方需 MDCG_CLEARANCE=private 才能写入")
551
+ # 自动 fix-pair 挖掘(对标 deja-vu:错误→修复)——as_proposals=True:
552
+ # 自动管线产物走审核队列,绝不直写 knowledge 层(§5.5 系统纪律)
553
+ if mine_fix_pairs and new_events and not dry_run:
554
+ result["fix_pairs"] = self.cg.mine_fix_pairs(
555
+ [{"role": e.get("role"), "text": e.get("text")}
556
+ for e in new_events], as_proposals=True)
557
+
558
+ if new_events and not dry_run:
559
+ d = self._load()
560
+ last = new_events[-1]
561
+ d["sources"][key] = {
562
+ "t": last.get("t") or last_t,
563
+ "seq": last.get("seq"),
564
+ "count": (wm.get("count") or 0) + written,
565
+ "updated_at": time.time(),
566
+ "path": getattr(source, "path", None),
567
+ }
568
+ self._save(d)
569
+ return result
570
+
571
+
572
+ # 生效条件:text 为 None 或假值时按 "" 参与 sha1;n 默认 12,返回 hexdigest 前 n 位(n=0 得空串)。
573
+ def _sig(text: str, n: int = 12) -> str:
574
+ import hashlib
575
+ return hashlib.sha1((text or "").encode("utf-8")).hexdigest()[:n]
576
+
577
+
578
+ # --------------------------------------------------------------------------
579
+ # 摄取分派(P0 · ingest op):按扩展名选摄取方式,单一入口吃多种文件
580
+ # --------------------------------------------------------------------------
581
+ #
582
+ # 三条摄取链:
583
+ # session —— 会话流(.jsonl):走 Ingestor(watermark + 去重 + fix-pair)
584
+ # doc —— 文档(.md/.txt/...):走 docindex.extract + refindex.add_items
585
+ # code —— 代码(.py/.ts/...):走 codeindex.extract + refindex.add_items
586
+ #
587
+ # 设计要点:
588
+ # - 注册表 INGEST_REGISTRY 是唯一真源:新增后缀只改这里。
589
+ # - dir 动作分链处理:目录里 doc / code / jsonl 混放时各链互不干扰。
590
+ # - 幂等:沿用 refindex.Ledger(size+mtime 水位)与 Ingestor watermark。
591
+ # - 预演:dry_run=True 只统计、不写入(对应计划「可预演」要求)。
592
+
593
+ INGEST_ACTIONS = ("file", "dir", "jsonl", "stat", "hive")
594
+
595
+ INGEST_REGISTRY = {
596
+ # 会话流
597
+ ".jsonl": "session", ".ndjson": "session",
598
+ # 文档
599
+ ".md": "doc", ".markdown": "doc", ".txt": "doc", ".rst": "doc",
600
+ ".html": "doc", ".htm": "doc",
601
+ # 代码
602
+ ".py": "code", ".js": "code", ".mjs": "code", ".ts": "code",
603
+ ".tsx": "code", ".jsx": "code", ".go": "code", ".rs": "code",
604
+ ".java": "code", ".kt": "code", ".swift": "code", ".rb": "code",
605
+ ".php": "code", ".cs": "code", ".cpp": "code", ".cc": "code",
606
+ ".c": "code", ".h": "code", ".hpp": "code",
607
+ }
608
+
609
+
610
+ # 生效条件:path 为 None 或空串时 splitext 得 "" 且未登记 → 返回 None;扩展名(小写)存在于 INGEST_REGISTRY 时返回其 kind。
611
+ def dispatch_of(path: str):
612
+ """按扩展名返回摄取方式(session / doc / code);未登记返回 None。"""
613
+ return INGEST_REGISTRY.get(os.path.splitext(path or "")[1].lower())
614
+
615
+
616
+ # 生效条件:required 形参 cg 总被存入;可选形参 sensitivity 为假值(None/空串)时内部 Ingestor 的 sensitivity 回落模块级常量 SESSION_SENSITIVITY,为真值时用传入的 sensitivity。
617
+ class FileDispatcher:
618
+ """单一入口吃多种文件:按扩展名分派到会话流 / 文档 / 代码三条摄取链。"""
619
+
620
+ # 生效条件:传入 cg 即成立,sensitivity 为假值(None/空串)时所用 Ingestor 回落 SESSION_SENSITIVITY,否则用传入值。
621
+ def __init__(self, cg, sensitivity=None):
622
+ self.cg = cg
623
+ # 会话链默认 sensitivity=private;调用方可显式覆盖(测试/受限环境)
624
+ self.ingestor = Ingestor(cg, sensitivity=sensitivity or SESSION_SENSITIVITY)
625
+
626
+ # ---- stat:看水位与支持面 ----
627
+
628
+ # 生效条件:from . import refindex 与 refindex.Ledger(self.cg.root).stat() 均不抛异常时返回该 stat 结果,抛任何异常时返回 {}。
629
+ def _ledger_stat(self):
630
+ try:
631
+ from . import refindex
632
+ return refindex.Ledger(self.cg.root).stat()
633
+ except Exception: # noqa: BLE001
634
+ return {}
635
+
636
+ # 生效条件:调用即返回由 INGEST_REGISTRY 按 kind 分组并排序后的 extensions、list(INGEST_ACTIONS)、self.ingestor.watermarks()、self._ledger_stat() 与固定 note。
637
+ def stat(self):
638
+ kinds = {}
639
+ for ext, kind in INGEST_REGISTRY.items():
640
+ kinds.setdefault(kind, []).append(ext)
641
+ return {"ok": True, "kind": "stat", "actions": list(INGEST_ACTIONS),
642
+ "extensions": {k: sorted(v) for k, v in sorted(kinds.items())},
643
+ "watermarks": self.ingestor.watermarks(),
644
+ "ledger": self._ledger_stat(),
645
+ "note": ("注册表是唯一真源:新增后缀只改 INGEST_REGISTRY。"
646
+ "watermarks=会话流水位;ledger=文档/代码文件水位。")}
647
+
648
+ # ---- 单文件 ----
649
+
650
+ # 生效条件:dispatch_of(path) 为 None 时返回 ok=False 的「不支持的后缀」结果;path 不是文件时返回 ok=False 的「文件不存在」结果;kind=="session" 时转 ingest_jsonl(path, dry_run=dry_run)(layer/sensitivity 不参与);其它 kind 转 _ingest_doc_or_code(path, kind, layer=layer, sensitivity=sensitivity, dry_run=dry_run)。
651
+ def ingest_file(self, path, layer=None, sensitivity=None, dry_run=False):
652
+ kind = dispatch_of(path)
653
+ if kind is None:
654
+ ext = os.path.splitext(path or "")[1] or "(无后缀)"
655
+ return {"ok": False, "path": path, "error": f"不支持的后缀:{ext}",
656
+ "supported": sorted(INGEST_REGISTRY)}
657
+ if not os.path.isfile(path):
658
+ return {"ok": False, "path": path, "error": "文件不存在"}
659
+ if kind == "session":
660
+ return self.ingest_jsonl(path, dry_run=dry_run)
661
+ return self._ingest_doc_or_code(path, kind, layer=layer,
662
+ sensitivity=sensitivity, dry_run=dry_run)
663
+
664
+ # 生效条件:kind=="code" 用 codeindex 否则用 docindex;mod.extract 抛 ValueError 时返回 ok=False 的「抽取失败」;dry_run 为真时只返回 items 计数与前 20 个 node_id 不写盘;否则经 refindex.add_items(kind 为 code_ref/doc_ref)写入并返回 indexed、ids[:20] 与 sensitivity。
665
+ def _ingest_doc_or_code(self, path, kind, layer=None, sensitivity=None,
666
+ dry_run=False):
667
+ from . import codeindex, docindex, refindex
668
+ mod = codeindex if kind == "code" else docindex
669
+ rel = os.path.basename(path)
670
+ with open(path, encoding="utf-8", errors="replace") as f:
671
+ src = f.read()
672
+ try:
673
+ items = mod.extract(src, path=rel,
674
+ suffix=os.path.splitext(path)[1].lower())
675
+ except ValueError as exc:
676
+ return {"ok": False, "path": path, "error": f"抽取失败:{exc}"}
677
+ items = items or []
678
+ if dry_run:
679
+ return {"ok": True, "dry_run": True, "kind": kind, "path": path,
680
+ "items": len(items),
681
+ "ids": [mod.node_id(i) for i in items[:20]],
682
+ "note": "预演:只抽取计数,未写盘"}
683
+ ref_kind = "code_ref" if kind == "code" else "doc_ref"
684
+ ids, sens = refindex.add_items(
685
+ self.cg, items, kind=ref_kind,
686
+ root=os.path.dirname(path) or ".", layer=layer,
687
+ sensitivity=sensitivity)
688
+ return {"ok": True, "kind": kind, "path": path, "items": len(items),
689
+ "indexed": len(ids), "ids": ids[:20], "sensitivity": sens}
690
+
691
+ # ---- 目录 ----
692
+
693
+ # 生效条件:调用即 os.walk(root) 统计每个文件名经 dispatch_of 得到的 kind(无匹配记 "unsupported")并返回 dry_run 预演计数结果。
694
+ def _dry_dir(self, root):
695
+ counts = {}
696
+ for _dp, _dn, fns in os.walk(root):
697
+ for name in fns:
698
+ k = dispatch_of(name) or "unsupported"
699
+ counts[k] = counts.get(k, 0) + 1
700
+ return {"ok": True, "dry_run": True, "kind": "dir", "root": root,
701
+ "counts": counts,
702
+ "note": "预演:仅统计各链文件数,未做任何写入"}
703
+
704
+ # 生效条件:root 非目录时返回 ok=False 的「目录不存在」;dry_run 为真时返回 _dry_dir(root);否则对 doc_ref/code_ref 两链各以 patterns/max_files/max_items/incremental/ledger 调 refindex.index_dir 与 add_items,并把 root 下 **/*.jsonl 前 max_files 个逐个 ingest_jsonl 后返回 out。
705
+ def ingest_dir(self, root, layer=None, sensitivity=None, patterns=None,
706
+ max_files=500, max_items=2000, incremental=False,
707
+ dry_run=False):
708
+ from . import refindex
709
+ if not os.path.isdir(root):
710
+ return {"ok": False, "error": f"目录不存在:{root}"}
711
+ if dry_run:
712
+ return self._dry_dir(root)
713
+ ledger = refindex.Ledger(self.cg.root)
714
+ out = {"ok": True, "kind": "dir", "root": root, "chains": {}}
715
+ for ref_kind, key in (("doc_ref", "doc"), ("code_ref", "code")):
716
+ items, errors, stats = refindex.index_dir(
717
+ root, kind=ref_kind, patterns=patterns, max_files=max_files,
718
+ max_items=max_items, incremental=incremental, ledger=ledger)
719
+ ids, sens = refindex.add_items(self.cg, items, kind=ref_kind,
720
+ root=root, layer=layer,
721
+ sensitivity=sensitivity)
722
+ out["chains"][key] = {
723
+ "indexed": len(ids), "errors": len(errors),
724
+ "files": stats.get("files"), "truncated": stats.get("truncated"),
725
+ "skipped_unchanged": stats.get("skipped_unchanged", 0),
726
+ "skipped_suffixes": stats.get("skipped_suffixes", []),
727
+ "sensitivity": sens}
728
+ # 会话流(.jsonl)逐个增量摄取
729
+ jsons = sorted(glob.glob(os.path.join(root, "**", "*.jsonl"),
730
+ recursive=True))[:max_files]
731
+ ses = []
732
+ for p in jsons:
733
+ r = self.ingest_jsonl(p)
734
+ ses.append({"path": p, "written": r.get("written", 0),
735
+ "new_events": r.get("new_events", 0)})
736
+ out["chains"]["session"] = {"files": len(jsons), "results": ses}
737
+ return out
738
+
739
+ # ---- 会话流 ----
740
+
741
+ @staticmethod
742
+ # 生效条件:required 形参 path 能被 open(...,encoding="utf-8",errors="replace") 打开时读前 50000 字符,head 含 '"user/message"'、'"assistant/message"'、'"tool/call"' 任一标记则返回 DSHSessionSource(path),否则返回 JsonlSource(path);打开抛 OSError 时直接返回 JsonlSource(path)。
743
+ def _auto_source(path):
744
+ """通用 JSONL vs DSH 会话:按内容探测,避免调用方选错源类型。"""
745
+ try:
746
+ with open(path, encoding="utf-8", errors="replace") as f:
747
+ head = f.read(50000)
748
+ except OSError:
749
+ return JsonlSource(path)
750
+ for marker in ('"user/message"', '"assistant/message"', '"tool/call"'):
751
+ if marker in head:
752
+ return DSHSessionSource(path)
753
+ return JsonlSource(path)
754
+
755
+ # 生效条件:path 不是文件时返回 ok=False 的「文件不存在」;否则经 _auto_source(path) 选源后调 self.ingestor.ingest(src, dry_run=dry_run, max_events=max_events),并补上 ok=True/kind/path/source_class 后返回。
756
+ def ingest_jsonl(self, path, dry_run=False, max_events=None):
757
+ if not os.path.isfile(path):
758
+ return {"ok": False, "error": f"文件不存在:{path}"}
759
+ src = self._auto_source(path)
760
+ res = self.ingestor.ingest(src, dry_run=dry_run, max_events=max_events)
761
+ res.update({"ok": True, "kind": "session", "path": path,
762
+ "source_class": type(src).__name__})
763
+ return res
764
+
765
+
766
+ # 生效条件:cg 必填;action 为 None/空串时 (action or "stat") 归为 stat;file/jsonl 缺 path 返回 ok=False;dir 的 max_files/max_items 走 int(x or 500)/int(x or 2000),传 0 也变 500/2000;未知 action 抛 ValueError。
767
+ def run(cg, action: str = "stat", **kw):
768
+ """ingest op 唯一入口。"""
769
+ act = (action or "stat").strip().lower()
770
+ d = FileDispatcher(cg, sensitivity=kw.get("sensitivity"))
771
+ if act == "stat":
772
+ return d.stat()
773
+ if act == "file":
774
+ p = kw.get("path")
775
+ if not p:
776
+ return {"ok": False, "error": "file 动作需要 path"}
777
+ return d.ingest_file(p, layer=kw.get("layer"),
778
+ sensitivity=kw.get("sensitivity"),
779
+ dry_run=bool(kw.get("dry_run")))
780
+ if act == "dir":
781
+ p = kw.get("path") or cg.root
782
+ return d.ingest_dir(p, layer=kw.get("layer"),
783
+ sensitivity=kw.get("sensitivity"),
784
+ patterns=kw.get("patterns"),
785
+ max_files=int(kw.get("max_files") or 500),
786
+ max_items=int(kw.get("max_items") or 2000),
787
+ incremental=bool(kw.get("incremental")),
788
+ dry_run=bool(kw.get("dry_run")))
789
+ if act == "jsonl":
790
+ p = kw.get("path")
791
+ if not p:
792
+ return {"ok": False, "error": "jsonl 动作需要 path"}
793
+ return d.ingest_jsonl(p, dry_run=bool(kw.get("dry_run")),
794
+ max_events=kw.get("max_events"))
795
+ if act == "hive":
796
+ # M6:蜂巢任务事件源(hive/jobs → contextual)。root 解析链:显式 path >
797
+ # env MDCG_HIVE_JOBS > 报错指引(fail-closed,不猜仓库布局)。
798
+ # §5.5 硬纪律:mine_fix_pairs=False(先落账后挖矿——自动挖掘产物直写
799
+ # knowledge 层违反双轨制,实测 0→2 污染);默认密级 internal,
800
+ # error 事件由源层 per-event 覆写 private(失败细节可能含路径/配置)。
801
+ root = kw.get("path") or os.environ.get("MDCG_HIVE_JOBS")
802
+ if not root:
803
+ return {"ok": False, "error": (
804
+ "hive 动作需要 path(hive/jobs 目录),或设 env MDCG_HIVE_JOBS——"
805
+ "不猜测仓库布局(fail-closed)")}
806
+ src = HiveJobsSource(root, error_sensitivity=kw.get("error_sensitivity")
807
+ if kw.get("error_sensitivity") is not None else "private")
808
+ rep = Ingestor(cg, layer=kw.get("layer") or "contextual",
809
+ sensitivity=kw.get("sensitivity") or "internal"
810
+ ).ingest(src, mine_fix_pairs=False,
811
+ max_events=kw.get("max_events"),
812
+ dry_run=bool(kw.get("dry_run")))
813
+ rep["source"] = src.key()
814
+ rep["skipped"] = src.skipped
815
+ return rep
583
816
  raise ValueError(f"未知 ingest action:{action!r}(允许 {INGEST_ACTIONS})")