@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/refindex.py CHANGED
@@ -1,834 +1,834 @@
1
- # -*- coding: utf-8 -*-
2
- """md_cg · 统一 ref 协议 + 索引水位(增量)+ 漂移/悬空巡检
3
-
4
- 对照 `docs/mdcg/认知图_索引与工程规范化_计划_v0.1.md` 的 R3(修 D + 修 F):
5
-
6
- **D 漂移 / 悬空检测(本模块 `check_refs`)**
7
- 扫描带 `code_ref` / `doc_ref` 的节点,回两类问题:
8
- · `stale` —— 源文件被改(区间哈希不再匹配)
9
- · `dangling` —— 源文件被删(索引指向不存在的文件)
10
- 巡检**只读**、**不抛**、**不改源文件**;修复动作是「重跑 index_code / index_doc」,
11
- 因为索引是派生物(对齐 sustain.heal 的既有边界)。
12
-
13
- **F 全量重扫 + 静默截断(本模块 `Ledger` + `index_dir`)**
14
- `<root>/_refindex.json` 是 ref 索引水位(抄 `sources.Ingestor` 的 `_sources.json` 范式),
15
- 以**源文件绝对路径**为键,记每个源文件的 (size, mtime) 与节点区间 + 它所属的**源大域
16
- root**;`incremental=True` 时未变文件**不再读盘重切**,直接跳过(`skipped_unchanged`)
17
- ——这就是「不全量重扫」。
18
- 键用绝对路径、且逐文件记 root,是因为一份认知图可以索引多个大域:只按 rel 记会在同名
19
- 文件上互相覆盖,巡检时若拿认知图根去拼路径则会把一切都误判成 dangling。
20
- 截断(`max_files` / `max_items`)由 codeindex / docindex 显式上报,本模块把
21
- 「最近一次索引被截断」写进水位,交给 `sustain.diagnose` 巡检看见(不再静默)。
22
-
23
- **为什么回读要收进本模块**
24
- `op=ref` 的回读与 `check_refs` 的判定**必须共用同一实现**,否则会出现
25
- 「回读说没漂、巡检说有漂」。与 `region_hash` 的教训同源:区间哈希只允许一份实现,
26
- 这里连「怎么判定 ok / stale / dangling」也只允许一份。
27
-
28
- 零第三方依赖。
29
- """
30
- from __future__ import annotations
31
-
32
- import json
33
- import os
34
- import time
35
-
36
- from .fsutil import atomic_write
37
-
38
- SCHEMA = 2 # v2:水位以「源文件绝对路径」为键(v1 按 rel 会跨大域撞名)
39
- LEDGER_FILE = "_refindex.json"
40
- REF_KEYS = ("code_ref", "doc_ref")
41
- MAX_CHECK = 2000 # 巡检节点上限(超出报 truncated,不静默截断)
42
- STATUSES = ("ok", "stale", "dangling", "unresolved", "error")
43
-
44
-
45
- # 生效条件:给定 fp 时返回 os.path.abspath(fp or '')(fp 为空/None 则返回当前目录的绝对路径),作为水位键以绝对路径保证不同 root 下同名文件不互相覆盖。
46
- def _src_key(fp: str) -> str:
47
- """水位的键 = 源文件绝对路径。
48
-
49
- 不能用 rel:一份认知图可以索引多个大域(不同 root),只按 rel 记会在
50
- `alpha.py` 这种同名文件上互相覆盖——水位被静默丢掉,巡检就漏报。
51
- """
52
- return os.path.abspath(fp or "")
53
-
54
-
55
- # 生效条件:无 required 形参,任何调用都返回 round(time.time(), 1),把时间戳压到 1 位小数以稳定 `_refindex.json` 字节数。
56
- def _now() -> float:
57
- """时间戳压到 1 位小数:让 `_refindex.json` 字节数稳定(重跑不涨),
58
- 同时保留足够的「多久以前」信息(float 的最短 repr 保证小数位固定为 1)。"""
59
- return round(time.time(), 1)
60
-
61
-
62
- # --------------------------------------------------------------------------
63
- # 提取器注册表(统一调度:调用方只说 kind,不说「用哪个模块」)
64
- # --------------------------------------------------------------------------
65
-
66
- # 生效条件:kind == 'code_ref' 返回 codeindex、kind == 'doc_ref' 返回 docindex,其他 kind 抛 ValueError(提示支持 REF_KEYS)。
67
- def _mod(kind: str):
68
- from . import codeindex, docindex
69
- if kind == "code_ref":
70
- return codeindex
71
- if kind == "doc_ref":
72
- return docindex
73
- raise ValueError(f"未知 ref kind:{kind!r}(支持 {REF_KEYS})")
74
-
75
-
76
- # 生效条件:无 required 形参,调用即返回 {'code_ref': {'suffixes': tuple(codeindex.SUFFIX)}, 'doc_ref': {'suffixes': tuple(docindex.SUFFIX)}}。
77
- def registry() -> dict:
78
- """后缀 → kind 的注册表(code / doc 各一份提取器)。"""
79
- from . import codeindex, docindex
80
- return {
81
- "code_ref": {"suffixes": tuple(codeindex.SUFFIX)},
82
- "doc_ref": {"suffixes": tuple(docindex.SUFFIX)},
83
- }
84
-
85
-
86
- # 生效条件:path 的小写后缀在 codeindex.EXTRACTORS 中返回 'code_ref',在 docindex.SUFFIX 中返回 'doc_ref',无后缀或均不匹配返回 ''。
87
- def kind_of_path(path: str) -> str:
88
- """按后缀判 kind;无提取器返回 ''(由调用方决定是报错还是跳过)。"""
89
- from . import codeindex, docindex
90
- ext = os.path.splitext(path or "")[1].lower()
91
- if not ext:
92
- return ""
93
- if ext in codeindex.EXTRACTORS:
94
- return "code_ref"
95
- if ext in docindex.SUFFIX:
96
- return "doc_ref"
97
- return ""
98
-
99
-
100
- # 生效条件:source 为待提取文本,kind 非空或 path 后缀能推出 kind 时返回 _mod(k).extract(source, path),推不出 kind 时抛 ValueError。
101
- def extract(source: str, path: str = "", kind: str = ""):
102
- """统一提取入口:按 kind(或从 path 推断)分发到对应 extractor。"""
103
- k = kind or kind_of_path(path)
104
- if not k:
105
- ext = os.path.splitext(path or "")[1] or "<none>"
106
- raise ValueError(f"无索引提取器(suffix={ext})")
107
- return _mod(k).extract(source, path)
108
-
109
-
110
- # 生效条件:kind 为 'code_ref'/'doc_ref' 时返回 _mod(kind).node_id(item),其他 kind 由 _mod 抛 ValueError。
111
- def node_id_of(item: dict, kind: str) -> str:
112
- return _mod(kind).node_id(item)
113
-
114
-
115
- # 生效条件:kind 为 'code_ref'/'doc_ref' 时返回 _mod(kind).render(item),其他 kind 由 _mod 抛 ValueError。
116
- def render_of(item: dict, kind: str) -> str:
117
- return _mod(kind).render(item)
118
-
119
-
120
- # 生效条件:node 的 frontmatter 中 REF_KEYS 命中且值为非空 dict 时返回 {'ref': ref, 'ref_kind': kind},否则返回 {'ref': None, 'ref_kind': ''}。
121
- def ref_fields(node) -> dict:
122
- """节点 → 检索结果要带的两字段(读侧只加字段,不改召回逻辑)。"""
123
- kind, ref = ref_of(node)
124
- return {"ref": ref, "ref_kind": kind} if ref else {"ref": None, "ref_kind": ""}
125
-
126
-
127
- # 生效条件:node 的 frontmatter 按 REF_KEYS 顺序取到第一个非空 dict 时返回 (k, r),否则返回 ('', None)。
128
- def ref_of(node) -> tuple:
129
- """从节点 frontmatter 取 ref:返回 (kind, ref) 或 ('', None)。"""
130
- fm = (node or {}).get("frontmatter") or {}
131
- for k in REF_KEYS:
132
- r = fm.get(k)
133
- if isinstance(r, dict) and r:
134
- return k, r
135
- return "", None
136
-
137
-
138
- # --------------------------------------------------------------------------
139
- # 索引水位(_refindex.json):增量 + 截断留痕
140
- # --------------------------------------------------------------------------
141
-
142
- # 生效条件:以 root 为必填实参构造,实例化即置 self.root=root、self.path=os.path.join(root, LEDGER_FILE)、self._d=None;
143
- class Ledger:
144
- """`<root>/_refindex.json`:每个源文件的 (size, mtime) 水位 + 节点区间。"""
145
-
146
- # 生效条件:当传入 root 时,self.root 取该 root,self.path 为 os.path.join(root, LEDGER_FILE),self._d 置为 None;
147
- def __init__(self, root: str):
148
- self.root = root
149
- self.path = os.path.join(root, LEDGER_FILE)
150
- self._d = None
151
-
152
- # 生效条件:当 self._d is not None 时直接返回 self._d;否则读取 self.path 的 JSON,仅当 obj 是 dict 且 obj.get("schema") == SCHEMA 且 obj.get("files") 是 dict 时用 obj,否则(含 OSError/ValueError、结构不符)回落为 {"schema": SCHEMA, "updated_at": 0.0, "files": {}} 并缓存返回;
153
- def load(self) -> dict:
154
- if self._d is not None:
155
- return self._d
156
- d = None
157
- try:
158
- with open(self.path, "r", encoding="utf-8") as f:
159
- obj = json.load(f)
160
- if isinstance(obj, dict) and obj.get("schema") == SCHEMA \
161
- and isinstance(obj.get("files"), dict):
162
- d = obj
163
- except (OSError, ValueError):
164
- d = None
165
- self._d = d or {"schema": SCHEMA, "updated_at": 0.0, "files": {}}
166
- return self._d
167
-
168
- # 生效条件:传入 rel、fp 时,若 self.load()["files"].get(_src_key(fp)) 缺失或为假值、或 os.stat(fp) 抛 OSError、或条目 e.get("size") != st.st_size,则返回 False;否则返回 abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6(mtime 缺失或假值时按 0.0);
169
- def is_fresh(self, rel: str, fp: str) -> bool:
170
- """源文件自上次索引后未变(size + mtime 双等)→ 可跳过不重切。"""
171
- e = self.load()["files"].get(_src_key(fp))
172
- if not e:
173
- return False
174
- try:
175
- st = os.stat(fp)
176
- except OSError:
177
- return False
178
- if e.get("size") != st.st_size:
179
- return False
180
- return abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6
181
-
182
- # 生效条件:当 rel、fp、kind、nodes 传入且 os.stat(fp) 成功时,向 self.load()["files"][_src_key(fp)] 写条目,其中 root 为 root if root else os.path.dirname(key)、path 为 rel、kind 为 kind、size/mtime 取 st、nodes 为每项 n.get("id")/n.get("lineno")/n.get("end")/n.get("hash");os.stat(fp) 抛 OSError 时不写入;
183
- def record(self, rel: str, fp: str, kind: str, nodes,
184
- root: str = None) -> None:
185
- """记一个源文件的水位(节点区间用于判 stale)。
186
-
187
- `root` 是**源**大域的根(≠ 认知图根):巡检要拿它拼 `root/rel` 才能
188
- 找到源文件,缺了它就会把「源在别处」误判成 dangling。
189
- """
190
- key = _src_key(fp)
191
- try:
192
- st = os.stat(fp)
193
- except OSError:
194
- return
195
- self.load()["files"][key] = {
196
- "root": root if root else os.path.dirname(key),
197
- "path": rel,
198
- "size": st.st_size,
199
- "mtime": st.st_mtime,
200
- "kind": kind,
201
- "nodes": [
202
- {"id": n.get("id"), "lineno": n.get("lineno"),
203
- "end": n.get("end"), "hash": n.get("hash")}
204
- for n in nodes
205
- ],
206
- }
207
-
208
- # 生效条件:当传入 fp 时,self.load()["files"].pop(_src_key(fp), None),即删除对应键(不存在也静默);
209
- def drop(self, fp: str) -> None:
210
- self.load()["files"].pop(_src_key(fp), None)
211
-
212
- # 生效条件:当传入 root、kind、seen 时,对 self.load()["files"] 中满足 os.path.abspath(e.get("root") or "") == os.path.abspath(root) 且 e.get("kind") == kind 且键 k 不在 seen 的条目删除,返回删除数量;
213
- def reconcile(self, root: str, kind: str, seen) -> int:
214
- """一次**完整**索引后对账:本 (root, kind) 下没被扫到的旧条目剪掉。
215
-
216
- 否则「源文件被删 → 索引悬空 → heal 重建」之后条目还在,巡检就永远报
217
- dangling,heal 是治不好的。`seen` 是本次真正走过(提取成功或判定未变)
218
- 的源文件键集合。**截断的索引不能对账**——没扫完不等于剩下的都消失了。
219
- """
220
- r = os.path.abspath(root)
221
- files = self.load()["files"]
222
- dead = [k for k, e in files.items()
223
- if os.path.abspath(e.get("root") or "") == r
224
- and e.get("kind") == kind and k not in seen]
225
- for k in dead:
226
- files.pop(k, None)
227
- return len(dead)
228
-
229
- # 生效条件:对 load()["files"] 中「条目 root(为假值时用 os.path.dirname(键) 兜底)不是目录」的条目逐一 pop 并返回删除条数,无匹配时返回 0。
230
- def prune(self) -> int:
231
- """剪掉「源大域已不存在」的条目(整个目录被搬走/删除)。
232
-
233
- 这类条目已不可能再被任何大域索引到,留着只会在巡检里报永不消失的
234
- dangling;而节点自带的 ref 仍会兜底探测,所以剪掉不会漏报真实悬空。
235
- """
236
- files = self.load()["files"]
237
- dead = [k for k, e in files.items()
238
- if not os.path.isdir(e.get("root") or os.path.dirname(k))]
239
- for k in dead:
240
- files.pop(k, None)
241
- return len(dead)
242
-
243
- # 生效条件:当 kind、root、files、indexed、truncated 传入时,self.load()["last_index"] 被设为含 ts=_now()、kind、root、files、indexed、truncated=bool(truncated)、truncated_reason=reason or "" 的字典;reason 为假值(默认 ""/None)时 truncated_reason 回落 "";
244
- def note_index(self, *, kind: str, root: str, files: int, indexed: int,
245
- truncated: bool, reason: str = "") -> None:
246
- """记「最近一次索引」结果——截断在这里留痕,供 diagnose 看见。"""
247
- self.load()["last_index"] = {
248
- "ts": _now(), "kind": kind, "root": root, "files": files,
249
- "indexed": indexed, "truncated": bool(truncated),
250
- "truncated_reason": reason or "",
251
- }
252
-
253
- # 生效条件:无参数调用即生效,取 self.load() 结果把 updated_at 置为 _now(),再以 atomic_write 把 json.dumps(..., ensure_ascii=False, indent=1, sort_keys=True) 写入 self.path,无返回值。
254
- def save(self) -> None:
255
- d = self.load()
256
- d["updated_at"] = _now()
257
- atomic_write(self.path, json.dumps(d, ensure_ascii=False,
258
- indent=1, sort_keys=True))
259
-
260
- # 生效条件:无参数调用即生效,返回含 path、schema、load()["files"] 条目数、nodes 总数(各条目 nodes 列表长度之和)、updated_at、exists=os.path.isfile(self.path) 的 out;age_s 在 updated_at 为假值(0.0)时为 None,否则为 max(0.0, time.time()-up);仅当 load() 的 last_index 为 dict 时才并入 out["last_index"]。
261
- def summary(self) -> dict:
262
- d = self.load()
263
- files = d.get("files") or {}
264
- nodes = sum(len(e.get("nodes") or []) for e in files.values())
265
- up = float(d.get("updated_at") or 0.0)
266
- out = {
267
- "path": self.path,
268
- "schema": d.get("schema"),
269
- "files": len(files),
270
- "nodes": nodes,
271
- "updated_at": up,
272
- "age_s": None if not up else max(0.0, time.time() - up),
273
- "exists": os.path.isfile(self.path),
274
- }
275
- if isinstance(d.get("last_index"), dict):
276
- out["last_index"] = d["last_index"]
277
- return out
278
-
279
-
280
- # --------------------------------------------------------------------------
281
- # 统一 index_dir:调度 + 水位 + 落盘(供 op=index_code / op=index_doc / heal 共用)
282
- # --------------------------------------------------------------------------
283
-
284
- # 生效条件:root 为源大域根、kind 为 'code_ref'/'doc_ref' 时经 _mod(kind) 调度底层 index_dir 并返回 (items, errors, stats);ledger 非空时逐文件 record,incremental 为真时跳过 ledger.is_fresh 为真的文件,且 stats 未截断时执行 reconcile。
285
- def index_dir(root: str, *, kind: str, patterns=None, max_files: int = 500,
286
- max_items: int = 2000, incremental: bool = False,
287
- ledger: "Ledger" = None, skip_dirs=None):
288
- """按 kind 调度 codeindex / docindex 的全量(或增量)索引。
289
-
290
- incremental=True 且给了 ledger 时:未变文件跳过(`skipped_unchanged`)。
291
- 返回 (items, errors, stats),与底层 index_dir 的返回一致(多一个
292
- `skipped_unchanged`)。
293
-
294
- `skip_dirs` 透传给底层:**追加**排除、只增不减(内置 `.git`/`.venv`/
295
- `node_modules` 等不可被关闭),见 `codeindex.skip_matcher`。实际排掉了哪些目录
296
- 由 `stats["skipped_dirs"]` 回报,仍不静默。
297
- """
298
- mod = _mod(kind)
299
- fresh = None
300
- on_file = None
301
- seen = set() # 本次真正走过的源文件(用于对账)
302
- if ledger is not None:
303
- if incremental:
304
- # 生效条件:当 rel、fp 传入时,ok = ledger.is_fresh(rel, fp);若 ok 为真则将 _src_key(fp) 加入 seen 并返回 ok,若 ok 为假则直接返回 False;
305
- def fresh(rel, fp): # noqa: E306
306
- ok = ledger.is_fresh(rel, fp)
307
- if ok:
308
- seen.add(_src_key(fp))
309
- return ok
310
-
311
- # 生效条件:当 rel、fp、got 传入时,将 _src_key(fp) 加入 seen,并以 root=root 调用 ledger.record(rel, fp, kind, [{"id": node_id_of(it, kind), "lineno": it.get("lineno"), "end": it.get("end"), "hash": it.get("hash")} for it in got]);
312
- def on_file(rel, fp, got): # noqa: E306
313
- seen.add(_src_key(fp))
314
- ledger.record(rel, fp, kind,
315
- [{"id": node_id_of(it, kind), "lineno": it.get("lineno"),
316
- "end": it.get("end"), "hash": it.get("hash")} for it in got],
317
- root=root)
318
-
319
- items, errors, stats = mod.index_dir(
320
- root, patterns=patterns, max_files=max_files, max_items=max_items,
321
- fresh=fresh, on_file=on_file, skip_dirs=skip_dirs,
322
- )
323
- if ledger is not None:
324
- ledger.prune()
325
- if not stats.get("truncated"):
326
- # 没扫完就不能对账:截断时「没见到」不等于「源已消失」。
327
- ledger.reconcile(root, kind, seen)
328
- ledger.note_index(kind=kind, root=root, files=stats.get("files", 0),
329
- indexed=len(items), truncated=bool(stats.get("truncated")),
330
- reason=stats.get("truncated_reason") or "")
331
- ledger.save()
332
- return items, errors, stats
333
-
334
-
335
- # 生效条件:it['path'] 非空时返回其首段 path.split('/')[0] 作为 domain 键,path 为空返回 'orphan'。
336
- def _domain_of(it: dict) -> str:
337
- """条目 → 路由域键(供 `tags` 的 `domain:` 显式声明)。
338
-
339
- 与 `observation_position` **分开**:position 是给人读的条件文本(「本地
340
- 源码仓(大域=md_cg)」),domain 是给 `routing.route_key` 直取的短键。
341
- 两者混成一个字段就会重演普查里的退化:实例名嵌进条件字段 → 3037 桶 /
342
- 3048 节点(99.9% 单例桶),路由等于失效。
343
-
344
- 取 path 首段,与改造前 `normalize_domain(observation_position)` 的产物
345
- **逐字相同**,故本次加标签不改变任何既有节点的分桶结果。
346
- """
347
- path = it.get("path") or ""
348
- return path.split("/")[0] or "orphan"
349
-
350
-
351
- # 生效条件:kind == 'code_ref' 时按 codeindex.node_id/render 写入 cg(tags 含 'code'、code_ref=_code_ref(it, root)),kind == 'doc_ref' 时按 docindex 写入(tags 含 'doc'、doc_ref=_doc_ref(it, root)、密级取自 docindex.sensitivity_for(it['path'], sensitivity)),其他 kind 抛 ValueError,返回 (ids, sens)。
352
- def add_items(cg, items, *, kind: str, root: str, layer=None, sensitivity=None,
353
- layer_of=None):
354
- """把索引条目写进认知图(code / doc 的落盘细节收在这里,唯一实现)。
355
-
356
- - `layer=None` → 默认 `knowledge`(与代码节点同层,保证进默认召回)。
357
- - `layer_of(nid)` 可逐节点覆盖 layer(heal 重建时保留原层)。
358
- - doc 节点:密级走 `docindex.sensitivity_for`(只可能更严);返回密级分布。
359
- - `condition_space` 走 `codeindex/docindex.condition_space`,与正文的
360
- `# 生效条件:` 行**同源**——改造前此处只写 `observation_position` 单槽,
361
- 而单槽不是生效条件,于是 frontmatter 的条件空间形同未声明。
362
- 返回 (ids, sens_counts)。
363
- """
364
- from . import codeindex, docindex
365
- ids, sens = [], {}
366
- for it in items:
367
- if kind == "code_ref":
368
- nid = codeindex.node_id(it)
369
- cg.add(
370
- nid, codeindex.render(it),
371
- layer=(layer_of(nid) if layer_of else None) or layer or "knowledge",
372
- tags=["code", "code:" + it.get("kind", ""),
373
- "domain:" + _domain_of(it)],
374
- condition_space=codeindex.condition_space(it),
375
- verification_basis=it.get("basis") or "compiler",
376
- code_ref=_code_ref(it, root),
377
- )
378
- elif kind == "doc_ref":
379
- nid = docindex.node_id(it)
380
- level = it.get("level")
381
- s, _basis = docindex.sensitivity_for(it.get("path") or "", sensitivity)
382
- sens[s] = sens.get(s, 0) + 1
383
- cg.add(
384
- nid, docindex.render(it),
385
- layer=(layer_of(nid) if layer_of else None) or layer or "knowledge",
386
- tags=["doc", "doc:md", f"level:{level}",
387
- "domain:" + _domain_of(it)],
388
- condition_space=docindex.condition_space(it),
389
- verification_basis="data",
390
- sensitivity=s,
391
- doc_ref=_doc_ref(it, root),
392
- )
393
- else:
394
- raise ValueError(f"未知 ref kind:{kind!r}")
395
- ids.append(nid)
396
- return ids, sens
397
-
398
-
399
- # 生效条件:把入参 root 原样写入返回 dict 的 'root',path/name/kind/lineno/end/lang/hash 按 it.get 取值(缺省 None),precise 取 bool(it.get('precise', True)),render_version 取传入值(传入 None 时延迟 import codeindex 取 codeindex.RENDER_VERSION,保证与 render 契约**同源**、无第二处硬编码)。
400
- def _code_ref(it: dict, root: str, render_version=None) -> dict:
401
- if render_version is None: # 直接调用点的兜底:与 render 产物同源
402
- from . import codeindex
403
- render_version = codeindex.RENDER_VERSION
404
- return {
405
- "path": it.get("path"), "name": it.get("name"),
406
- "kind": it.get("kind"), "lineno": it.get("lineno"), "end": it.get("end"),
407
- "lang": it.get("lang"), "precise": bool(it.get("precise", True)),
408
- "hash": it.get("hash"), "root": root,
409
- "render_version": render_version,
410
- }
411
-
412
-
413
- # 生效条件:把入参 root 原样写入返回 dict 的 'root',path/heading/heading_path/level/lineno/end/anchor/hash/lang 按 it.get 取值(缺省 None),precise 取 bool(it.get('precise', True))。
414
- def _doc_ref(it: dict, root: str) -> dict:
415
- return {
416
- "path": it.get("path"), "heading": it.get("heading"),
417
- "heading_path": it.get("heading_path"), "level": it.get("level"),
418
- "lineno": it.get("lineno"), "end": it.get("end"),
419
- "anchor": it.get("anchor"), "hash": it.get("hash"), "lang": it.get("lang"),
420
- "precise": bool(it.get("precise", True)), "root": root,
421
- }
422
-
423
-
424
- # --------------------------------------------------------------------------
425
- # 回读(唯一实现:op=ref 与 check_refs 共用)
426
- # --------------------------------------------------------------------------
427
-
428
- # 生效条件:传入 ref 为假值(如 None/{})时按 {} 处理,rel 取 ref.get("path") or "";root 与 ref.get("root") 均为假值时返回含 ref/path/status:"unresolved"/ok:False/error:"ref 未记录 root..." 的 out;否则用 root or ref.get("root") 与 rel 拼 fp,os.path.isfile(fp) 为假时返回 status:"dangling"、stale:True,读取抛 OSError/UnicodeDecodeError 时返回 status:"error";读取成功时 lineno 取 int(ref.get("lineno") or 1)(假值回落 1)、end 取 int(ref.get("end") or lineno)(假值回落 lineno),ref.get("hash") 为 None 时 match=None、ok=True、status:"ok",ref.get("hash") 为真值且等于 region_hash 时 ok=True/status:"ok"、不等时 ok=False/status:"stale",ref.get("hash") 为假值但非 None(如 ""/0/False)时 ok=False/status:"stale";with_text 为真时 out["text"] 取 lines[max(0,lineno-1):max(max(0,lineno-1),end)] 的 join;
429
- def probe_ref(ref: dict, *, root: str = None, with_text: bool = False) -> dict:
430
- """只读探测单个 ref 的状态(不回读整篇,除非 with_text)。"""
431
- from . import codeindex
432
- ref = ref or {}
433
- rel = ref.get("path") or ""
434
- r = root or ref.get("root") or ""
435
- base = {"ref": ref, "path": rel, "status": "unresolved", "ok": False}
436
- if not r:
437
- return {**base, "error": "ref 未记录 root,请显式传 root 参数"
438
- "(索引里存的是相对 root 的 path)"}
439
- fp = os.path.join(r, rel)
440
- base["root"] = r
441
- if not os.path.isfile(fp):
442
- return {**base, "status": "dangling", "abspath": fp, "stale": True,
443
- "error": f"源文件不存在(索引已悬空):{fp}"}
444
- try:
445
- with open(fp, "r", encoding="utf-8") as f:
446
- lines = f.read().split("\n")
447
- except (OSError, UnicodeDecodeError) as exc:
448
- return {**base, "status": "error", "abspath": fp,
449
- "error": f"读取失败:{exc}"}
450
- total = len(lines)
451
- lineno = int(ref.get("lineno") or 1)
452
- end = int(ref.get("end") or lineno)
453
- got = codeindex.region_hash(lines, lineno, end)
454
- expect = ref.get("hash")
455
- match = (got == expect) if expect else None
456
- out = {
457
- **base, "abspath": fp, "total_lines": total,
458
- "hash": got, "hash_expected": expect, "hash_match": match,
459
- "stale": bool(expect) and not match,
460
- "ok": expect is None or bool(match),
461
- "status": "ok" if (expect is None or match) else "stale",
462
- }
463
- if with_text:
464
- lo = max(0, lineno - 1)
465
- out["text"] = "\n".join(lines[lo:max(lo, end)])
466
- return out
467
-
468
-
469
- # 生效条件:ref 经 probe_ref(root=root, with_text=True) 后 status 为 'ok'/'stale' 时返回 ok=True 及 text/total_lines/hash/hash_match/stale/precise,status 为 'unresolved'/'error'/'dangling' 时返回 ok=False 与 error。
470
- def read_ref(ref: dict, *, root: str = None, ref_kind: str = "ref") -> dict:
471
- """按 ref 回读源区间——`op=ref` 与 `check_refs` 的唯一实现。"""
472
- p = probe_ref(ref, root=root, with_text=True)
473
- base = {"ref": ref, "ref_kind": ref_kind}
474
- if p["status"] in ("unresolved", "error", "dangling"):
475
- out = {**base, "ok": False, "error": p["error"]}
476
- if p["status"] == "dangling":
477
- out["stale"] = True
478
- return out
479
- return {
480
- **base, "ok": True, "text": p["text"], "total_lines": p["total_lines"],
481
- "hash": p["hash"], "hash_expected": p["hash_expected"],
482
- "hash_match": p["hash_match"], "stale": p["stale"],
483
- "precise": bool((ref or {}).get("precise", True)),
484
- "note": "按 ref 区间回读;hash_match=False 说明源已改动,"
485
- "重跑 index_code / index_doc 重建",
486
- }
487
-
488
-
489
- # 生效条件:cg 的 index['nodes'] 非空时汇总 stale/dangling/unresolved/errors 并返回 ok =(无 stale 且无 dangling);only_tagged 为真时只探测 tags 含 'code'/'doc' 的节点,ledger 非空时先走 (size, mtime) 快路径。
490
- def check_refs(cg, *, ledger: "Ledger" = None, max_nodes: int = MAX_CHECK,
491
- only_tagged: bool = True) -> dict:
492
- """漂移 / 悬空巡检(只读、不抛)。
493
-
494
- 优先走 ledger 的 (size, mtime) 快路径:未变文件**不读盘**直接判 ok;
495
- 变了的文件读一次、按记录区间重算哈希判 stale。
496
- ledger 覆盖不到的节点(早期索引 / 未开增量)再回退逐节点探测。
497
-
498
- `only_tagged=True`(默认)只探测带 `code` / `doc` 标签的节点——ref 只由
499
- `index_code` / `index_doc` 产生,两者都会打这两个标签;这样巡检不必为每条
500
- 记忆节点都读一次盘。需要穷举(含手写 ref)时传 False。
501
- """
502
- nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
503
- stale, dangling, unresolved, errors = [], [], [], []
504
- covered = set()
505
-
506
- # 生效条件:当 nid、ref、kind、rel 传入时,p = probe_ref(ref) 后按 p["status"] 分派:为 "dangling" 时把含 node_id/ref_kind/path/lineno/end/error 的 row 加入 dangling,为 "stale" 时补 hash_expected/hash 加入 stale,为 "unresolved" 时加入 unresolved,为 "error" 时加入 errors;其他状态不加入;
507
- def _probe_one(nid, ref, kind, rel):
508
- p = probe_ref(ref)
509
- row = {"node_id": nid, "ref_kind": kind, "path": rel,
510
- "lineno": ref.get("lineno"), "end": ref.get("end"),
511
- "error": p.get("error", "")}
512
- if p["status"] == "dangling":
513
- dangling.append(row)
514
- elif p["status"] == "stale":
515
- row["hash_expected"] = p.get("hash_expected")
516
- row["hash"] = p.get("hash")
517
- stale.append(row)
518
- elif p["status"] == "unresolved":
519
- unresolved.append(row)
520
- elif p["status"] == "error":
521
- errors.append(row)
522
-
523
- # 快路径:ledger 记录的文件(键 = 源文件绝对路径,天然跨大域不撞名)
524
- if ledger is not None:
525
- for key, e in sorted((ledger.load().get("files") or {}).items()):
526
- rec_nodes = [n for n in (e.get("nodes") or [])
527
- if n.get("id") in nodes]
528
- if not rec_nodes:
529
- continue
530
- covered.update(n.get("id") for n in rec_nodes)
531
- rel = e.get("path") or ""
532
- # 必须用**每个源文件自己的 root**(索引时的源大域),不能用 ledger.root:
533
- # 后者是认知图根,拿它拼路径会指向不存在的位置、把一切都误判成 dangling。
534
- src_root = e.get("root") or os.path.dirname(key)
535
- try:
536
- st = os.stat(key)
537
- unchanged = (e.get("size") == st.st_size
538
- and abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6)
539
- except OSError:
540
- unchanged = False
541
- if not unchanged:
542
- kind = e.get("kind") or kind_of_path(rel)
543
- for n in rec_nodes:
544
- _probe_one(n.get("id"), {**n, "path": rel, "root": src_root},
545
- kind, rel)
546
-
547
- # 回退:ledger 未覆盖的索引节点
548
- # 生效条件:当 nid 传入时,若 only_tagged 为假值立即返回 True;否则取 (nodes.get(nid) or {}).get("tags") or [],仅当其中存在 "code" 或 "doc" 返回 True,否则返回 False;
549
- def _candidate(nid):
550
- if not only_tagged:
551
- return True
552
- tags = (nodes.get(nid) or {}).get("tags") or []
553
- return any(t in ("code", "doc") for t in tags)
554
-
555
- todo = [nid for nid in nodes if nid not in covered and _candidate(nid)]
556
- truncated = len(todo) > max_nodes
557
- checked = 0
558
- for nid in todo[:max_nodes]:
559
- try:
560
- node = cg.get(nid)
561
- except Exception:
562
- continue
563
- if not node:
564
- continue
565
- kind, ref = ref_of(node)
566
- if not ref:
567
- continue
568
- checked += 1
569
- try:
570
- _probe_one(nid, ref, kind, ref.get("path") or "")
571
- except Exception as exc: # 巡检不抛
572
- errors.append({"node_id": nid, "ref_kind": kind, "error": str(exc)})
573
-
574
- return {
575
- "ok": not (stale or dangling),
576
- "status": "ok" if not (stale or dangling) else ("dangling" if dangling else "stale"),
577
- "checked": checked + len(covered),
578
- "ledger_files": len((ledger.load().get("files") or {})) if ledger else 0,
579
- "stale": stale, "dangling": dangling,
580
- "unresolved": unresolved, "errors": errors,
581
- "truncated": truncated, "max_nodes": max_nodes,
582
- }
583
-
584
-
585
- # --------------------------------------------------------------------------
586
- # 节点级对账:孤儿清退 + 悬空清退
587
- #
588
- # 水位层的 `Ledger.reconcile` 只剪**水位条目**、`prune` 只剪「源大域已消失」
589
- # 的条目,两者都不碰**节点**。于是节点层的两类残留无人处置:
590
- # · 孤儿(同一文档的过期代)——标题路径一变 id 全量重算,旧代与新代并存;
591
- # · 悬空(源文件已删)——回读必然失败,巡检永远报 dangling。
592
- # 本节的唯一实现同时供 `op=index_code / index_doc`(孤儿)与
593
- # `op=ref action=prune`(悬空)使用,避免两处各写一套口径。
594
- # --------------------------------------------------------------------------
595
-
596
- # 生效条件:返回 os.path.normcase(os.path.abspath(str(p or ''))),即 p 为 None/空串时返回当前目录的归一绝对路径。
597
- def _norm_root(p) -> str:
598
- """root 归一:同一目录的大小写/分隔符差异不得影响「同一大域」判定。"""
599
- return os.path.normcase(os.path.abspath(str(p or "")))
600
-
601
-
602
- # 生效条件:a 与 b 都非空且 _norm_root(a) == _norm_root(b) 时返回 True,否则(含 TypeError/ValueError)返回 False。
603
- def _same_root(a, b) -> bool:
604
- try:
605
- return bool(a) and bool(b) and _norm_root(a) == _norm_root(b)
606
- except (TypeError, ValueError):
607
- return False
608
-
609
-
610
- # 生效条件:返回 str(p or '').replace('\\', '/').lstrip('./'),即 p 为 None/空串时返回 ''。
611
- def _norm_rel(p) -> str:
612
- return str(p or "").replace("\\", "/").lstrip("./")
613
-
614
-
615
- # 生效条件:cg 具备可调用的 forget 方法时对 plan 中每个 nid 调 cg.forget(nid, why),返回 (成功 id 列表, 被拦下/失败的 {node_id, error} 列表);cg 无 forget 时返回 ([], plan 中每 nid 一条错误)。
616
- def _forget_many(cg, plan, why: str) -> tuple:
617
- """逐条软删(进 trash/、写删除清单、可 restore);受保护节点拦下不删。
618
-
619
- `forget` 属 **MdCGOS 层**能力(保护裁决 + 回收站 + 删除清单),基础层
620
- `MdCG` 没有任何删除原语。缺能力时**明确报错、不静默跳过**——否则
621
- 「清退了 N 条」看着成功、实际一条没删(P27 §10 实测过这个坑)。
622
- """
623
- fn = getattr(cg, "forget", None)
624
- if not callable(fn):
625
- err = (f"{type(cg).__name__} 无 forget 能力(对账须能删除节点);"
626
- f"生产路径是 MdCGOS,测试请用 MdCGOS")
627
- return [], [{"node_id": nid, "error": err} for nid in sorted(plan)]
628
- done, blocked = [], []
629
- for nid in sorted(plan):
630
- try:
631
- res = fn(nid, why) or {}
632
- except Exception as exc: # ProtectionError 等 → 拦下,不越权
633
- blocked.append({"node_id": nid, "error": str(exc)[:120]})
634
- continue
635
- if res.get("ok"):
636
- done.append(nid)
637
- else:
638
- blocked.append({"node_id": nid, "error": str(res.get("error"))[:120]})
639
- return done, blocked
640
-
641
-
642
- # 生效条件:cg 具备可调用的 _unstage 时对 ghosts 逐个调用并收集成功 id(单条异常跳过),cg 无该能力时返回 [](不假装成功)。
643
- def _drop_ghosts(cg, ghosts) -> list:
644
- """摘除幽灵条目的索引记录(节点文件已不存在,没有可软删的实体)。
645
-
646
- 走 `MdCG._unstage`:它顺带落删除记录,保证幽灵不会再次从分片日志里复活。
647
- 基础层没有该能力时如实返回空列表(不假装成功)。
648
- """
649
- drop = getattr(cg, "_unstage", None)
650
- if not callable(drop):
651
- return []
652
- out = []
653
- for nid in ghosts:
654
- try:
655
- drop(nid)
656
- out.append(nid)
657
- except Exception: # 单条失败不拖垮整批
658
- continue
659
- return out
660
-
661
-
662
- # 生效条件:items 中同 kind 的节点若其 ref['path'] 命中本次 items 的文件、id 不在本次产出内且 ref['root'] 与入参 root 同一(_same_root),则列入清退计划;dry_run 为真只返回计划,否则经 _forget_many 软删;items 为空时返回 count 0。
663
- def prune_orphans(cg, *, kind: str, root: str, items, dry_run: bool = False,
664
- reason: str = "") -> dict:
665
- """清退「同 root + 同 path,但已不在本次产出里」的**过期代**节点。
666
-
667
- 为什么必须有:`node_id = sha1(相对path + "#" + heading_path)`,而
668
- `add_items` 只做**同 id 幂等 upsert**——文档标题结构一变,整篇 id 全量
669
- 重算,旧代节点无人清退,与新代并存(同一文档召回两份,且旧代引用的区间
670
- 已失效)。截断的索引由调用方负责不调用本函数(没扫完 ≠ 剩下的都过期)。
671
-
672
- 范围**只限本次真正重切过的文件**(`items` 的 path):增量索引跳过的未变
673
- 文件不在 items 里,其节点不进判定——否则会把完好的节点整片误删。
674
- """
675
- nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
676
- touched: dict = {}
677
- for it in items or []:
678
- rel = _norm_rel(it.get("path"))
679
- if rel:
680
- touched.setdefault(rel, set()).add(node_id_of(it, kind))
681
- base = {"scanned": len(nodes), "touched_files": len(touched),
682
- "dry_run": bool(dry_run)}
683
- if not touched:
684
- return {**base, "ok": True, "count": 0, "pruned": [],
685
- "skipped_protected": []}
686
-
687
- # ⚠ ref 只存在于**节点 frontmatter**里;`cg.index['nodes']` 是元数据快照
688
- # (path/layer/tags/…,见 mdcg._scan_nodes),**不含 ref**。因此必须
689
- # `cg.get(nid)` 取回节点再 ref_of —— 否则 ref_of 恒返回 ('', None)、
690
- # 整个对账静默失效(P27 §10 实测过这个坑)。标签预筛与 check_refs 同口径,
691
- # 免得为全库每条记忆都读一次盘。
692
- tag = "doc" if kind == "doc_ref" else "code"
693
- plan = []
694
- for nid, e in nodes.items():
695
- if tag not in ((e or {}).get("tags") or []):
696
- continue
697
- try:
698
- node = cg.get(nid)
699
- except Exception:
700
- continue
701
- if not node:
702
- continue
703
- k, ref = ref_of(node)
704
- if k != kind or not isinstance(ref, dict):
705
- continue
706
- keep = touched.get(_norm_rel(ref.get("path")))
707
- if keep is None or nid in keep or not _same_root(ref.get("root"), root):
708
- continue
709
- plan.append(nid)
710
-
711
- why = reason or ("索引重建:本节点已不在同文档新代产出中(标题路径变更致 "
712
- "node_id 重算),清退过期代以消除重复召回")
713
- if dry_run:
714
- return {**base, "ok": True, "count": len(plan), "pruned": sorted(plan)[:50],
715
- "skipped_protected": [], "reason": why}
716
- done, blocked = _forget_many(cg, plan, why)
717
- return {**base, "ok": True, "count": len(done), "pruned": done[:50],
718
- "skipped_protected": blocked[:20], "reason": why}
719
-
720
-
721
- # 生效条件:cg 中带 'code'/'doc' 标签且 ref 带 root 的节点经 probe_ref 判为 'dangling' 时列入清退计划;only_roots 为空时另收集 cg.root 下取不到对应节点 .md 的幽灵条目经 _drop_ghosts 摘除;dry_run 为真只返回计划。
722
- def prune_dangling(cg, *, only_roots=None, dry_run: bool = False,
723
- max_nodes: int = MAX_CHECK, reason: str = "") -> dict:
724
- """清退**悬空**节点:ref 指向的源文件已删除,回读必然失败。
725
-
726
- `check_refs` 只报告不处置(其原话是「悬空需人工处置」),本函数就是那个
727
- 出口——已删脚本、被搬走的文档留下的残留节点一次清掉,而不是逐条手工
728
- `forget`。判定与巡检共用 `probe_ref` 的唯一实现,口径不会打架。
729
-
730
- 另清**幽灵条目**(ghosts):索引有条目、节点文件却不存在。它们是历史
731
- 「删除只摘内存索引、不落盘」的遗留——`cg.get` 取不回 → 悬空清退够不着它,
732
- 而 `check_refs` 走 ledger 会一直报 → dangling 永不归零。判据只用唯一真源
733
- (节点 .md 不存在即脏索引),与「索引是派生物」的宣言一致;`only_roots`
734
- 非空时跳过(幽灵条目无 ref,无法归因到某个源大域)。
735
- """
736
- nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
737
- ghosts = []
738
- if not only_roots:
739
- for nid, e in list(nodes.items()):
740
- tags = (e or {}).get("tags") or []
741
- if not any(t in ("code", "doc") for t in tags):
742
- continue
743
- path = (e or {}).get("path") or ""
744
- if path and not os.path.exists(os.path.join(cg.root, path)):
745
- ghosts.append(nid)
746
- # 同 prune_orphans:ref 只在节点 frontmatter 里,索引条目里没有,
747
- # 必须 cg.get 取回节点再 ref_of(否则恒空、静默不删)。
748
- todo = []
749
- for nid, e in nodes.items():
750
- tags = (e or {}).get("tags") or []
751
- if not any(t in ("code", "doc") for t in tags):
752
- continue
753
- try:
754
- node = cg.get(nid)
755
- except Exception:
756
- continue
757
- if not node:
758
- continue
759
- _k, ref = ref_of(node)
760
- if not ref or not ref.get("root"):
761
- continue
762
- if only_roots and not any(_same_root(ref.get("root"), r) for r in only_roots):
763
- continue
764
- todo.append((nid, ref))
765
- truncated = len(todo) > max_nodes or len(ghosts) > max_nodes
766
- plan = []
767
- for nid, ref in todo[:max_nodes]:
768
- try:
769
- if probe_ref(ref).get("status") == "dangling":
770
- plan.append(nid)
771
- except Exception: # 探测失败不算悬空(宁可不删)
772
- continue
773
-
774
- why = reason or ("源文件已删除,索引节点悬空(回读必然失败),"
775
- "清退以消除永不消失的 dangling")
776
- ghost_plan = sorted(ghosts)[:max_nodes]
777
- base = {"scanned": len(nodes), "candidates": len(plan),
778
- "ghosts": len(ghost_plan),
779
- "dry_run": bool(dry_run), "truncated": truncated,
780
- "max_nodes": max_nodes}
781
- if dry_run:
782
- return {**base, "ok": True, "count": len(plan), "pruned": sorted(plan)[:50],
783
- "ghost_pruned": ghost_plan[:50],
784
- "skipped_protected": [], "reason": why}
785
- done, blocked = _forget_many(cg, plan, why)
786
- dropped = _drop_ghosts(cg, ghost_plan)
787
- return {**base, "ok": True, "count": len(done), "pruned": done[:50],
788
- "ghost_pruned": dropped[:50],
789
- "skipped_protected": blocked[:20], "reason": why}
790
-
791
-
792
- # 生效条件:cg 节点按 ref['root'] 与 kind 分组后逐组以 index_dir(incremental=False, ledger=ledger) 重切、再以 add_items(layer_of=原 layer) 重建,返回 {'ok','roots','groups','indexed','errors','truncated'};only_roots 非 None 时只处理其中列出的 root。
793
- def rebuild(cg, *, ledger: "Ledger" = None, only_roots=None, max_files: int = 500,
794
- max_items: int = 2000) -> dict:
795
- """按 ref 记录的 root 重建索引(sustain.heal 的修复动作)。
796
-
797
- 只重跑出了问题的 root(`only_roots`),逐节点**保留原 layer**;
798
- doc 密级用默认策略重算(默认只可能更严,不会放松)。
799
- """
800
- nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
801
- groups = {}
802
- for nid in nodes:
803
- try:
804
- node = cg.get(nid)
805
- except Exception:
806
- continue
807
- kind, ref = ref_of(node)
808
- if not ref or not ref.get("root"):
809
- continue
810
- r = ref["root"]
811
- if only_roots is not None and r not in only_roots:
812
- continue
813
- groups.setdefault((r, kind), 0)
814
- groups[(r, kind)] += 1
815
-
816
- out = {"ok": True, "roots": sorted({r for r, _ in groups}),
817
- "groups": len(groups), "indexed": 0, "errors": [], "truncated": False}
818
- for (root, kind) in sorted(groups):
819
- try:
820
- items, errors, stats = index_dir(
821
- root, kind=kind, max_files=max_files, max_items=max_items,
822
- incremental=False, ledger=ledger)
823
- ids, _sens = add_items(
824
- cg, items, kind=kind, root=root,
825
- layer_of=lambda nid: (nodes.get(nid) or {}).get("layer"))
826
- out["indexed"] += len(ids)
827
- out["errors"].extend(errors)
828
- out["truncated"] = out["truncated"] or bool(stats.get("truncated"))
829
- except Exception as exc: # 自愈不抛
830
- out["ok"] = False
831
- out["errors"].append(f"{root} [{kind}]:{exc}")
832
- if ledger is not None:
833
- ledger.save()
1
+ # -*- coding: utf-8 -*-
2
+ """md_cg · 统一 ref 协议 + 索引水位(增量)+ 漂移/悬空巡检
3
+
4
+ 对照 `docs/mdcg/认知图_索引与工程规范化_计划_v0.1.md` 的 R3(修 D + 修 F):
5
+
6
+ **D 漂移 / 悬空检测(本模块 `check_refs`)**
7
+ 扫描带 `code_ref` / `doc_ref` 的节点,回两类问题:
8
+ · `stale` —— 源文件被改(区间哈希不再匹配)
9
+ · `dangling` —— 源文件被删(索引指向不存在的文件)
10
+ 巡检**只读**、**不抛**、**不改源文件**;修复动作是「重跑 index_code / index_doc」,
11
+ 因为索引是派生物(对齐 sustain.heal 的既有边界)。
12
+
13
+ **F 全量重扫 + 静默截断(本模块 `Ledger` + `index_dir`)**
14
+ `<root>/_refindex.json` 是 ref 索引水位(抄 `sources.Ingestor` 的 `_sources.json` 范式),
15
+ 以**源文件绝对路径**为键,记每个源文件的 (size, mtime) 与节点区间 + 它所属的**源大域
16
+ root**;`incremental=True` 时未变文件**不再读盘重切**,直接跳过(`skipped_unchanged`)
17
+ ——这就是「不全量重扫」。
18
+ 键用绝对路径、且逐文件记 root,是因为一份认知图可以索引多个大域:只按 rel 记会在同名
19
+ 文件上互相覆盖,巡检时若拿认知图根去拼路径则会把一切都误判成 dangling。
20
+ 截断(`max_files` / `max_items`)由 codeindex / docindex 显式上报,本模块把
21
+ 「最近一次索引被截断」写进水位,交给 `sustain.diagnose` 巡检看见(不再静默)。
22
+
23
+ **为什么回读要收进本模块**
24
+ `op=ref` 的回读与 `check_refs` 的判定**必须共用同一实现**,否则会出现
25
+ 「回读说没漂、巡检说有漂」。与 `region_hash` 的教训同源:区间哈希只允许一份实现,
26
+ 这里连「怎么判定 ok / stale / dangling」也只允许一份。
27
+
28
+ 零第三方依赖。
29
+ """
30
+ from __future__ import annotations
31
+
32
+ import json
33
+ import os
34
+ import time
35
+
36
+ from .fsutil import atomic_write
37
+
38
+ SCHEMA = 2 # v2:水位以「源文件绝对路径」为键(v1 按 rel 会跨大域撞名)
39
+ LEDGER_FILE = "_refindex.json"
40
+ REF_KEYS = ("code_ref", "doc_ref")
41
+ MAX_CHECK = 2000 # 巡检节点上限(超出报 truncated,不静默截断)
42
+ STATUSES = ("ok", "stale", "dangling", "unresolved", "error")
43
+
44
+
45
+ # 生效条件:给定 fp 时返回 os.path.abspath(fp or '')(fp 为空/None 则返回当前目录的绝对路径),作为水位键以绝对路径保证不同 root 下同名文件不互相覆盖。
46
+ def _src_key(fp: str) -> str:
47
+ """水位的键 = 源文件绝对路径。
48
+
49
+ 不能用 rel:一份认知图可以索引多个大域(不同 root),只按 rel 记会在
50
+ `alpha.py` 这种同名文件上互相覆盖——水位被静默丢掉,巡检就漏报。
51
+ """
52
+ return os.path.abspath(fp or "")
53
+
54
+
55
+ # 生效条件:无 required 形参,任何调用都返回 round(time.time(), 1),把时间戳压到 1 位小数以稳定 `_refindex.json` 字节数。
56
+ def _now() -> float:
57
+ """时间戳压到 1 位小数:让 `_refindex.json` 字节数稳定(重跑不涨),
58
+ 同时保留足够的「多久以前」信息(float 的最短 repr 保证小数位固定为 1)。"""
59
+ return round(time.time(), 1)
60
+
61
+
62
+ # --------------------------------------------------------------------------
63
+ # 提取器注册表(统一调度:调用方只说 kind,不说「用哪个模块」)
64
+ # --------------------------------------------------------------------------
65
+
66
+ # 生效条件:kind == 'code_ref' 返回 codeindex、kind == 'doc_ref' 返回 docindex,其他 kind 抛 ValueError(提示支持 REF_KEYS)。
67
+ def _mod(kind: str):
68
+ from . import codeindex, docindex
69
+ if kind == "code_ref":
70
+ return codeindex
71
+ if kind == "doc_ref":
72
+ return docindex
73
+ raise ValueError(f"未知 ref kind:{kind!r}(支持 {REF_KEYS})")
74
+
75
+
76
+ # 生效条件:无 required 形参,调用即返回 {'code_ref': {'suffixes': tuple(codeindex.SUFFIX)}, 'doc_ref': {'suffixes': tuple(docindex.SUFFIX)}}。
77
+ def registry() -> dict:
78
+ """后缀 → kind 的注册表(code / doc 各一份提取器)。"""
79
+ from . import codeindex, docindex
80
+ return {
81
+ "code_ref": {"suffixes": tuple(codeindex.SUFFIX)},
82
+ "doc_ref": {"suffixes": tuple(docindex.SUFFIX)},
83
+ }
84
+
85
+
86
+ # 生效条件:path 的小写后缀在 codeindex.EXTRACTORS 中返回 'code_ref',在 docindex.SUFFIX 中返回 'doc_ref',无后缀或均不匹配返回 ''。
87
+ def kind_of_path(path: str) -> str:
88
+ """按后缀判 kind;无提取器返回 ''(由调用方决定是报错还是跳过)。"""
89
+ from . import codeindex, docindex
90
+ ext = os.path.splitext(path or "")[1].lower()
91
+ if not ext:
92
+ return ""
93
+ if ext in codeindex.EXTRACTORS:
94
+ return "code_ref"
95
+ if ext in docindex.SUFFIX:
96
+ return "doc_ref"
97
+ return ""
98
+
99
+
100
+ # 生效条件:source 为待提取文本,kind 非空或 path 后缀能推出 kind 时返回 _mod(k).extract(source, path),推不出 kind 时抛 ValueError。
101
+ def extract(source: str, path: str = "", kind: str = ""):
102
+ """统一提取入口:按 kind(或从 path 推断)分发到对应 extractor。"""
103
+ k = kind or kind_of_path(path)
104
+ if not k:
105
+ ext = os.path.splitext(path or "")[1] or "<none>"
106
+ raise ValueError(f"无索引提取器(suffix={ext})")
107
+ return _mod(k).extract(source, path)
108
+
109
+
110
+ # 生效条件:kind 为 'code_ref'/'doc_ref' 时返回 _mod(kind).node_id(item),其他 kind 由 _mod 抛 ValueError。
111
+ def node_id_of(item: dict, kind: str) -> str:
112
+ return _mod(kind).node_id(item)
113
+
114
+
115
+ # 生效条件:kind 为 'code_ref'/'doc_ref' 时返回 _mod(kind).render(item),其他 kind 由 _mod 抛 ValueError。
116
+ def render_of(item: dict, kind: str) -> str:
117
+ return _mod(kind).render(item)
118
+
119
+
120
+ # 生效条件:node 的 frontmatter 中 REF_KEYS 命中且值为非空 dict 时返回 {'ref': ref, 'ref_kind': kind},否则返回 {'ref': None, 'ref_kind': ''}。
121
+ def ref_fields(node) -> dict:
122
+ """节点 → 检索结果要带的两字段(读侧只加字段,不改召回逻辑)。"""
123
+ kind, ref = ref_of(node)
124
+ return {"ref": ref, "ref_kind": kind} if ref else {"ref": None, "ref_kind": ""}
125
+
126
+
127
+ # 生效条件:node 的 frontmatter 按 REF_KEYS 顺序取到第一个非空 dict 时返回 (k, r),否则返回 ('', None)。
128
+ def ref_of(node) -> tuple:
129
+ """从节点 frontmatter 取 ref:返回 (kind, ref) 或 ('', None)。"""
130
+ fm = (node or {}).get("frontmatter") or {}
131
+ for k in REF_KEYS:
132
+ r = fm.get(k)
133
+ if isinstance(r, dict) and r:
134
+ return k, r
135
+ return "", None
136
+
137
+
138
+ # --------------------------------------------------------------------------
139
+ # 索引水位(_refindex.json):增量 + 截断留痕
140
+ # --------------------------------------------------------------------------
141
+
142
+ # 生效条件:以 root 为必填实参构造,实例化即置 self.root=root、self.path=os.path.join(root, LEDGER_FILE)、self._d=None;
143
+ class Ledger:
144
+ """`<root>/_refindex.json`:每个源文件的 (size, mtime) 水位 + 节点区间。"""
145
+
146
+ # 生效条件:当传入 root 时,self.root 取该 root,self.path 为 os.path.join(root, LEDGER_FILE),self._d 置为 None;
147
+ def __init__(self, root: str):
148
+ self.root = root
149
+ self.path = os.path.join(root, LEDGER_FILE)
150
+ self._d = None
151
+
152
+ # 生效条件:当 self._d is not None 时直接返回 self._d;否则读取 self.path 的 JSON,仅当 obj 是 dict 且 obj.get("schema") == SCHEMA 且 obj.get("files") 是 dict 时用 obj,否则(含 OSError/ValueError、结构不符)回落为 {"schema": SCHEMA, "updated_at": 0.0, "files": {}} 并缓存返回;
153
+ def load(self) -> dict:
154
+ if self._d is not None:
155
+ return self._d
156
+ d = None
157
+ try:
158
+ with open(self.path, "r", encoding="utf-8") as f:
159
+ obj = json.load(f)
160
+ if isinstance(obj, dict) and obj.get("schema") == SCHEMA \
161
+ and isinstance(obj.get("files"), dict):
162
+ d = obj
163
+ except (OSError, ValueError):
164
+ d = None
165
+ self._d = d or {"schema": SCHEMA, "updated_at": 0.0, "files": {}}
166
+ return self._d
167
+
168
+ # 生效条件:传入 rel、fp 时,若 self.load()["files"].get(_src_key(fp)) 缺失或为假值、或 os.stat(fp) 抛 OSError、或条目 e.get("size") != st.st_size,则返回 False;否则返回 abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6(mtime 缺失或假值时按 0.0);
169
+ def is_fresh(self, rel: str, fp: str) -> bool:
170
+ """源文件自上次索引后未变(size + mtime 双等)→ 可跳过不重切。"""
171
+ e = self.load()["files"].get(_src_key(fp))
172
+ if not e:
173
+ return False
174
+ try:
175
+ st = os.stat(fp)
176
+ except OSError:
177
+ return False
178
+ if e.get("size") != st.st_size:
179
+ return False
180
+ return abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6
181
+
182
+ # 生效条件:当 rel、fp、kind、nodes 传入且 os.stat(fp) 成功时,向 self.load()["files"][_src_key(fp)] 写条目,其中 root 为 root if root else os.path.dirname(key)、path 为 rel、kind 为 kind、size/mtime 取 st、nodes 为每项 n.get("id")/n.get("lineno")/n.get("end")/n.get("hash");os.stat(fp) 抛 OSError 时不写入;
183
+ def record(self, rel: str, fp: str, kind: str, nodes,
184
+ root: str = None) -> None:
185
+ """记一个源文件的水位(节点区间用于判 stale)。
186
+
187
+ `root` 是**源**大域的根(≠ 认知图根):巡检要拿它拼 `root/rel` 才能
188
+ 找到源文件,缺了它就会把「源在别处」误判成 dangling。
189
+ """
190
+ key = _src_key(fp)
191
+ try:
192
+ st = os.stat(fp)
193
+ except OSError:
194
+ return
195
+ self.load()["files"][key] = {
196
+ "root": root if root else os.path.dirname(key),
197
+ "path": rel,
198
+ "size": st.st_size,
199
+ "mtime": st.st_mtime,
200
+ "kind": kind,
201
+ "nodes": [
202
+ {"id": n.get("id"), "lineno": n.get("lineno"),
203
+ "end": n.get("end"), "hash": n.get("hash")}
204
+ for n in nodes
205
+ ],
206
+ }
207
+
208
+ # 生效条件:当传入 fp 时,self.load()["files"].pop(_src_key(fp), None),即删除对应键(不存在也静默);
209
+ def drop(self, fp: str) -> None:
210
+ self.load()["files"].pop(_src_key(fp), None)
211
+
212
+ # 生效条件:当传入 root、kind、seen 时,对 self.load()["files"] 中满足 os.path.abspath(e.get("root") or "") == os.path.abspath(root) 且 e.get("kind") == kind 且键 k 不在 seen 的条目删除,返回删除数量;
213
+ def reconcile(self, root: str, kind: str, seen) -> int:
214
+ """一次**完整**索引后对账:本 (root, kind) 下没被扫到的旧条目剪掉。
215
+
216
+ 否则「源文件被删 → 索引悬空 → heal 重建」之后条目还在,巡检就永远报
217
+ dangling,heal 是治不好的。`seen` 是本次真正走过(提取成功或判定未变)
218
+ 的源文件键集合。**截断的索引不能对账**——没扫完不等于剩下的都消失了。
219
+ """
220
+ r = os.path.abspath(root)
221
+ files = self.load()["files"]
222
+ dead = [k for k, e in files.items()
223
+ if os.path.abspath(e.get("root") or "") == r
224
+ and e.get("kind") == kind and k not in seen]
225
+ for k in dead:
226
+ files.pop(k, None)
227
+ return len(dead)
228
+
229
+ # 生效条件:对 load()["files"] 中「条目 root(为假值时用 os.path.dirname(键) 兜底)不是目录」的条目逐一 pop 并返回删除条数,无匹配时返回 0。
230
+ def prune(self) -> int:
231
+ """剪掉「源大域已不存在」的条目(整个目录被搬走/删除)。
232
+
233
+ 这类条目已不可能再被任何大域索引到,留着只会在巡检里报永不消失的
234
+ dangling;而节点自带的 ref 仍会兜底探测,所以剪掉不会漏报真实悬空。
235
+ """
236
+ files = self.load()["files"]
237
+ dead = [k for k, e in files.items()
238
+ if not os.path.isdir(e.get("root") or os.path.dirname(k))]
239
+ for k in dead:
240
+ files.pop(k, None)
241
+ return len(dead)
242
+
243
+ # 生效条件:当 kind、root、files、indexed、truncated 传入时,self.load()["last_index"] 被设为含 ts=_now()、kind、root、files、indexed、truncated=bool(truncated)、truncated_reason=reason or "" 的字典;reason 为假值(默认 ""/None)时 truncated_reason 回落 "";
244
+ def note_index(self, *, kind: str, root: str, files: int, indexed: int,
245
+ truncated: bool, reason: str = "") -> None:
246
+ """记「最近一次索引」结果——截断在这里留痕,供 diagnose 看见。"""
247
+ self.load()["last_index"] = {
248
+ "ts": _now(), "kind": kind, "root": root, "files": files,
249
+ "indexed": indexed, "truncated": bool(truncated),
250
+ "truncated_reason": reason or "",
251
+ }
252
+
253
+ # 生效条件:无参数调用即生效,取 self.load() 结果把 updated_at 置为 _now(),再以 atomic_write 把 json.dumps(..., ensure_ascii=False, indent=1, sort_keys=True) 写入 self.path,无返回值。
254
+ def save(self) -> None:
255
+ d = self.load()
256
+ d["updated_at"] = _now()
257
+ atomic_write(self.path, json.dumps(d, ensure_ascii=False,
258
+ indent=1, sort_keys=True))
259
+
260
+ # 生效条件:无参数调用即生效,返回含 path、schema、load()["files"] 条目数、nodes 总数(各条目 nodes 列表长度之和)、updated_at、exists=os.path.isfile(self.path) 的 out;age_s 在 updated_at 为假值(0.0)时为 None,否则为 max(0.0, time.time()-up);仅当 load() 的 last_index 为 dict 时才并入 out["last_index"]。
261
+ def summary(self) -> dict:
262
+ d = self.load()
263
+ files = d.get("files") or {}
264
+ nodes = sum(len(e.get("nodes") or []) for e in files.values())
265
+ up = float(d.get("updated_at") or 0.0)
266
+ out = {
267
+ "path": self.path,
268
+ "schema": d.get("schema"),
269
+ "files": len(files),
270
+ "nodes": nodes,
271
+ "updated_at": up,
272
+ "age_s": None if not up else max(0.0, time.time() - up),
273
+ "exists": os.path.isfile(self.path),
274
+ }
275
+ if isinstance(d.get("last_index"), dict):
276
+ out["last_index"] = d["last_index"]
277
+ return out
278
+
279
+
280
+ # --------------------------------------------------------------------------
281
+ # 统一 index_dir:调度 + 水位 + 落盘(供 op=index_code / op=index_doc / heal 共用)
282
+ # --------------------------------------------------------------------------
283
+
284
+ # 生效条件:root 为源大域根、kind 为 'code_ref'/'doc_ref' 时经 _mod(kind) 调度底层 index_dir 并返回 (items, errors, stats);ledger 非空时逐文件 record,incremental 为真时跳过 ledger.is_fresh 为真的文件,且 stats 未截断时执行 reconcile。
285
+ def index_dir(root: str, *, kind: str, patterns=None, max_files: int = 500,
286
+ max_items: int = 2000, incremental: bool = False,
287
+ ledger: "Ledger" = None, skip_dirs=None):
288
+ """按 kind 调度 codeindex / docindex 的全量(或增量)索引。
289
+
290
+ incremental=True 且给了 ledger 时:未变文件跳过(`skipped_unchanged`)。
291
+ 返回 (items, errors, stats),与底层 index_dir 的返回一致(多一个
292
+ `skipped_unchanged`)。
293
+
294
+ `skip_dirs` 透传给底层:**追加**排除、只增不减(内置 `.git`/`.venv`/
295
+ `node_modules` 等不可被关闭),见 `codeindex.skip_matcher`。实际排掉了哪些目录
296
+ 由 `stats["skipped_dirs"]` 回报,仍不静默。
297
+ """
298
+ mod = _mod(kind)
299
+ fresh = None
300
+ on_file = None
301
+ seen = set() # 本次真正走过的源文件(用于对账)
302
+ if ledger is not None:
303
+ if incremental:
304
+ # 生效条件:当 rel、fp 传入时,ok = ledger.is_fresh(rel, fp);若 ok 为真则将 _src_key(fp) 加入 seen 并返回 ok,若 ok 为假则直接返回 False;
305
+ def fresh(rel, fp): # noqa: E306
306
+ ok = ledger.is_fresh(rel, fp)
307
+ if ok:
308
+ seen.add(_src_key(fp))
309
+ return ok
310
+
311
+ # 生效条件:当 rel、fp、got 传入时,将 _src_key(fp) 加入 seen,并以 root=root 调用 ledger.record(rel, fp, kind, [{"id": node_id_of(it, kind), "lineno": it.get("lineno"), "end": it.get("end"), "hash": it.get("hash")} for it in got]);
312
+ def on_file(rel, fp, got): # noqa: E306
313
+ seen.add(_src_key(fp))
314
+ ledger.record(rel, fp, kind,
315
+ [{"id": node_id_of(it, kind), "lineno": it.get("lineno"),
316
+ "end": it.get("end"), "hash": it.get("hash")} for it in got],
317
+ root=root)
318
+
319
+ items, errors, stats = mod.index_dir(
320
+ root, patterns=patterns, max_files=max_files, max_items=max_items,
321
+ fresh=fresh, on_file=on_file, skip_dirs=skip_dirs,
322
+ )
323
+ if ledger is not None:
324
+ ledger.prune()
325
+ if not stats.get("truncated"):
326
+ # 没扫完就不能对账:截断时「没见到」不等于「源已消失」。
327
+ ledger.reconcile(root, kind, seen)
328
+ ledger.note_index(kind=kind, root=root, files=stats.get("files", 0),
329
+ indexed=len(items), truncated=bool(stats.get("truncated")),
330
+ reason=stats.get("truncated_reason") or "")
331
+ ledger.save()
332
+ return items, errors, stats
333
+
334
+
335
+ # 生效条件:it['path'] 非空时返回其首段 path.split('/')[0] 作为 domain 键,path 为空返回 'orphan'。
336
+ def _domain_of(it: dict) -> str:
337
+ """条目 → 路由域键(供 `tags` 的 `domain:` 显式声明)。
338
+
339
+ 与 `observation_position` **分开**:position 是给人读的条件文本(「本地
340
+ 源码仓(大域=md_cg)」),domain 是给 `routing.route_key` 直取的短键。
341
+ 两者混成一个字段就会重演普查里的退化:实例名嵌进条件字段 → 3037 桶 /
342
+ 3048 节点(99.9% 单例桶),路由等于失效。
343
+
344
+ 取 path 首段,与改造前 `normalize_domain(observation_position)` 的产物
345
+ **逐字相同**,故本次加标签不改变任何既有节点的分桶结果。
346
+ """
347
+ path = it.get("path") or ""
348
+ return path.split("/")[0] or "orphan"
349
+
350
+
351
+ # 生效条件:kind == 'code_ref' 时按 codeindex.node_id/render 写入 cg(tags 含 'code'、code_ref=_code_ref(it, root)),kind == 'doc_ref' 时按 docindex 写入(tags 含 'doc'、doc_ref=_doc_ref(it, root)、密级取自 docindex.sensitivity_for(it['path'], sensitivity)),其他 kind 抛 ValueError,返回 (ids, sens)。
352
+ def add_items(cg, items, *, kind: str, root: str, layer=None, sensitivity=None,
353
+ layer_of=None):
354
+ """把索引条目写进认知图(code / doc 的落盘细节收在这里,唯一实现)。
355
+
356
+ - `layer=None` → 默认 `knowledge`(与代码节点同层,保证进默认召回)。
357
+ - `layer_of(nid)` 可逐节点覆盖 layer(heal 重建时保留原层)。
358
+ - doc 节点:密级走 `docindex.sensitivity_for`(只可能更严);返回密级分布。
359
+ - `condition_space` 走 `codeindex/docindex.condition_space`,与正文的
360
+ `# 生效条件:` 行**同源**——改造前此处只写 `observation_position` 单槽,
361
+ 而单槽不是生效条件,于是 frontmatter 的条件空间形同未声明。
362
+ 返回 (ids, sens_counts)。
363
+ """
364
+ from . import codeindex, docindex
365
+ ids, sens = [], {}
366
+ for it in items:
367
+ if kind == "code_ref":
368
+ nid = codeindex.node_id(it)
369
+ cg.add(
370
+ nid, codeindex.render(it),
371
+ layer=(layer_of(nid) if layer_of else None) or layer or "knowledge",
372
+ tags=["code", "code:" + it.get("kind", ""),
373
+ "domain:" + _domain_of(it)],
374
+ condition_space=codeindex.condition_space(it),
375
+ verification_basis=it.get("basis") or "compiler",
376
+ code_ref=_code_ref(it, root),
377
+ )
378
+ elif kind == "doc_ref":
379
+ nid = docindex.node_id(it)
380
+ level = it.get("level")
381
+ s, _basis = docindex.sensitivity_for(it.get("path") or "", sensitivity)
382
+ sens[s] = sens.get(s, 0) + 1
383
+ cg.add(
384
+ nid, docindex.render(it),
385
+ layer=(layer_of(nid) if layer_of else None) or layer or "knowledge",
386
+ tags=["doc", "doc:md", f"level:{level}",
387
+ "domain:" + _domain_of(it)],
388
+ condition_space=docindex.condition_space(it),
389
+ verification_basis="data",
390
+ sensitivity=s,
391
+ doc_ref=_doc_ref(it, root),
392
+ )
393
+ else:
394
+ raise ValueError(f"未知 ref kind:{kind!r}")
395
+ ids.append(nid)
396
+ return ids, sens
397
+
398
+
399
+ # 生效条件:把入参 root 原样写入返回 dict 的 'root',path/name/kind/lineno/end/lang/hash 按 it.get 取值(缺省 None),precise 取 bool(it.get('precise', True)),render_version 取传入值(传入 None 时延迟 import codeindex 取 codeindex.RENDER_VERSION,保证与 render 契约**同源**、无第二处硬编码)。
400
+ def _code_ref(it: dict, root: str, render_version=None) -> dict:
401
+ if render_version is None: # 直接调用点的兜底:与 render 产物同源
402
+ from . import codeindex
403
+ render_version = codeindex.RENDER_VERSION
404
+ return {
405
+ "path": it.get("path"), "name": it.get("name"),
406
+ "kind": it.get("kind"), "lineno": it.get("lineno"), "end": it.get("end"),
407
+ "lang": it.get("lang"), "precise": bool(it.get("precise", True)),
408
+ "hash": it.get("hash"), "root": root,
409
+ "render_version": render_version,
410
+ }
411
+
412
+
413
+ # 生效条件:把入参 root 原样写入返回 dict 的 'root',path/heading/heading_path/level/lineno/end/anchor/hash/lang 按 it.get 取值(缺省 None),precise 取 bool(it.get('precise', True))。
414
+ def _doc_ref(it: dict, root: str) -> dict:
415
+ return {
416
+ "path": it.get("path"), "heading": it.get("heading"),
417
+ "heading_path": it.get("heading_path"), "level": it.get("level"),
418
+ "lineno": it.get("lineno"), "end": it.get("end"),
419
+ "anchor": it.get("anchor"), "hash": it.get("hash"), "lang": it.get("lang"),
420
+ "precise": bool(it.get("precise", True)), "root": root,
421
+ }
422
+
423
+
424
+ # --------------------------------------------------------------------------
425
+ # 回读(唯一实现:op=ref 与 check_refs 共用)
426
+ # --------------------------------------------------------------------------
427
+
428
+ # 生效条件:传入 ref 为假值(如 None/{})时按 {} 处理,rel 取 ref.get("path") or "";root 与 ref.get("root") 均为假值时返回含 ref/path/status:"unresolved"/ok:False/error:"ref 未记录 root..." 的 out;否则用 root or ref.get("root") 与 rel 拼 fp,os.path.isfile(fp) 为假时返回 status:"dangling"、stale:True,读取抛 OSError/UnicodeDecodeError 时返回 status:"error";读取成功时 lineno 取 int(ref.get("lineno") or 1)(假值回落 1)、end 取 int(ref.get("end") or lineno)(假值回落 lineno),ref.get("hash") 为 None 时 match=None、ok=True、status:"ok",ref.get("hash") 为真值且等于 region_hash 时 ok=True/status:"ok"、不等时 ok=False/status:"stale",ref.get("hash") 为假值但非 None(如 ""/0/False)时 ok=False/status:"stale";with_text 为真时 out["text"] 取 lines[max(0,lineno-1):max(max(0,lineno-1),end)] 的 join;
429
+ def probe_ref(ref: dict, *, root: str = None, with_text: bool = False) -> dict:
430
+ """只读探测单个 ref 的状态(不回读整篇,除非 with_text)。"""
431
+ from . import codeindex
432
+ ref = ref or {}
433
+ rel = ref.get("path") or ""
434
+ r = root or ref.get("root") or ""
435
+ base = {"ref": ref, "path": rel, "status": "unresolved", "ok": False}
436
+ if not r:
437
+ return {**base, "error": "ref 未记录 root,请显式传 root 参数"
438
+ "(索引里存的是相对 root 的 path)"}
439
+ fp = os.path.join(r, rel)
440
+ base["root"] = r
441
+ if not os.path.isfile(fp):
442
+ return {**base, "status": "dangling", "abspath": fp, "stale": True,
443
+ "error": f"源文件不存在(索引已悬空):{fp}"}
444
+ try:
445
+ with open(fp, "r", encoding="utf-8") as f:
446
+ lines = f.read().split("\n")
447
+ except (OSError, UnicodeDecodeError) as exc:
448
+ return {**base, "status": "error", "abspath": fp,
449
+ "error": f"读取失败:{exc}"}
450
+ total = len(lines)
451
+ lineno = int(ref.get("lineno") or 1)
452
+ end = int(ref.get("end") or lineno)
453
+ got = codeindex.region_hash(lines, lineno, end)
454
+ expect = ref.get("hash")
455
+ match = (got == expect) if expect else None
456
+ out = {
457
+ **base, "abspath": fp, "total_lines": total,
458
+ "hash": got, "hash_expected": expect, "hash_match": match,
459
+ "stale": bool(expect) and not match,
460
+ "ok": expect is None or bool(match),
461
+ "status": "ok" if (expect is None or match) else "stale",
462
+ }
463
+ if with_text:
464
+ lo = max(0, lineno - 1)
465
+ out["text"] = "\n".join(lines[lo:max(lo, end)])
466
+ return out
467
+
468
+
469
+ # 生效条件:ref 经 probe_ref(root=root, with_text=True) 后 status 为 'ok'/'stale' 时返回 ok=True 及 text/total_lines/hash/hash_match/stale/precise,status 为 'unresolved'/'error'/'dangling' 时返回 ok=False 与 error。
470
+ def read_ref(ref: dict, *, root: str = None, ref_kind: str = "ref") -> dict:
471
+ """按 ref 回读源区间——`op=ref` 与 `check_refs` 的唯一实现。"""
472
+ p = probe_ref(ref, root=root, with_text=True)
473
+ base = {"ref": ref, "ref_kind": ref_kind}
474
+ if p["status"] in ("unresolved", "error", "dangling"):
475
+ out = {**base, "ok": False, "error": p["error"]}
476
+ if p["status"] == "dangling":
477
+ out["stale"] = True
478
+ return out
479
+ return {
480
+ **base, "ok": True, "text": p["text"], "total_lines": p["total_lines"],
481
+ "hash": p["hash"], "hash_expected": p["hash_expected"],
482
+ "hash_match": p["hash_match"], "stale": p["stale"],
483
+ "precise": bool((ref or {}).get("precise", True)),
484
+ "note": "按 ref 区间回读;hash_match=False 说明源已改动,"
485
+ "重跑 index_code / index_doc 重建",
486
+ }
487
+
488
+
489
+ # 生效条件:cg 的 index['nodes'] 非空时汇总 stale/dangling/unresolved/errors 并返回 ok =(无 stale 且无 dangling);only_tagged 为真时只探测 tags 含 'code'/'doc' 的节点,ledger 非空时先走 (size, mtime) 快路径。
490
+ def check_refs(cg, *, ledger: "Ledger" = None, max_nodes: int = MAX_CHECK,
491
+ only_tagged: bool = True) -> dict:
492
+ """漂移 / 悬空巡检(只读、不抛)。
493
+
494
+ 优先走 ledger 的 (size, mtime) 快路径:未变文件**不读盘**直接判 ok;
495
+ 变了的文件读一次、按记录区间重算哈希判 stale。
496
+ ledger 覆盖不到的节点(早期索引 / 未开增量)再回退逐节点探测。
497
+
498
+ `only_tagged=True`(默认)只探测带 `code` / `doc` 标签的节点——ref 只由
499
+ `index_code` / `index_doc` 产生,两者都会打这两个标签;这样巡检不必为每条
500
+ 记忆节点都读一次盘。需要穷举(含手写 ref)时传 False。
501
+ """
502
+ nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
503
+ stale, dangling, unresolved, errors = [], [], [], []
504
+ covered = set()
505
+
506
+ # 生效条件:当 nid、ref、kind、rel 传入时,p = probe_ref(ref) 后按 p["status"] 分派:为 "dangling" 时把含 node_id/ref_kind/path/lineno/end/error 的 row 加入 dangling,为 "stale" 时补 hash_expected/hash 加入 stale,为 "unresolved" 时加入 unresolved,为 "error" 时加入 errors;其他状态不加入;
507
+ def _probe_one(nid, ref, kind, rel):
508
+ p = probe_ref(ref)
509
+ row = {"node_id": nid, "ref_kind": kind, "path": rel,
510
+ "lineno": ref.get("lineno"), "end": ref.get("end"),
511
+ "error": p.get("error", "")}
512
+ if p["status"] == "dangling":
513
+ dangling.append(row)
514
+ elif p["status"] == "stale":
515
+ row["hash_expected"] = p.get("hash_expected")
516
+ row["hash"] = p.get("hash")
517
+ stale.append(row)
518
+ elif p["status"] == "unresolved":
519
+ unresolved.append(row)
520
+ elif p["status"] == "error":
521
+ errors.append(row)
522
+
523
+ # 快路径:ledger 记录的文件(键 = 源文件绝对路径,天然跨大域不撞名)
524
+ if ledger is not None:
525
+ for key, e in sorted((ledger.load().get("files") or {}).items()):
526
+ rec_nodes = [n for n in (e.get("nodes") or [])
527
+ if n.get("id") in nodes]
528
+ if not rec_nodes:
529
+ continue
530
+ covered.update(n.get("id") for n in rec_nodes)
531
+ rel = e.get("path") or ""
532
+ # 必须用**每个源文件自己的 root**(索引时的源大域),不能用 ledger.root:
533
+ # 后者是认知图根,拿它拼路径会指向不存在的位置、把一切都误判成 dangling。
534
+ src_root = e.get("root") or os.path.dirname(key)
535
+ try:
536
+ st = os.stat(key)
537
+ unchanged = (e.get("size") == st.st_size
538
+ and abs(float(e.get("mtime") or 0.0) - st.st_mtime) < 1e-6)
539
+ except OSError:
540
+ unchanged = False
541
+ if not unchanged:
542
+ kind = e.get("kind") or kind_of_path(rel)
543
+ for n in rec_nodes:
544
+ _probe_one(n.get("id"), {**n, "path": rel, "root": src_root},
545
+ kind, rel)
546
+
547
+ # 回退:ledger 未覆盖的索引节点
548
+ # 生效条件:当 nid 传入时,若 only_tagged 为假值立即返回 True;否则取 (nodes.get(nid) or {}).get("tags") or [],仅当其中存在 "code" 或 "doc" 返回 True,否则返回 False;
549
+ def _candidate(nid):
550
+ if not only_tagged:
551
+ return True
552
+ tags = (nodes.get(nid) or {}).get("tags") or []
553
+ return any(t in ("code", "doc") for t in tags)
554
+
555
+ todo = [nid for nid in nodes if nid not in covered and _candidate(nid)]
556
+ truncated = len(todo) > max_nodes
557
+ checked = 0
558
+ for nid in todo[:max_nodes]:
559
+ try:
560
+ node = cg.get(nid)
561
+ except Exception:
562
+ continue
563
+ if not node:
564
+ continue
565
+ kind, ref = ref_of(node)
566
+ if not ref:
567
+ continue
568
+ checked += 1
569
+ try:
570
+ _probe_one(nid, ref, kind, ref.get("path") or "")
571
+ except Exception as exc: # 巡检不抛
572
+ errors.append({"node_id": nid, "ref_kind": kind, "error": str(exc)})
573
+
574
+ return {
575
+ "ok": not (stale or dangling),
576
+ "status": "ok" if not (stale or dangling) else ("dangling" if dangling else "stale"),
577
+ "checked": checked + len(covered),
578
+ "ledger_files": len((ledger.load().get("files") or {})) if ledger else 0,
579
+ "stale": stale, "dangling": dangling,
580
+ "unresolved": unresolved, "errors": errors,
581
+ "truncated": truncated, "max_nodes": max_nodes,
582
+ }
583
+
584
+
585
+ # --------------------------------------------------------------------------
586
+ # 节点级对账:孤儿清退 + 悬空清退
587
+ #
588
+ # 水位层的 `Ledger.reconcile` 只剪**水位条目**、`prune` 只剪「源大域已消失」
589
+ # 的条目,两者都不碰**节点**。于是节点层的两类残留无人处置:
590
+ # · 孤儿(同一文档的过期代)——标题路径一变 id 全量重算,旧代与新代并存;
591
+ # · 悬空(源文件已删)——回读必然失败,巡检永远报 dangling。
592
+ # 本节的唯一实现同时供 `op=index_code / index_doc`(孤儿)与
593
+ # `op=ref action=prune`(悬空)使用,避免两处各写一套口径。
594
+ # --------------------------------------------------------------------------
595
+
596
+ # 生效条件:返回 os.path.normcase(os.path.abspath(str(p or ''))),即 p 为 None/空串时返回当前目录的归一绝对路径。
597
+ def _norm_root(p) -> str:
598
+ """root 归一:同一目录的大小写/分隔符差异不得影响「同一大域」判定。"""
599
+ return os.path.normcase(os.path.abspath(str(p or "")))
600
+
601
+
602
+ # 生效条件:a 与 b 都非空且 _norm_root(a) == _norm_root(b) 时返回 True,否则(含 TypeError/ValueError)返回 False。
603
+ def _same_root(a, b) -> bool:
604
+ try:
605
+ return bool(a) and bool(b) and _norm_root(a) == _norm_root(b)
606
+ except (TypeError, ValueError):
607
+ return False
608
+
609
+
610
+ # 生效条件:返回 str(p or '').replace('\\', '/').lstrip('./'),即 p 为 None/空串时返回 ''。
611
+ def _norm_rel(p) -> str:
612
+ return str(p or "").replace("\\", "/").lstrip("./")
613
+
614
+
615
+ # 生效条件:cg 具备可调用的 forget 方法时对 plan 中每个 nid 调 cg.forget(nid, why),返回 (成功 id 列表, 被拦下/失败的 {node_id, error} 列表);cg 无 forget 时返回 ([], plan 中每 nid 一条错误)。
616
+ def _forget_many(cg, plan, why: str) -> tuple:
617
+ """逐条软删(进 trash/、写删除清单、可 restore);受保护节点拦下不删。
618
+
619
+ `forget` 属 **MdCGOS 层**能力(保护裁决 + 回收站 + 删除清单),基础层
620
+ `MdCG` 没有任何删除原语。缺能力时**明确报错、不静默跳过**——否则
621
+ 「清退了 N 条」看着成功、实际一条没删(P27 §10 实测过这个坑)。
622
+ """
623
+ fn = getattr(cg, "forget", None)
624
+ if not callable(fn):
625
+ err = (f"{type(cg).__name__} 无 forget 能力(对账须能删除节点);"
626
+ f"生产路径是 MdCGOS,测试请用 MdCGOS")
627
+ return [], [{"node_id": nid, "error": err} for nid in sorted(plan)]
628
+ done, blocked = [], []
629
+ for nid in sorted(plan):
630
+ try:
631
+ res = fn(nid, why) or {}
632
+ except Exception as exc: # ProtectionError 等 → 拦下,不越权
633
+ blocked.append({"node_id": nid, "error": str(exc)[:120]})
634
+ continue
635
+ if res.get("ok"):
636
+ done.append(nid)
637
+ else:
638
+ blocked.append({"node_id": nid, "error": str(res.get("error"))[:120]})
639
+ return done, blocked
640
+
641
+
642
+ # 生效条件:cg 具备可调用的 _unstage 时对 ghosts 逐个调用并收集成功 id(单条异常跳过),cg 无该能力时返回 [](不假装成功)。
643
+ def _drop_ghosts(cg, ghosts) -> list:
644
+ """摘除幽灵条目的索引记录(节点文件已不存在,没有可软删的实体)。
645
+
646
+ 走 `MdCG._unstage`:它顺带落删除记录,保证幽灵不会再次从分片日志里复活。
647
+ 基础层没有该能力时如实返回空列表(不假装成功)。
648
+ """
649
+ drop = getattr(cg, "_unstage", None)
650
+ if not callable(drop):
651
+ return []
652
+ out = []
653
+ for nid in ghosts:
654
+ try:
655
+ drop(nid)
656
+ out.append(nid)
657
+ except Exception: # 单条失败不拖垮整批
658
+ continue
659
+ return out
660
+
661
+
662
+ # 生效条件:items 中同 kind 的节点若其 ref['path'] 命中本次 items 的文件、id 不在本次产出内且 ref['root'] 与入参 root 同一(_same_root),则列入清退计划;dry_run 为真只返回计划,否则经 _forget_many 软删;items 为空时返回 count 0。
663
+ def prune_orphans(cg, *, kind: str, root: str, items, dry_run: bool = False,
664
+ reason: str = "") -> dict:
665
+ """清退「同 root + 同 path,但已不在本次产出里」的**过期代**节点。
666
+
667
+ 为什么必须有:`node_id = sha1(相对path + "#" + heading_path)`,而
668
+ `add_items` 只做**同 id 幂等 upsert**——文档标题结构一变,整篇 id 全量
669
+ 重算,旧代节点无人清退,与新代并存(同一文档召回两份,且旧代引用的区间
670
+ 已失效)。截断的索引由调用方负责不调用本函数(没扫完 ≠ 剩下的都过期)。
671
+
672
+ 范围**只限本次真正重切过的文件**(`items` 的 path):增量索引跳过的未变
673
+ 文件不在 items 里,其节点不进判定——否则会把完好的节点整片误删。
674
+ """
675
+ nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
676
+ touched: dict = {}
677
+ for it in items or []:
678
+ rel = _norm_rel(it.get("path"))
679
+ if rel:
680
+ touched.setdefault(rel, set()).add(node_id_of(it, kind))
681
+ base = {"scanned": len(nodes), "touched_files": len(touched),
682
+ "dry_run": bool(dry_run)}
683
+ if not touched:
684
+ return {**base, "ok": True, "count": 0, "pruned": [],
685
+ "skipped_protected": []}
686
+
687
+ # ⚠ ref 只存在于**节点 frontmatter**里;`cg.index['nodes']` 是元数据快照
688
+ # (path/layer/tags/…,见 mdcg._scan_nodes),**不含 ref**。因此必须
689
+ # `cg.get(nid)` 取回节点再 ref_of —— 否则 ref_of 恒返回 ('', None)、
690
+ # 整个对账静默失效(P27 §10 实测过这个坑)。标签预筛与 check_refs 同口径,
691
+ # 免得为全库每条记忆都读一次盘。
692
+ tag = "doc" if kind == "doc_ref" else "code"
693
+ plan = []
694
+ for nid, e in nodes.items():
695
+ if tag not in ((e or {}).get("tags") or []):
696
+ continue
697
+ try:
698
+ node = cg.get(nid)
699
+ except Exception:
700
+ continue
701
+ if not node:
702
+ continue
703
+ k, ref = ref_of(node)
704
+ if k != kind or not isinstance(ref, dict):
705
+ continue
706
+ keep = touched.get(_norm_rel(ref.get("path")))
707
+ if keep is None or nid in keep or not _same_root(ref.get("root"), root):
708
+ continue
709
+ plan.append(nid)
710
+
711
+ why = reason or ("索引重建:本节点已不在同文档新代产出中(标题路径变更致 "
712
+ "node_id 重算),清退过期代以消除重复召回")
713
+ if dry_run:
714
+ return {**base, "ok": True, "count": len(plan), "pruned": sorted(plan)[:50],
715
+ "skipped_protected": [], "reason": why}
716
+ done, blocked = _forget_many(cg, plan, why)
717
+ return {**base, "ok": True, "count": len(done), "pruned": done[:50],
718
+ "skipped_protected": blocked[:20], "reason": why}
719
+
720
+
721
+ # 生效条件:cg 中带 'code'/'doc' 标签且 ref 带 root 的节点经 probe_ref 判为 'dangling' 时列入清退计划;only_roots 为空时另收集 cg.root 下取不到对应节点 .md 的幽灵条目经 _drop_ghosts 摘除;dry_run 为真只返回计划。
722
+ def prune_dangling(cg, *, only_roots=None, dry_run: bool = False,
723
+ max_nodes: int = MAX_CHECK, reason: str = "") -> dict:
724
+ """清退**悬空**节点:ref 指向的源文件已删除,回读必然失败。
725
+
726
+ `check_refs` 只报告不处置(其原话是「悬空需人工处置」),本函数就是那个
727
+ 出口——已删脚本、被搬走的文档留下的残留节点一次清掉,而不是逐条手工
728
+ `forget`。判定与巡检共用 `probe_ref` 的唯一实现,口径不会打架。
729
+
730
+ 另清**幽灵条目**(ghosts):索引有条目、节点文件却不存在。它们是历史
731
+ 「删除只摘内存索引、不落盘」的遗留——`cg.get` 取不回 → 悬空清退够不着它,
732
+ 而 `check_refs` 走 ledger 会一直报 → dangling 永不归零。判据只用唯一真源
733
+ (节点 .md 不存在即脏索引),与「索引是派生物」的宣言一致;`only_roots`
734
+ 非空时跳过(幽灵条目无 ref,无法归因到某个源大域)。
735
+ """
736
+ nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
737
+ ghosts = []
738
+ if not only_roots:
739
+ for nid, e in list(nodes.items()):
740
+ tags = (e or {}).get("tags") or []
741
+ if not any(t in ("code", "doc") for t in tags):
742
+ continue
743
+ path = (e or {}).get("path") or ""
744
+ if path and not os.path.exists(os.path.join(cg.root, path)):
745
+ ghosts.append(nid)
746
+ # 同 prune_orphans:ref 只在节点 frontmatter 里,索引条目里没有,
747
+ # 必须 cg.get 取回节点再 ref_of(否则恒空、静默不删)。
748
+ todo = []
749
+ for nid, e in nodes.items():
750
+ tags = (e or {}).get("tags") or []
751
+ if not any(t in ("code", "doc") for t in tags):
752
+ continue
753
+ try:
754
+ node = cg.get(nid)
755
+ except Exception:
756
+ continue
757
+ if not node:
758
+ continue
759
+ _k, ref = ref_of(node)
760
+ if not ref or not ref.get("root"):
761
+ continue
762
+ if only_roots and not any(_same_root(ref.get("root"), r) for r in only_roots):
763
+ continue
764
+ todo.append((nid, ref))
765
+ truncated = len(todo) > max_nodes or len(ghosts) > max_nodes
766
+ plan = []
767
+ for nid, ref in todo[:max_nodes]:
768
+ try:
769
+ if probe_ref(ref).get("status") == "dangling":
770
+ plan.append(nid)
771
+ except Exception: # 探测失败不算悬空(宁可不删)
772
+ continue
773
+
774
+ why = reason or ("源文件已删除,索引节点悬空(回读必然失败),"
775
+ "清退以消除永不消失的 dangling")
776
+ ghost_plan = sorted(ghosts)[:max_nodes]
777
+ base = {"scanned": len(nodes), "candidates": len(plan),
778
+ "ghosts": len(ghost_plan),
779
+ "dry_run": bool(dry_run), "truncated": truncated,
780
+ "max_nodes": max_nodes}
781
+ if dry_run:
782
+ return {**base, "ok": True, "count": len(plan), "pruned": sorted(plan)[:50],
783
+ "ghost_pruned": ghost_plan[:50],
784
+ "skipped_protected": [], "reason": why}
785
+ done, blocked = _forget_many(cg, plan, why)
786
+ dropped = _drop_ghosts(cg, ghost_plan)
787
+ return {**base, "ok": True, "count": len(done), "pruned": done[:50],
788
+ "ghost_pruned": dropped[:50],
789
+ "skipped_protected": blocked[:20], "reason": why}
790
+
791
+
792
+ # 生效条件:cg 节点按 ref['root'] 与 kind 分组后逐组以 index_dir(incremental=False, ledger=ledger) 重切、再以 add_items(layer_of=原 layer) 重建,返回 {'ok','roots','groups','indexed','errors','truncated'};only_roots 非 None 时只处理其中列出的 root。
793
+ def rebuild(cg, *, ledger: "Ledger" = None, only_roots=None, max_files: int = 500,
794
+ max_items: int = 2000) -> dict:
795
+ """按 ref 记录的 root 重建索引(sustain.heal 的修复动作)。
796
+
797
+ 只重跑出了问题的 root(`only_roots`),逐节点**保留原 layer**;
798
+ doc 密级用默认策略重算(默认只可能更严,不会放松)。
799
+ """
800
+ nodes = (getattr(cg, "index", {}) or {}).get("nodes") or {}
801
+ groups = {}
802
+ for nid in nodes:
803
+ try:
804
+ node = cg.get(nid)
805
+ except Exception:
806
+ continue
807
+ kind, ref = ref_of(node)
808
+ if not ref or not ref.get("root"):
809
+ continue
810
+ r = ref["root"]
811
+ if only_roots is not None and r not in only_roots:
812
+ continue
813
+ groups.setdefault((r, kind), 0)
814
+ groups[(r, kind)] += 1
815
+
816
+ out = {"ok": True, "roots": sorted({r for r, _ in groups}),
817
+ "groups": len(groups), "indexed": 0, "errors": [], "truncated": False}
818
+ for (root, kind) in sorted(groups):
819
+ try:
820
+ items, errors, stats = index_dir(
821
+ root, kind=kind, max_files=max_files, max_items=max_items,
822
+ incremental=False, ledger=ledger)
823
+ ids, _sens = add_items(
824
+ cg, items, kind=kind, root=root,
825
+ layer_of=lambda nid: (nodes.get(nid) or {}).get("layer"))
826
+ out["indexed"] += len(ids)
827
+ out["errors"].extend(errors)
828
+ out["truncated"] = out["truncated"] or bool(stats.get("truncated"))
829
+ except Exception as exc: # 自愈不抛
830
+ out["ok"] = False
831
+ out["errors"].append(f"{root} [{kind}]:{exc}")
832
+ if ledger is not None:
833
+ ledger.save()
834
834
  return out