@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
@@ -1,939 +1,939 @@
1
- # -*- coding: utf-8 -*-
2
- """记忆评审流水线 · M3 定位模块(D1 字段级定位,确定性、零写入)。
3
-
4
- 真源:docs/记忆评审系统_立项设计与施工交接_20260915.md §5 D1。
5
-
6
- D1 契约:
7
-
8
- locate(node_id, issue_hint) -> [{"field", "span", "issue_kind", "evidence"}, ...]
9
-
10
- * **field** —— frontmatter 字段名,或 `"content"`(正文)
11
- * **span** —— 命中处的字符区间 `[start, end)`(半开),**相对正文 content**;
12
- 纯字段级缺失无字符区间可指 → `None`
13
- * **issue_kind** —— dup / missing_field / stale / contradiction / weak_source / template_flow
14
- (`ISSUE_KINDS`=问题面);另有 `ADVISORY_KINDS`(观测面,可显式定位
15
- 但**不进默认全量、不进评审告警面**,见下「stale 与 observation_aged」)
16
- * **evidence** —— 白箱判据:为什么算问题(人可复核、机器可断言)
17
-
18
- 定位的职责边界:M2 的意见说「这一条有问题」,D1 回答「问题在这一条的哪个字段/哪一段」。
19
- 故本模块只做**确定性可判定**的事——语义级矛盾(正文自相矛盾)**不猜**,如实返回
20
- BLINDSPOT 项交回 LLM 意见层(`status="blindspot"`,见 `locate()` 返回值)。
21
-
22
- 口径全部引用既有唯一真源,**不在本模块另立一份**(防「术语双写法」漂移):
23
-
24
- =============== ==================================================================
25
- 字段/正文结构 ``nodefile``(CCG_MARKS / CONDITION_SLOTS / VERIFICATION_BASIS /
26
- loads / content_hash / is_placeholder_text / time_window_text)
27
- 模板骨架 ``writelimit._skeleton`` / ``MIN_SKELETON``
28
- 赛道×基底相容 ``crosscheck``(与 M1 `ruleset._chk_basis_licensed` 同源)
29
- 字段层门限 ``ruleset``(即 `rules/*.json` 的 `matcher.layer`,见
30
- `field_layer_scope`——**派生**,不在此另立一份)
31
- 索引快照 ``conformance.load_index``
32
- =============== ==================================================================
33
-
34
- 与 M1(`ruleset` 机械层)的用词归并:D1 说 `dup`,M1 的 `dup_hash_group` 说
35
- `dup_content`——同一件事两种写法,由 `KIND_ALIASES` 单一归口(`canonical_kind`)。
36
-
37
- 超出 D1 最小契约的**增强键**(供人工核对直接跳文件/跳行,不影响四键契约):
38
- `line`(节点文件原文 1-based 行号)、`sentence`(正文句索引,0 基)、
39
- `snippet`(命中片段)、`rule`(触发定位的判据名,审计用)、`peer`(同组对照节点)、
40
- `cause`(`contradiction` 的**成因**判定,见 `hash_mismatch_cause`——同一条命中
41
- 可能是「真不一致」也可能是「索引快照滞后」,两种成因不分会让复核者误判为数据损坏)。
42
-
43
- **`stale` 与 `observation_aged` 的判据源分工(2026-09-16 修正)**:
44
-
45
- * `stale`(**问题面**)—— **依赖存在性**:`code_ref`/`doc_ref` 指的源文件已不存在
46
- (悬空)或已漂移(区间哈希不符)。判据直接调 `refindex.probe_ref`,**不另立一份**
47
- (防「术语双写法」漂移)。代码知识的真值挂在源文件上,源没了/改了才是适用边界越出。
48
- * `observation_aged`(**观测面**,`ADVISORY_KINDS`,**不进默认全量**)——
49
- `condition_space.time_window` 已过。这是**观测时刻**不是失效声明:写入端
50
- (`mdcg.add`,见其 `OBSERVATION_WINDOW_SEC`)在未给 time_window 时以**写入时刻**
51
- 自动填 1 小时窗,故真库中「已过期」绝大多数是「写入超过 1 小时」,与知识是否失效
52
- 无关——把它当问题投进评审告警面就是系统性误报(实测:全库 1453 条过期里 1434 条
53
- 属默认窗填充)。需要时效判定用 `stale`(依赖存在性),不是这里。
54
- """
55
- from __future__ import annotations
56
-
57
- import argparse
58
- import json
59
- import os
60
- import re
61
- import sys
62
- import time
63
-
64
- from .. import conformance as CF
65
- from .. import crosscheck as CC
66
- from .. import nodefile as NF
67
- from .. import refindex as RI
68
- from .. import writelimit as WL
69
- from . import ruleset as RS
70
-
71
- __all__ = ["ISSUE_KINDS", "ADVISORY_KINDS", "KIND_ALIASES", "canonical_kind", "locate",
72
- "locate_many", "locate_package", "sentence_spans", "mark_spans", "load_node",
73
- "main", "field_layer_scope", "hash_mismatch_cause", "HASH_CAUSES"]
74
-
75
- #: D1 的问题类别(真源:立项文档 §5 D1)——**问题面**:命中即「知识有问题」
76
- ISSUE_KINDS = ("dup", "missing_field", "stale", "contradiction",
77
- "weak_source", "template_flow")
78
-
79
- #: **观测面**(咨询级):可显式定位,但**不进默认全量定位、不进评审告警面**。
80
- #: 判据:本类命中说的是「观测手段的副产品」(如「这条记忆写入超过 1 小时」),
81
- #: 不是「知识有问题」——把它混进 ISSUE_KINDS 会让每一次写入在 1 小时后自动变成
82
- #: 待评审问题(系统性误报)。故与问题面分表,只有显式点名才产出。
83
- ADVISORY_KINDS = ("observation_aged",)
84
-
85
- #: M1(ruleset 机械层)与 D1 的用词归并——两处说的是同一件事,不许各说各话
86
- KIND_ALIASES = {"dup_content": "dup", # M1 `dup_hash_group`
87
- "missing": "missing_field",
88
- "flow": "template_flow",
89
- "template_flow_digits_only": "template_flow"}
90
-
91
- #: 扫描 frontmatter 的字段清单(真源字段名取自 nodefile 口径,不自由发明)
92
- FM_SCAN_FIELDS = ("role", "tags", "importance", "evidence_count",
93
- "verification_basis", "lifecycle_state", "condition_space")
94
-
95
- #: 判为「同模板流水」所需的最少同骨架句数——单句偶合不算流水(防误报)
96
- MIN_FLOW_SENTENCES = 2
97
-
98
- #: 一句话摘要的最大长度
99
- SNIPPET_MAX = 120
100
-
101
- #: 索引快照与节点文件的 mtime 差**小于**此值 → 视为同一批写入,不判「快照滞后」
102
- MTIME_TOLERANCE = 1.0
103
-
104
- #: `content_hash` 不一致的**成因**(确定性判据,不猜;见 `hash_mismatch_cause`)
105
- HASH_CAUSES = ("index_lag", "true_mismatch", "unknown")
106
-
107
- #: `refindex.probe_ref` 的五态里,哪些构成「依赖已失效」(`stale` 判据,见 `_loc_stale`):
108
- #: 只有这两态说明**载体真的没了/改了**;`unresolved`(拿不到 root)与 `error`(读盘失败)
109
- #: 是观测手段不足,按「不猜」纪律不判。
110
- _REF_DEAD_STATUSES = ("dangling", "stale")
111
-
112
- #: 字段层门限缓存(规则库是包内数据文件,进程内不变 → 惰性算一次)
113
- _FIELD_LAYER_CACHE = None
114
-
115
-
116
- # 生效条件:kind 经 str(kind or "").strip() 得 k(None/空串等假值 → 空串 ""),k 命中 KIND_ALIASES 时返回其规范名,否则原样返回 k。
117
- def canonical_kind(kind) -> str:
118
- """M1/D1 用词 → D1 规范名(未知原样返回,不假装认路)。"""
119
- k = str(kind or "").strip()
120
- return KIND_ALIASES.get(k, k)
121
-
122
-
123
- # 生效条件:仅当 rules 与 rules_dir 均为 None 且模块级 _FIELD_LAYER_CACHE 非 None 时直接返回该缓存;否则遍历 rules(dict 取其 "rules" 键的列表、非 dict 直接 list(rules))或 rules_dir 经 RS.load_rules 取得的 rules 列表,从限了 matcher.layer 的 mechanical 规则中收集 field_absent 的 spec["field"] 与 evidence_zero 的 evidence_count,返回 {字段: 排序列},且只在 rules 与 rules_dir 均为 None 时写回 _FIELD_LAYER_CACHE。
124
- def field_layer_scope(rules=None, rules_dir=None) -> dict:
125
- """字段 → 适用层清单(**派生**自 M1 规则库,不在此另立一份)。
126
-
127
- 真源 = `md_cg/mreview/rules/*.json` 的 `matcher.layer`:只收录**限了层**的
128
- 机械规则字段(`field_absent` 取 `spec["field"]`、`evidence_zero` 取
129
- `evidence_count`);未限层的规则 = 全层适用,不入表。
130
-
131
- 理由:字段范围两处各写一份 ⇒ 必然漂移(同「术语双写法」)。派生使 M1 改规则
132
- 时 M3 自动跟随。规则库损坏 → `load_rules` 抛错(fail-closed,不静默降级成全层)。
133
- """
134
- global _FIELD_LAYER_CACHE
135
- if rules is None and rules_dir is None and _FIELD_LAYER_CACHE is not None:
136
- return _FIELD_LAYER_CACHE
137
- if isinstance(rules, dict):
138
- rl = list(rules.get("rules") or [])
139
- elif rules is not None:
140
- rl = list(rules)
141
- else:
142
- rl = list(RS.load_rules(rules_dir)["rules"])
143
- scope = {}
144
- for r in rl:
145
- layers = set((r.get("matcher") or {}).get("layer") or [])
146
- if not layers:
147
- continue
148
- for spec in r.get("mechanical") or []:
149
- chk = spec.get("check")
150
- if chk == "field_absent" and spec.get("field"):
151
- scope.setdefault(str(spec["field"]), set()).update(layers)
152
- elif chk == "evidence_zero":
153
- scope.setdefault("evidence_count", set()).update(layers)
154
- out = {k: sorted(v) for k, v in sorted(scope.items())}
155
- if rules is None and rules_dir is None:
156
- _FIELD_LAYER_CACHE = out
157
- return out
158
-
159
-
160
- # 生效条件:(scope or {}).get(field) 为假值(scope 为 None、空 dict 或 field 不在其中)→ 返回 True;want 为真值时仅当 str((meta or {}).get("layer")) 在 set(want) 内返回 True,否则 False。
161
- def _layer_ok(field, scope, meta) -> bool:
162
- """字段层门限:**限层的字段只在该层的节点上检查**。
163
-
164
- 口径与 M1 `ruleset._scope`(`str(e.get("layer")) in want`)逐字一致——缺层
165
- (`None`)不在任何层清单内 ⇒ 不检查。两处判据同源,防「越层误报」:
166
- M1 不在某层报的问题,M3 也不该报。
167
- """
168
- want = (scope or {}).get(field)
169
- if not want:
170
- return True
171
- return str((meta or {}).get("layer")) in set(want)
172
-
173
-
174
- # 生效条件:os.path.getmtime(path) 成功则返回该 mtime;抛 OSError 或 TypeError → 返回 None。
175
- def _mtime(path):
176
- """文件 mtime(不可读 → `None`,不假装知道)。"""
177
- try:
178
- return os.path.getmtime(path)
179
- except (OSError, TypeError):
180
- return None
181
-
182
-
183
- # 生效条件:root 或 rel(path 为真值时取 path,否则取 meta 的 "path")为空 → 返回 ("unknown", "无 root/path 可用,取不到盘上 mtime,成因未判定");两者都有时取节点文件与 CF.INDEX_FILE 的 mtime,任一为 None → 返回 ("unknown", "节点文件或索引快照不可读,成因未判定"),节点 mtime 减索引 mtime 之差 > MTIME_TOLERANCE → 返回 ("index_lag", …),否则返回 ("true_mismatch", …)。
184
- def hash_mismatch_cause(meta, *, root=None, path=None) -> tuple:
185
- """`content_hash` 声明值 ≠ 正文实算值 → **成因**判定(靠 mtime 证据,不猜)。
186
-
187
- → `(cause, note)`;cause ∈ `HASH_CAUSES`;取不到盘上证据 → `"unknown"`。
188
-
189
- 写路径真源(`md_cg/mdcg.py`):写走 `_stage()` 落**分片日志**,`rebuild_index()`
190
- 才写 `_index.json` 快照——故「节点文件比索引快照新」是**正常写路径现象**
191
- (快照滞后),不是文件被篡改;把两种成因合并成一句「索引与文件不一致」会让
192
- 复核者误判为数据损坏(实测滞后可达 14s 量级)。
193
- """
194
- rel = path if path else (meta or {}).get("path")
195
- if not root or not rel:
196
- return "unknown", "无 root/path 可用,取不到盘上 mtime,成因未判定"
197
- fp = str(rel) if os.path.isabs(str(rel)) else os.path.join(str(root), str(rel))
198
- fm_t = _mtime(fp)
199
- idx_t = _mtime(os.path.join(str(root), CF.INDEX_FILE))
200
- if fm_t is None or idx_t is None:
201
- return "unknown", "节点文件或索引快照不可读,成因未判定"
202
- delta = fm_t - idx_t
203
- if delta > MTIME_TOLERANCE:
204
- return "index_lag", ("节点文件比索引快照新 %.2fs——分片日志已写、_index.json "
205
- "未 rebuild(快照滞后属正常写路径现象)" % delta)
206
- return "true_mismatch", ("节点文件不晚于索引快照(差 %.2fs)——非快照滞后,"
207
- "索引声明与文件真源相抵触" % delta)
208
-
209
-
210
- # ---------------------------- 基础工具(纯函数) ----------------------------
211
-
212
- # 生效条件:v 为 list/tuple/dict 时返回 not v(空容器 → True);其余类型返回 v is None 或 str(v).strip() 为 "" 或 "None"。
213
- def _blank(v) -> bool:
214
- if isinstance(v, (list, tuple, dict)):
215
- return not v
216
- return v is None or str(v).strip() in ("", "None")
217
-
218
-
219
- # 生效条件:int(float(v)) 可算(含数字字符串)则返回该整数;抛 TypeError 或 ValueError(含 v 为 None、非数字串)→ 返回 None。
220
- def _as_int(v):
221
- try:
222
- return int(float(v))
223
- except (TypeError, ValueError):
224
- return None
225
-
226
-
227
- # 生效条件:float(v) 可算则返回该浮点数;抛 TypeError 或 ValueError(含 v 为 None、非数字串)→ 返回 None。
228
- def _as_float(v):
229
- try:
230
- return float(v)
231
- except (TypeError, ValueError):
232
- return None
233
-
234
-
235
- _SENT_END = "。!?;!?;\n"
236
-
237
-
238
- # 生效条件:content 为假值(None/空串)按 "" 处理,逐字符命中 _SENT_END 切出且 seg.strip() 非空的段以 len(out) 为序号追加 (索引, start, i+1, 原文),末尾 tail.strip() 非空亦追加;无合格段 → 返回空列表。
239
- def sentence_spans(content: str) -> list:
240
- """正文 → `[(句索引, start, end, 原文)]`;空句不编号(索引连续,确定性)。"""
241
- text, out, start = content or "", [], 0
242
- for i, ch in enumerate(text):
243
- if ch in _SENT_END:
244
- seg = text[start:i + 1]
245
- if seg.strip():
246
- out.append((len(out), start, i + 1, seg))
247
- start = i + 1
248
- tail = text[start:]
249
- if tail.strip():
250
- out.append((len(out), start, len(text), tail))
251
- return out
252
-
253
-
254
- #: 正文要素行:`# 功能名:…` / `# 生效条件: …`(间隔与分隔符两形态都收)
255
- _RE_MARK = re.compile(r"^#[ \t]*(?P<mark>[^::\n]{1,16})[::][^\n]*", re.M)
256
-
257
-
258
- # 生效条件:在 content(假值 → "")上用模块级 _RE_MARK 迭代匹配,以 m.group("mark").strip() 为键 setdefault 记下首次出现的 [m.start(), m.end());无匹配(含 content 为假值)→ 返回空 dict。
259
- def mark_spans(content: str) -> dict:
260
- """正文 CCG 要素行 → `{要素名: [start, end)}`;同要素取首次出现(确定性)。"""
261
- out = {}
262
- for m in _RE_MARK.finditer(content or ""):
263
- out.setdefault(m.group("mark").strip(), [m.start(), m.end()])
264
- return out
265
-
266
-
267
- # 生效条件:(text or "").split("\n") 的每行 strip 后非空、不以 "#" 开头且含 ":" 时,取首个冒号前的键 k,k 非空且尚未入表则记 (1-based 行号, [该行起始偏移, 起始偏移+len(line)]);text 为假值或无合格行 → 返回空 dict。
268
- def key_line_spans(text: str) -> dict:
269
- """节点文件原文 → `{键: (1-based 行号, [start, end))}`。
270
-
271
- 只认**无 `#` 前缀**且含 `:` 的行——即 frontmatter 行;`---` 分隔线无冒号,
272
- 自然被排除。同键取首次出现。
273
- """
274
- out, pos = {}, 0
275
- for i, line in enumerate((text or "").split("\n")):
276
- s = line.strip()
277
- if s and not s.startswith("#") and ":" in s:
278
- k = s.split(":", 1)[0].strip()
279
- if k and k not in out:
280
- out[k] = (i + 1, [pos, pos + len(line)])
281
- pos += len(line) + 1
282
- return out
283
-
284
-
285
- # 生效条件:index 为 dict 时其 "nodes" 为 dict 则返回 index["nodes"],否则返回 index 本身;index 为 None/非 dict 时调 CF.load_index(root),抛 OSError 或 ValueError → 返回 {},返回值为 dict 且其 "nodes" 为 dict → 返回该 "nodes",是 dict → 原样返回,否则返回 {}。
286
- def _index(root, index=None) -> dict:
287
- """索引节点表 `{node_id: meta}`(兼容 load_index 的 `{"nodes": …}` 形态)。"""
288
- if isinstance(index, dict):
289
- return index.get("nodes") if isinstance(index.get("nodes"), dict) else index
290
- try:
291
- idx = CF.load_index(root)
292
- except (OSError, ValueError):
293
- return {}
294
- if isinstance(idx, dict) and isinstance(idx.get("nodes"), dict):
295
- return idx["nodes"]
296
- return idx if isinstance(idx, dict) else {}
297
-
298
-
299
- # 生效条件:_index(root, index) 中 node_id 对应值非 dict → 返回 None;否则用 meta.get("path")(绝对路径直接用,否则 join(root, str(rel or "")))读文件——OSError 时返回 content/text 为 None、fm 为 {} 的 dict,成功则把 NF.loads(text) 得到的 content 与 fm(假值 → {})连同 meta/path/text 一并返回。
300
- def load_node(node_id, root, *, index=None) -> dict:
301
- """读一个节点(索引 meta + 文件 frontmatter + 正文原文)。
302
-
303
- → `{"node_id","meta","fm","content","text","path"}`;索引无此项 → `None`。
304
- `meta` 是**索引快照**(含索引侧 content_hash,矛盾判定要用),`fm` 是**文件真源**。
305
- """
306
- nodes = _index(root, index)
307
- meta = nodes.get(node_id)
308
- if not isinstance(meta, dict):
309
- return None
310
- rel = meta.get("path")
311
- fp = rel if (rel and os.path.isabs(str(rel))) else os.path.join(root, str(rel or ""))
312
- try:
313
- with open(fp, encoding="utf-8") as f:
314
- text = f.read()
315
- except OSError:
316
- return {"node_id": node_id, "meta": dict(meta), "fm": {}, "content": None,
317
- "text": None, "path": fp}
318
- fm, content = NF.loads(text)
319
- return {"node_id": node_id, "meta": dict(meta), "fm": fm or {}, "content": content,
320
- "text": text, "path": fp}
321
-
322
-
323
- # 生效条件:span 为 None → 返回 "";否则取 (text or "")[span[0]:span[1]] 去空白并把换行替换为 "⏎",长度超 SNIPPET_MAX 时截断并追加 "…"。
324
- def _snippet(text, span) -> str:
325
- """命中片段(供人工核对肉眼确认「指的是不是这一句」)。"""
326
- if span is None:
327
- return ""
328
- seg = (text or "")[span[0]:span[1]].strip().replace("\n", "⏎")
329
- return seg[:SNIPPET_MAX] + ("…" if len(seg) > SNIPPET_MAX else "")
330
-
331
-
332
- # 生效条件:任意 node_id/kind/field/span/evidence 均原样写入返回 dict 的 node_id/issue_kind/field/span/evidence 键,可选 rule/line/sentence/snippet/peer/cause/severity 未传时为 None、status 未传时为 "located",不做任何校验。
333
- def _hit(node_id, kind, field, span, evidence, *, rule=None, line=None,
334
- sentence=None, snippet=None, peer=None, cause=None, status="located",
335
- severity=None) -> dict:
336
- """命中行(四键契约 + 增强键)。
337
-
338
- `severity` 是**观测面专用**的等级标注(`ADVISORY_KINDS` 命中带 `"info"`):
339
- 问题面命中不带(问题即问题,无等级可降);本键让下游一眼分清「这是观测
340
- 副产品」与「这是待修的问题」,不必靠 issue_kind 名字去猜。
341
- """
342
- return {"node_id": node_id, "field": field, "span": span, "issue_kind": kind,
343
- "evidence": evidence, "rule": rule, "line": line, "sentence": sentence,
344
- "snippet": snippet, "peer": peer, "cause": cause, "status": status,
345
- "severity": severity}
346
-
347
-
348
- # 生效条件:text 为真值、span 与 content 均非 None 且 text.find(content)>=0 时,返回 text.count("\n",0,min(off+span[0],len(text)))+1 的 1-based 行号;text 假值或 span/content 为 None 或 content 未找到时返回 None。
349
- def _line_of(text, content, span):
350
- """正文区间 → 节点文件原文的 1-based 行号(供人工核对直接跳文件)。"""
351
- if not text or span is None or content is None:
352
- return None
353
- off = text.find(content)
354
- if off < 0:
355
- return None
356
- return text.count("\n", 0, min(off + span[0], len(text))) + 1
357
-
358
-
359
- # ---------------------------- 定位器注册表 ----------------------------
360
- #
361
- # 每个定位器是**纯函数**:只吃 (node_id, meta, content, ctx),只吐 hits。
362
- # 新增判据 = 加函数 + 注册,不改 locate 主流程(与 M1「引擎冻结、规则可增删」同构)。
363
- # ctx = {"fm", "text", "root", "index", "peers", "now"}
364
-
365
- # 生效条件:在 node_id/meta/content/ctx 下,对 FM_SCAN_FIELDS 中除 verification_basis、condition_space 外且 _layer_ok(f,scope,meta) 为真的字段,若 meta.get(f) 与 ctx.get("fm") 或 {} 中的同名字段皆 _blank 则记 field_absent;对过 _layer_ok 且 _as_int(meta.get("evidence_count"))==0 记 evidence_zero;对 meta.get("importance") 非 None 且 _as_float 为 None 或不在 [0.0,1.0] 记 field_invalid;对 condition_space 经 meta 或回落 ctx.get("fm") 后 NF.condition_space_missing 非空记 condition_slots;对正文缺 NF.CCG_MARKS 行记 ccg_incomplete,返回这些命中列表。
366
- def _loc_missing_field(node_id, meta, content, ctx):
367
- """frontmatter/正文结构字段为空(判据与 M1 `field_absent`/`evidence_zero` 同源)。
368
-
369
- **层门限**:限层的字段(如 `role`/`evidence_count` 限 `knowledge`)只在
370
- 该层节点上检查——与 M1 `_scope` 同口径,否则 M1 不报的问题会被 M3 越层报出。
371
- """
372
- fm = ctx.get("fm") or {}
373
- scope = ctx.get("field_layers")
374
- if scope is None:
375
- scope = field_layer_scope()
376
- lines = key_line_spans(ctx.get("text") or "")
377
- marks = mark_spans(content or "")
378
- hits = []
379
-
380
- for f in FM_SCAN_FIELDS:
381
- if f in ("verification_basis", "condition_space"):
382
- continue # 基底空属 weak_source;四槽缺失单独报(要点名缺哪几槽)
383
- if not _layer_ok(f, scope, meta):
384
- continue # 越层不报(层门限真源 = M1 规则库 matcher.layer)
385
- if _blank(meta.get(f)) and _blank(fm.get(f)):
386
- hits.append(_hit(node_id, "missing_field", f, None,
387
- "字段 %s 为空(索引与文件两处皆空)" % f,
388
- rule="field_absent",
389
- line=(lines.get(f) or (None, None))[0]))
390
-
391
- if _layer_ok("evidence_count", scope, meta) \
392
- and _as_int(meta.get("evidence_count")) == 0:
393
- hits.append(_hit(node_id, "missing_field", "evidence_count", None,
394
- "evidence_count=0(无验证证据计数)", rule="evidence_zero",
395
- line=(lines.get("evidence_count") or (None, None))[0]))
396
-
397
- imp = meta.get("importance")
398
- if imp is not None:
399
- fv = _as_float(imp)
400
- if fv is None or not 0.0 <= fv <= 1.0:
401
- hits.append(_hit(node_id, "missing_field", "importance", None,
402
- "importance=%r 不在 [0,1](字段存在但不可用)"
403
- % (imp,), rule="field_invalid",
404
- line=(lines.get("importance") or (None, None))[0]))
405
-
406
- cs = meta.get("condition_space")
407
- if not isinstance(cs, dict):
408
- cs = fm.get("condition_space")
409
- miss = NF.condition_space_missing(cs)
410
- if miss:
411
- span = marks.get("生效条件")
412
- hits.append(_hit(node_id, "missing_field", "condition_space", span,
413
- "条件空间缺槽 %s(%d/4 已声明)——四槽不全不构成生效条件"
414
- % ("/".join(miss), len(NF.CONDITION_SLOTS) - len(miss)),
415
- rule="condition_slots",
416
- snippet=_snippet(content, span)))
417
-
418
- # 正文 CCG 要素缺行:行不存在 → 无字符区间可指(span=None),点名缺哪几行
419
- missing_marks = [m for m in NF.CCG_MARKS if m not in marks]
420
- if missing_marks:
421
- hits.append(_hit(node_id, "missing_field", "content", None,
422
- "正文缺 CCG 要素行 %s(# 要素名:值)——缺声明即缺证据"
423
- % "/".join(missing_marks), rule="ccg_incomplete"))
424
- return hits
425
-
426
-
427
- # 生效条件:meta.get("verification_basis") 为空时回落 ctx.get("fm") 或 {} 的 verification_basis,若两者皆 _blank 返回 basis_absent 命中;非空但 str(basis) 不在 NF.VERIFICATION_BASIS 返回 basis_enum 命中;在枚举内时按 CC.classify_track 依 meta.get("layer")、meta.get("tags") 与 content 判赛道,若 CC.basis_licensed 为假返回 basis_licensed 命中;否则返回 []。
428
- def _loc_weak_source(node_id, meta, content, ctx):
429
- """验证基底缺失/越枚举/与赛道不相容(与 M1 `basis_licensed` 判据逐字同源)。"""
430
- fm = ctx.get("fm") or {}
431
- basis = meta.get("verification_basis")
432
- if _blank(basis):
433
- basis = fm.get("verification_basis")
434
- _ln = key_line_spans(ctx.get("text") or "").get("verification_basis")
435
- ln = (_ln or (None, None))[0]
436
-
437
- if _blank(basis):
438
- return [_hit(node_id, "weak_source", "verification_basis", None,
439
- "verification_basis 为空——无验证基底的断言不可裁决",
440
- rule="basis_absent", line=ln)]
441
- if str(basis) not in NF.VERIFICATION_BASIS:
442
- return [_hit(node_id, "weak_source", "verification_basis", None,
443
- "基底 %s 不在合法枚举(%s)"
444
- % (basis, "/".join(NF.VERIFICATION_BASIS)),
445
- rule="basis_enum", line=ln)]
446
- track = CC.classify_track({"layer": meta.get("layer"),
447
- "tags": meta.get("tags") or []}, content or "")
448
- if not CC.basis_licensed(track, basis):
449
- allow = "/".join(CC.allowed_basis(track)) or "—"
450
- return [_hit(node_id, "weak_source", "verification_basis", None,
451
- "赛道 %s(政策 %s)不允许基底 %s;允许 %s"
452
- % (track, CC.source_policy(track) or "—", basis, allow),
453
- rule="basis_licensed", line=ln)]
454
- return []
455
-
456
-
457
- # 生效条件:ctx.get("fm") 或 {} 中 code_ref 或 doc_ref 为非空 dict 且 RI.probe_ref(ref) 返回的 status 属于 _REF_DEAD_STATUSES(dangling 或 stale)时,返回对应 stale 命中(dangling 归因载体消失、stale 归因区间哈希不符);否则返回 []。
458
- def _loc_stale(node_id, meta, content, ctx):
459
- """**依赖存在性**——声明的载体(源文件)已不存在 / 已漂移(真正的适用边界越出)。
460
-
461
- 知识的真值挂在它描述的对象上:代码知识挂在源文件的符号上,源没了或改了,
462
- 知识才真的不再成立。判据与 `refindex` **同源**(`probe_ref`,不另立一份):
463
-
464
- ============== ====================================== ==========
465
- probe_ref 态 含义 本判据
466
- ============== ====================================== ==========
467
- ``ok`` 源文件在、区间哈希匹配 不判
468
- ``stale`` 源文件在、区间哈希不符(源被改,漂移) **stale**
469
- ``dangling`` 源文件不存在(载体消失) **stale**
470
- ``unresolved`` 拿不到 root(判不了) 不判(不猜)
471
- ``error`` 读盘失败 不判(不猜)
472
- ============== ====================================== ==========
473
-
474
- 后两态是**观测手段不足**,不是「依赖已失效」——按「不猜」纪律如实不报。
475
-
476
- **时间窗不在此处判**(2026-09-16 修正的原缺陷):`condition_space.time_window`
477
- 是写入端**观测时刻**栏(`mdcg.add` 未给时自动填「写入时刻 +1h」,见其
478
- `OBSERVATION_WINDOW_SEC`),不是知识的有效期——拿它判「适用边界越出」,
479
- 等于把「这条记忆写入超过 1 小时」冒充为「失效」。真实适用前提由知识自己
480
- 声明在 `code_ref`/`doc_ref` 上,故改由依赖存在性裁决;观测时刻本身不丢,
481
- 降级为 `observation_aged`(观测面,见 `ADVISORY_KINDS`)。
482
- """
483
- fm = ctx.get("fm") or {}
484
- hits = []
485
- for key in ("code_ref", "doc_ref"):
486
- ref = fm.get(key)
487
- if not isinstance(ref, dict) or not ref:
488
- continue
489
- # root 基准必须取自 ref 自身(`_code_ref`/`_doc_ref` 落的是**源大域** root,
490
- # path 相对它);`ctx["root"]` 是**认知图** root——拿它去拼会把每一条依赖
491
- # 都误判成悬空。ref 无 root(旧节点)时 probe 返回 `unresolved` → 不判。
492
- probe = RI.probe_ref(ref)
493
- st = str(probe.get("status") or "")
494
- if st not in _REF_DEAD_STATUSES:
495
- continue
496
- rel = str(ref.get("path") or "?")
497
- if st == "dangling":
498
- ev = ("依赖的源文件已不存在(%s:%s)——知识所指的载体消失,适用边界越出"
499
- % (key, rel))
500
- else:
501
- ev = ("依赖的源文件已漂移(%s:%s):索引区间哈希 %s ≠ 现算 %s——"
502
- "源被改动,知识所指的符号可能已不是原物"
503
- % (key, rel, probe.get("hash_expected"), probe.get("hash")))
504
- hits.append(_hit(node_id, "stale", key, None, ev, rule="ref_" + st))
505
- return hits
506
-
507
-
508
- # 生效条件:从 meta.get("condition_space") 或回落 ctx.get("fm") 的 condition_space(须为 dict)取 time_window,缺则取 meta.get("time_window"),若 NF.is_full_time_window(tw) 或 float(tw[1]) 抛 TypeError/ValueError/IndexError/KeyError 或 ctx.get("now") 为 None 或 hi>=float(now) 则返回 [];否则返回一条 severity="info" 的 observation_aged 命中。
509
- def _loc_observation_aged(node_id, meta, content, ctx):
510
- """时间窗已过——**这是观测时刻,不是失效声明**(观测面,不进告警面)。
511
-
512
- 时间窗来源链与 `_loc_missing_field` 的槽检查同构:`meta.condition_space`
513
- (显式注入面)→ `fm.condition_space`(文件真源)→ `meta.time_window`
514
- (索引快照字段)。**索引快照只透传 `time_window`**(`mdcg.py` `_stage` 口径:
515
- 时空字段入快照、condition_space 整块不入),故第三跳必留。
516
-
517
- **不进默认全量**(不在 `ISSUE_KINDS`):写入端未给 time_window 时以写入时刻
518
- 自动填 1 小时窗,故真库「已过期」绝大多数是「写入超过 1 小时」,与知识是否
519
- 失效无关。观测时刻本身仍如实报(审计有用),只是**不冒充问题**——命中带
520
- `severity="info"` 标记它是咨询级。
521
- """
522
- fm = ctx.get("fm") or {}
523
- cs = meta.get("condition_space")
524
- if not isinstance(cs, dict):
525
- csf = fm.get("condition_space")
526
- cs = csf if isinstance(csf, dict) else None
527
- tw = (cs or {}).get("time_window")
528
- if tw is None:
529
- tw = meta.get("time_window")
530
- if NF.is_full_time_window(tw):
531
- return [] # 全时窗是**合法声明**(任意时刻成立),不判
532
- try:
533
- hi = float(tw[1])
534
- except (TypeError, ValueError, IndexError, KeyError):
535
- return []
536
- now = ctx.get("now")
537
- if now is None or hi >= float(now):
538
- return []
539
- span = mark_spans(content or "").get("生效条件")
540
- return [_hit(node_id, "observation_aged", "condition_space", span,
541
- "观测时间窗 %s 已过(hi=%.0f < now=%.0f)——这是**观测时刻**,"
542
- "不是失效声明(知识不因观测过去而失效);时效判定见 stale"
543
- % (NF.time_window_text(tw) or "?", hi, float(now)),
544
- rule="observation_window_passed", severity="info",
545
- line=_line_of(ctx.get("text"), content, span),
546
- snippet=_snippet(content, span))]
547
-
548
-
549
- # 生效条件:h 取 str(meta.get("content_hash") or "").strip(),若 _blank(h) 则 h=NF.content_hash(content 或 "");对 ctx.get("peers") 或 [] 中 node_id 不等于本节点且 ph=str(p.get("hash") or "") 或回落 NF.content_hash(p.get("content") or "") 后非空且等于 h 的 peer 计入 matched;matched 非空时返回一条 dup 命中(span=[0,len(body)]),否则返回 []。
550
- def _loc_dup(node_id, meta, content, ctx):
551
- """同组同内容指纹(与 M1 `dup_hash_group` 同判据;D1 名 `dup`)。"""
552
- body = content if content is not None else ""
553
- h = str(meta.get("content_hash") or "").strip()
554
- if _blank(h):
555
- h = NF.content_hash(body) # 索引无指纹 → 用正文实算(诚实标注来源)
556
- matched = []
557
- for p in ctx.get("peers") or []:
558
- if p.get("node_id") == node_id:
559
- continue
560
- ph = str(p.get("hash") or "") or NF.content_hash(p.get("content") or "")
561
- if ph and ph == h:
562
- matched.append(p)
563
- if not matched:
564
- return []
565
- span = [0, len(body)]
566
- return [_hit(node_id, "dup", "content_hash", span,
567
- "与 %s 同内容指纹 %s(同类共 %d 条),整篇正文重复"
568
- % (matched[0].get("node_id"), h, len(matched) + 1),
569
- rule="dup_hash_group", peer=matched[0].get("node_id"),
570
- snippet=_snippet(body, span))]
571
-
572
-
573
- # 生效条件:对 ctx.get("peers") 或 [] 中每个非自身 peer,按 sentence_spans 对齐 content 与 peer content 的句子,当同一句位上去数字骨架相同(WL._skeleton(seg) 非空、长度 ≥ WL.MIN_SKELETON 且等于 peer 骨架)且 seg.strip() != ptext.strip() 的句子数达到 MIN_FLOW_SENTENCES 时,为这些句各返回一条 template_flow 命中;否则返回 []。
574
- def _loc_template_flow(node_id, meta, content, ctx):
575
- """同模板流水:与同组节点逐句「去数字骨架相同、字面不同」→ 指向那些句。
576
-
577
- 与 M1 `template_flow_digits_only` 同判据(都走 `writelimit._skeleton`),
578
- 差别只在产物:M1 给一条条目级 issue,D1 给出**具体是哪几句**(span + 句索引)。
579
- 单句偶合不算流水 → 需同组至少 `MIN_FLOW_SENTENCES` 句同时成立。
580
- """
581
- tgt = sentence_spans(content or "")
582
- hits = []
583
- for p in ctx.get("peers") or []:
584
- if p.get("node_id") == node_id:
585
- continue
586
- pseg = sentence_spans(p.get("content") or "")
587
- same = []
588
- for (i, s, e, seg) in tgt:
589
- if i >= len(pseg):
590
- break
591
- ptext = pseg[i][3]
592
- sk = WL._skeleton(seg)
593
- if not sk or len(sk) < WL.MIN_SKELETON or sk != WL._skeleton(ptext):
594
- continue
595
- if seg.strip() == ptext.strip():
596
- continue # 字面全同属整篇重复(dup),不是「只换数字」的流水
597
- same.append((i, s, e, seg, sk))
598
- if len(same) < MIN_FLOW_SENTENCES:
599
- continue
600
- for (i, s, e, seg, sk) in same:
601
- hits.append(_hit(node_id, "template_flow", "content", [s, e],
602
- "同模板流水(与 %s 第 %d 句去数字骨架相同、仅字面数值不同):%s"
603
- % (p.get("node_id"), i, sk[:40]),
604
- rule="template_flow_digits_only", sentence=i,
605
- peer=p.get("node_id"),
606
- line=_line_of(ctx.get("text"), content, [s, e]),
607
- snippet=seg.strip()[:SNIPPET_MAX]))
608
- return hits
609
-
610
-
611
- # 生效条件:meta.get("content_hash") 非 _blank 且 content 非 None 且 h != NF.content_hash(content) 时,返回一条 hash_declared_vs_actual 的 contradiction 命中(cause 由 hash_mismatch_cause 依 meta/ctx.get("root")/ctx.get("path") 判);ctx.get("fm") 或 {} 的 id 非 _blank 且 str(fm_id) != str(node_id) 时额外返回一条 id_declared_vs_index 命中;两者皆不成立返回 []。
612
- def _loc_contradiction(node_id, meta, content, ctx):
613
- """**确定性**矛盾:声明与事实不符(指纹 / 标识)。
614
-
615
- 语义级矛盾(正文自相矛盾)不在此处——那是 LLM 的活,D1 不猜(见 `SEMANTIC_ONLY`)。
616
- 指纹不一致**必须带成因**(`cause`):`index_lag`(快照滞后,属正常写路径现象,
617
- 复核者无需修数据)与 `true_mismatch`(真不一致,需查)处置完全不同。
618
- """
619
- hits = []
620
- lines = key_line_spans(ctx.get("text") or "")
621
- h = str(meta.get("content_hash") or "").strip()
622
- if not _blank(h) and content is not None:
623
- actual = NF.content_hash(content)
624
- if h != actual:
625
- cause, note = hash_mismatch_cause(meta, root=ctx.get("root"),
626
- path=ctx.get("path"))
627
- hits.append(_hit(node_id, "contradiction", "content_hash", None,
628
- "索引声明指纹 %s ≠ 正文实算 %s(%s)"
629
- % (h, actual, note), rule="hash_declared_vs_actual",
630
- cause=cause,
631
- line=(lines.get("content_hash") or (None, None))[0]))
632
- fm_id = (ctx.get("fm") or {}).get("id")
633
- if not _blank(fm_id) and str(fm_id) != str(node_id):
634
- hits.append(_hit(node_id, "contradiction", "id", None,
635
- "文件 frontmatter id=%s ≠ 索引键 %s" % (fm_id, node_id),
636
- rule="id_declared_vs_index",
637
- line=(lines.get("id") or (None, None))[0]))
638
- return hits
639
-
640
-
641
- LOCATORS = {
642
- "missing_field": _loc_missing_field,
643
- "weak_source": _loc_weak_source,
644
- "stale": _loc_stale,
645
- "observation_aged": _loc_observation_aged,
646
- "dup": _loc_dup,
647
- "template_flow": _loc_template_flow,
648
- "contradiction": _loc_contradiction,
649
- }
650
-
651
- #: D1 **不定位**的判定:属语义层,交回 LLM 意见(不猜 = 不编造字符区间)
652
- SEMANTIC_ONLY = {
653
- "contradiction_semantic": "正文自相矛盾属语义判定;D1 只做确定性矛盾"
654
- "(索引指纹 / 标识与事实不符)",
655
- }
656
-
657
-
658
- # 生效条件:始终返回 SEMANTIC_ONLY 中以 str(kind or "").strip() 为键查得的值,缺键时返回空串(空串表示可定位)。
659
- def blindspot_reason(kind) -> str:
660
- """D1 拒绝定位的类别 → 理由(空串表示可定位)。"""
661
- return SEMANTIC_ONLY.get(str(kind or "").strip(), "")
662
-
663
-
664
- # ---------------------------- 主入口 ----------------------------
665
-
666
- # 生效条件:items 中每项按 content=it.get("content") or ""、hash=str(it.get("hash") or "").strip() 或回落 NF.content_hash(content)、sk=WL.template_signature(content) or "" 预处理后,返回 {node_id: [同 hash 或同非空 sk 的其他 peer]};items 为空/None 返回空 dict。
667
- def build_peers(items) -> dict:
668
- """`[{node_id, content, hash?, …}]` → `{node_id: [peer, …]}`。
669
-
670
- 只有**同内容指纹**或**同模板骨架**才互为对照——不是「同一批就是一组」。
671
- 组的作用域由调用方给定(批 / 包),与 M1 机械层「组 = 同一个包」的口径一致。
672
- """
673
- prep = []
674
- for it in items or []:
675
- c = it.get("content") or ""
676
- prep.append({"node_id": it.get("node_id"), "content": c, "meta": it.get("meta"),
677
- "text": it.get("text"), "fm": it.get("fm"),
678
- "hash": str(it.get("hash") or "").strip() or NF.content_hash(c),
679
- "sk": WL.template_signature(c) or ""})
680
- by_hash, by_sk = {}, {}
681
- for p in prep:
682
- by_hash.setdefault(p["hash"], []).append(p)
683
- if p["sk"]:
684
- by_sk.setdefault(p["sk"], []).append(p)
685
- out = {}
686
- for p in prep:
687
- grp = {q["node_id"]: q for q in by_hash.get(p["hash"], [])
688
- if q["node_id"] != p["node_id"]}
689
- for q in (by_sk.get(p["sk"], []) if p["sk"] else []):
690
- if q["node_id"] != p["node_id"]:
691
- grp.setdefault(q["node_id"], q)
692
- out[p["node_id"]] = list(grp.values())
693
- return out
694
-
695
-
696
- # 生效条件:对 issue_hint(None/str/dict 或其 list/tuple)逐 hint 分类,返回 (kinds, fields, blind):能 canonical_kind 到 ISSUE_KINDS 或 ADVISORY_KINDS 的入 kinds,blindspot_reason 非空的入 blind,其余非空 kind 与 field 入 fields;issue_hint 为 None 时 kinds/fields 为空集、blind 为空列表。
697
- def _norm_hint(issue_hint):
698
- """issue_hint → `(kinds, fields, blindspot)`。
699
-
700
- 形态:`None`(全量)|str(issue_kind 或 field)|dict(issue_kind/field)
701
- |上述的 list。
702
- """
703
- kinds, fields, blind = set(), set(), []
704
- hints = issue_hint if isinstance(issue_hint, (list, tuple)) else [issue_hint]
705
- for h in hints:
706
- if h is None:
707
- continue
708
- if isinstance(h, dict):
709
- s = h.get("issue_kind") or h.get("kind") or ""
710
- f = h.get("field")
711
- else:
712
- s, f = str(h).strip(), None
713
- if s:
714
- why = blindspot_reason(s)
715
- if why:
716
- blind.append("%s:%s" % (s, why))
717
- else:
718
- ck = canonical_kind(s)
719
- if ck in ISSUE_KINDS or ck in ADVISORY_KINDS:
720
- kinds.add(ck)
721
- else:
722
- fields.add(s) # 不是类别 → 当作字段过滤
723
- if f:
724
- fields.add(str(f).strip())
725
- return kinds, fields, blind
726
-
727
-
728
- # 生效条件:当 meta 与 content 均非 None 时不读盘;否则 root 为 None 抛 ValueError,经 load_node 读不到节点返回 {"hits": [], "blindspot": blind+["节点 %s 不在索引"%node_id], "load": None};读到时补 meta/content/text/fm/path,按 kinds 非空时只跑 sorted(kinds)、否则 fields 非空或 issue_hint 为 None 时跑 sorted(ISSUE_KINDS)、否则 run=[] 执行 LOCATORS,对命中按 fields 过滤、补 line 并排序后返回 {"hits": hits, "blindspot": blind, "load": {...}}。
729
- def locate_ex(node_id, issue_hint=None, *, meta=None, content=None, fm=None, text=None,
730
- root=None, index=None, peers=None, now=None, path=None,
731
- field_layers=None) -> dict:
732
- """D1 定位(完整返回):`{"hits": [...], "blindspot": [...], "load": {...}}`。
733
-
734
- `meta`/`content` 显式给出则**不读盘**(纯函数用法——测试与批量复用都走这条);
735
- 否则 `root` 必填,经 `load_node` 从索引+文件取。返回的 hits 已按
736
- `(issue_kind, field, span起点, peer)` 排序,**确定性可复算**。
737
-
738
- `path`(节点文件绝对路径——供 `contradiction` 判成因)、`field_layers`
739
- (字段层门限,缺省 `field_layer_scope()`)可显式注入,便于纯函数测试。
740
- """
741
- kinds, fields, blind = _norm_hint(issue_hint)
742
- if meta is None or content is None:
743
- if root is None:
744
- raise ValueError("locate 需要 meta+content,或给出 root 以便读节点")
745
- nd = load_node(node_id, root, index=index)
746
- if nd is None:
747
- return {"hits": [], "blindspot": blind + ["节点 %s 不在索引" % node_id],
748
- "load": None}
749
- if meta is None:
750
- meta = nd["meta"]
751
- if content is None:
752
- content = nd["content"]
753
- if text is None:
754
- text = nd["text"]
755
- if fm is None:
756
- fm = nd["fm"]
757
- if path is None:
758
- path = nd.get("path")
759
- meta = meta or {}
760
- ctx = {"fm": fm or {}, "text": text, "root": root, "index": index,
761
- "peers": peers or [], "now": time.time() if now is None else float(now),
762
- "path": path, "field_layers": field_layers}
763
- hits = []
764
- # 定位范围:显式类别 → 只跑该类;给出字段 → 全量再由字段收窄;
765
- # 无提示 → 全量。**提示全部落 blindspot 时范围为空**——不借机报别的类
766
- # (否则「给了语义提示」会顺手退回全量,与「不猜」纪律相悖)。
767
- if kinds:
768
- run = sorted(kinds)
769
- elif fields or issue_hint is None:
770
- run = sorted(ISSUE_KINDS)
771
- else:
772
- run = []
773
- for k in run:
774
- fn = LOCATORS.get(k)
775
- if fn is None:
776
- continue
777
- for h in fn(node_id, meta, content, ctx):
778
- if fields and h.get("field") not in fields:
779
- continue
780
- if h.get("line") is None and h.get("span") is not None:
781
- h["line"] = _line_of(text, content, h["span"])
782
- hits.append(h)
783
- hits.sort(key=lambda h: (h["issue_kind"], str(h.get("field") or ""),
784
- -1 if h.get("span") is None else h["span"][0],
785
- str(h.get("peer") or "")))
786
- return {"hits": hits, "blindspot": blind,
787
- "load": {"node_id": node_id, "meta": meta,
788
- "content_len": len(content or "")}}
789
-
790
-
791
- # 生效条件:给定 node_id(必填)与可选 issue_hint 及 kw 后,直接返回 locate_ex(node_id, issue_hint, **kw) 结果的 "hits" 列表。
792
- def locate(node_id, issue_hint=None, **kw) -> list:
793
- """**D1 契约入口**:`locate(node_id, issue_hint) -> [{field, span, issue_kind, evidence}]`。"""
794
- return locate_ex(node_id, issue_hint, **kw)["hits"]
795
-
796
-
797
- # 生效条件:给定 hits 与 key 后,返回 hits 中每个 h.get(key) 字符串化取值到出现次数的字典(键升序);hits 为 None/空时返回空 dict。
798
- def _counts(hits, key) -> dict:
799
- out = {}
800
- for h in hits or []:
801
- k = h.get(key)
802
- out[str(k)] = out.get(str(k), 0) + 1
803
- return dict(sorted(out.items()))
804
-
805
-
806
- # 生效条件:items 显式给出时不读盘;items 为 None 时 root 为 None 抛 ValueError,否则按 node_ids 逐节点 load_node 组装 items(读不到的记入 missing);随后对每个 item 调 locate_ex 汇总 hits 或 clean,并返回含 nodes/hits/by_kind/by_field/by_rule/clean/missing 的 dict。
807
- def locate_many(node_ids=None, *, root=None, index=None, items=None,
808
- issue_hint=None, now=None) -> dict:
809
- """批量定位:**同一批内**互为对照(`dup` / `template_flow` 的组 = 本批)。
810
-
811
- `items` 显式给出 `[{node_id, content, meta?, text?, fm?, hash?}]` 则不再读盘。
812
- 返回 `{nodes, hits, by_kind, by_field, clean, missing}`;`clean` 是**无命中**的节点、
813
- `missing` 是**读不到**的节点——两者分开报(「没问题」≠「没看」)。
814
- """
815
- missing = []
816
- if items is None:
817
- if root is None:
818
- raise ValueError("locate_many 需要 root(读盘)或 items(显式给出)")
819
- items = []
820
- for nid in node_ids or []:
821
- nd = load_node(nid, root, index=index)
822
- if nd is None:
823
- missing.append(nid)
824
- continue
825
- items.append({"node_id": nid, "content": nd["content"] or "",
826
- "text": nd["text"], "fm": nd["fm"], "meta": nd["meta"],
827
- "path": nd.get("path"),
828
- "hash": nd["meta"].get("content_hash")})
829
- peers = build_peers(items)
830
- now = time.time() if now is None else float(now)
831
- hits, clean = [], []
832
- for it in items:
833
- nid = it["node_id"]
834
- ex = locate_ex(nid, issue_hint, meta=it.get("meta") or {},
835
- content=it.get("content") or "", fm=it.get("fm"),
836
- text=it.get("text"), peers=peers.get(nid) or [], now=now,
837
- root=root, path=it.get("path"))
838
- if ex["hits"]:
839
- hits.extend(ex["hits"])
840
- else:
841
- clean.append(nid)
842
- hits.sort(key=lambda h: (str(h.get("node_id")), h["issue_kind"],
843
- str(h.get("field") or ""),
844
- -1 if h.get("span") is None else h["span"][0]))
845
- return {"nodes": len(items), "hits": hits, "by_kind": _counts(hits, "issue_kind"),
846
- "by_field": _counts(hits, "field"), "by_rule": _counts(hits, "rule"),
847
- "clean": clean, "missing": missing}
848
-
849
-
850
- # 生效条件:从 pkg.get("entries") or [] 取条目并跳过 e.get("node_id") 为空者,root 非 None 时逐个 load_node 取正文,取不到时回落 e.get("excerpt") or "",组装 items 调 locate_many 后返回其结果并附加 bundle_id 与 entries 数量;pkg 为 None 时 entries 为空列表。
851
- def locate_package(pkg, *, root=None, index=None, issue_hint=None, now=None) -> dict:
852
- """M1 包 → 定位汇总(组的作用域 = 本包,与 M1 机械层同口径)。
853
-
854
- 正文优先读盘(准确);文件不可读时退回包内 `excerpt`(可能截断——宁可降级也不
855
- 假装读到了全文,`missing` 会如实记下读不到的节点)。
856
- """
857
- entries = (pkg or {}).get("entries") or []
858
- items = []
859
- for e in entries:
860
- nid = e.get("node_id")
861
- if not nid:
862
- continue
863
- nd = load_node(nid, root, index=index) if root else None
864
- body = (nd or {}).get("content")
865
- items.append({"node_id": nid,
866
- "content": body if body is not None else (e.get("excerpt") or ""),
867
- "text": (nd or {}).get("text"), "fm": (nd or {}).get("fm"),
868
- "meta": (nd or {}).get("meta") or e,
869
- "path": (nd or {}).get("path") or e.get("path"),
870
- "hash": ((nd or {}).get("meta") or {}).get("content_hash")
871
- or e.get("content_hash")})
872
- out = locate_many(items=items, issue_hint=issue_hint, now=now)
873
- out["bundle_id"] = (pkg or {}).get("bundle_id")
874
- out["entries"] = len(entries)
875
- return out
876
-
877
-
878
- # 生效条件:给定 hits 后返回 total=len(hits or [])、去重 node_id 数、排序后的 node_ids、以及 by_kind/by_field/by_rule 计数;hits 为 None/空时 total=0、nodes=0、node_ids=[]。
879
- def summary(hits) -> dict:
880
- """命中汇总(审计留痕用)。"""
881
- nodes = sorted({str(h.get("node_id")) for h in hits or []})
882
- return {"total": len(hits or []), "nodes": len(nodes), "node_ids": nodes,
883
- "by_kind": _counts(hits, "issue_kind"), "by_field": _counts(hits, "field"),
884
- "by_rule": _counts(hits, "rule")}
885
-
886
-
887
- # 生效条件:给定 hits 后逐条生成 Markdown 行并返回表头加各行:field/line/snippet 取 h.get(...) or "—"(假值回落 "—"),span 为 None 时显示 "—" 否则 "起-止",evidence 取 (h.get("evidence") or "").replace("|","\\|")(假值回落空串)。
888
- def markdown_table(hits) -> str:
889
- """人工核对清单:每行一条命中,末列留空供核对者填判定(D1 验收抽样用)。"""
890
- head = ("| # | node_id | issue_kind | field | span | line | 片段 | evidence | 人工判定 |\n"
891
- "|---|---|---|---|---|---|---|---|---|\n")
892
- rows = []
893
- for i, h in enumerate(hits or [], 1):
894
- sp = "—" if h.get("span") is None else "%d-%d" % (h["span"][0], h["span"][1])
895
- rows.append("| %d | %s | %s | %s | %s | %s | %s | %s | |"
896
- % (i, h.get("node_id"), h.get("issue_kind"), h.get("field") or "—",
897
- sp, h.get("line") or "—",
898
- (h.get("snippet") or "—").replace("|", "\\|"),
899
- (h.get("evidence") or "").replace("|", "\\|")))
900
- return head + "\n".join(rows)
901
-
902
-
903
- # 生效条件:解析 argv(缺省 sys.argv)后,--root 为假值(含默认 os.environ.get("MDCG_ROOT") 为 None 或空串)时打印提示并返回 2;--node 追加列表为空时打印提示并返回 2;否则以 --root 与 --node 调 locate_many,并按 --json 或 --markdown 输出后返回 0,两者皆无则逐行打印命中与汇总后返回 0。
904
- def main(argv=None) -> int:
905
- ap = argparse.ArgumentParser(prog="python -m md_cg.mreview.locate",
906
- description="记忆评审 M3 · D1 字段级定位")
907
- ap.add_argument("--root", default=os.environ.get("MDCG_ROOT"),
908
- help="真源根(缺省读环境变量 MDCG_ROOT)")
909
- ap.add_argument("--node", action="append", default=[],
910
- help="节点 id(可多次;同一批内互为对照)")
911
- ap.add_argument("--kind", action="append", default=[], help="只定位该类问题")
912
- ap.add_argument("--field", action="append", default=[], help="只保留该字段的命中")
913
- ap.add_argument("--json", action="store_true", help="输出 JSON(缺省人读表)")
914
- ap.add_argument("--markdown", action="store_true", help="输出人工核对清单")
915
- a = ap.parse_args(argv)
916
- if not a.root:
917
- print("需要 --root 或环境变量 MDCG_ROOT", file=sys.stderr)
918
- return 2
919
- if not a.node:
920
- print("需要 --node <id>(可多次)", file=sys.stderr)
921
- return 2
922
- rep = locate_many(a.node, root=a.root,
923
- issue_hint=(list(a.kind) + list(a.field)) or None)
924
- if a.json:
925
- print(json.dumps(rep, ensure_ascii=False, indent=2))
926
- return 0
927
- if a.markdown:
928
- print(markdown_table(rep["hits"]))
929
- return 0
930
- for h in rep["hits"]:
931
- print("%-14s %-20s %-14s %s" % (h["node_id"], h["field"], h["issue_kind"],
932
- h["evidence"]))
933
- print("-- 命中 %d 条|干净 %d 条|缺项 %d 条"
934
- % (len(rep["hits"]), len(rep["clean"]), len(rep["missing"])))
935
- return 0
936
-
937
-
938
- if __name__ == "__main__":
939
- sys.exit(main())
1
+ # -*- coding: utf-8 -*-
2
+ """记忆评审流水线 · M3 定位模块(D1 字段级定位,确定性、零写入)。
3
+
4
+ 真源:docs/记忆评审系统_立项设计与施工交接_20260915.md §5 D1。
5
+
6
+ D1 契约:
7
+
8
+ locate(node_id, issue_hint) -> [{"field", "span", "issue_kind", "evidence"}, ...]
9
+
10
+ * **field** —— frontmatter 字段名,或 `"content"`(正文)
11
+ * **span** —— 命中处的字符区间 `[start, end)`(半开),**相对正文 content**;
12
+ 纯字段级缺失无字符区间可指 → `None`
13
+ * **issue_kind** —— dup / missing_field / stale / contradiction / weak_source / template_flow
14
+ (`ISSUE_KINDS`=问题面);另有 `ADVISORY_KINDS`(观测面,可显式定位
15
+ 但**不进默认全量、不进评审告警面**,见下「stale 与 observation_aged」)
16
+ * **evidence** —— 白箱判据:为什么算问题(人可复核、机器可断言)
17
+
18
+ 定位的职责边界:M2 的意见说「这一条有问题」,D1 回答「问题在这一条的哪个字段/哪一段」。
19
+ 故本模块只做**确定性可判定**的事——语义级矛盾(正文自相矛盾)**不猜**,如实返回
20
+ BLINDSPOT 项交回 LLM 意见层(`status="blindspot"`,见 `locate()` 返回值)。
21
+
22
+ 口径全部引用既有唯一真源,**不在本模块另立一份**(防「术语双写法」漂移):
23
+
24
+ =============== ==================================================================
25
+ 字段/正文结构 ``nodefile``(CCG_MARKS / CONDITION_SLOTS / VERIFICATION_BASIS /
26
+ loads / content_hash / is_placeholder_text / time_window_text)
27
+ 模板骨架 ``writelimit._skeleton`` / ``MIN_SKELETON``
28
+ 赛道×基底相容 ``crosscheck``(与 M1 `ruleset._chk_basis_licensed` 同源)
29
+ 字段层门限 ``ruleset``(即 `rules/*.json` 的 `matcher.layer`,见
30
+ `field_layer_scope`——**派生**,不在此另立一份)
31
+ 索引快照 ``conformance.load_index``
32
+ =============== ==================================================================
33
+
34
+ 与 M1(`ruleset` 机械层)的用词归并:D1 说 `dup`,M1 的 `dup_hash_group` 说
35
+ `dup_content`——同一件事两种写法,由 `KIND_ALIASES` 单一归口(`canonical_kind`)。
36
+
37
+ 超出 D1 最小契约的**增强键**(供人工核对直接跳文件/跳行,不影响四键契约):
38
+ `line`(节点文件原文 1-based 行号)、`sentence`(正文句索引,0 基)、
39
+ `snippet`(命中片段)、`rule`(触发定位的判据名,审计用)、`peer`(同组对照节点)、
40
+ `cause`(`contradiction` 的**成因**判定,见 `hash_mismatch_cause`——同一条命中
41
+ 可能是「真不一致」也可能是「索引快照滞后」,两种成因不分会让复核者误判为数据损坏)。
42
+
43
+ **`stale` 与 `observation_aged` 的判据源分工(2026-09-16 修正)**:
44
+
45
+ * `stale`(**问题面**)—— **依赖存在性**:`code_ref`/`doc_ref` 指的源文件已不存在
46
+ (悬空)或已漂移(区间哈希不符)。判据直接调 `refindex.probe_ref`,**不另立一份**
47
+ (防「术语双写法」漂移)。代码知识的真值挂在源文件上,源没了/改了才是适用边界越出。
48
+ * `observation_aged`(**观测面**,`ADVISORY_KINDS`,**不进默认全量**)——
49
+ `condition_space.time_window` 已过。这是**观测时刻**不是失效声明:写入端
50
+ (`mdcg.add`,见其 `OBSERVATION_WINDOW_SEC`)在未给 time_window 时以**写入时刻**
51
+ 自动填 1 小时窗,故真库中「已过期」绝大多数是「写入超过 1 小时」,与知识是否失效
52
+ 无关——把它当问题投进评审告警面就是系统性误报(实测:全库 1453 条过期里 1434 条
53
+ 属默认窗填充)。需要时效判定用 `stale`(依赖存在性),不是这里。
54
+ """
55
+ from __future__ import annotations
56
+
57
+ import argparse
58
+ import json
59
+ import os
60
+ import re
61
+ import sys
62
+ import time
63
+
64
+ from .. import conformance as CF
65
+ from .. import crosscheck as CC
66
+ from .. import nodefile as NF
67
+ from .. import refindex as RI
68
+ from .. import writelimit as WL
69
+ from . import ruleset as RS
70
+
71
+ __all__ = ["ISSUE_KINDS", "ADVISORY_KINDS", "KIND_ALIASES", "canonical_kind", "locate",
72
+ "locate_many", "locate_package", "sentence_spans", "mark_spans", "load_node",
73
+ "main", "field_layer_scope", "hash_mismatch_cause", "HASH_CAUSES"]
74
+
75
+ #: D1 的问题类别(真源:立项文档 §5 D1)——**问题面**:命中即「知识有问题」
76
+ ISSUE_KINDS = ("dup", "missing_field", "stale", "contradiction",
77
+ "weak_source", "template_flow")
78
+
79
+ #: **观测面**(咨询级):可显式定位,但**不进默认全量定位、不进评审告警面**。
80
+ #: 判据:本类命中说的是「观测手段的副产品」(如「这条记忆写入超过 1 小时」),
81
+ #: 不是「知识有问题」——把它混进 ISSUE_KINDS 会让每一次写入在 1 小时后自动变成
82
+ #: 待评审问题(系统性误报)。故与问题面分表,只有显式点名才产出。
83
+ ADVISORY_KINDS = ("observation_aged",)
84
+
85
+ #: M1(ruleset 机械层)与 D1 的用词归并——两处说的是同一件事,不许各说各话
86
+ KIND_ALIASES = {"dup_content": "dup", # M1 `dup_hash_group`
87
+ "missing": "missing_field",
88
+ "flow": "template_flow",
89
+ "template_flow_digits_only": "template_flow"}
90
+
91
+ #: 扫描 frontmatter 的字段清单(真源字段名取自 nodefile 口径,不自由发明)
92
+ FM_SCAN_FIELDS = ("role", "tags", "importance", "evidence_count",
93
+ "verification_basis", "lifecycle_state", "condition_space")
94
+
95
+ #: 判为「同模板流水」所需的最少同骨架句数——单句偶合不算流水(防误报)
96
+ MIN_FLOW_SENTENCES = 2
97
+
98
+ #: 一句话摘要的最大长度
99
+ SNIPPET_MAX = 120
100
+
101
+ #: 索引快照与节点文件的 mtime 差**小于**此值 → 视为同一批写入,不判「快照滞后」
102
+ MTIME_TOLERANCE = 1.0
103
+
104
+ #: `content_hash` 不一致的**成因**(确定性判据,不猜;见 `hash_mismatch_cause`)
105
+ HASH_CAUSES = ("index_lag", "true_mismatch", "unknown")
106
+
107
+ #: `refindex.probe_ref` 的五态里,哪些构成「依赖已失效」(`stale` 判据,见 `_loc_stale`):
108
+ #: 只有这两态说明**载体真的没了/改了**;`unresolved`(拿不到 root)与 `error`(读盘失败)
109
+ #: 是观测手段不足,按「不猜」纪律不判。
110
+ _REF_DEAD_STATUSES = ("dangling", "stale")
111
+
112
+ #: 字段层门限缓存(规则库是包内数据文件,进程内不变 → 惰性算一次)
113
+ _FIELD_LAYER_CACHE = None
114
+
115
+
116
+ # 生效条件:kind 经 str(kind or "").strip() 得 k(None/空串等假值 → 空串 ""),k 命中 KIND_ALIASES 时返回其规范名,否则原样返回 k。
117
+ def canonical_kind(kind) -> str:
118
+ """M1/D1 用词 → D1 规范名(未知原样返回,不假装认路)。"""
119
+ k = str(kind or "").strip()
120
+ return KIND_ALIASES.get(k, k)
121
+
122
+
123
+ # 生效条件:仅当 rules 与 rules_dir 均为 None 且模块级 _FIELD_LAYER_CACHE 非 None 时直接返回该缓存;否则遍历 rules(dict 取其 "rules" 键的列表、非 dict 直接 list(rules))或 rules_dir 经 RS.load_rules 取得的 rules 列表,从限了 matcher.layer 的 mechanical 规则中收集 field_absent 的 spec["field"] 与 evidence_zero 的 evidence_count,返回 {字段: 排序列},且只在 rules 与 rules_dir 均为 None 时写回 _FIELD_LAYER_CACHE。
124
+ def field_layer_scope(rules=None, rules_dir=None) -> dict:
125
+ """字段 → 适用层清单(**派生**自 M1 规则库,不在此另立一份)。
126
+
127
+ 真源 = `md_cg/mreview/rules/*.json` 的 `matcher.layer`:只收录**限了层**的
128
+ 机械规则字段(`field_absent` 取 `spec["field"]`、`evidence_zero` 取
129
+ `evidence_count`);未限层的规则 = 全层适用,不入表。
130
+
131
+ 理由:字段范围两处各写一份 ⇒ 必然漂移(同「术语双写法」)。派生使 M1 改规则
132
+ 时 M3 自动跟随。规则库损坏 → `load_rules` 抛错(fail-closed,不静默降级成全层)。
133
+ """
134
+ global _FIELD_LAYER_CACHE
135
+ if rules is None and rules_dir is None and _FIELD_LAYER_CACHE is not None:
136
+ return _FIELD_LAYER_CACHE
137
+ if isinstance(rules, dict):
138
+ rl = list(rules.get("rules") or [])
139
+ elif rules is not None:
140
+ rl = list(rules)
141
+ else:
142
+ rl = list(RS.load_rules(rules_dir)["rules"])
143
+ scope = {}
144
+ for r in rl:
145
+ layers = set((r.get("matcher") or {}).get("layer") or [])
146
+ if not layers:
147
+ continue
148
+ for spec in r.get("mechanical") or []:
149
+ chk = spec.get("check")
150
+ if chk == "field_absent" and spec.get("field"):
151
+ scope.setdefault(str(spec["field"]), set()).update(layers)
152
+ elif chk == "evidence_zero":
153
+ scope.setdefault("evidence_count", set()).update(layers)
154
+ out = {k: sorted(v) for k, v in sorted(scope.items())}
155
+ if rules is None and rules_dir is None:
156
+ _FIELD_LAYER_CACHE = out
157
+ return out
158
+
159
+
160
+ # 生效条件:(scope or {}).get(field) 为假值(scope 为 None、空 dict 或 field 不在其中)→ 返回 True;want 为真值时仅当 str((meta or {}).get("layer")) 在 set(want) 内返回 True,否则 False。
161
+ def _layer_ok(field, scope, meta) -> bool:
162
+ """字段层门限:**限层的字段只在该层的节点上检查**。
163
+
164
+ 口径与 M1 `ruleset._scope`(`str(e.get("layer")) in want`)逐字一致——缺层
165
+ (`None`)不在任何层清单内 ⇒ 不检查。两处判据同源,防「越层误报」:
166
+ M1 不在某层报的问题,M3 也不该报。
167
+ """
168
+ want = (scope or {}).get(field)
169
+ if not want:
170
+ return True
171
+ return str((meta or {}).get("layer")) in set(want)
172
+
173
+
174
+ # 生效条件:os.path.getmtime(path) 成功则返回该 mtime;抛 OSError 或 TypeError → 返回 None。
175
+ def _mtime(path):
176
+ """文件 mtime(不可读 → `None`,不假装知道)。"""
177
+ try:
178
+ return os.path.getmtime(path)
179
+ except (OSError, TypeError):
180
+ return None
181
+
182
+
183
+ # 生效条件:root 或 rel(path 为真值时取 path,否则取 meta 的 "path")为空 → 返回 ("unknown", "无 root/path 可用,取不到盘上 mtime,成因未判定");两者都有时取节点文件与 CF.INDEX_FILE 的 mtime,任一为 None → 返回 ("unknown", "节点文件或索引快照不可读,成因未判定"),节点 mtime 减索引 mtime 之差 > MTIME_TOLERANCE → 返回 ("index_lag", …),否则返回 ("true_mismatch", …)。
184
+ def hash_mismatch_cause(meta, *, root=None, path=None) -> tuple:
185
+ """`content_hash` 声明值 ≠ 正文实算值 → **成因**判定(靠 mtime 证据,不猜)。
186
+
187
+ → `(cause, note)`;cause ∈ `HASH_CAUSES`;取不到盘上证据 → `"unknown"`。
188
+
189
+ 写路径真源(`md_cg/mdcg.py`):写走 `_stage()` 落**分片日志**,`rebuild_index()`
190
+ 才写 `_index.json` 快照——故「节点文件比索引快照新」是**正常写路径现象**
191
+ (快照滞后),不是文件被篡改;把两种成因合并成一句「索引与文件不一致」会让
192
+ 复核者误判为数据损坏(实测滞后可达 14s 量级)。
193
+ """
194
+ rel = path if path else (meta or {}).get("path")
195
+ if not root or not rel:
196
+ return "unknown", "无 root/path 可用,取不到盘上 mtime,成因未判定"
197
+ fp = str(rel) if os.path.isabs(str(rel)) else os.path.join(str(root), str(rel))
198
+ fm_t = _mtime(fp)
199
+ idx_t = _mtime(os.path.join(str(root), CF.INDEX_FILE))
200
+ if fm_t is None or idx_t is None:
201
+ return "unknown", "节点文件或索引快照不可读,成因未判定"
202
+ delta = fm_t - idx_t
203
+ if delta > MTIME_TOLERANCE:
204
+ return "index_lag", ("节点文件比索引快照新 %.2fs——分片日志已写、_index.json "
205
+ "未 rebuild(快照滞后属正常写路径现象)" % delta)
206
+ return "true_mismatch", ("节点文件不晚于索引快照(差 %.2fs)——非快照滞后,"
207
+ "索引声明与文件真源相抵触" % delta)
208
+
209
+
210
+ # ---------------------------- 基础工具(纯函数) ----------------------------
211
+
212
+ # 生效条件:v 为 list/tuple/dict 时返回 not v(空容器 → True);其余类型返回 v is None 或 str(v).strip() 为 "" 或 "None"。
213
+ def _blank(v) -> bool:
214
+ if isinstance(v, (list, tuple, dict)):
215
+ return not v
216
+ return v is None or str(v).strip() in ("", "None")
217
+
218
+
219
+ # 生效条件:int(float(v)) 可算(含数字字符串)则返回该整数;抛 TypeError 或 ValueError(含 v 为 None、非数字串)→ 返回 None。
220
+ def _as_int(v):
221
+ try:
222
+ return int(float(v))
223
+ except (TypeError, ValueError):
224
+ return None
225
+
226
+
227
+ # 生效条件:float(v) 可算则返回该浮点数;抛 TypeError 或 ValueError(含 v 为 None、非数字串)→ 返回 None。
228
+ def _as_float(v):
229
+ try:
230
+ return float(v)
231
+ except (TypeError, ValueError):
232
+ return None
233
+
234
+
235
+ _SENT_END = "。!?;!?;\n"
236
+
237
+
238
+ # 生效条件:content 为假值(None/空串)按 "" 处理,逐字符命中 _SENT_END 切出且 seg.strip() 非空的段以 len(out) 为序号追加 (索引, start, i+1, 原文),末尾 tail.strip() 非空亦追加;无合格段 → 返回空列表。
239
+ def sentence_spans(content: str) -> list:
240
+ """正文 → `[(句索引, start, end, 原文)]`;空句不编号(索引连续,确定性)。"""
241
+ text, out, start = content or "", [], 0
242
+ for i, ch in enumerate(text):
243
+ if ch in _SENT_END:
244
+ seg = text[start:i + 1]
245
+ if seg.strip():
246
+ out.append((len(out), start, i + 1, seg))
247
+ start = i + 1
248
+ tail = text[start:]
249
+ if tail.strip():
250
+ out.append((len(out), start, len(text), tail))
251
+ return out
252
+
253
+
254
+ #: 正文要素行:`# 功能名:…` / `# 生效条件: …`(间隔与分隔符两形态都收)
255
+ _RE_MARK = re.compile(r"^#[ \t]*(?P<mark>[^::\n]{1,16})[::][^\n]*", re.M)
256
+
257
+
258
+ # 生效条件:在 content(假值 → "")上用模块级 _RE_MARK 迭代匹配,以 m.group("mark").strip() 为键 setdefault 记下首次出现的 [m.start(), m.end());无匹配(含 content 为假值)→ 返回空 dict。
259
+ def mark_spans(content: str) -> dict:
260
+ """正文 CCG 要素行 → `{要素名: [start, end)}`;同要素取首次出现(确定性)。"""
261
+ out = {}
262
+ for m in _RE_MARK.finditer(content or ""):
263
+ out.setdefault(m.group("mark").strip(), [m.start(), m.end()])
264
+ return out
265
+
266
+
267
+ # 生效条件:(text or "").split("\n") 的每行 strip 后非空、不以 "#" 开头且含 ":" 时,取首个冒号前的键 k,k 非空且尚未入表则记 (1-based 行号, [该行起始偏移, 起始偏移+len(line)]);text 为假值或无合格行 → 返回空 dict。
268
+ def key_line_spans(text: str) -> dict:
269
+ """节点文件原文 → `{键: (1-based 行号, [start, end))}`。
270
+
271
+ 只认**无 `#` 前缀**且含 `:` 的行——即 frontmatter 行;`---` 分隔线无冒号,
272
+ 自然被排除。同键取首次出现。
273
+ """
274
+ out, pos = {}, 0
275
+ for i, line in enumerate((text or "").split("\n")):
276
+ s = line.strip()
277
+ if s and not s.startswith("#") and ":" in s:
278
+ k = s.split(":", 1)[0].strip()
279
+ if k and k not in out:
280
+ out[k] = (i + 1, [pos, pos + len(line)])
281
+ pos += len(line) + 1
282
+ return out
283
+
284
+
285
+ # 生效条件:index 为 dict 时其 "nodes" 为 dict 则返回 index["nodes"],否则返回 index 本身;index 为 None/非 dict 时调 CF.load_index(root),抛 OSError 或 ValueError → 返回 {},返回值为 dict 且其 "nodes" 为 dict → 返回该 "nodes",是 dict → 原样返回,否则返回 {}。
286
+ def _index(root, index=None) -> dict:
287
+ """索引节点表 `{node_id: meta}`(兼容 load_index 的 `{"nodes": …}` 形态)。"""
288
+ if isinstance(index, dict):
289
+ return index.get("nodes") if isinstance(index.get("nodes"), dict) else index
290
+ try:
291
+ idx = CF.load_index(root)
292
+ except (OSError, ValueError):
293
+ return {}
294
+ if isinstance(idx, dict) and isinstance(idx.get("nodes"), dict):
295
+ return idx["nodes"]
296
+ return idx if isinstance(idx, dict) else {}
297
+
298
+
299
+ # 生效条件:_index(root, index) 中 node_id 对应值非 dict → 返回 None;否则用 meta.get("path")(绝对路径直接用,否则 join(root, str(rel or "")))读文件——OSError 时返回 content/text 为 None、fm 为 {} 的 dict,成功则把 NF.loads(text) 得到的 content 与 fm(假值 → {})连同 meta/path/text 一并返回。
300
+ def load_node(node_id, root, *, index=None) -> dict:
301
+ """读一个节点(索引 meta + 文件 frontmatter + 正文原文)。
302
+
303
+ → `{"node_id","meta","fm","content","text","path"}`;索引无此项 → `None`。
304
+ `meta` 是**索引快照**(含索引侧 content_hash,矛盾判定要用),`fm` 是**文件真源**。
305
+ """
306
+ nodes = _index(root, index)
307
+ meta = nodes.get(node_id)
308
+ if not isinstance(meta, dict):
309
+ return None
310
+ rel = meta.get("path")
311
+ fp = rel if (rel and os.path.isabs(str(rel))) else os.path.join(root, str(rel or ""))
312
+ try:
313
+ with open(fp, encoding="utf-8") as f:
314
+ text = f.read()
315
+ except OSError:
316
+ return {"node_id": node_id, "meta": dict(meta), "fm": {}, "content": None,
317
+ "text": None, "path": fp}
318
+ fm, content = NF.loads(text)
319
+ return {"node_id": node_id, "meta": dict(meta), "fm": fm or {}, "content": content,
320
+ "text": text, "path": fp}
321
+
322
+
323
+ # 生效条件:span 为 None → 返回 "";否则取 (text or "")[span[0]:span[1]] 去空白并把换行替换为 "⏎",长度超 SNIPPET_MAX 时截断并追加 "…"。
324
+ def _snippet(text, span) -> str:
325
+ """命中片段(供人工核对肉眼确认「指的是不是这一句」)。"""
326
+ if span is None:
327
+ return ""
328
+ seg = (text or "")[span[0]:span[1]].strip().replace("\n", "⏎")
329
+ return seg[:SNIPPET_MAX] + ("…" if len(seg) > SNIPPET_MAX else "")
330
+
331
+
332
+ # 生效条件:任意 node_id/kind/field/span/evidence 均原样写入返回 dict 的 node_id/issue_kind/field/span/evidence 键,可选 rule/line/sentence/snippet/peer/cause/severity 未传时为 None、status 未传时为 "located",不做任何校验。
333
+ def _hit(node_id, kind, field, span, evidence, *, rule=None, line=None,
334
+ sentence=None, snippet=None, peer=None, cause=None, status="located",
335
+ severity=None) -> dict:
336
+ """命中行(四键契约 + 增强键)。
337
+
338
+ `severity` 是**观测面专用**的等级标注(`ADVISORY_KINDS` 命中带 `"info"`):
339
+ 问题面命中不带(问题即问题,无等级可降);本键让下游一眼分清「这是观测
340
+ 副产品」与「这是待修的问题」,不必靠 issue_kind 名字去猜。
341
+ """
342
+ return {"node_id": node_id, "field": field, "span": span, "issue_kind": kind,
343
+ "evidence": evidence, "rule": rule, "line": line, "sentence": sentence,
344
+ "snippet": snippet, "peer": peer, "cause": cause, "status": status,
345
+ "severity": severity}
346
+
347
+
348
+ # 生效条件:text 为真值、span 与 content 均非 None 且 text.find(content)>=0 时,返回 text.count("\n",0,min(off+span[0],len(text)))+1 的 1-based 行号;text 假值或 span/content 为 None 或 content 未找到时返回 None。
349
+ def _line_of(text, content, span):
350
+ """正文区间 → 节点文件原文的 1-based 行号(供人工核对直接跳文件)。"""
351
+ if not text or span is None or content is None:
352
+ return None
353
+ off = text.find(content)
354
+ if off < 0:
355
+ return None
356
+ return text.count("\n", 0, min(off + span[0], len(text))) + 1
357
+
358
+
359
+ # ---------------------------- 定位器注册表 ----------------------------
360
+ #
361
+ # 每个定位器是**纯函数**:只吃 (node_id, meta, content, ctx),只吐 hits。
362
+ # 新增判据 = 加函数 + 注册,不改 locate 主流程(与 M1「引擎冻结、规则可增删」同构)。
363
+ # ctx = {"fm", "text", "root", "index", "peers", "now"}
364
+
365
+ # 生效条件:在 node_id/meta/content/ctx 下,对 FM_SCAN_FIELDS 中除 verification_basis、condition_space 外且 _layer_ok(f,scope,meta) 为真的字段,若 meta.get(f) 与 ctx.get("fm") 或 {} 中的同名字段皆 _blank 则记 field_absent;对过 _layer_ok 且 _as_int(meta.get("evidence_count"))==0 记 evidence_zero;对 meta.get("importance") 非 None 且 _as_float 为 None 或不在 [0.0,1.0] 记 field_invalid;对 condition_space 经 meta 或回落 ctx.get("fm") 后 NF.condition_space_missing 非空记 condition_slots;对正文缺 NF.CCG_MARKS 行记 ccg_incomplete,返回这些命中列表。
366
+ def _loc_missing_field(node_id, meta, content, ctx):
367
+ """frontmatter/正文结构字段为空(判据与 M1 `field_absent`/`evidence_zero` 同源)。
368
+
369
+ **层门限**:限层的字段(如 `role`/`evidence_count` 限 `knowledge`)只在
370
+ 该层节点上检查——与 M1 `_scope` 同口径,否则 M1 不报的问题会被 M3 越层报出。
371
+ """
372
+ fm = ctx.get("fm") or {}
373
+ scope = ctx.get("field_layers")
374
+ if scope is None:
375
+ scope = field_layer_scope()
376
+ lines = key_line_spans(ctx.get("text") or "")
377
+ marks = mark_spans(content or "")
378
+ hits = []
379
+
380
+ for f in FM_SCAN_FIELDS:
381
+ if f in ("verification_basis", "condition_space"):
382
+ continue # 基底空属 weak_source;四槽缺失单独报(要点名缺哪几槽)
383
+ if not _layer_ok(f, scope, meta):
384
+ continue # 越层不报(层门限真源 = M1 规则库 matcher.layer)
385
+ if _blank(meta.get(f)) and _blank(fm.get(f)):
386
+ hits.append(_hit(node_id, "missing_field", f, None,
387
+ "字段 %s 为空(索引与文件两处皆空)" % f,
388
+ rule="field_absent",
389
+ line=(lines.get(f) or (None, None))[0]))
390
+
391
+ if _layer_ok("evidence_count", scope, meta) \
392
+ and _as_int(meta.get("evidence_count")) == 0:
393
+ hits.append(_hit(node_id, "missing_field", "evidence_count", None,
394
+ "evidence_count=0(无验证证据计数)", rule="evidence_zero",
395
+ line=(lines.get("evidence_count") or (None, None))[0]))
396
+
397
+ imp = meta.get("importance")
398
+ if imp is not None:
399
+ fv = _as_float(imp)
400
+ if fv is None or not 0.0 <= fv <= 1.0:
401
+ hits.append(_hit(node_id, "missing_field", "importance", None,
402
+ "importance=%r 不在 [0,1](字段存在但不可用)"
403
+ % (imp,), rule="field_invalid",
404
+ line=(lines.get("importance") or (None, None))[0]))
405
+
406
+ cs = meta.get("condition_space")
407
+ if not isinstance(cs, dict):
408
+ cs = fm.get("condition_space")
409
+ miss = NF.condition_space_missing(cs)
410
+ if miss:
411
+ span = marks.get("生效条件")
412
+ hits.append(_hit(node_id, "missing_field", "condition_space", span,
413
+ "条件空间缺槽 %s(%d/4 已声明)——四槽不全不构成生效条件"
414
+ % ("/".join(miss), len(NF.CONDITION_SLOTS) - len(miss)),
415
+ rule="condition_slots",
416
+ snippet=_snippet(content, span)))
417
+
418
+ # 正文 CCG 要素缺行:行不存在 → 无字符区间可指(span=None),点名缺哪几行
419
+ missing_marks = [m for m in NF.CCG_MARKS if m not in marks]
420
+ if missing_marks:
421
+ hits.append(_hit(node_id, "missing_field", "content", None,
422
+ "正文缺 CCG 要素行 %s(# 要素名:值)——缺声明即缺证据"
423
+ % "/".join(missing_marks), rule="ccg_incomplete"))
424
+ return hits
425
+
426
+
427
+ # 生效条件:meta.get("verification_basis") 为空时回落 ctx.get("fm") 或 {} 的 verification_basis,若两者皆 _blank 返回 basis_absent 命中;非空但 str(basis) 不在 NF.VERIFICATION_BASIS 返回 basis_enum 命中;在枚举内时按 CC.classify_track 依 meta.get("layer")、meta.get("tags") 与 content 判赛道,若 CC.basis_licensed 为假返回 basis_licensed 命中;否则返回 []。
428
+ def _loc_weak_source(node_id, meta, content, ctx):
429
+ """验证基底缺失/越枚举/与赛道不相容(与 M1 `basis_licensed` 判据逐字同源)。"""
430
+ fm = ctx.get("fm") or {}
431
+ basis = meta.get("verification_basis")
432
+ if _blank(basis):
433
+ basis = fm.get("verification_basis")
434
+ _ln = key_line_spans(ctx.get("text") or "").get("verification_basis")
435
+ ln = (_ln or (None, None))[0]
436
+
437
+ if _blank(basis):
438
+ return [_hit(node_id, "weak_source", "verification_basis", None,
439
+ "verification_basis 为空——无验证基底的断言不可裁决",
440
+ rule="basis_absent", line=ln)]
441
+ if str(basis) not in NF.VERIFICATION_BASIS:
442
+ return [_hit(node_id, "weak_source", "verification_basis", None,
443
+ "基底 %s 不在合法枚举(%s)"
444
+ % (basis, "/".join(NF.VERIFICATION_BASIS)),
445
+ rule="basis_enum", line=ln)]
446
+ track = CC.classify_track({"layer": meta.get("layer"),
447
+ "tags": meta.get("tags") or []}, content or "")
448
+ if not CC.basis_licensed(track, basis):
449
+ allow = "/".join(CC.allowed_basis(track)) or "—"
450
+ return [_hit(node_id, "weak_source", "verification_basis", None,
451
+ "赛道 %s(政策 %s)不允许基底 %s;允许 %s"
452
+ % (track, CC.source_policy(track) or "—", basis, allow),
453
+ rule="basis_licensed", line=ln)]
454
+ return []
455
+
456
+
457
+ # 生效条件:ctx.get("fm") 或 {} 中 code_ref 或 doc_ref 为非空 dict 且 RI.probe_ref(ref) 返回的 status 属于 _REF_DEAD_STATUSES(dangling 或 stale)时,返回对应 stale 命中(dangling 归因载体消失、stale 归因区间哈希不符);否则返回 []。
458
+ def _loc_stale(node_id, meta, content, ctx):
459
+ """**依赖存在性**——声明的载体(源文件)已不存在 / 已漂移(真正的适用边界越出)。
460
+
461
+ 知识的真值挂在它描述的对象上:代码知识挂在源文件的符号上,源没了或改了,
462
+ 知识才真的不再成立。判据与 `refindex` **同源**(`probe_ref`,不另立一份):
463
+
464
+ ============== ====================================== ==========
465
+ probe_ref 态 含义 本判据
466
+ ============== ====================================== ==========
467
+ ``ok`` 源文件在、区间哈希匹配 不判
468
+ ``stale`` 源文件在、区间哈希不符(源被改,漂移) **stale**
469
+ ``dangling`` 源文件不存在(载体消失) **stale**
470
+ ``unresolved`` 拿不到 root(判不了) 不判(不猜)
471
+ ``error`` 读盘失败 不判(不猜)
472
+ ============== ====================================== ==========
473
+
474
+ 后两态是**观测手段不足**,不是「依赖已失效」——按「不猜」纪律如实不报。
475
+
476
+ **时间窗不在此处判**(2026-09-16 修正的原缺陷):`condition_space.time_window`
477
+ 是写入端**观测时刻**栏(`mdcg.add` 未给时自动填「写入时刻 +1h」,见其
478
+ `OBSERVATION_WINDOW_SEC`),不是知识的有效期——拿它判「适用边界越出」,
479
+ 等于把「这条记忆写入超过 1 小时」冒充为「失效」。真实适用前提由知识自己
480
+ 声明在 `code_ref`/`doc_ref` 上,故改由依赖存在性裁决;观测时刻本身不丢,
481
+ 降级为 `observation_aged`(观测面,见 `ADVISORY_KINDS`)。
482
+ """
483
+ fm = ctx.get("fm") or {}
484
+ hits = []
485
+ for key in ("code_ref", "doc_ref"):
486
+ ref = fm.get(key)
487
+ if not isinstance(ref, dict) or not ref:
488
+ continue
489
+ # root 基准必须取自 ref 自身(`_code_ref`/`_doc_ref` 落的是**源大域** root,
490
+ # path 相对它);`ctx["root"]` 是**认知图** root——拿它去拼会把每一条依赖
491
+ # 都误判成悬空。ref 无 root(旧节点)时 probe 返回 `unresolved` → 不判。
492
+ probe = RI.probe_ref(ref)
493
+ st = str(probe.get("status") or "")
494
+ if st not in _REF_DEAD_STATUSES:
495
+ continue
496
+ rel = str(ref.get("path") or "?")
497
+ if st == "dangling":
498
+ ev = ("依赖的源文件已不存在(%s:%s)——知识所指的载体消失,适用边界越出"
499
+ % (key, rel))
500
+ else:
501
+ ev = ("依赖的源文件已漂移(%s:%s):索引区间哈希 %s ≠ 现算 %s——"
502
+ "源被改动,知识所指的符号可能已不是原物"
503
+ % (key, rel, probe.get("hash_expected"), probe.get("hash")))
504
+ hits.append(_hit(node_id, "stale", key, None, ev, rule="ref_" + st))
505
+ return hits
506
+
507
+
508
+ # 生效条件:从 meta.get("condition_space") 或回落 ctx.get("fm") 的 condition_space(须为 dict)取 time_window,缺则取 meta.get("time_window"),若 NF.is_full_time_window(tw) 或 float(tw[1]) 抛 TypeError/ValueError/IndexError/KeyError 或 ctx.get("now") 为 None 或 hi>=float(now) 则返回 [];否则返回一条 severity="info" 的 observation_aged 命中。
509
+ def _loc_observation_aged(node_id, meta, content, ctx):
510
+ """时间窗已过——**这是观测时刻,不是失效声明**(观测面,不进告警面)。
511
+
512
+ 时间窗来源链与 `_loc_missing_field` 的槽检查同构:`meta.condition_space`
513
+ (显式注入面)→ `fm.condition_space`(文件真源)→ `meta.time_window`
514
+ (索引快照字段)。**索引快照只透传 `time_window`**(`mdcg.py` `_stage` 口径:
515
+ 时空字段入快照、condition_space 整块不入),故第三跳必留。
516
+
517
+ **不进默认全量**(不在 `ISSUE_KINDS`):写入端未给 time_window 时以写入时刻
518
+ 自动填 1 小时窗,故真库「已过期」绝大多数是「写入超过 1 小时」,与知识是否
519
+ 失效无关。观测时刻本身仍如实报(审计有用),只是**不冒充问题**——命中带
520
+ `severity="info"` 标记它是咨询级。
521
+ """
522
+ fm = ctx.get("fm") or {}
523
+ cs = meta.get("condition_space")
524
+ if not isinstance(cs, dict):
525
+ csf = fm.get("condition_space")
526
+ cs = csf if isinstance(csf, dict) else None
527
+ tw = (cs or {}).get("time_window")
528
+ if tw is None:
529
+ tw = meta.get("time_window")
530
+ if NF.is_full_time_window(tw):
531
+ return [] # 全时窗是**合法声明**(任意时刻成立),不判
532
+ try:
533
+ hi = float(tw[1])
534
+ except (TypeError, ValueError, IndexError, KeyError):
535
+ return []
536
+ now = ctx.get("now")
537
+ if now is None or hi >= float(now):
538
+ return []
539
+ span = mark_spans(content or "").get("生效条件")
540
+ return [_hit(node_id, "observation_aged", "condition_space", span,
541
+ "观测时间窗 %s 已过(hi=%.0f < now=%.0f)——这是**观测时刻**,"
542
+ "不是失效声明(知识不因观测过去而失效);时效判定见 stale"
543
+ % (NF.time_window_text(tw) or "?", hi, float(now)),
544
+ rule="observation_window_passed", severity="info",
545
+ line=_line_of(ctx.get("text"), content, span),
546
+ snippet=_snippet(content, span))]
547
+
548
+
549
+ # 生效条件:h 取 str(meta.get("content_hash") or "").strip(),若 _blank(h) 则 h=NF.content_hash(content 或 "");对 ctx.get("peers") 或 [] 中 node_id 不等于本节点且 ph=str(p.get("hash") or "") 或回落 NF.content_hash(p.get("content") or "") 后非空且等于 h 的 peer 计入 matched;matched 非空时返回一条 dup 命中(span=[0,len(body)]),否则返回 []。
550
+ def _loc_dup(node_id, meta, content, ctx):
551
+ """同组同内容指纹(与 M1 `dup_hash_group` 同判据;D1 名 `dup`)。"""
552
+ body = content if content is not None else ""
553
+ h = str(meta.get("content_hash") or "").strip()
554
+ if _blank(h):
555
+ h = NF.content_hash(body) # 索引无指纹 → 用正文实算(诚实标注来源)
556
+ matched = []
557
+ for p in ctx.get("peers") or []:
558
+ if p.get("node_id") == node_id:
559
+ continue
560
+ ph = str(p.get("hash") or "") or NF.content_hash(p.get("content") or "")
561
+ if ph and ph == h:
562
+ matched.append(p)
563
+ if not matched:
564
+ return []
565
+ span = [0, len(body)]
566
+ return [_hit(node_id, "dup", "content_hash", span,
567
+ "与 %s 同内容指纹 %s(同类共 %d 条),整篇正文重复"
568
+ % (matched[0].get("node_id"), h, len(matched) + 1),
569
+ rule="dup_hash_group", peer=matched[0].get("node_id"),
570
+ snippet=_snippet(body, span))]
571
+
572
+
573
+ # 生效条件:对 ctx.get("peers") 或 [] 中每个非自身 peer,按 sentence_spans 对齐 content 与 peer content 的句子,当同一句位上去数字骨架相同(WL._skeleton(seg) 非空、长度 ≥ WL.MIN_SKELETON 且等于 peer 骨架)且 seg.strip() != ptext.strip() 的句子数达到 MIN_FLOW_SENTENCES 时,为这些句各返回一条 template_flow 命中;否则返回 []。
574
+ def _loc_template_flow(node_id, meta, content, ctx):
575
+ """同模板流水:与同组节点逐句「去数字骨架相同、字面不同」→ 指向那些句。
576
+
577
+ 与 M1 `template_flow_digits_only` 同判据(都走 `writelimit._skeleton`),
578
+ 差别只在产物:M1 给一条条目级 issue,D1 给出**具体是哪几句**(span + 句索引)。
579
+ 单句偶合不算流水 → 需同组至少 `MIN_FLOW_SENTENCES` 句同时成立。
580
+ """
581
+ tgt = sentence_spans(content or "")
582
+ hits = []
583
+ for p in ctx.get("peers") or []:
584
+ if p.get("node_id") == node_id:
585
+ continue
586
+ pseg = sentence_spans(p.get("content") or "")
587
+ same = []
588
+ for (i, s, e, seg) in tgt:
589
+ if i >= len(pseg):
590
+ break
591
+ ptext = pseg[i][3]
592
+ sk = WL._skeleton(seg)
593
+ if not sk or len(sk) < WL.MIN_SKELETON or sk != WL._skeleton(ptext):
594
+ continue
595
+ if seg.strip() == ptext.strip():
596
+ continue # 字面全同属整篇重复(dup),不是「只换数字」的流水
597
+ same.append((i, s, e, seg, sk))
598
+ if len(same) < MIN_FLOW_SENTENCES:
599
+ continue
600
+ for (i, s, e, seg, sk) in same:
601
+ hits.append(_hit(node_id, "template_flow", "content", [s, e],
602
+ "同模板流水(与 %s 第 %d 句去数字骨架相同、仅字面数值不同):%s"
603
+ % (p.get("node_id"), i, sk[:40]),
604
+ rule="template_flow_digits_only", sentence=i,
605
+ peer=p.get("node_id"),
606
+ line=_line_of(ctx.get("text"), content, [s, e]),
607
+ snippet=seg.strip()[:SNIPPET_MAX]))
608
+ return hits
609
+
610
+
611
+ # 生效条件:meta.get("content_hash") 非 _blank 且 content 非 None 且 h != NF.content_hash(content) 时,返回一条 hash_declared_vs_actual 的 contradiction 命中(cause 由 hash_mismatch_cause 依 meta/ctx.get("root")/ctx.get("path") 判);ctx.get("fm") 或 {} 的 id 非 _blank 且 str(fm_id) != str(node_id) 时额外返回一条 id_declared_vs_index 命中;两者皆不成立返回 []。
612
+ def _loc_contradiction(node_id, meta, content, ctx):
613
+ """**确定性**矛盾:声明与事实不符(指纹 / 标识)。
614
+
615
+ 语义级矛盾(正文自相矛盾)不在此处——那是 LLM 的活,D1 不猜(见 `SEMANTIC_ONLY`)。
616
+ 指纹不一致**必须带成因**(`cause`):`index_lag`(快照滞后,属正常写路径现象,
617
+ 复核者无需修数据)与 `true_mismatch`(真不一致,需查)处置完全不同。
618
+ """
619
+ hits = []
620
+ lines = key_line_spans(ctx.get("text") or "")
621
+ h = str(meta.get("content_hash") or "").strip()
622
+ if not _blank(h) and content is not None:
623
+ actual = NF.content_hash(content)
624
+ if h != actual:
625
+ cause, note = hash_mismatch_cause(meta, root=ctx.get("root"),
626
+ path=ctx.get("path"))
627
+ hits.append(_hit(node_id, "contradiction", "content_hash", None,
628
+ "索引声明指纹 %s ≠ 正文实算 %s(%s)"
629
+ % (h, actual, note), rule="hash_declared_vs_actual",
630
+ cause=cause,
631
+ line=(lines.get("content_hash") or (None, None))[0]))
632
+ fm_id = (ctx.get("fm") or {}).get("id")
633
+ if not _blank(fm_id) and str(fm_id) != str(node_id):
634
+ hits.append(_hit(node_id, "contradiction", "id", None,
635
+ "文件 frontmatter id=%s ≠ 索引键 %s" % (fm_id, node_id),
636
+ rule="id_declared_vs_index",
637
+ line=(lines.get("id") or (None, None))[0]))
638
+ return hits
639
+
640
+
641
+ LOCATORS = {
642
+ "missing_field": _loc_missing_field,
643
+ "weak_source": _loc_weak_source,
644
+ "stale": _loc_stale,
645
+ "observation_aged": _loc_observation_aged,
646
+ "dup": _loc_dup,
647
+ "template_flow": _loc_template_flow,
648
+ "contradiction": _loc_contradiction,
649
+ }
650
+
651
+ #: D1 **不定位**的判定:属语义层,交回 LLM 意见(不猜 = 不编造字符区间)
652
+ SEMANTIC_ONLY = {
653
+ "contradiction_semantic": "正文自相矛盾属语义判定;D1 只做确定性矛盾"
654
+ "(索引指纹 / 标识与事实不符)",
655
+ }
656
+
657
+
658
+ # 生效条件:始终返回 SEMANTIC_ONLY 中以 str(kind or "").strip() 为键查得的值,缺键时返回空串(空串表示可定位)。
659
+ def blindspot_reason(kind) -> str:
660
+ """D1 拒绝定位的类别 → 理由(空串表示可定位)。"""
661
+ return SEMANTIC_ONLY.get(str(kind or "").strip(), "")
662
+
663
+
664
+ # ---------------------------- 主入口 ----------------------------
665
+
666
+ # 生效条件:items 中每项按 content=it.get("content") or ""、hash=str(it.get("hash") or "").strip() 或回落 NF.content_hash(content)、sk=WL.template_signature(content) or "" 预处理后,返回 {node_id: [同 hash 或同非空 sk 的其他 peer]};items 为空/None 返回空 dict。
667
+ def build_peers(items) -> dict:
668
+ """`[{node_id, content, hash?, …}]` → `{node_id: [peer, …]}`。
669
+
670
+ 只有**同内容指纹**或**同模板骨架**才互为对照——不是「同一批就是一组」。
671
+ 组的作用域由调用方给定(批 / 包),与 M1 机械层「组 = 同一个包」的口径一致。
672
+ """
673
+ prep = []
674
+ for it in items or []:
675
+ c = it.get("content") or ""
676
+ prep.append({"node_id": it.get("node_id"), "content": c, "meta": it.get("meta"),
677
+ "text": it.get("text"), "fm": it.get("fm"),
678
+ "hash": str(it.get("hash") or "").strip() or NF.content_hash(c),
679
+ "sk": WL.template_signature(c) or ""})
680
+ by_hash, by_sk = {}, {}
681
+ for p in prep:
682
+ by_hash.setdefault(p["hash"], []).append(p)
683
+ if p["sk"]:
684
+ by_sk.setdefault(p["sk"], []).append(p)
685
+ out = {}
686
+ for p in prep:
687
+ grp = {q["node_id"]: q for q in by_hash.get(p["hash"], [])
688
+ if q["node_id"] != p["node_id"]}
689
+ for q in (by_sk.get(p["sk"], []) if p["sk"] else []):
690
+ if q["node_id"] != p["node_id"]:
691
+ grp.setdefault(q["node_id"], q)
692
+ out[p["node_id"]] = list(grp.values())
693
+ return out
694
+
695
+
696
+ # 生效条件:对 issue_hint(None/str/dict 或其 list/tuple)逐 hint 分类,返回 (kinds, fields, blind):能 canonical_kind 到 ISSUE_KINDS 或 ADVISORY_KINDS 的入 kinds,blindspot_reason 非空的入 blind,其余非空 kind 与 field 入 fields;issue_hint 为 None 时 kinds/fields 为空集、blind 为空列表。
697
+ def _norm_hint(issue_hint):
698
+ """issue_hint → `(kinds, fields, blindspot)`。
699
+
700
+ 形态:`None`(全量)|str(issue_kind 或 field)|dict(issue_kind/field)
701
+ |上述的 list。
702
+ """
703
+ kinds, fields, blind = set(), set(), []
704
+ hints = issue_hint if isinstance(issue_hint, (list, tuple)) else [issue_hint]
705
+ for h in hints:
706
+ if h is None:
707
+ continue
708
+ if isinstance(h, dict):
709
+ s = h.get("issue_kind") or h.get("kind") or ""
710
+ f = h.get("field")
711
+ else:
712
+ s, f = str(h).strip(), None
713
+ if s:
714
+ why = blindspot_reason(s)
715
+ if why:
716
+ blind.append("%s:%s" % (s, why))
717
+ else:
718
+ ck = canonical_kind(s)
719
+ if ck in ISSUE_KINDS or ck in ADVISORY_KINDS:
720
+ kinds.add(ck)
721
+ else:
722
+ fields.add(s) # 不是类别 → 当作字段过滤
723
+ if f:
724
+ fields.add(str(f).strip())
725
+ return kinds, fields, blind
726
+
727
+
728
+ # 生效条件:当 meta 与 content 均非 None 时不读盘;否则 root 为 None 抛 ValueError,经 load_node 读不到节点返回 {"hits": [], "blindspot": blind+["节点 %s 不在索引"%node_id], "load": None};读到时补 meta/content/text/fm/path,按 kinds 非空时只跑 sorted(kinds)、否则 fields 非空或 issue_hint 为 None 时跑 sorted(ISSUE_KINDS)、否则 run=[] 执行 LOCATORS,对命中按 fields 过滤、补 line 并排序后返回 {"hits": hits, "blindspot": blind, "load": {...}}。
729
+ def locate_ex(node_id, issue_hint=None, *, meta=None, content=None, fm=None, text=None,
730
+ root=None, index=None, peers=None, now=None, path=None,
731
+ field_layers=None) -> dict:
732
+ """D1 定位(完整返回):`{"hits": [...], "blindspot": [...], "load": {...}}`。
733
+
734
+ `meta`/`content` 显式给出则**不读盘**(纯函数用法——测试与批量复用都走这条);
735
+ 否则 `root` 必填,经 `load_node` 从索引+文件取。返回的 hits 已按
736
+ `(issue_kind, field, span起点, peer)` 排序,**确定性可复算**。
737
+
738
+ `path`(节点文件绝对路径——供 `contradiction` 判成因)、`field_layers`
739
+ (字段层门限,缺省 `field_layer_scope()`)可显式注入,便于纯函数测试。
740
+ """
741
+ kinds, fields, blind = _norm_hint(issue_hint)
742
+ if meta is None or content is None:
743
+ if root is None:
744
+ raise ValueError("locate 需要 meta+content,或给出 root 以便读节点")
745
+ nd = load_node(node_id, root, index=index)
746
+ if nd is None:
747
+ return {"hits": [], "blindspot": blind + ["节点 %s 不在索引" % node_id],
748
+ "load": None}
749
+ if meta is None:
750
+ meta = nd["meta"]
751
+ if content is None:
752
+ content = nd["content"]
753
+ if text is None:
754
+ text = nd["text"]
755
+ if fm is None:
756
+ fm = nd["fm"]
757
+ if path is None:
758
+ path = nd.get("path")
759
+ meta = meta or {}
760
+ ctx = {"fm": fm or {}, "text": text, "root": root, "index": index,
761
+ "peers": peers or [], "now": time.time() if now is None else float(now),
762
+ "path": path, "field_layers": field_layers}
763
+ hits = []
764
+ # 定位范围:显式类别 → 只跑该类;给出字段 → 全量再由字段收窄;
765
+ # 无提示 → 全量。**提示全部落 blindspot 时范围为空**——不借机报别的类
766
+ # (否则「给了语义提示」会顺手退回全量,与「不猜」纪律相悖)。
767
+ if kinds:
768
+ run = sorted(kinds)
769
+ elif fields or issue_hint is None:
770
+ run = sorted(ISSUE_KINDS)
771
+ else:
772
+ run = []
773
+ for k in run:
774
+ fn = LOCATORS.get(k)
775
+ if fn is None:
776
+ continue
777
+ for h in fn(node_id, meta, content, ctx):
778
+ if fields and h.get("field") not in fields:
779
+ continue
780
+ if h.get("line") is None and h.get("span") is not None:
781
+ h["line"] = _line_of(text, content, h["span"])
782
+ hits.append(h)
783
+ hits.sort(key=lambda h: (h["issue_kind"], str(h.get("field") or ""),
784
+ -1 if h.get("span") is None else h["span"][0],
785
+ str(h.get("peer") or "")))
786
+ return {"hits": hits, "blindspot": blind,
787
+ "load": {"node_id": node_id, "meta": meta,
788
+ "content_len": len(content or "")}}
789
+
790
+
791
+ # 生效条件:给定 node_id(必填)与可选 issue_hint 及 kw 后,直接返回 locate_ex(node_id, issue_hint, **kw) 结果的 "hits" 列表。
792
+ def locate(node_id, issue_hint=None, **kw) -> list:
793
+ """**D1 契约入口**:`locate(node_id, issue_hint) -> [{field, span, issue_kind, evidence}]`。"""
794
+ return locate_ex(node_id, issue_hint, **kw)["hits"]
795
+
796
+
797
+ # 生效条件:给定 hits 与 key 后,返回 hits 中每个 h.get(key) 字符串化取值到出现次数的字典(键升序);hits 为 None/空时返回空 dict。
798
+ def _counts(hits, key) -> dict:
799
+ out = {}
800
+ for h in hits or []:
801
+ k = h.get(key)
802
+ out[str(k)] = out.get(str(k), 0) + 1
803
+ return dict(sorted(out.items()))
804
+
805
+
806
+ # 生效条件:items 显式给出时不读盘;items 为 None 时 root 为 None 抛 ValueError,否则按 node_ids 逐节点 load_node 组装 items(读不到的记入 missing);随后对每个 item 调 locate_ex 汇总 hits 或 clean,并返回含 nodes/hits/by_kind/by_field/by_rule/clean/missing 的 dict。
807
+ def locate_many(node_ids=None, *, root=None, index=None, items=None,
808
+ issue_hint=None, now=None) -> dict:
809
+ """批量定位:**同一批内**互为对照(`dup` / `template_flow` 的组 = 本批)。
810
+
811
+ `items` 显式给出 `[{node_id, content, meta?, text?, fm?, hash?}]` 则不再读盘。
812
+ 返回 `{nodes, hits, by_kind, by_field, clean, missing}`;`clean` 是**无命中**的节点、
813
+ `missing` 是**读不到**的节点——两者分开报(「没问题」≠「没看」)。
814
+ """
815
+ missing = []
816
+ if items is None:
817
+ if root is None:
818
+ raise ValueError("locate_many 需要 root(读盘)或 items(显式给出)")
819
+ items = []
820
+ for nid in node_ids or []:
821
+ nd = load_node(nid, root, index=index)
822
+ if nd is None:
823
+ missing.append(nid)
824
+ continue
825
+ items.append({"node_id": nid, "content": nd["content"] or "",
826
+ "text": nd["text"], "fm": nd["fm"], "meta": nd["meta"],
827
+ "path": nd.get("path"),
828
+ "hash": nd["meta"].get("content_hash")})
829
+ peers = build_peers(items)
830
+ now = time.time() if now is None else float(now)
831
+ hits, clean = [], []
832
+ for it in items:
833
+ nid = it["node_id"]
834
+ ex = locate_ex(nid, issue_hint, meta=it.get("meta") or {},
835
+ content=it.get("content") or "", fm=it.get("fm"),
836
+ text=it.get("text"), peers=peers.get(nid) or [], now=now,
837
+ root=root, path=it.get("path"))
838
+ if ex["hits"]:
839
+ hits.extend(ex["hits"])
840
+ else:
841
+ clean.append(nid)
842
+ hits.sort(key=lambda h: (str(h.get("node_id")), h["issue_kind"],
843
+ str(h.get("field") or ""),
844
+ -1 if h.get("span") is None else h["span"][0]))
845
+ return {"nodes": len(items), "hits": hits, "by_kind": _counts(hits, "issue_kind"),
846
+ "by_field": _counts(hits, "field"), "by_rule": _counts(hits, "rule"),
847
+ "clean": clean, "missing": missing}
848
+
849
+
850
+ # 生效条件:从 pkg.get("entries") or [] 取条目并跳过 e.get("node_id") 为空者,root 非 None 时逐个 load_node 取正文,取不到时回落 e.get("excerpt") or "",组装 items 调 locate_many 后返回其结果并附加 bundle_id 与 entries 数量;pkg 为 None 时 entries 为空列表。
851
+ def locate_package(pkg, *, root=None, index=None, issue_hint=None, now=None) -> dict:
852
+ """M1 包 → 定位汇总(组的作用域 = 本包,与 M1 机械层同口径)。
853
+
854
+ 正文优先读盘(准确);文件不可读时退回包内 `excerpt`(可能截断——宁可降级也不
855
+ 假装读到了全文,`missing` 会如实记下读不到的节点)。
856
+ """
857
+ entries = (pkg or {}).get("entries") or []
858
+ items = []
859
+ for e in entries:
860
+ nid = e.get("node_id")
861
+ if not nid:
862
+ continue
863
+ nd = load_node(nid, root, index=index) if root else None
864
+ body = (nd or {}).get("content")
865
+ items.append({"node_id": nid,
866
+ "content": body if body is not None else (e.get("excerpt") or ""),
867
+ "text": (nd or {}).get("text"), "fm": (nd or {}).get("fm"),
868
+ "meta": (nd or {}).get("meta") or e,
869
+ "path": (nd or {}).get("path") or e.get("path"),
870
+ "hash": ((nd or {}).get("meta") or {}).get("content_hash")
871
+ or e.get("content_hash")})
872
+ out = locate_many(items=items, issue_hint=issue_hint, now=now)
873
+ out["bundle_id"] = (pkg or {}).get("bundle_id")
874
+ out["entries"] = len(entries)
875
+ return out
876
+
877
+
878
+ # 生效条件:给定 hits 后返回 total=len(hits or [])、去重 node_id 数、排序后的 node_ids、以及 by_kind/by_field/by_rule 计数;hits 为 None/空时 total=0、nodes=0、node_ids=[]。
879
+ def summary(hits) -> dict:
880
+ """命中汇总(审计留痕用)。"""
881
+ nodes = sorted({str(h.get("node_id")) for h in hits or []})
882
+ return {"total": len(hits or []), "nodes": len(nodes), "node_ids": nodes,
883
+ "by_kind": _counts(hits, "issue_kind"), "by_field": _counts(hits, "field"),
884
+ "by_rule": _counts(hits, "rule")}
885
+
886
+
887
+ # 生效条件:给定 hits 后逐条生成 Markdown 行并返回表头加各行:field/line/snippet 取 h.get(...) or "—"(假值回落 "—"),span 为 None 时显示 "—" 否则 "起-止",evidence 取 (h.get("evidence") or "").replace("|","\\|")(假值回落空串)。
888
+ def markdown_table(hits) -> str:
889
+ """人工核对清单:每行一条命中,末列留空供核对者填判定(D1 验收抽样用)。"""
890
+ head = ("| # | node_id | issue_kind | field | span | line | 片段 | evidence | 人工判定 |\n"
891
+ "|---|---|---|---|---|---|---|---|---|\n")
892
+ rows = []
893
+ for i, h in enumerate(hits or [], 1):
894
+ sp = "—" if h.get("span") is None else "%d-%d" % (h["span"][0], h["span"][1])
895
+ rows.append("| %d | %s | %s | %s | %s | %s | %s | %s | |"
896
+ % (i, h.get("node_id"), h.get("issue_kind"), h.get("field") or "—",
897
+ sp, h.get("line") or "—",
898
+ (h.get("snippet") or "—").replace("|", "\\|"),
899
+ (h.get("evidence") or "").replace("|", "\\|")))
900
+ return head + "\n".join(rows)
901
+
902
+
903
+ # 生效条件:解析 argv(缺省 sys.argv)后,--root 为假值(含默认 os.environ.get("MDCG_ROOT") 为 None 或空串)时打印提示并返回 2;--node 追加列表为空时打印提示并返回 2;否则以 --root 与 --node 调 locate_many,并按 --json 或 --markdown 输出后返回 0,两者皆无则逐行打印命中与汇总后返回 0。
904
+ def main(argv=None) -> int:
905
+ ap = argparse.ArgumentParser(prog="python -m md_cg.mreview.locate",
906
+ description="记忆评审 M3 · D1 字段级定位")
907
+ ap.add_argument("--root", default=os.environ.get("MDCG_ROOT"),
908
+ help="真源根(缺省读环境变量 MDCG_ROOT)")
909
+ ap.add_argument("--node", action="append", default=[],
910
+ help="节点 id(可多次;同一批内互为对照)")
911
+ ap.add_argument("--kind", action="append", default=[], help="只定位该类问题")
912
+ ap.add_argument("--field", action="append", default=[], help="只保留该字段的命中")
913
+ ap.add_argument("--json", action="store_true", help="输出 JSON(缺省人读表)")
914
+ ap.add_argument("--markdown", action="store_true", help="输出人工核对清单")
915
+ a = ap.parse_args(argv)
916
+ if not a.root:
917
+ print("需要 --root 或环境变量 MDCG_ROOT", file=sys.stderr)
918
+ return 2
919
+ if not a.node:
920
+ print("需要 --node <id>(可多次)", file=sys.stderr)
921
+ return 2
922
+ rep = locate_many(a.node, root=a.root,
923
+ issue_hint=(list(a.kind) + list(a.field)) or None)
924
+ if a.json:
925
+ print(json.dumps(rep, ensure_ascii=False, indent=2))
926
+ return 0
927
+ if a.markdown:
928
+ print(markdown_table(rep["hits"]))
929
+ return 0
930
+ for h in rep["hits"]:
931
+ print("%-14s %-20s %-14s %s" % (h["node_id"], h["field"], h["issue_kind"],
932
+ h["evidence"]))
933
+ print("-- 命中 %d 条|干净 %d 条|缺项 %d 条"
934
+ % (len(rep["hits"]), len(rep["clean"]), len(rep["missing"])))
935
+ return 0
936
+
937
+
938
+ if __name__ == "__main__":
939
+ sys.exit(main())