@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
@@ -1,1098 +1,1098 @@
1
- # -*- coding: utf-8 -*-
2
- """批量核对(P34/P35):工单 → 反思候选 → 白箱闸门 → 验证否决 → 来源执照 → 落库。
3
-
4
- 设计约束(与计划一致):
5
-
6
- * **不新增 MCP op**:本模块是「模块 + `_cli`」形态(同 `backfill.py`),子代理经
7
- `python -m md_cg.crosscheck …` 调用;令牌决定权限边界。
8
- * **防自证**:反思单元(`reflect`)与验证单元(`verify`)必须为**不同执行者**;
9
- 验证单元**只能否决、不能新增**候选(越界字段在白箱闸门直接丢弃)。
10
- * **来源执照**:文科(`humanities`)只接受来源一致性档(`textbook`/`public_kb`);
11
- 理科(`science`)只接受可复现档(`compiler`/`test`/`measurement`/`formal_proof`/`data`);
12
- 赛道未定(`undetermined`)或无来源一律不写,保持 DEFER 并登记待补。
13
- * **诚实边界**:占位空壳节点(`骨架锚点`/`内容待填充`)不进接线,转待填充工单;
14
- 拿不出证据的节点绝不写 `verification_basis`。
15
- * **留痕可回滚**:写入前记录 `_crosscheck.jsonl`,`rollback` 仅在「当前值 == 写入值」
16
- 时撤销,否则计入 conflict 跳过。
17
-
18
- `_cli` 之外的所有函数都是纯逻辑(无网络、无第三方依赖),便于离线回归。
19
- """
20
-
21
- from __future__ import annotations
22
-
23
- import argparse
24
- import json
25
- import os
26
- import re
27
- import sys
28
- import time
29
- from collections import OrderedDict
30
-
31
- from . import crypto, nodefile
32
- from .backfill import (BASIS_ENUM_DEFAULT, BASIS_TEXT, INTERNAL_LAYERS,
33
- SKIP_LAYERS, SKIP_TAGS, _as_cg, _as_text, _comment,
34
- _ensure_comment, _entry_id, _readable_guard,
35
- _remove_ccg_line, _sha, derive_fields)
36
- from .consolidate import _has_ccg_line, _upsert_ccg_line
37
- from .fsutil import append_jsonl, read_jsonl
38
- from .mdcos import _ccg_field
39
-
40
- # ---- 常量 ----------------------------------------------------------------
41
-
42
- CROSSCHECK_LOG = "_crosscheck.jsonl"
43
- CROSSCHECK_BATCH = "crosscheck"
44
-
45
- REFLECT_UNIT = "reflect"
46
- VERIFY_UNIT = "verify"
47
-
48
- # 可写字段(本管线只动这一处,杜绝越界面)
49
- WRITABLE_FIELDS = ("验证方式",)
50
- FIELD_NORMALIZE = {
51
- "验证方式": "验证方式",
52
- "verification_basis": "验证方式",
53
- "验证": "验证方式",
54
- "verification": "验证方式",
55
- }
56
-
57
- # 赛道 → 来源执照策略
58
- SOURCE_POLICY = {"science": "reproducible", "humanities": "consistency"}
59
-
60
- # 学科关键词(赛道判定;两栖词不入表 → undetermined,宁缺勿猜)
61
- SCIENCE_SUBJECT_HINTS = (
62
- "数学", "物理", "化学", "生物", "科学", "信息技术", "通用技术", "计算机",
63
- )
64
- HUMANITIES_SUBJECT_HINTS = (
65
- "语文", "历史", "政治", "道德与法治", "思想政治", "思想品德", "英语",
66
- "文学", "哲学", "艺术", "音乐", "美术",
67
- )
68
-
69
- # 基底枚举 → 可读标签(条件化表述引用)
70
- BASIS_LABEL = {
71
- "compiler": "编译器/静态检查",
72
- "test": "单元测试",
73
- "measurement": "实测数据",
74
- "formal_proof": "形式化证明",
75
- "data": "数据统计",
76
- "textbook": "人教版教材",
77
- "public_kb": "公开知识库",
78
- "other": "人工评审",
79
- }
80
-
81
- CLAIM_FIELDS = ("功能名", "生效条件", "子功能", "执行")
82
- B_CLAIM = "B_valuation"
83
- A_CLAIM = "A_fact"
84
-
85
- # B 型(评价性断言)标记词:只收「明显是价值判断/修饰」的表达,避免把事实误判。
86
- VALUATION_MARKERS = (
87
- "结晶", "瑰宝", "杰作", "卓越", "杰出", "伟大", "不朽", "巅峰", "典范",
88
- "精华", "珍品", "璀璨", "辉煌", "丰碑", "博大精深", "源远流长",
89
- "不可估量", "无与伦比", "举足轻重", "首屈一指", "独树一帜", "别具一格",
90
- "最优秀", "极富", "令人叹为观止", "不可磨灭", "辉煌成就", "灿烂", "崇高",
91
- "非凡",
92
- )
93
- VALUATION_PATTERNS = (
94
- re.compile(r"被誉为|被称[之为]|堪称|不愧[为是]"),
95
- re.compile(r"是[^,。;\n]{0,24}的(结晶|瑰宝|杰作|典范|精华|骄傲|象征|丰碑)"),
96
- )
97
-
98
- CONDITION_MARK = "〔来源限定〕"
99
-
100
-
101
- # ---- 赛道与来源执照 ------------------------------------------------------
102
-
103
- # 生效条件:给定 fm,若显式 track/discipline_type 命中枚举则返回对应赛道;否则用相关元数据与正文 CCG 字段匹配提示词,返回 humanities/science/undetermined。
104
- def classify_track(fm: dict, content: str = "") -> str:
105
- """判定节点赛道:`humanities` / `science` / `undetermined`。
106
-
107
- 只看**已声明**的元数据(显式字段 > 学科标记),不扫正文散文,避免误判。
108
- """
109
- fm = fm or {}
110
- explicit = str(fm.get("track") or fm.get("discipline_type") or "").strip().lower()
111
- if explicit in ("humanities", "文科", "arts"):
112
- return "humanities"
113
- if explicit in ("science", "理科", "stem"):
114
- return "science"
115
- st = fm.get("state_attributes")
116
- name = _as_text(st.get("name")) if isinstance(st, dict) else ""
117
- parts = [
118
- name,
119
- _as_text(fm.get("title")),
120
- _as_text(fm.get("discipline")),
121
- _as_text(fm.get("subject")),
122
- " ".join(str(t) for t in (fm.get("tags") or [])),
123
- _ccg_field(content, "功能名"),
124
- _ccg_field(content, "执行"),
125
- _ccg_field(content, "子功能"),
126
- ]
127
- text = " ".join(p for p in parts if p)
128
- has_sci = any(k in text for k in SCIENCE_SUBJECT_HINTS)
129
- has_hum = any(k in text for k in HUMANITIES_SUBJECT_HINTS)
130
- if has_sci and not has_hum:
131
- return "science"
132
- if has_hum and not has_sci:
133
- return "humanities"
134
- return "undetermined"
135
-
136
-
137
- # 生效条件:给定 track,返回 SOURCE_POLICY 中映射的策略名;未知 track 返回空串。
138
- def source_policy(track: str) -> str:
139
- """赛道 → 来源策略名(空串表示不可判定,应 DEFER)。"""
140
- return SOURCE_POLICY.get(track or "", "")
141
-
142
-
143
- # 生效条件:给定 track,若为 science 返回 REPRODUCIBLE_BASIS,若为 humanities 返回 CONSISTENCY_BASIS,否则返回 ()。
144
- def allowed_basis(track: str) -> tuple:
145
- if track == "science":
146
- return tuple(nodefile.REPRODUCIBLE_BASIS)
147
- if track == "humanities":
148
- return tuple(nodefile.CONSISTENCY_BASIS)
149
- return ()
150
-
151
-
152
- # 生效条件:给定 track 与 basis,当 basis 非空且其字符串形式属于 allowed_basis(track) 时返回 True,否则 False。
153
- def basis_licensed(track: str, basis) -> bool:
154
- """来源执照:理科要可复现证据,文科要来源一致性;赛道未定一律不发放。"""
155
- return bool(basis) and str(basis) in allowed_basis(track)
156
-
157
-
158
- # 生效条件:给定 field,返回 FIELD_NORMALIZE 映射值;未知字段返回空串。
159
- def normalize_field(field) -> str:
160
- return FIELD_NORMALIZE.get(str(field or "").strip(), "")
161
-
162
-
163
- # 生效条件:给定 v,若为 None 返回 [];否则将单值或列表转为去除空白后非空字符串的列表。
164
- def _as_source(v) -> list:
165
- if v is None:
166
- return []
167
- items = list(v) if isinstance(v, (list, tuple)) else [v]
168
- return [str(x).strip() for x in items if str(x).strip()]
169
-
170
-
171
- # ---- B 型识别与条件化改写 ------------------------------------------------
172
-
173
- # 生效条件:给定 text,若含 VALUATION_MARKERS 或匹配 VALUATION_PATTERNS 则返回 B_CLAIM,否则 A_CLAIM。
174
- def claim_type(text) -> str:
175
- """`A_fact`(事实性)或 `B_valuation`(评价性断言)。"""
176
- s = str(text or "")
177
- if not s.strip():
178
- return A_CLAIM
179
- if any(m in s for m in VALUATION_MARKERS):
180
- return B_CLAIM
181
- if any(p.search(s) for p in VALUATION_PATTERNS):
182
- return B_CLAIM
183
- return A_CLAIM
184
-
185
-
186
- # 生效条件:给定 text,返回其去除首尾空白后是否以 CONDITION_MARK 开头。
187
- def is_conditioned(text) -> bool:
188
- return str(text or "").strip().startswith(CONDITION_MARK)
189
-
190
-
191
- # 生效条件:给定 text、label、source,若 text 非空且 label 非空且 source 解析后非空,则返回带 CONDITION_MARK 的来源限定表述;已条件化原样返回;否则 None。
192
- def conditioned_claim(text, label, source):
193
- """把评价性断言改写为**带来源限定的条件表述**;缺来源/标签则返回 `None`(不写)。
194
-
195
- 形态:`〔来源限定〕据<来源标签>(<来源>)的表述:<原文>`——
196
- 原文完整保留(可追溯),前缀显式声明「这是某来源的表述」而非无条件事实。
197
- 已条件化的文本原样返回(幂等)。
198
- """
199
- body = str(text or "").strip()
200
- if not body or not label:
201
- return None
202
- if is_conditioned(body):
203
- return body
204
- src = ";".join(_as_source(source))
205
- if not src:
206
- return None
207
- return f"{CONDITION_MARK}据{label}({src})的表述:{body}"
208
-
209
-
210
- # 生效条件:给定 fm 与 content,提取 CCG 声明字段、comment 值与正文长句,返回断言列表,每项含 text/type/where/field。
211
- def extract_claims(fm: dict, content: str) -> list:
212
- """提取可核对断言:CCG 声明字段 + comment 值 + 正文长句。
213
-
214
- 每条:`{"text", "type", "where", "field"}`;`where` ∈ ccg/comment/body。
215
- 占位标记不成为断言。
216
- """
217
- out, seen = [], set()
218
-
219
- # 生效条件:仅当 str(text or "").strip() 得到的 s 长度 >= 4、s 不在 seen 中、且 nodefile.is_placeholder_text(s) 为假时,把 {text: s, type: claim_type(s), where, field} 追加进 out 并把 s 加入 seen,否则直接返回(field 默认 "")。
220
- def _push(text, where, field=""):
221
- s = str(text or "").strip()
222
- if len(s) < 4 or s in seen or nodefile.is_placeholder_text(s):
223
- return
224
- seen.add(s)
225
- out.append({"text": s, "type": claim_type(s), "where": where,
226
- "field": field})
227
-
228
- for f in CLAIM_FIELDS:
229
- v = _ccg_field(content, f)
230
- if v:
231
- _push(v, "ccg", f)
232
- c = _comment(fm)
233
- for f in CLAIM_FIELDS:
234
- v = c.get(f)
235
- if isinstance(v, list):
236
- for item in v:
237
- _push(item, "comment", f)
238
- elif v:
239
- _push(v, "comment", f)
240
- for line in (content or "").split("\n"):
241
- raw = line.strip()
242
- if not raw:
243
- continue
244
- body = raw.lstrip("#").strip()
245
- if body.split(":", 1)[0].strip() in nodefile.CCG_MARKS:
246
- continue # 声明行已按 CCG 字段处理,不重复断言
247
- for sent in re.split(r"[。!?]", body):
248
- s = sent.strip()
249
- if len(s) >= 8 and not s.startswith(CONDITION_MARK):
250
- _push(s, "body", "")
251
- return out
252
-
253
-
254
- # 生效条件:给定 fm、content、claim、new_text,按 claim.where 定位并在唯一匹配时替换断言返回 (content, True),否则返回 (content, False)。
255
- def _rewrite_claim(fm: dict, content: str, claim: dict, new_text: str):
256
- """节点内定位并替换一条断言 → `(content, ok)`;定位不唯一则 fail-closed 不动。"""
257
- where, field = claim.get("where"), claim.get("field")
258
- before = str(claim.get("text") or "")
259
- if where == "ccg" and field:
260
- if _ccg_field(content, field).strip() == before.strip():
261
- return _upsert_ccg_line(content, field, new_text), True
262
- return content, False
263
- if where == "comment" and field:
264
- c = _comment(fm)
265
- v = c.get(field)
266
- if isinstance(v, list):
267
- if before in v:
268
- c[field] = [new_text if x == before else x for x in v]
269
- return content, True
270
- return content, False
271
- if str(v or "").strip() == before.strip():
272
- c[field] = new_text
273
- return content, True
274
- return content, False
275
- if where == "body":
276
- if before and content.count(before) == 1:
277
- return content.replace(before, new_text), True
278
- return content, False
279
- return content, False
280
-
281
-
282
- # ---- 工单 ----------------------------------------------------------------
283
-
284
- # 生效条件:给定 fm 与 content,返回缺失项列表:verification_basis 无效则加入该名,正文无 "# 验证方式" 行则加入该名。
285
- def _need(fm: dict, content: str) -> list:
286
- need = []
287
- if not nodefile.verification_basis_valid(fm):
288
- need.append("verification_basis")
289
- if not _has_ccg_line(content, "验证方式"):
290
- need.append("验证方式")
291
- return need
292
-
293
-
294
- # 生效条件:给定 nid、e、fm、content,返回含 id、layer、track、claims、need、source_policy 的工单行字典。
295
- def _worklist_row(nid: str, e: dict, fm: dict, content: str) -> dict:
296
- track = classify_track(fm, content)
297
- return {
298
- "id": nid,
299
- "layer": e.get("layer"),
300
- "track": track,
301
- "claims": extract_claims(fm, content),
302
- "need": _need(fm, content),
303
- "source_policy": source_policy(track),
304
- }
305
-
306
-
307
- # 生效条件:给定 fm 与 content,若正文或 comment 中声明的执行字段为占位文本则返回 True;未声明执行时以正文整体占位判定。
308
- def _is_placeholder_shell(fm: dict, content: str) -> bool:
309
- """空壳判定:核心可执行内容未被填充 → 禁止接线(不得把「待填充」固化成事实)。
310
-
311
- 口径(宁漏判不误判):
312
- 1. 已声明 `执行`(正文 `# 执行:` 行优先,其次 `state_attributes.comment.执行`)
313
- 且值为占位标记 → 空壳;
314
- 2. 未声明 `执行` 时,以正文整体是否为空/占位标记为准——无 comment 但正文写实的
315
- `kp_archaeo_*` 类节点因此不被误判为空壳。
316
- """
317
- decl = _ccg_field(content, "执行") or _as_text(_comment(fm).get("执行"))
318
- if decl:
319
- return nodefile.is_placeholder_text(decl)
320
- return nodefile.is_placeholder_text(content)
321
-
322
-
323
- # 生效条件:给定 cg,逐节点按 layer/ids/prefix 过滤后产出状态为 skip(internal/denied/locked/derived/present/placeholder/unreadable 等)或 row 的扫描结果。
324
- def _scan(cg, layer=None, ids=None, prefix=None):
325
- """逐节点产出扫描结果:`{"status", "reason"?, "id", "row"?}`。
326
-
327
- `prefix`:只纳入 id 以该前缀开头的节点(真实库以 `kp_` 收窄到用户知识节点,
328
- 避免 `node_`/`note_`/`imgpart_` 等派生记忆混入工单);**不计数**,与 `layer` 同理。
329
- """
330
- want = set(ids) if ids else None
331
- for nid, e in list((cg.index.get("nodes") or {}).items()):
332
- if want is not None and nid not in want:
333
- continue
334
- if prefix and not str(nid).startswith(prefix):
335
- continue
336
- if layer and e.get("layer") != layer:
337
- continue
338
- if e.get("layer") in INTERNAL_LAYERS:
339
- yield {"status": "skip", "reason": "internal", "id": nid}
340
- continue
341
- if not _readable_guard(cg, e):
342
- yield {"status": "skip", "reason": "denied", "id": nid}
343
- continue
344
- fm, content = cg._read(e)
345
- if fm is None:
346
- yield {"status": "skip", "reason": "unreadable", "id": nid}
347
- continue
348
- if crypto.is_encrypted(content):
349
- yield {"status": "skip", "reason": "locked", "id": nid}
350
- continue
351
- if e.get("layer") in SKIP_LAYERS or any(
352
- t in SKIP_TAGS for t in (fm.get("tags") or [])):
353
- yield {"status": "skip", "reason": "derived", "id": nid}
354
- continue
355
- need = _need(fm, content)
356
- if not need: # 已齐备
357
- yield {"status": "skip", "reason": "present", "id": nid}
358
- continue
359
- ph = []
360
- der = derive_fields(fm, content, placeholder_out=ph)
361
- if _is_placeholder_shell(fm, content): # 空壳:转待填充工单
362
- yield {"status": "skip", "reason": "placeholder", "id": nid,
363
- "placeholder_fields": ph}
364
- continue
365
- yield {"status": "row", "id": nid,
366
- "row": _worklist_row(nid, e, fm, content)}
367
-
368
-
369
- _SKIP_KEY = {"locked": "skipped_locked", "derived": "skipped_derived",
370
- "present": "skipped_present", "denied": "skipped_denied",
371
- "unreadable": "skipped_unreadable", "internal": "skipped_internal"}
372
-
373
-
374
- # 生效条件:给定 x(路径或 MdCGOS),只读扫描并生成缺 verification_basis 或 "# 验证方式" 的节点工单,返回统计 rep。
375
- def build_worklist(x, layer=None, limit=None, ids=None, prefix=None) -> dict:
376
- """生成核对工单(只读):缺 `verification_basis`/`验证方式` 的节点入列。
377
-
378
- `prefix`:按 id 前缀收窄范围(真实库用 `kp_`,与计划交付边界一致);
379
- 内部 `anchor`/`self` 层一律出局(计入 `skipped_internal`)。
380
- """
381
- cg = _as_cg(x)
382
- rep = {"root": cg.root, "dry_run": True, "action": "crosscheck_worklist",
383
- "prefix": prefix, "nodes_scanned": 0, "skipped_locked": 0,
384
- "skipped_derived": 0, "skipped_internal": 0, "skipped_present": 0,
385
- "skipped_denied": 0, "skipped_placeholder": 0,
386
- "skipped_unreadable": 0, "placeholder_ids": [], "undetermined": 0,
387
- "targeted": 0, "items": []}
388
- for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
389
- rep["nodes_scanned"] += 1
390
- if scan["status"] == "skip":
391
- reason = scan["reason"]
392
- if reason == "placeholder":
393
- rep["skipped_placeholder"] += 1
394
- rep["placeholder_ids"].append(scan["id"])
395
- else:
396
- key = _SKIP_KEY.get(reason)
397
- if key:
398
- rep[key] += 1
399
- continue
400
- row = scan["row"]
401
- if row["track"] == "undetermined":
402
- rep["undetermined"] += 1
403
- rep["targeted"] += 1
404
- if limit is None or len(rep["items"]) < limit:
405
- rep["items"].append(row)
406
- rep["planned_ids"] = [r["id"] for r in rep["items"]]
407
- return rep
408
-
409
-
410
- # ---- 子代理接口(提示词 + 解析) -----------------------------------------
411
-
412
- _REFLECT_TEMPLATE = """你是认知图节点的**反思单元**(reflect)。为节点补齐「验证方式」与其验证基底。
413
- 赛道:{track};来源策略:{policy};待补字段:{need}
414
- 节点标题:{title}
415
- 已声明断言:
416
- {claims}
417
- 正文:
418
- {body}
419
-
420
- 只输出 JSON 数组,元素形如:
421
- {{"field":"验证方式","value":"<一句可核对的验证方式声明>","basis":"<基底枚举>","source":["<教材版本+章节 或 公开知识库条目地址>"],"verdict":"accept|defer","reason":"<理由>"}}
422
- 硬约束:
423
- 1. 文科(humanities)basis 只能是 textbook / public_kb;
424
- 2. 理科(science)basis 只能是 compiler / test / measurement / formal_proof / data;
425
- 3. 来源必须可追溯(教材名称+章节,或公开知识库条目地址);拿不出来就把 verdict 置 defer、source 留空;
426
- 4. 只能补 field=验证方式,禁止新增其它字段。"""
427
-
428
- _VERIFY_TEMPLATE = """你是独立**验证单元**(verify)。对下列候选逐条复核:来源是否真实可追溯、基底是否与赛道相容。
429
- 赛道:{track};来源策略:{policy}
430
- 候选(JSON):
431
- {candidates}
432
-
433
- 只输出 JSON 数组,元素形如:
434
- {{"field":"验证方式","value":"<原样回填候选 value>","verdict":"accept|drop|defer","reason":"<理由>"}}
435
- 硬约束:你只能否决(drop)或存疑(defer),**不得新增候选、不得改写 value**。"""
436
-
437
-
438
- # 生效条件:给定 row、fm、content,用 row 的 track/source_policy/need/claims 与 fm 标题、content 前 1200 字符填充反思模板并返回字符串。
439
- def reflect_prompt(row: dict, fm: dict, content: str) -> str:
440
- claims = "\n".join(f"- [{c['type']}] {c['text']}" for c in (row.get("claims") or []))
441
- return _REFLECT_TEMPLATE.format(
442
- track=row.get("track"), policy=row.get("source_policy") or "(未定)",
443
- need="、".join(row.get("need") or []),
444
- title=_as_text(fm.get("title")) or row.get("id"),
445
- claims=claims or "(无)",
446
- body=(content or "")[:1200])
447
-
448
-
449
- # 生效条件:给定 row 与 rows,把候选字段、值、依据、来源序列化为 JSON 并填充验证模板返回字符串。
450
- def verify_prompt(row: dict, rows: list) -> str:
451
- cands = [{"field": r.get("field"), "value": r.get("value"),
452
- "basis": r.get("basis"), "source": r.get("source")} for r in rows]
453
- return _VERIFY_TEMPLATE.format(
454
- track=row.get("track"), policy=row.get("source_policy") or "(未定)",
455
- candidates=json.dumps(cands, ensure_ascii=False))
456
-
457
-
458
- # 生效条件:raw 经 str(raw or "") 得 s 后,want_list 为真时先试 s 首个 "[" 至末个 "]"、再试首个 "{" 至末个 "}"(want_list 假值时只试花括号),区间可被 json.loads 解析且结果为 list 时原样返回该 list;结果为 dict 时按 rows/items/verdicts/candidates/data 顺序取首个 obj.get(key) 为 list 的 obj[key],都不满足则返回 [obj],非 list/dict 或区间缺失、解析抛 ValueError 时继续下一组括号,全部落空(含 raw 为假值使 s 为空串)返回 []。
459
- def _extract_json(raw, want_list=True):
460
- """从模型输出里抽取 JSON(容忍代码围栏与前后废话)。"""
461
- s = str(raw or "")
462
- pairs = ([("[", "]")] if want_list else []) + [("{", "}")]
463
- for op, cl in pairs:
464
- i, j = s.find(op), s.rfind(cl)
465
- if i < 0 or j <= i:
466
- continue
467
- try:
468
- obj = json.loads(s[i:j + 1])
469
- except ValueError:
470
- continue
471
- if isinstance(obj, list):
472
- return obj
473
- if isinstance(obj, dict):
474
- for key in ("rows", "items", "verdicts", "candidates", "data"):
475
- if isinstance(obj.get(key), list):
476
- return obj[key]
477
- return [obj]
478
- return []
479
-
480
-
481
- # 生效条件:给定 item,若为 dict 则规范化 field/value/basis/source/verdict/reason 后返回字典,否则返回 {}。
482
- def _norm_row(item) -> dict:
483
- if not isinstance(item, dict):
484
- return {}
485
- return {
486
- "field": normalize_field(item.get("field")) or str(item.get("field") or "").strip(),
487
- "value": _as_text(item.get("value")),
488
- "basis": str(item.get("basis") or "").strip(),
489
- "source": _as_source(item.get("source")),
490
- "verdict": str(item.get("verdict") or "").strip().lower(),
491
- "reason": str(item.get("reason") or "").strip(),
492
- }
493
-
494
-
495
- # 生效条件:给定 raw,解析 JSON 行并保留 field 为“验证方式”或“verification_basis”(统一为“验证方式”)的行,返回列表。
496
- def parse_reflect_rows(raw) -> list:
497
- out = []
498
- for item in _extract_json(raw, want_list=True):
499
- r = _norm_row(item)
500
- if not r or r["field"] not in ("验证方式", "verification_basis"):
501
- continue
502
- r["field"] = "验证方式"
503
- out.append(r)
504
- return out
505
-
506
-
507
- # 生效条件:遍历 _extract_json(raw, want_list=False)(只认花括号 JSON)的结果,仅当 item 经 _norm_row 后为真且 r["field"] 非空时产出 {field,value,verdict,reason} 四键行,否则跳过(raw 无可解析花括号对象时 out 为空列表)。
508
- def parse_verify_rows(raw) -> list:
509
- out = []
510
- for item in _extract_json(raw, want_list=False):
511
- r = _norm_row(item)
512
- if not r or not r["field"]:
513
- continue
514
- out.append({k: r[k] for k in ("field", "value", "verdict", "reason")})
515
- return out
516
-
517
-
518
- # ---- 白箱闸门与双单元折叠 ------------------------------------------------
519
-
520
- # 生效条件:逐行处理 rows,仅当 normalize_field(r.get("field")) 落在 WRITABLE_FIELDS、basis_licensed(track, r.get("basis")) 为真、r.get("source") 为真、且 r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "") 非空时进入 kept(附 verdict="accept"),否则该行带对应 reason 进入 gated。
521
- def gate_rows(rows: list, track: str) -> tuple:
522
- """零模型白箱闸门:字段越界 / 来源执照不通过 / 无来源 → 一律降级为 defer。
523
-
524
- 返回 `(kept, gated)`;`kept` 只含「执照齐全」的候选,可进验证单元。
525
- """
526
- kept, gated = [], []
527
- for r in rows:
528
- f = normalize_field(r.get("field"))
529
- row = dict(r, field=f)
530
- if f not in WRITABLE_FIELDS:
531
- gated.append(dict(row, reason=f"越界字段:{r.get('field')}"))
532
- continue
533
- if not basis_licensed(track, r.get("basis")):
534
- gated.append(dict(row, reason=f"{track or '未定赛道'} 不接受基底 {r.get('basis') or '(缺)'}"))
535
- continue
536
- if not r.get("source"):
537
- gated.append(dict(row, reason="无来源,不写"))
538
- continue
539
- value = r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "")
540
- if not value:
541
- gated.append(dict(row, reason="无验证方式声明"))
542
- continue
543
- kept.append(dict(row, value=value, verdict="accept"))
544
- return kept, gated
545
-
546
-
547
- # 生效条件:按 (r.get("id"), normalize_field(r.get("field")) or r.get("field"), r.get("value")) 分组后,组内缺 unit==REFLECT_UNIT 或 unit==VERIFY_UNIT 的行时进 deferred,否则 verify 侧出现 verdict=="drop" 即进 dropped(veto 优先),再否则仅当 reflect 与 verify 各存在 verdict=="accept" 时才进 accepted(附 units),其余进 deferred。
548
- def fold_verdicts(rows: list) -> tuple:
549
- """把两单元裁决折叠为可落库结论 → `(accepted, deferred, dropped)`。
550
-
551
- 接受条件:同 `(id, field, value)` 同时存在 reflect-accept 与 verify-accept;
552
- 任一 verify-drop 即否决(veto 优先)。
553
- """
554
- groups = OrderedDict()
555
- for r in rows:
556
- key = (r.get("id"), normalize_field(r.get("field")) or r.get("field"),
557
- r.get("value"))
558
- groups.setdefault(key, []).append(r)
559
- accepted, deferred, dropped = [], [], []
560
- for (nid, field, value), rs in groups.items():
561
- refl = [r for r in rs if r.get("unit") == REFLECT_UNIT]
562
- ver = [r for r in rs if r.get("unit") == VERIFY_UNIT]
563
- base = {"id": nid, "field": field or "验证方式", "value": value,
564
- "basis": next((r.get("basis") for r in refl if r.get("basis")), ""),
565
- "source": next((r.get("source") for r in refl if r.get("source")), []),
566
- "track": next((r.get("track") for r in refl if r.get("track")), "")}
567
- if not refl or not ver:
568
- base["reason"] = "缺" + ("反思裁决" if not refl else "验证裁决")
569
- deferred.append(base)
570
- continue
571
- if any(r.get("verdict") == "drop" for r in ver):
572
- base["reason"] = next((r.get("reason") for r in ver
573
- if r.get("verdict") == "drop"), "验证单元否决")
574
- dropped.append(base)
575
- continue
576
- ra = any(r.get("verdict") == "accept" for r in refl)
577
- va = any(r.get("verdict") == "accept" for r in ver)
578
- if ra and va:
579
- base["units"] = {
580
- "reflect": sorted({str(r.get("actor") or REFLECT_UNIT) for r in refl}),
581
- "verify": sorted({str(r.get("actor") or VERIFY_UNIT) for r in ver}),
582
- }
583
- accepted.append(base)
584
- else:
585
- base["reason"] = f"单元未确认(reflect={ra}, verify={va})"
586
- deferred.append(base)
587
- return accepted, deferred, dropped
588
-
589
-
590
- # 生效条件:当 rows 中 unit==REFLECT_UNIT 与 unit==VERIFY_UNIT 的执行者经 str(x.get("actor") or "") 后存在相同的非空值(空串被 discard)时返回 True,否则返回 False。
591
- def detect_self_verify(rows: list) -> bool:
592
- """同一执行者同时充当反思与验证 = 自证(禁止)。"""
593
- r = {str(x.get("actor") or "") for x in rows if x.get("unit") == REFLECT_UNIT}
594
- v = {str(x.get("actor") or "") for x in rows if x.get("unit") == VERIFY_UNIT}
595
- r.discard("")
596
- v.discard("")
597
- return bool(r & v)
598
-
599
-
600
- # ---- 落库写入 ------------------------------------------------------------
601
-
602
- # 生效条件:verdicts 为 None 时返回 None;verdicts 为 dict 时对每个键值把 (rs or []) 中的 dict 元素收为 {str(nid): [...]};否则遍历 verdicts or [],仅当元素为 dict 且 str(r.get("id") or "") 非空时按该 id 追加到对应列表。
603
- def _norm_verdicts(verdicts):
604
- """外部裁决(子代理落盘)→ `{id: [rows]}`。"""
605
- if verdicts is None:
606
- return None
607
- out = OrderedDict()
608
- if isinstance(verdicts, dict):
609
- for nid, rs in verdicts.items():
610
- out[str(nid)] = [dict(x) for x in (rs or []) if isinstance(x, dict)]
611
- return out
612
- for r in verdicts or []:
613
- if not isinstance(r, dict):
614
- continue
615
- nid = str(r.get("id") or "")
616
- if nid:
617
- out.setdefault(nid, []).append(dict(r))
618
- return out
619
-
620
-
621
- # 生效条件:accepted 非空(取 accepted[0])时,先以 basis=str(a.get("basis") or BASIS_ENUM_DEFAULT) 与 value=str(a.get("value") or BASIS_TEXT.get(basis, "")).strip() 写「验证方式」行与 comment,之后才在 nodefile.verification_basis_valid(fm) 为真时把 basis 换成 fm.get("verification_basis")、否则把该 basis 写入 fm["verification_basis"];condition_claims 为真时仅对 row.get("claims") 中 type==B_CLAIM 且未被 is_conditioned 的条目做条件化改写,返回含 fm_before、content_hash_before 等留痕的 dict。
622
- def _apply_node(cg, nid, e, fm, content, accepted, row, batch, actor,
623
- condition_claims=True):
624
- """把一个节点的已接受结论写入 md,返回留痕记录(含回滚所需现场)。"""
625
- a = accepted[0]
626
- basis = str(a.get("basis") or BASIS_ENUM_DEFAULT)
627
- source = _as_source(a.get("source"))
628
- content_before = content
629
- value = str(a.get("value") or BASIS_TEXT.get(basis, "")).strip()
630
-
631
- vb_before = {"had": "verification_basis" in fm,
632
- "value": fm.get("verification_basis")}
633
- prov_before = {"had": "verification_evidence" in fm,
634
- "value": fm.get("verification_evidence")}
635
- line_before = {"had": _has_ccg_line(content, "验证方式"),
636
- "value": _ccg_field(content, "验证方式")}
637
- comment0 = _comment(fm)
638
- cv_before = {"had": "验证方式" in comment0, "value": comment0.get("验证方式")}
639
-
640
- # 1) 落「验证方式」规范行 + comment + verification_basis(已有合法基底不覆盖)
641
- content = _upsert_ccg_line(content, "验证方式", value)
642
- _ensure_comment(fm)["验证方式"] = value
643
- if nodefile.verification_basis_valid(fm):
644
- basis = str(fm.get("verification_basis"))
645
- else:
646
- fm["verification_basis"] = basis
647
-
648
- # 2) B 型评价断言 → 条件化表述(A 型保持原样;无来源已由闸门挡掉)
649
- conditioned = []
650
- if condition_claims:
651
- label = BASIS_LABEL.get(basis, basis)
652
- for c in row.get("claims") or []:
653
- if c.get("type") != B_CLAIM or is_conditioned(c.get("text")):
654
- continue
655
- new = conditioned_claim(c.get("text"), label, source)
656
- if not new or new == c.get("text"):
657
- continue
658
- nc, ok = _rewrite_claim(fm, content, c, new)
659
- if not ok:
660
- continue
661
- content = nc
662
- conditioned.append({"where": c.get("where"), "field": c.get("field") or "",
663
- "before": c.get("text"), "after": new})
664
-
665
- wid = _sha(f"{nid}|{batch}|{time.time()}")
666
- fm["verification_evidence"] = {
667
- "at": round(time.time(), 3), "batch": batch, "basis": basis,
668
- "source": source, "track": row.get("track"),
669
- "policy": row.get("source_policy"), "write_id": wid,
670
- "units": a.get("units") or {},
671
- "conditioned": len(conditioned),
672
- }
673
-
674
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
675
- durable=True)
676
- return {
677
- "action": "crosscheck", "ts": time.time(), "batch": batch,
678
- "actor": actor, "entry_id": _entry_id(batch, nid), "write_id": wid,
679
- "node": nid, "layer": e.get("layer"), "track": row.get("track"),
680
- "policy": row.get("source_policy"), "basis": basis,
681
- "verification_value": value, "source": source,
682
- "units": a.get("units") or {},
683
- "fm_before": {"verification_basis": vb_before,
684
- "verification_evidence": prov_before,
685
- "comment_verification": cv_before},
686
- "verification_line_before": line_before,
687
- "claims_conditioned": conditioned,
688
- "content_hash_before": _sha(content_before),
689
- "content_hash_after": _sha(content),
690
- }
691
-
692
-
693
- # ---- 主流程 --------------------------------------------------------------
694
-
695
- # 生效条件:reflect_fn(reflect_prompt(scan_row, fm, content)) 经 parse_reflect_rows 得到非空候选时返回 (rrows, vrows),rrows 为空则返回 ([], []);verify_fn 为 None 时 vrows 为空列表,非 None 时由 parse_verify_rows(verify_fn(verify_prompt(scan_row, rrows))) 生成、每行 value 为 c.get("value") or rrows 中同 field 的 value、再回落 ""。
696
- def _rows_for(scan_row, fm, content, reflect_fn, verify_fn, r_actor, v_actor):
697
- """调用两单元子代理,返回合并后的裁决行(reflect + verify)。"""
698
- prompt = reflect_prompt(scan_row, fm, content)
699
- cands = parse_reflect_rows(reflect_fn(prompt))
700
- rrows = [dict(c, id=scan_row["id"], unit=REFLECT_UNIT, actor=r_actor,
701
- track=scan_row["track"]) for c in cands]
702
- if not rrows:
703
- return [], []
704
- vrows = []
705
- if verify_fn is not None:
706
- vp = verify_prompt(scan_row, rrows)
707
- vrows = [dict(c, id=scan_row["id"], unit=VERIFY_UNIT, actor=v_actor,
708
- track=scan_row["track"],
709
- value=c.get("value") or next(
710
- (x.get("value") for x in rrows
711
- if x.get("field") == c.get("field")), ""))
712
- for c in parse_verify_rows(verify_fn(vp))]
713
- return rrows, vrows
714
-
715
-
716
- # 生效条件:对 pre 中每个 r,str(r.get("unit") or REFLECT_UNIT).strip().lower() 等于 VERIFY_UNIT 时进 vrows,否则(含 unit 缺失回落到 REFLECT_UNIT 及任何其他取值)进 rrows,两组行均覆盖 id=nid、unit、track=track。
717
- def _rows_from_verdicts(nid, track, pre):
718
- """从外部裁决中拆出 (reflect, verify) 两组行。"""
719
- rrows, vrows = [], []
720
- for r in pre:
721
- unit = str(r.get("unit") or REFLECT_UNIT).strip().lower()
722
- base = dict(r, id=nid, unit=unit, track=track)
723
- (vrows if unit == VERIFY_UNIT else rrows).append(base)
724
- return rrows, vrows
725
-
726
-
727
- # 生效条件:x 经 _as_cg 解析且 batch = batch or CROSSCHECK_BATCH 后逐节点扫描,裁决来源按 vmap(verdicts 归一化后非 None)→ reflect_fn 非 None → 二者皆无记 no_reflect 三条分支取行;allow_self_verify=False 时同执行者自证记 self_verify_disallowed,再经 gate_rows 闸门与 require_verify 后 fold_verdicts,仅 apply=True 才 _apply_node 写盘并在有写入时 cg.rebuild_index;limit 非 None 且已达标数 >= limit 时用 continue 跳过(非终止)。
728
- def crosscheck(x, layer=None, limit=None, ids=None, reflect_fn=None,
729
- verify_fn=None, verdicts=None, apply=False,
730
- batch=CROSSCHECK_BATCH, actor=None, require_verify=True,
731
- allow_self_verify=False, reflect_actor=None, verify_actor=None,
732
- condition_claims=True, verbose=True, prefix=None) -> dict:
733
- """批量核对主流程:工单 → 反思候选 → 白箱闸门 → 验证否决 → 落库。
734
-
735
- `reflect_fn`/`verify_fn`:可注入的子代理函数(接收提示词、返回 JSON 文本);
736
- `verdicts`:子代理离线产出的裁决行(`[VERDICT_ROW]` 或 `{id: [rows]}`),
737
- 二选一。`apply=True` 才写盘。
738
- """
739
- cg = _as_cg(x)
740
- batch = batch or CROSSCHECK_BATCH
741
- vmap = _norm_verdicts(verdicts)
742
- r_actor = reflect_actor or getattr(reflect_fn, "__name__", "") or REFLECT_UNIT
743
- v_actor = verify_actor or getattr(verify_fn, "__name__", "") or VERIFY_UNIT
744
-
745
- rep = {"root": cg.root, "dry_run": not apply, "action": "crosscheck",
746
- "batch": batch, "actor": actor, "prefix": prefix, "nodes_scanned": 0,
747
- "targeted": 0,
748
- "accepted": 0, "rejected": 0, "deferred": 0, "written": 0,
749
- "claims_conditioned": 0, "skipped_locked": 0, "skipped_derived": 0,
750
- "skipped_internal": 0, "skipped_present": 0, "skipped_denied": 0,
751
- "skipped_unreadable": 0, "skipped_placeholder": 0,
752
- "placeholder_ids": [], "undetermined": 0,
753
- "reasons": {}, "samples": [], "entry_ids": []}
754
-
755
- # 生效条件:无条件执行 rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1(reason 缺键时按 .get 的第二参数 0 起算),返回 None。
756
- def _bump(reason):
757
- rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
758
-
759
- # 生效条件:仅当外层 verbose 为真且 len(rep["samples"]) < 20 时把 {kind, id: nid, detail} 追加进 rep["samples"],否则不追加(已达 20 条即停止采样)。
760
- def _sample(kind, nid, detail=""):
761
- if verbose and len(rep["samples"]) < 20:
762
- rep["samples"].append({"kind": kind, "id": nid, "detail": detail})
763
-
764
- seen_targets = 0
765
- for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
766
- rep["nodes_scanned"] += 1
767
- if scan["status"] == "skip":
768
- reason = scan["reason"]
769
- if reason == "placeholder":
770
- rep["skipped_placeholder"] += 1
771
- rep["placeholder_ids"].append(scan["id"])
772
- else:
773
- key = _SKIP_KEY.get(reason)
774
- if key:
775
- rep[key] += 1
776
- continue
777
- row = scan["row"]
778
- if row["track"] == "undetermined":
779
- rep["undetermined"] += 1
780
- if limit is not None and seen_targets >= limit:
781
- continue
782
- seen_targets += 1
783
- rep["targeted"] += 1
784
- nid = row["id"]
785
- e = cg.index["nodes"].get(nid)
786
- fm, content = cg._read(e) if e else (None, None)
787
- if fm is None or crypto.is_encrypted(content):
788
- rep["skipped_locked"] += 1
789
- continue
790
-
791
- # ---- 取两单元裁决 ----
792
- if vmap is not None:
793
- pre = vmap.get(nid)
794
- if not pre:
795
- rep["deferred"] += 1
796
- _bump("no_verdict")
797
- continue
798
- rrows, vrows = _rows_from_verdicts(nid, row["track"], pre)
799
- elif reflect_fn is not None:
800
- try:
801
- rrows, vrows = _rows_for(row, fm, content, reflect_fn,
802
- verify_fn, r_actor, v_actor)
803
- except Exception as exc: # noqa: BLE001
804
- rep["deferred"] += 1
805
- _bump(f"unit_error:{type(exc).__name__}")
806
- continue
807
- else:
808
- rep["deferred"] += 1
809
- _bump("no_reflect")
810
- continue
811
-
812
- # 自证:同一执行者既反思又验证 → 拒收
813
- if not allow_self_verify and detect_self_verify(rrows + vrows):
814
- rep["deferred"] += 1
815
- _bump("self_verify_disallowed")
816
- _sample("self_verify", nid)
817
- continue
818
-
819
- # ---- 白箱闸门(对反思候选;验证行只做字段归位) ----
820
- kept, gated = gate_rows(rrows, row["track"])
821
- for g in gated:
822
- _bump(f"gate:{g.get('reason')[:24]}")
823
- if not kept:
824
- rep["deferred"] += 1
825
- _bump("no_candidate")
826
- _sample("gated", nid, gated[0].get("reason") if gated else "")
827
- continue
828
- if require_verify and not vrows:
829
- rep["deferred"] += 1
830
- _bump("verify_unavailable")
831
- continue
832
-
833
- rows = kept + vrows
834
- accepted, deferred, dropped = fold_verdicts(rows)
835
- if dropped and not accepted:
836
- rep["rejected"] += 1
837
- _bump("verify_veto")
838
- _sample("veto", nid, dropped[0].get("reason", ""))
839
- continue
840
- if not accepted:
841
- rep["deferred"] += 1
842
- _bump("verdict_deferred")
843
- _sample("deferred", nid, deferred[0].get("reason", "") if deferred else "")
844
- continue
845
-
846
- rep["accepted"] += 1
847
- rep["claims_conditioned"] += sum(
848
- 1 for c in (row.get("claims") or []) if c.get("type") == B_CLAIM)
849
- if apply:
850
- rec = _apply_node(cg, nid, e, fm, content, accepted, row, batch,
851
- actor, condition_claims=condition_claims)
852
- append_jsonl(_log_path(cg), rec)
853
- rep["written"] += 1
854
- rep["entry_ids"].append(rec["entry_id"])
855
- _sample("accepted", nid, accepted[0].get("basis", ""))
856
-
857
- if rep["written"]:
858
- cg.rebuild_index()
859
- return rep
860
-
861
-
862
- # ---- 留痕查询 / 回滚 -----------------------------------------------------
863
-
864
- # 生效条件:无条件返回 os.path.join(cg.root, CROSSCHECK_LOG)(以 cg.root 与常量 CROSSCHECK_LOG 拼接,无分支)。
865
- def _log_path(cg) -> str:
866
- return os.path.join(cg.root, CROSSCHECK_LOG)
867
-
868
-
869
- # 生效条件:box 非 dict 时返回 False;box 为 dict 且 key=="comment_verification" 时按 box.get("had") 为真则把 comment 的「验证方式」设为 box.get("value")、否则删除该键并返回 True;其他 key 时 had 为真赋 fm[key]=value、否则 fm.pop(key, None) 并返回 True。
870
- def _reattach(fm: dict, content: str, box: dict, key: str):
871
- """把 `fm_before[key]` 现场还原到 fm,返回是否发生还原。"""
872
- if not isinstance(box, dict):
873
- return False
874
- had, value = box.get("had"), box.get("value")
875
- if key == "comment_verification":
876
- c = _ensure_comment(fm)
877
- if had:
878
- c["验证方式"] = value
879
- else:
880
- c.pop("验证方式", None)
881
- return True
882
- if had:
883
- fm[key] = value
884
- else:
885
- fm.pop(key, None)
886
- return True
887
-
888
-
889
- # 生效条件:仅当 str(c.get("after") or "") 非空,且分别满足 where=="ccg" 且 field 真值且 _ccg_field(content, field).strip()==after.strip()(用 before 覆盖该行)、where=="comment" 且 field 真值且 comment 该 field 为含 after 的 list 或 str(v or "").strip()==after.strip()(改为 before)、where=="body" 且 after 出现在 content 中(替换首个匹配)时返回 (content, True);其余情形(含 where 为其他值、字段缺失、当前值不等于写入值)返回 (content, False)。
890
- def _rewind_claim(fm: dict, content: str, c: dict):
891
- """撤销一条条件化改写(仅当前值 == 写入值时才动)→ `(content, ok)`。"""
892
- where, field = c.get("where"), c.get("field")
893
- before, after = str(c.get("before") or ""), str(c.get("after") or "")
894
- if not after:
895
- return content, False
896
- if where == "ccg" and field:
897
- if _ccg_field(content, field).strip() == after.strip():
898
- return _upsert_ccg_line(content, field, before), True
899
- return content, False
900
- if where == "comment" and field:
901
- cc = _comment(fm)
902
- v = cc.get(field)
903
- if isinstance(v, list):
904
- if after in v:
905
- cc[field] = [before if x == after else x for x in v]
906
- return content, True
907
- return content, False
908
- if str(v or "").strip() == after.strip():
909
- cc[field] = before
910
- return content, True
911
- return content, False
912
- if where == "body":
913
- if after in content:
914
- return content.replace(after, before, 1), True
915
- return content, False
916
- return content, False
917
-
918
-
919
- # 生效条件:x 经 _as_cg 后,对 read_jsonl(_log_path(cg)) 中 action=="crosscheck"、batch 为 None 或等于参数 batch、且 entry_ids 为假值不做 id 过滤(为真值时仅取 entry_id 在集合中的)的记录逐条处理:node 缺失或已处理则跳过,索引无该 node 或 cg._read 得 fm 为 None 或 crypto.is_encrypted(content) 为真时 skipped_drift 加一,write_id 双方非空且不等时 conflict 加一,否则撤销 claims_conditioned、在当前「验证方式」行非空且等于 rec 的 verification_value 时撤销该行、再按 fm_before 还原,reverted 为空则 conflict 加一,非空则写回节点、追加 crosscheck_rollback 日志、reverted 与 entry_ids 加一,最终 reverted 非零时 cg.rebuild_index(),返回 rep;
920
- def rollback(x, batch=None, entry_ids=None, actor=None) -> dict:
921
- """按留痕反向应用:撤销核对写入(当前值 ≠ 写入值时跳过,计入 conflict)。"""
922
- cg = _as_cg(x)
923
- want = set(entry_ids) if entry_ids else None
924
- done = set()
925
- rep = {"root": cg.root, "dry_run": False, "action": "crosscheck_rollback",
926
- "batch": batch, "actor": actor, "planned": 0, "reverted": 0,
927
- "skipped_drift": 0, "conflict": 0, "entry_ids": []}
928
- recs = [r for r in (read_jsonl(_log_path(cg)) or [])
929
- if r.get("action") == "crosscheck"
930
- and (batch is None or r.get("batch") == batch)
931
- and (want is None or r.get("entry_id") in want)]
932
- rep["planned"] = len(recs)
933
- for rec in recs:
934
- nid = rec.get("node")
935
- if not nid or nid in done:
936
- continue
937
- e = cg.index["nodes"].get(nid)
938
- if not e:
939
- rep["skipped_drift"] += 1
940
- continue
941
- fm, content = cg._read(e)
942
- if fm is None or crypto.is_encrypted(content):
943
- rep["skipped_drift"] += 1
944
- continue
945
- # 写入现场校验:write_id 一致才回滚(防「写入后又被改过」被误撤)
946
- wid = (fm.get("verification_evidence") or {}).get("write_id")
947
- if wid and rec.get("write_id") and wid != rec.get("write_id"):
948
- rep["conflict"] += 1
949
- continue
950
- # 1) 撤销条件化改写(先于验证方式行,避免行被覆盖影响定位)
951
- reverted = []
952
- for c in rec.get("claims_conditioned") or []:
953
- content, ok = _rewind_claim(fm, content, c)
954
- if ok:
955
- reverted.append(c.get("field") or c.get("where"))
956
- # 2) 撤销「验证方式」行
957
- lb = rec.get("verification_line_before") or {}
958
- cur_line = _ccg_field(content, "验证方式")
959
- if cur_line.strip() and cur_line.strip() == str(
960
- rec.get("verification_value") or "").strip():
961
- content = (_upsert_ccg_line(content, "验证方式", lb.get("value") or "")
962
- if lb.get("had") else _remove_ccg_line(content, "验证方式"))
963
- reverted.append("验证方式")
964
- # 3) 还原 frontmatter 现场
965
- for key, box in (rec.get("fm_before") or {}).items():
966
- _reattach(fm, content, box, key)
967
- if not reverted:
968
- rep["conflict"] += 1
969
- continue
970
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
971
- durable=True)
972
- append_jsonl(_log_path(cg), {
973
- "action": "crosscheck_rollback", "ts": time.time(),
974
- "batch": rec.get("batch"), "actor": actor,
975
- "entry_id": rec.get("entry_id"), "node": nid,
976
- "reverted": reverted, "content_hash_after": _sha(content)})
977
- rep["reverted"] += 1
978
- rep["entry_ids"].append(rec.get("entry_id"))
979
- done.add(nid)
980
- if rep["reverted"]:
981
- cg.rebuild_index()
982
- return rep
983
-
984
-
985
- # 生效条件:遍历 _log_path(cg) 的记录时,action 为真值只留 rec.get("action")==action 的行、batch 为真值只留 rec.get("batch")==batch 的行;limit 非 None 且 limit>=0 时按 recs[-limit:] 截取(limit 为 0 时 [-0:] 即整表不被削减),否则保留全部;返回 {'root','total','returned','records'}。
986
- def history(x, limit=100, action=None, batch=None) -> dict:
987
- cg = _as_cg(x)
988
- recs = []
989
- for rec in read_jsonl(_log_path(cg)) or []:
990
- if action and rec.get("action") != action:
991
- continue
992
- if batch and rec.get("batch") != batch:
993
- continue
994
- recs.append(rec)
995
- total = len(recs)
996
- if limit is not None and limit >= 0:
997
- recs = recs[-limit:]
998
- return {"root": cg.root, "total": total, "returned": len(recs),
999
- "records": recs}
1000
-
1001
-
1002
- # ---- 权限与 CLI ----------------------------------------------------------
1003
-
1004
- # 生效条件:principal 为 None 时返回 False;否则仅当 principal.expired() 为假、principal.can_write 为真、且 principal.allows_layer("knowledge") 为真时返回 True,期间任一步抛 Exception 亦返回 False。
1005
- def can_write_knowledge(principal) -> bool:
1006
- """落 knowledge 层必须持有可写该层的令牌(designer 派生);否则 fail-closed。"""
1007
- if principal is None:
1008
- return False
1009
- try:
1010
- if principal.expired() or not principal.can_write:
1011
- return False
1012
- return bool(principal.allows_layer("knowledge"))
1013
- except Exception: # noqa: BLE001
1014
- return False
1015
-
1016
-
1017
- # 生效条件:path 为假值(空串/None)返回 None;path 不存在则 raise SystemExit;已存在且读取文本 strip 后为空串返回 [],非空时整段 json.loads 成功即返回该值,抛 ValueError 时按行解析(跳过空行与 "//" 开头行)返回行列表。
1018
- def _load_verdicts(path: str):
1019
- if not path:
1020
- return None
1021
- if not os.path.exists(path):
1022
- raise SystemExit(f"裁决文件不存在:{path}")
1023
- with open(path, "r", encoding="utf-8") as fh:
1024
- text = fh.read().strip()
1025
- if not text:
1026
- return []
1027
- try:
1028
- return json.loads(text)
1029
- except ValueError:
1030
- rows = []
1031
- for line in text.splitlines():
1032
- line = line.strip()
1033
- if not line or line.startswith("//"):
1034
- continue
1035
- rows.append(json.loads(line))
1036
- return rows
1037
-
1038
-
1039
- # 生效条件:argv(为 None 时由 argparse 读 sys.argv)解析后按 --action 分派——worklist 调 build_worklist,history 调 history(--limit 默认 None,为 None 时传 100),rollback 在 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 rollback,crosscheck 在 --apply 为真且 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 crosscheck;--token(默认 os.environ.get("MDCG_TOKEN") or "")为真值时先 tokens.verify_token 校验、失败抛 SystemExit;最后打印 rep 并返回 0;
1040
- def _cli(argv=None) -> int:
1041
- ap = argparse.ArgumentParser(
1042
- prog="python -m md_cg.crosscheck",
1043
- description="kp_ 批量核对管线(工单/核对/回滚/留痕)")
1044
- ap.add_argument("--root", default=os.environ.get("MDCG_ROOT") or ".")
1045
- ap.add_argument("--token", default=os.environ.get("MDCG_TOKEN") or "")
1046
- ap.add_argument("--token-file", default=None)
1047
- ap.add_argument("--action", default="worklist",
1048
- choices=("worklist", "crosscheck", "rollback", "history"))
1049
- ap.add_argument("--verdicts", default="", help="子代理裁决 JSON/JSONL 路径")
1050
- ap.add_argument("--apply", action="store_true", help="真正写盘(默认 dry-run)")
1051
- ap.add_argument("--batch", default=CROSSCHECK_BATCH)
1052
- ap.add_argument("--limit", type=int, default=None)
1053
- ap.add_argument("--layer", default=None)
1054
- ap.add_argument("--prefix", default=None, help="按 id 前缀收窄(真实库用 kp_)")
1055
- ap.add_argument("--ids", default="", help="逗号分隔节点 id")
1056
- ap.add_argument("--no-verify", action="store_true", help="允许无验证单元(不建议)")
1057
- ap.add_argument("--allow-self-verify", action="store_true")
1058
- ap.add_argument("--reflect-actor", default=None)
1059
- ap.add_argument("--verify-actor", default=None)
1060
- args = ap.parse_args(argv)
1061
-
1062
- from . import tokens
1063
- principal = None
1064
- if args.token:
1065
- try:
1066
- principal = tokens.verify_token(args.token, path=args.token_file)
1067
- except tokens.TokenError as exc:
1068
- raise SystemExit(f"令牌校验失败:{exc}")
1069
- actor = getattr(principal, "actor", None)
1070
- ids = [s.strip() for s in args.ids.split(",") if s.strip()] or None
1071
-
1072
- if args.action == "worklist":
1073
- rep = build_worklist(args.root, layer=args.layer, limit=args.limit,
1074
- ids=ids, prefix=args.prefix)
1075
- elif args.action == "history":
1076
- rep = history(args.root, limit=args.limit if args.limit is not None else 100,
1077
- batch=args.batch)
1078
- elif args.action == "rollback":
1079
- if not can_write_knowledge(principal):
1080
- raise SystemExit("权限不足:回滚需要可写 knowledge 层的令牌")
1081
- rep = rollback(args.root, batch=args.batch, actor=actor)
1082
- else:
1083
- if args.apply and not can_write_knowledge(principal):
1084
- raise SystemExit("权限不足:落库需要可写 knowledge 层的令牌(designer 派生)")
1085
- rep = crosscheck(args.root, layer=args.layer, limit=args.limit, ids=ids,
1086
- prefix=args.prefix,
1087
- verdicts=_load_verdicts(args.verdicts), apply=args.apply,
1088
- batch=args.batch,
1089
- require_verify=not args.no_verify,
1090
- allow_self_verify=args.allow_self_verify,
1091
- reflect_actor=args.reflect_actor,
1092
- verify_actor=args.verify_actor, actor=actor)
1093
- print(json.dumps(rep, ensure_ascii=False, indent=2))
1094
- return 0
1095
-
1096
-
1097
- if __name__ == "__main__": # pragma: no cover
1
+ # -*- coding: utf-8 -*-
2
+ """批量核对(P34/P35):工单 → 反思候选 → 白箱闸门 → 验证否决 → 来源执照 → 落库。
3
+
4
+ 设计约束(与计划一致):
5
+
6
+ * **不新增 MCP op**:本模块是「模块 + `_cli`」形态(同 `backfill.py`),子代理经
7
+ `python -m md_cg.crosscheck …` 调用;令牌决定权限边界。
8
+ * **防自证**:反思单元(`reflect`)与验证单元(`verify`)必须为**不同执行者**;
9
+ 验证单元**只能否决、不能新增**候选(越界字段在白箱闸门直接丢弃)。
10
+ * **来源执照**:文科(`humanities`)只接受来源一致性档(`textbook`/`public_kb`);
11
+ 理科(`science`)只接受可复现档(`compiler`/`test`/`measurement`/`formal_proof`/`data`);
12
+ 赛道未定(`undetermined`)或无来源一律不写,保持 DEFER 并登记待补。
13
+ * **诚实边界**:占位空壳节点(`骨架锚点`/`内容待填充`)不进接线,转待填充工单;
14
+ 拿不出证据的节点绝不写 `verification_basis`。
15
+ * **留痕可回滚**:写入前记录 `_crosscheck.jsonl`,`rollback` 仅在「当前值 == 写入值」
16
+ 时撤销,否则计入 conflict 跳过。
17
+
18
+ `_cli` 之外的所有函数都是纯逻辑(无网络、无第三方依赖),便于离线回归。
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import argparse
24
+ import json
25
+ import os
26
+ import re
27
+ import sys
28
+ import time
29
+ from collections import OrderedDict
30
+
31
+ from . import crypto, nodefile
32
+ from .backfill import (BASIS_ENUM_DEFAULT, BASIS_TEXT, INTERNAL_LAYERS,
33
+ SKIP_LAYERS, SKIP_TAGS, _as_cg, _as_text, _comment,
34
+ _ensure_comment, _entry_id, _readable_guard,
35
+ _remove_ccg_line, _sha, derive_fields)
36
+ from .consolidate import _has_ccg_line, _upsert_ccg_line
37
+ from .fsutil import append_jsonl, read_jsonl
38
+ from .mdcos import _ccg_field
39
+
40
+ # ---- 常量 ----------------------------------------------------------------
41
+
42
+ CROSSCHECK_LOG = "_crosscheck.jsonl"
43
+ CROSSCHECK_BATCH = "crosscheck"
44
+
45
+ REFLECT_UNIT = "reflect"
46
+ VERIFY_UNIT = "verify"
47
+
48
+ # 可写字段(本管线只动这一处,杜绝越界面)
49
+ WRITABLE_FIELDS = ("验证方式",)
50
+ FIELD_NORMALIZE = {
51
+ "验证方式": "验证方式",
52
+ "verification_basis": "验证方式",
53
+ "验证": "验证方式",
54
+ "verification": "验证方式",
55
+ }
56
+
57
+ # 赛道 → 来源执照策略
58
+ SOURCE_POLICY = {"science": "reproducible", "humanities": "consistency"}
59
+
60
+ # 学科关键词(赛道判定;两栖词不入表 → undetermined,宁缺勿猜)
61
+ SCIENCE_SUBJECT_HINTS = (
62
+ "数学", "物理", "化学", "生物", "科学", "信息技术", "通用技术", "计算机",
63
+ )
64
+ HUMANITIES_SUBJECT_HINTS = (
65
+ "语文", "历史", "政治", "道德与法治", "思想政治", "思想品德", "英语",
66
+ "文学", "哲学", "艺术", "音乐", "美术",
67
+ )
68
+
69
+ # 基底枚举 → 可读标签(条件化表述引用)
70
+ BASIS_LABEL = {
71
+ "compiler": "编译器/静态检查",
72
+ "test": "单元测试",
73
+ "measurement": "实测数据",
74
+ "formal_proof": "形式化证明",
75
+ "data": "数据统计",
76
+ "textbook": "人教版教材",
77
+ "public_kb": "公开知识库",
78
+ "other": "人工评审",
79
+ }
80
+
81
+ CLAIM_FIELDS = ("功能名", "生效条件", "子功能", "执行")
82
+ B_CLAIM = "B_valuation"
83
+ A_CLAIM = "A_fact"
84
+
85
+ # B 型(评价性断言)标记词:只收「明显是价值判断/修饰」的表达,避免把事实误判。
86
+ VALUATION_MARKERS = (
87
+ "结晶", "瑰宝", "杰作", "卓越", "杰出", "伟大", "不朽", "巅峰", "典范",
88
+ "精华", "珍品", "璀璨", "辉煌", "丰碑", "博大精深", "源远流长",
89
+ "不可估量", "无与伦比", "举足轻重", "首屈一指", "独树一帜", "别具一格",
90
+ "最优秀", "极富", "令人叹为观止", "不可磨灭", "辉煌成就", "灿烂", "崇高",
91
+ "非凡",
92
+ )
93
+ VALUATION_PATTERNS = (
94
+ re.compile(r"被誉为|被称[之为]|堪称|不愧[为是]"),
95
+ re.compile(r"是[^,。;\n]{0,24}的(结晶|瑰宝|杰作|典范|精华|骄傲|象征|丰碑)"),
96
+ )
97
+
98
+ CONDITION_MARK = "〔来源限定〕"
99
+
100
+
101
+ # ---- 赛道与来源执照 ------------------------------------------------------
102
+
103
+ # 生效条件:给定 fm,若显式 track/discipline_type 命中枚举则返回对应赛道;否则用相关元数据与正文 CCG 字段匹配提示词,返回 humanities/science/undetermined。
104
+ def classify_track(fm: dict, content: str = "") -> str:
105
+ """判定节点赛道:`humanities` / `science` / `undetermined`。
106
+
107
+ 只看**已声明**的元数据(显式字段 > 学科标记),不扫正文散文,避免误判。
108
+ """
109
+ fm = fm or {}
110
+ explicit = str(fm.get("track") or fm.get("discipline_type") or "").strip().lower()
111
+ if explicit in ("humanities", "文科", "arts"):
112
+ return "humanities"
113
+ if explicit in ("science", "理科", "stem"):
114
+ return "science"
115
+ st = fm.get("state_attributes")
116
+ name = _as_text(st.get("name")) if isinstance(st, dict) else ""
117
+ parts = [
118
+ name,
119
+ _as_text(fm.get("title")),
120
+ _as_text(fm.get("discipline")),
121
+ _as_text(fm.get("subject")),
122
+ " ".join(str(t) for t in (fm.get("tags") or [])),
123
+ _ccg_field(content, "功能名"),
124
+ _ccg_field(content, "执行"),
125
+ _ccg_field(content, "子功能"),
126
+ ]
127
+ text = " ".join(p for p in parts if p)
128
+ has_sci = any(k in text for k in SCIENCE_SUBJECT_HINTS)
129
+ has_hum = any(k in text for k in HUMANITIES_SUBJECT_HINTS)
130
+ if has_sci and not has_hum:
131
+ return "science"
132
+ if has_hum and not has_sci:
133
+ return "humanities"
134
+ return "undetermined"
135
+
136
+
137
+ # 生效条件:给定 track,返回 SOURCE_POLICY 中映射的策略名;未知 track 返回空串。
138
+ def source_policy(track: str) -> str:
139
+ """赛道 → 来源策略名(空串表示不可判定,应 DEFER)。"""
140
+ return SOURCE_POLICY.get(track or "", "")
141
+
142
+
143
+ # 生效条件:给定 track,若为 science 返回 REPRODUCIBLE_BASIS,若为 humanities 返回 CONSISTENCY_BASIS,否则返回 ()。
144
+ def allowed_basis(track: str) -> tuple:
145
+ if track == "science":
146
+ return tuple(nodefile.REPRODUCIBLE_BASIS)
147
+ if track == "humanities":
148
+ return tuple(nodefile.CONSISTENCY_BASIS)
149
+ return ()
150
+
151
+
152
+ # 生效条件:给定 track 与 basis,当 basis 非空且其字符串形式属于 allowed_basis(track) 时返回 True,否则 False。
153
+ def basis_licensed(track: str, basis) -> bool:
154
+ """来源执照:理科要可复现证据,文科要来源一致性;赛道未定一律不发放。"""
155
+ return bool(basis) and str(basis) in allowed_basis(track)
156
+
157
+
158
+ # 生效条件:给定 field,返回 FIELD_NORMALIZE 映射值;未知字段返回空串。
159
+ def normalize_field(field) -> str:
160
+ return FIELD_NORMALIZE.get(str(field or "").strip(), "")
161
+
162
+
163
+ # 生效条件:给定 v,若为 None 返回 [];否则将单值或列表转为去除空白后非空字符串的列表。
164
+ def _as_source(v) -> list:
165
+ if v is None:
166
+ return []
167
+ items = list(v) if isinstance(v, (list, tuple)) else [v]
168
+ return [str(x).strip() for x in items if str(x).strip()]
169
+
170
+
171
+ # ---- B 型识别与条件化改写 ------------------------------------------------
172
+
173
+ # 生效条件:给定 text,若含 VALUATION_MARKERS 或匹配 VALUATION_PATTERNS 则返回 B_CLAIM,否则 A_CLAIM。
174
+ def claim_type(text) -> str:
175
+ """`A_fact`(事实性)或 `B_valuation`(评价性断言)。"""
176
+ s = str(text or "")
177
+ if not s.strip():
178
+ return A_CLAIM
179
+ if any(m in s for m in VALUATION_MARKERS):
180
+ return B_CLAIM
181
+ if any(p.search(s) for p in VALUATION_PATTERNS):
182
+ return B_CLAIM
183
+ return A_CLAIM
184
+
185
+
186
+ # 生效条件:给定 text,返回其去除首尾空白后是否以 CONDITION_MARK 开头。
187
+ def is_conditioned(text) -> bool:
188
+ return str(text or "").strip().startswith(CONDITION_MARK)
189
+
190
+
191
+ # 生效条件:给定 text、label、source,若 text 非空且 label 非空且 source 解析后非空,则返回带 CONDITION_MARK 的来源限定表述;已条件化原样返回;否则 None。
192
+ def conditioned_claim(text, label, source):
193
+ """把评价性断言改写为**带来源限定的条件表述**;缺来源/标签则返回 `None`(不写)。
194
+
195
+ 形态:`〔来源限定〕据<来源标签>(<来源>)的表述:<原文>`——
196
+ 原文完整保留(可追溯),前缀显式声明「这是某来源的表述」而非无条件事实。
197
+ 已条件化的文本原样返回(幂等)。
198
+ """
199
+ body = str(text or "").strip()
200
+ if not body or not label:
201
+ return None
202
+ if is_conditioned(body):
203
+ return body
204
+ src = ";".join(_as_source(source))
205
+ if not src:
206
+ return None
207
+ return f"{CONDITION_MARK}据{label}({src})的表述:{body}"
208
+
209
+
210
+ # 生效条件:给定 fm 与 content,提取 CCG 声明字段、comment 值与正文长句,返回断言列表,每项含 text/type/where/field。
211
+ def extract_claims(fm: dict, content: str) -> list:
212
+ """提取可核对断言:CCG 声明字段 + comment 值 + 正文长句。
213
+
214
+ 每条:`{"text", "type", "where", "field"}`;`where` ∈ ccg/comment/body。
215
+ 占位标记不成为断言。
216
+ """
217
+ out, seen = [], set()
218
+
219
+ # 生效条件:仅当 str(text or "").strip() 得到的 s 长度 >= 4、s 不在 seen 中、且 nodefile.is_placeholder_text(s) 为假时,把 {text: s, type: claim_type(s), where, field} 追加进 out 并把 s 加入 seen,否则直接返回(field 默认 "")。
220
+ def _push(text, where, field=""):
221
+ s = str(text or "").strip()
222
+ if len(s) < 4 or s in seen or nodefile.is_placeholder_text(s):
223
+ return
224
+ seen.add(s)
225
+ out.append({"text": s, "type": claim_type(s), "where": where,
226
+ "field": field})
227
+
228
+ for f in CLAIM_FIELDS:
229
+ v = _ccg_field(content, f)
230
+ if v:
231
+ _push(v, "ccg", f)
232
+ c = _comment(fm)
233
+ for f in CLAIM_FIELDS:
234
+ v = c.get(f)
235
+ if isinstance(v, list):
236
+ for item in v:
237
+ _push(item, "comment", f)
238
+ elif v:
239
+ _push(v, "comment", f)
240
+ for line in (content or "").split("\n"):
241
+ raw = line.strip()
242
+ if not raw:
243
+ continue
244
+ body = raw.lstrip("#").strip()
245
+ if body.split(":", 1)[0].strip() in nodefile.CCG_MARKS:
246
+ continue # 声明行已按 CCG 字段处理,不重复断言
247
+ for sent in re.split(r"[。!?]", body):
248
+ s = sent.strip()
249
+ if len(s) >= 8 and not s.startswith(CONDITION_MARK):
250
+ _push(s, "body", "")
251
+ return out
252
+
253
+
254
+ # 生效条件:给定 fm、content、claim、new_text,按 claim.where 定位并在唯一匹配时替换断言返回 (content, True),否则返回 (content, False)。
255
+ def _rewrite_claim(fm: dict, content: str, claim: dict, new_text: str):
256
+ """节点内定位并替换一条断言 → `(content, ok)`;定位不唯一则 fail-closed 不动。"""
257
+ where, field = claim.get("where"), claim.get("field")
258
+ before = str(claim.get("text") or "")
259
+ if where == "ccg" and field:
260
+ if _ccg_field(content, field).strip() == before.strip():
261
+ return _upsert_ccg_line(content, field, new_text), True
262
+ return content, False
263
+ if where == "comment" and field:
264
+ c = _comment(fm)
265
+ v = c.get(field)
266
+ if isinstance(v, list):
267
+ if before in v:
268
+ c[field] = [new_text if x == before else x for x in v]
269
+ return content, True
270
+ return content, False
271
+ if str(v or "").strip() == before.strip():
272
+ c[field] = new_text
273
+ return content, True
274
+ return content, False
275
+ if where == "body":
276
+ if before and content.count(before) == 1:
277
+ return content.replace(before, new_text), True
278
+ return content, False
279
+ return content, False
280
+
281
+
282
+ # ---- 工单 ----------------------------------------------------------------
283
+
284
+ # 生效条件:给定 fm 与 content,返回缺失项列表:verification_basis 无效则加入该名,正文无 "# 验证方式" 行则加入该名。
285
+ def _need(fm: dict, content: str) -> list:
286
+ need = []
287
+ if not nodefile.verification_basis_valid(fm):
288
+ need.append("verification_basis")
289
+ if not _has_ccg_line(content, "验证方式"):
290
+ need.append("验证方式")
291
+ return need
292
+
293
+
294
+ # 生效条件:给定 nid、e、fm、content,返回含 id、layer、track、claims、need、source_policy 的工单行字典。
295
+ def _worklist_row(nid: str, e: dict, fm: dict, content: str) -> dict:
296
+ track = classify_track(fm, content)
297
+ return {
298
+ "id": nid,
299
+ "layer": e.get("layer"),
300
+ "track": track,
301
+ "claims": extract_claims(fm, content),
302
+ "need": _need(fm, content),
303
+ "source_policy": source_policy(track),
304
+ }
305
+
306
+
307
+ # 生效条件:给定 fm 与 content,若正文或 comment 中声明的执行字段为占位文本则返回 True;未声明执行时以正文整体占位判定。
308
+ def _is_placeholder_shell(fm: dict, content: str) -> bool:
309
+ """空壳判定:核心可执行内容未被填充 → 禁止接线(不得把「待填充」固化成事实)。
310
+
311
+ 口径(宁漏判不误判):
312
+ 1. 已声明 `执行`(正文 `# 执行:` 行优先,其次 `state_attributes.comment.执行`)
313
+ 且值为占位标记 → 空壳;
314
+ 2. 未声明 `执行` 时,以正文整体是否为空/占位标记为准——无 comment 但正文写实的
315
+ `kp_archaeo_*` 类节点因此不被误判为空壳。
316
+ """
317
+ decl = _ccg_field(content, "执行") or _as_text(_comment(fm).get("执行"))
318
+ if decl:
319
+ return nodefile.is_placeholder_text(decl)
320
+ return nodefile.is_placeholder_text(content)
321
+
322
+
323
+ # 生效条件:给定 cg,逐节点按 layer/ids/prefix 过滤后产出状态为 skip(internal/denied/locked/derived/present/placeholder/unreadable 等)或 row 的扫描结果。
324
+ def _scan(cg, layer=None, ids=None, prefix=None):
325
+ """逐节点产出扫描结果:`{"status", "reason"?, "id", "row"?}`。
326
+
327
+ `prefix`:只纳入 id 以该前缀开头的节点(真实库以 `kp_` 收窄到用户知识节点,
328
+ 避免 `node_`/`note_`/`imgpart_` 等派生记忆混入工单);**不计数**,与 `layer` 同理。
329
+ """
330
+ want = set(ids) if ids else None
331
+ for nid, e in list((cg.index.get("nodes") or {}).items()):
332
+ if want is not None and nid not in want:
333
+ continue
334
+ if prefix and not str(nid).startswith(prefix):
335
+ continue
336
+ if layer and e.get("layer") != layer:
337
+ continue
338
+ if e.get("layer") in INTERNAL_LAYERS:
339
+ yield {"status": "skip", "reason": "internal", "id": nid}
340
+ continue
341
+ if not _readable_guard(cg, e):
342
+ yield {"status": "skip", "reason": "denied", "id": nid}
343
+ continue
344
+ fm, content = cg._read(e)
345
+ if fm is None:
346
+ yield {"status": "skip", "reason": "unreadable", "id": nid}
347
+ continue
348
+ if crypto.is_encrypted(content):
349
+ yield {"status": "skip", "reason": "locked", "id": nid}
350
+ continue
351
+ if e.get("layer") in SKIP_LAYERS or any(
352
+ t in SKIP_TAGS for t in (fm.get("tags") or [])):
353
+ yield {"status": "skip", "reason": "derived", "id": nid}
354
+ continue
355
+ need = _need(fm, content)
356
+ if not need: # 已齐备
357
+ yield {"status": "skip", "reason": "present", "id": nid}
358
+ continue
359
+ ph = []
360
+ der = derive_fields(fm, content, placeholder_out=ph)
361
+ if _is_placeholder_shell(fm, content): # 空壳:转待填充工单
362
+ yield {"status": "skip", "reason": "placeholder", "id": nid,
363
+ "placeholder_fields": ph}
364
+ continue
365
+ yield {"status": "row", "id": nid,
366
+ "row": _worklist_row(nid, e, fm, content)}
367
+
368
+
369
+ _SKIP_KEY = {"locked": "skipped_locked", "derived": "skipped_derived",
370
+ "present": "skipped_present", "denied": "skipped_denied",
371
+ "unreadable": "skipped_unreadable", "internal": "skipped_internal"}
372
+
373
+
374
+ # 生效条件:给定 x(路径或 MdCGOS),只读扫描并生成缺 verification_basis 或 "# 验证方式" 的节点工单,返回统计 rep。
375
+ def build_worklist(x, layer=None, limit=None, ids=None, prefix=None) -> dict:
376
+ """生成核对工单(只读):缺 `verification_basis`/`验证方式` 的节点入列。
377
+
378
+ `prefix`:按 id 前缀收窄范围(真实库用 `kp_`,与计划交付边界一致);
379
+ 内部 `anchor`/`self` 层一律出局(计入 `skipped_internal`)。
380
+ """
381
+ cg = _as_cg(x)
382
+ rep = {"root": cg.root, "dry_run": True, "action": "crosscheck_worklist",
383
+ "prefix": prefix, "nodes_scanned": 0, "skipped_locked": 0,
384
+ "skipped_derived": 0, "skipped_internal": 0, "skipped_present": 0,
385
+ "skipped_denied": 0, "skipped_placeholder": 0,
386
+ "skipped_unreadable": 0, "placeholder_ids": [], "undetermined": 0,
387
+ "targeted": 0, "items": []}
388
+ for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
389
+ rep["nodes_scanned"] += 1
390
+ if scan["status"] == "skip":
391
+ reason = scan["reason"]
392
+ if reason == "placeholder":
393
+ rep["skipped_placeholder"] += 1
394
+ rep["placeholder_ids"].append(scan["id"])
395
+ else:
396
+ key = _SKIP_KEY.get(reason)
397
+ if key:
398
+ rep[key] += 1
399
+ continue
400
+ row = scan["row"]
401
+ if row["track"] == "undetermined":
402
+ rep["undetermined"] += 1
403
+ rep["targeted"] += 1
404
+ if limit is None or len(rep["items"]) < limit:
405
+ rep["items"].append(row)
406
+ rep["planned_ids"] = [r["id"] for r in rep["items"]]
407
+ return rep
408
+
409
+
410
+ # ---- 子代理接口(提示词 + 解析) -----------------------------------------
411
+
412
+ _REFLECT_TEMPLATE = """你是认知图节点的**反思单元**(reflect)。为节点补齐「验证方式」与其验证基底。
413
+ 赛道:{track};来源策略:{policy};待补字段:{need}
414
+ 节点标题:{title}
415
+ 已声明断言:
416
+ {claims}
417
+ 正文:
418
+ {body}
419
+
420
+ 只输出 JSON 数组,元素形如:
421
+ {{"field":"验证方式","value":"<一句可核对的验证方式声明>","basis":"<基底枚举>","source":["<教材版本+章节 或 公开知识库条目地址>"],"verdict":"accept|defer","reason":"<理由>"}}
422
+ 硬约束:
423
+ 1. 文科(humanities)basis 只能是 textbook / public_kb;
424
+ 2. 理科(science)basis 只能是 compiler / test / measurement / formal_proof / data;
425
+ 3. 来源必须可追溯(教材名称+章节,或公开知识库条目地址);拿不出来就把 verdict 置 defer、source 留空;
426
+ 4. 只能补 field=验证方式,禁止新增其它字段。"""
427
+
428
+ _VERIFY_TEMPLATE = """你是独立**验证单元**(verify)。对下列候选逐条复核:来源是否真实可追溯、基底是否与赛道相容。
429
+ 赛道:{track};来源策略:{policy}
430
+ 候选(JSON):
431
+ {candidates}
432
+
433
+ 只输出 JSON 数组,元素形如:
434
+ {{"field":"验证方式","value":"<原样回填候选 value>","verdict":"accept|drop|defer","reason":"<理由>"}}
435
+ 硬约束:你只能否决(drop)或存疑(defer),**不得新增候选、不得改写 value**。"""
436
+
437
+
438
+ # 生效条件:给定 row、fm、content,用 row 的 track/source_policy/need/claims 与 fm 标题、content 前 1200 字符填充反思模板并返回字符串。
439
+ def reflect_prompt(row: dict, fm: dict, content: str) -> str:
440
+ claims = "\n".join(f"- [{c['type']}] {c['text']}" for c in (row.get("claims") or []))
441
+ return _REFLECT_TEMPLATE.format(
442
+ track=row.get("track"), policy=row.get("source_policy") or "(未定)",
443
+ need="、".join(row.get("need") or []),
444
+ title=_as_text(fm.get("title")) or row.get("id"),
445
+ claims=claims or "(无)",
446
+ body=(content or "")[:1200])
447
+
448
+
449
+ # 生效条件:给定 row 与 rows,把候选字段、值、依据、来源序列化为 JSON 并填充验证模板返回字符串。
450
+ def verify_prompt(row: dict, rows: list) -> str:
451
+ cands = [{"field": r.get("field"), "value": r.get("value"),
452
+ "basis": r.get("basis"), "source": r.get("source")} for r in rows]
453
+ return _VERIFY_TEMPLATE.format(
454
+ track=row.get("track"), policy=row.get("source_policy") or "(未定)",
455
+ candidates=json.dumps(cands, ensure_ascii=False))
456
+
457
+
458
+ # 生效条件:raw 经 str(raw or "") 得 s 后,want_list 为真时先试 s 首个 "[" 至末个 "]"、再试首个 "{" 至末个 "}"(want_list 假值时只试花括号),区间可被 json.loads 解析且结果为 list 时原样返回该 list;结果为 dict 时按 rows/items/verdicts/candidates/data 顺序取首个 obj.get(key) 为 list 的 obj[key],都不满足则返回 [obj],非 list/dict 或区间缺失、解析抛 ValueError 时继续下一组括号,全部落空(含 raw 为假值使 s 为空串)返回 []。
459
+ def _extract_json(raw, want_list=True):
460
+ """从模型输出里抽取 JSON(容忍代码围栏与前后废话)。"""
461
+ s = str(raw or "")
462
+ pairs = ([("[", "]")] if want_list else []) + [("{", "}")]
463
+ for op, cl in pairs:
464
+ i, j = s.find(op), s.rfind(cl)
465
+ if i < 0 or j <= i:
466
+ continue
467
+ try:
468
+ obj = json.loads(s[i:j + 1])
469
+ except ValueError:
470
+ continue
471
+ if isinstance(obj, list):
472
+ return obj
473
+ if isinstance(obj, dict):
474
+ for key in ("rows", "items", "verdicts", "candidates", "data"):
475
+ if isinstance(obj.get(key), list):
476
+ return obj[key]
477
+ return [obj]
478
+ return []
479
+
480
+
481
+ # 生效条件:给定 item,若为 dict 则规范化 field/value/basis/source/verdict/reason 后返回字典,否则返回 {}。
482
+ def _norm_row(item) -> dict:
483
+ if not isinstance(item, dict):
484
+ return {}
485
+ return {
486
+ "field": normalize_field(item.get("field")) or str(item.get("field") or "").strip(),
487
+ "value": _as_text(item.get("value")),
488
+ "basis": str(item.get("basis") or "").strip(),
489
+ "source": _as_source(item.get("source")),
490
+ "verdict": str(item.get("verdict") or "").strip().lower(),
491
+ "reason": str(item.get("reason") or "").strip(),
492
+ }
493
+
494
+
495
+ # 生效条件:给定 raw,解析 JSON 行并保留 field 为“验证方式”或“verification_basis”(统一为“验证方式”)的行,返回列表。
496
+ def parse_reflect_rows(raw) -> list:
497
+ out = []
498
+ for item in _extract_json(raw, want_list=True):
499
+ r = _norm_row(item)
500
+ if not r or r["field"] not in ("验证方式", "verification_basis"):
501
+ continue
502
+ r["field"] = "验证方式"
503
+ out.append(r)
504
+ return out
505
+
506
+
507
+ # 生效条件:遍历 _extract_json(raw, want_list=False)(只认花括号 JSON)的结果,仅当 item 经 _norm_row 后为真且 r["field"] 非空时产出 {field,value,verdict,reason} 四键行,否则跳过(raw 无可解析花括号对象时 out 为空列表)。
508
+ def parse_verify_rows(raw) -> list:
509
+ out = []
510
+ for item in _extract_json(raw, want_list=False):
511
+ r = _norm_row(item)
512
+ if not r or not r["field"]:
513
+ continue
514
+ out.append({k: r[k] for k in ("field", "value", "verdict", "reason")})
515
+ return out
516
+
517
+
518
+ # ---- 白箱闸门与双单元折叠 ------------------------------------------------
519
+
520
+ # 生效条件:逐行处理 rows,仅当 normalize_field(r.get("field")) 落在 WRITABLE_FIELDS、basis_licensed(track, r.get("basis")) 为真、r.get("source") 为真、且 r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "") 非空时进入 kept(附 verdict="accept"),否则该行带对应 reason 进入 gated。
521
+ def gate_rows(rows: list, track: str) -> tuple:
522
+ """零模型白箱闸门:字段越界 / 来源执照不通过 / 无来源 → 一律降级为 defer。
523
+
524
+ 返回 `(kept, gated)`;`kept` 只含「执照齐全」的候选,可进验证单元。
525
+ """
526
+ kept, gated = [], []
527
+ for r in rows:
528
+ f = normalize_field(r.get("field"))
529
+ row = dict(r, field=f)
530
+ if f not in WRITABLE_FIELDS:
531
+ gated.append(dict(row, reason=f"越界字段:{r.get('field')}"))
532
+ continue
533
+ if not basis_licensed(track, r.get("basis")):
534
+ gated.append(dict(row, reason=f"{track or '未定赛道'} 不接受基底 {r.get('basis') or '(缺)'}"))
535
+ continue
536
+ if not r.get("source"):
537
+ gated.append(dict(row, reason="无来源,不写"))
538
+ continue
539
+ value = r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "")
540
+ if not value:
541
+ gated.append(dict(row, reason="无验证方式声明"))
542
+ continue
543
+ kept.append(dict(row, value=value, verdict="accept"))
544
+ return kept, gated
545
+
546
+
547
+ # 生效条件:按 (r.get("id"), normalize_field(r.get("field")) or r.get("field"), r.get("value")) 分组后,组内缺 unit==REFLECT_UNIT 或 unit==VERIFY_UNIT 的行时进 deferred,否则 verify 侧出现 verdict=="drop" 即进 dropped(veto 优先),再否则仅当 reflect 与 verify 各存在 verdict=="accept" 时才进 accepted(附 units),其余进 deferred。
548
+ def fold_verdicts(rows: list) -> tuple:
549
+ """把两单元裁决折叠为可落库结论 → `(accepted, deferred, dropped)`。
550
+
551
+ 接受条件:同 `(id, field, value)` 同时存在 reflect-accept 与 verify-accept;
552
+ 任一 verify-drop 即否决(veto 优先)。
553
+ """
554
+ groups = OrderedDict()
555
+ for r in rows:
556
+ key = (r.get("id"), normalize_field(r.get("field")) or r.get("field"),
557
+ r.get("value"))
558
+ groups.setdefault(key, []).append(r)
559
+ accepted, deferred, dropped = [], [], []
560
+ for (nid, field, value), rs in groups.items():
561
+ refl = [r for r in rs if r.get("unit") == REFLECT_UNIT]
562
+ ver = [r for r in rs if r.get("unit") == VERIFY_UNIT]
563
+ base = {"id": nid, "field": field or "验证方式", "value": value,
564
+ "basis": next((r.get("basis") for r in refl if r.get("basis")), ""),
565
+ "source": next((r.get("source") for r in refl if r.get("source")), []),
566
+ "track": next((r.get("track") for r in refl if r.get("track")), "")}
567
+ if not refl or not ver:
568
+ base["reason"] = "缺" + ("反思裁决" if not refl else "验证裁决")
569
+ deferred.append(base)
570
+ continue
571
+ if any(r.get("verdict") == "drop" for r in ver):
572
+ base["reason"] = next((r.get("reason") for r in ver
573
+ if r.get("verdict") == "drop"), "验证单元否决")
574
+ dropped.append(base)
575
+ continue
576
+ ra = any(r.get("verdict") == "accept" for r in refl)
577
+ va = any(r.get("verdict") == "accept" for r in ver)
578
+ if ra and va:
579
+ base["units"] = {
580
+ "reflect": sorted({str(r.get("actor") or REFLECT_UNIT) for r in refl}),
581
+ "verify": sorted({str(r.get("actor") or VERIFY_UNIT) for r in ver}),
582
+ }
583
+ accepted.append(base)
584
+ else:
585
+ base["reason"] = f"单元未确认(reflect={ra}, verify={va})"
586
+ deferred.append(base)
587
+ return accepted, deferred, dropped
588
+
589
+
590
+ # 生效条件:当 rows 中 unit==REFLECT_UNIT 与 unit==VERIFY_UNIT 的执行者经 str(x.get("actor") or "") 后存在相同的非空值(空串被 discard)时返回 True,否则返回 False。
591
+ def detect_self_verify(rows: list) -> bool:
592
+ """同一执行者同时充当反思与验证 = 自证(禁止)。"""
593
+ r = {str(x.get("actor") or "") for x in rows if x.get("unit") == REFLECT_UNIT}
594
+ v = {str(x.get("actor") or "") for x in rows if x.get("unit") == VERIFY_UNIT}
595
+ r.discard("")
596
+ v.discard("")
597
+ return bool(r & v)
598
+
599
+
600
+ # ---- 落库写入 ------------------------------------------------------------
601
+
602
+ # 生效条件:verdicts 为 None 时返回 None;verdicts 为 dict 时对每个键值把 (rs or []) 中的 dict 元素收为 {str(nid): [...]};否则遍历 verdicts or [],仅当元素为 dict 且 str(r.get("id") or "") 非空时按该 id 追加到对应列表。
603
+ def _norm_verdicts(verdicts):
604
+ """外部裁决(子代理落盘)→ `{id: [rows]}`。"""
605
+ if verdicts is None:
606
+ return None
607
+ out = OrderedDict()
608
+ if isinstance(verdicts, dict):
609
+ for nid, rs in verdicts.items():
610
+ out[str(nid)] = [dict(x) for x in (rs or []) if isinstance(x, dict)]
611
+ return out
612
+ for r in verdicts or []:
613
+ if not isinstance(r, dict):
614
+ continue
615
+ nid = str(r.get("id") or "")
616
+ if nid:
617
+ out.setdefault(nid, []).append(dict(r))
618
+ return out
619
+
620
+
621
+ # 生效条件:accepted 非空(取 accepted[0])时,先以 basis=str(a.get("basis") or BASIS_ENUM_DEFAULT) 与 value=str(a.get("value") or BASIS_TEXT.get(basis, "")).strip() 写「验证方式」行与 comment,之后才在 nodefile.verification_basis_valid(fm) 为真时把 basis 换成 fm.get("verification_basis")、否则把该 basis 写入 fm["verification_basis"];condition_claims 为真时仅对 row.get("claims") 中 type==B_CLAIM 且未被 is_conditioned 的条目做条件化改写,返回含 fm_before、content_hash_before 等留痕的 dict。
622
+ def _apply_node(cg, nid, e, fm, content, accepted, row, batch, actor,
623
+ condition_claims=True):
624
+ """把一个节点的已接受结论写入 md,返回留痕记录(含回滚所需现场)。"""
625
+ a = accepted[0]
626
+ basis = str(a.get("basis") or BASIS_ENUM_DEFAULT)
627
+ source = _as_source(a.get("source"))
628
+ content_before = content
629
+ value = str(a.get("value") or BASIS_TEXT.get(basis, "")).strip()
630
+
631
+ vb_before = {"had": "verification_basis" in fm,
632
+ "value": fm.get("verification_basis")}
633
+ prov_before = {"had": "verification_evidence" in fm,
634
+ "value": fm.get("verification_evidence")}
635
+ line_before = {"had": _has_ccg_line(content, "验证方式"),
636
+ "value": _ccg_field(content, "验证方式")}
637
+ comment0 = _comment(fm)
638
+ cv_before = {"had": "验证方式" in comment0, "value": comment0.get("验证方式")}
639
+
640
+ # 1) 落「验证方式」规范行 + comment + verification_basis(已有合法基底不覆盖)
641
+ content = _upsert_ccg_line(content, "验证方式", value)
642
+ _ensure_comment(fm)["验证方式"] = value
643
+ if nodefile.verification_basis_valid(fm):
644
+ basis = str(fm.get("verification_basis"))
645
+ else:
646
+ fm["verification_basis"] = basis
647
+
648
+ # 2) B 型评价断言 → 条件化表述(A 型保持原样;无来源已由闸门挡掉)
649
+ conditioned = []
650
+ if condition_claims:
651
+ label = BASIS_LABEL.get(basis, basis)
652
+ for c in row.get("claims") or []:
653
+ if c.get("type") != B_CLAIM or is_conditioned(c.get("text")):
654
+ continue
655
+ new = conditioned_claim(c.get("text"), label, source)
656
+ if not new or new == c.get("text"):
657
+ continue
658
+ nc, ok = _rewrite_claim(fm, content, c, new)
659
+ if not ok:
660
+ continue
661
+ content = nc
662
+ conditioned.append({"where": c.get("where"), "field": c.get("field") or "",
663
+ "before": c.get("text"), "after": new})
664
+
665
+ wid = _sha(f"{nid}|{batch}|{time.time()}")
666
+ fm["verification_evidence"] = {
667
+ "at": round(time.time(), 3), "batch": batch, "basis": basis,
668
+ "source": source, "track": row.get("track"),
669
+ "policy": row.get("source_policy"), "write_id": wid,
670
+ "units": a.get("units") or {},
671
+ "conditioned": len(conditioned),
672
+ }
673
+
674
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
675
+ durable=True)
676
+ return {
677
+ "action": "crosscheck", "ts": time.time(), "batch": batch,
678
+ "actor": actor, "entry_id": _entry_id(batch, nid), "write_id": wid,
679
+ "node": nid, "layer": e.get("layer"), "track": row.get("track"),
680
+ "policy": row.get("source_policy"), "basis": basis,
681
+ "verification_value": value, "source": source,
682
+ "units": a.get("units") or {},
683
+ "fm_before": {"verification_basis": vb_before,
684
+ "verification_evidence": prov_before,
685
+ "comment_verification": cv_before},
686
+ "verification_line_before": line_before,
687
+ "claims_conditioned": conditioned,
688
+ "content_hash_before": _sha(content_before),
689
+ "content_hash_after": _sha(content),
690
+ }
691
+
692
+
693
+ # ---- 主流程 --------------------------------------------------------------
694
+
695
+ # 生效条件:reflect_fn(reflect_prompt(scan_row, fm, content)) 经 parse_reflect_rows 得到非空候选时返回 (rrows, vrows),rrows 为空则返回 ([], []);verify_fn 为 None 时 vrows 为空列表,非 None 时由 parse_verify_rows(verify_fn(verify_prompt(scan_row, rrows))) 生成、每行 value 为 c.get("value") or rrows 中同 field 的 value、再回落 ""。
696
+ def _rows_for(scan_row, fm, content, reflect_fn, verify_fn, r_actor, v_actor):
697
+ """调用两单元子代理,返回合并后的裁决行(reflect + verify)。"""
698
+ prompt = reflect_prompt(scan_row, fm, content)
699
+ cands = parse_reflect_rows(reflect_fn(prompt))
700
+ rrows = [dict(c, id=scan_row["id"], unit=REFLECT_UNIT, actor=r_actor,
701
+ track=scan_row["track"]) for c in cands]
702
+ if not rrows:
703
+ return [], []
704
+ vrows = []
705
+ if verify_fn is not None:
706
+ vp = verify_prompt(scan_row, rrows)
707
+ vrows = [dict(c, id=scan_row["id"], unit=VERIFY_UNIT, actor=v_actor,
708
+ track=scan_row["track"],
709
+ value=c.get("value") or next(
710
+ (x.get("value") for x in rrows
711
+ if x.get("field") == c.get("field")), ""))
712
+ for c in parse_verify_rows(verify_fn(vp))]
713
+ return rrows, vrows
714
+
715
+
716
+ # 生效条件:对 pre 中每个 r,str(r.get("unit") or REFLECT_UNIT).strip().lower() 等于 VERIFY_UNIT 时进 vrows,否则(含 unit 缺失回落到 REFLECT_UNIT 及任何其他取值)进 rrows,两组行均覆盖 id=nid、unit、track=track。
717
+ def _rows_from_verdicts(nid, track, pre):
718
+ """从外部裁决中拆出 (reflect, verify) 两组行。"""
719
+ rrows, vrows = [], []
720
+ for r in pre:
721
+ unit = str(r.get("unit") or REFLECT_UNIT).strip().lower()
722
+ base = dict(r, id=nid, unit=unit, track=track)
723
+ (vrows if unit == VERIFY_UNIT else rrows).append(base)
724
+ return rrows, vrows
725
+
726
+
727
+ # 生效条件:x 经 _as_cg 解析且 batch = batch or CROSSCHECK_BATCH 后逐节点扫描,裁决来源按 vmap(verdicts 归一化后非 None)→ reflect_fn 非 None → 二者皆无记 no_reflect 三条分支取行;allow_self_verify=False 时同执行者自证记 self_verify_disallowed,再经 gate_rows 闸门与 require_verify 后 fold_verdicts,仅 apply=True 才 _apply_node 写盘并在有写入时 cg.rebuild_index;limit 非 None 且已达标数 >= limit 时用 continue 跳过(非终止)。
728
+ def crosscheck(x, layer=None, limit=None, ids=None, reflect_fn=None,
729
+ verify_fn=None, verdicts=None, apply=False,
730
+ batch=CROSSCHECK_BATCH, actor=None, require_verify=True,
731
+ allow_self_verify=False, reflect_actor=None, verify_actor=None,
732
+ condition_claims=True, verbose=True, prefix=None) -> dict:
733
+ """批量核对主流程:工单 → 反思候选 → 白箱闸门 → 验证否决 → 落库。
734
+
735
+ `reflect_fn`/`verify_fn`:可注入的子代理函数(接收提示词、返回 JSON 文本);
736
+ `verdicts`:子代理离线产出的裁决行(`[VERDICT_ROW]` 或 `{id: [rows]}`),
737
+ 二选一。`apply=True` 才写盘。
738
+ """
739
+ cg = _as_cg(x)
740
+ batch = batch or CROSSCHECK_BATCH
741
+ vmap = _norm_verdicts(verdicts)
742
+ r_actor = reflect_actor or getattr(reflect_fn, "__name__", "") or REFLECT_UNIT
743
+ v_actor = verify_actor or getattr(verify_fn, "__name__", "") or VERIFY_UNIT
744
+
745
+ rep = {"root": cg.root, "dry_run": not apply, "action": "crosscheck",
746
+ "batch": batch, "actor": actor, "prefix": prefix, "nodes_scanned": 0,
747
+ "targeted": 0,
748
+ "accepted": 0, "rejected": 0, "deferred": 0, "written": 0,
749
+ "claims_conditioned": 0, "skipped_locked": 0, "skipped_derived": 0,
750
+ "skipped_internal": 0, "skipped_present": 0, "skipped_denied": 0,
751
+ "skipped_unreadable": 0, "skipped_placeholder": 0,
752
+ "placeholder_ids": [], "undetermined": 0,
753
+ "reasons": {}, "samples": [], "entry_ids": []}
754
+
755
+ # 生效条件:无条件执行 rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1(reason 缺键时按 .get 的第二参数 0 起算),返回 None。
756
+ def _bump(reason):
757
+ rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
758
+
759
+ # 生效条件:仅当外层 verbose 为真且 len(rep["samples"]) < 20 时把 {kind, id: nid, detail} 追加进 rep["samples"],否则不追加(已达 20 条即停止采样)。
760
+ def _sample(kind, nid, detail=""):
761
+ if verbose and len(rep["samples"]) < 20:
762
+ rep["samples"].append({"kind": kind, "id": nid, "detail": detail})
763
+
764
+ seen_targets = 0
765
+ for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
766
+ rep["nodes_scanned"] += 1
767
+ if scan["status"] == "skip":
768
+ reason = scan["reason"]
769
+ if reason == "placeholder":
770
+ rep["skipped_placeholder"] += 1
771
+ rep["placeholder_ids"].append(scan["id"])
772
+ else:
773
+ key = _SKIP_KEY.get(reason)
774
+ if key:
775
+ rep[key] += 1
776
+ continue
777
+ row = scan["row"]
778
+ if row["track"] == "undetermined":
779
+ rep["undetermined"] += 1
780
+ if limit is not None and seen_targets >= limit:
781
+ continue
782
+ seen_targets += 1
783
+ rep["targeted"] += 1
784
+ nid = row["id"]
785
+ e = cg.index["nodes"].get(nid)
786
+ fm, content = cg._read(e) if e else (None, None)
787
+ if fm is None or crypto.is_encrypted(content):
788
+ rep["skipped_locked"] += 1
789
+ continue
790
+
791
+ # ---- 取两单元裁决 ----
792
+ if vmap is not None:
793
+ pre = vmap.get(nid)
794
+ if not pre:
795
+ rep["deferred"] += 1
796
+ _bump("no_verdict")
797
+ continue
798
+ rrows, vrows = _rows_from_verdicts(nid, row["track"], pre)
799
+ elif reflect_fn is not None:
800
+ try:
801
+ rrows, vrows = _rows_for(row, fm, content, reflect_fn,
802
+ verify_fn, r_actor, v_actor)
803
+ except Exception as exc: # noqa: BLE001
804
+ rep["deferred"] += 1
805
+ _bump(f"unit_error:{type(exc).__name__}")
806
+ continue
807
+ else:
808
+ rep["deferred"] += 1
809
+ _bump("no_reflect")
810
+ continue
811
+
812
+ # 自证:同一执行者既反思又验证 → 拒收
813
+ if not allow_self_verify and detect_self_verify(rrows + vrows):
814
+ rep["deferred"] += 1
815
+ _bump("self_verify_disallowed")
816
+ _sample("self_verify", nid)
817
+ continue
818
+
819
+ # ---- 白箱闸门(对反思候选;验证行只做字段归位) ----
820
+ kept, gated = gate_rows(rrows, row["track"])
821
+ for g in gated:
822
+ _bump(f"gate:{g.get('reason')[:24]}")
823
+ if not kept:
824
+ rep["deferred"] += 1
825
+ _bump("no_candidate")
826
+ _sample("gated", nid, gated[0].get("reason") if gated else "")
827
+ continue
828
+ if require_verify and not vrows:
829
+ rep["deferred"] += 1
830
+ _bump("verify_unavailable")
831
+ continue
832
+
833
+ rows = kept + vrows
834
+ accepted, deferred, dropped = fold_verdicts(rows)
835
+ if dropped and not accepted:
836
+ rep["rejected"] += 1
837
+ _bump("verify_veto")
838
+ _sample("veto", nid, dropped[0].get("reason", ""))
839
+ continue
840
+ if not accepted:
841
+ rep["deferred"] += 1
842
+ _bump("verdict_deferred")
843
+ _sample("deferred", nid, deferred[0].get("reason", "") if deferred else "")
844
+ continue
845
+
846
+ rep["accepted"] += 1
847
+ rep["claims_conditioned"] += sum(
848
+ 1 for c in (row.get("claims") or []) if c.get("type") == B_CLAIM)
849
+ if apply:
850
+ rec = _apply_node(cg, nid, e, fm, content, accepted, row, batch,
851
+ actor, condition_claims=condition_claims)
852
+ append_jsonl(_log_path(cg), rec)
853
+ rep["written"] += 1
854
+ rep["entry_ids"].append(rec["entry_id"])
855
+ _sample("accepted", nid, accepted[0].get("basis", ""))
856
+
857
+ if rep["written"]:
858
+ cg.rebuild_index()
859
+ return rep
860
+
861
+
862
+ # ---- 留痕查询 / 回滚 -----------------------------------------------------
863
+
864
+ # 生效条件:无条件返回 os.path.join(cg.root, CROSSCHECK_LOG)(以 cg.root 与常量 CROSSCHECK_LOG 拼接,无分支)。
865
+ def _log_path(cg) -> str:
866
+ return os.path.join(cg.root, CROSSCHECK_LOG)
867
+
868
+
869
+ # 生效条件:box 非 dict 时返回 False;box 为 dict 且 key=="comment_verification" 时按 box.get("had") 为真则把 comment 的「验证方式」设为 box.get("value")、否则删除该键并返回 True;其他 key 时 had 为真赋 fm[key]=value、否则 fm.pop(key, None) 并返回 True。
870
+ def _reattach(fm: dict, content: str, box: dict, key: str):
871
+ """把 `fm_before[key]` 现场还原到 fm,返回是否发生还原。"""
872
+ if not isinstance(box, dict):
873
+ return False
874
+ had, value = box.get("had"), box.get("value")
875
+ if key == "comment_verification":
876
+ c = _ensure_comment(fm)
877
+ if had:
878
+ c["验证方式"] = value
879
+ else:
880
+ c.pop("验证方式", None)
881
+ return True
882
+ if had:
883
+ fm[key] = value
884
+ else:
885
+ fm.pop(key, None)
886
+ return True
887
+
888
+
889
+ # 生效条件:仅当 str(c.get("after") or "") 非空,且分别满足 where=="ccg" 且 field 真值且 _ccg_field(content, field).strip()==after.strip()(用 before 覆盖该行)、where=="comment" 且 field 真值且 comment 该 field 为含 after 的 list 或 str(v or "").strip()==after.strip()(改为 before)、where=="body" 且 after 出现在 content 中(替换首个匹配)时返回 (content, True);其余情形(含 where 为其他值、字段缺失、当前值不等于写入值)返回 (content, False)。
890
+ def _rewind_claim(fm: dict, content: str, c: dict):
891
+ """撤销一条条件化改写(仅当前值 == 写入值时才动)→ `(content, ok)`。"""
892
+ where, field = c.get("where"), c.get("field")
893
+ before, after = str(c.get("before") or ""), str(c.get("after") or "")
894
+ if not after:
895
+ return content, False
896
+ if where == "ccg" and field:
897
+ if _ccg_field(content, field).strip() == after.strip():
898
+ return _upsert_ccg_line(content, field, before), True
899
+ return content, False
900
+ if where == "comment" and field:
901
+ cc = _comment(fm)
902
+ v = cc.get(field)
903
+ if isinstance(v, list):
904
+ if after in v:
905
+ cc[field] = [before if x == after else x for x in v]
906
+ return content, True
907
+ return content, False
908
+ if str(v or "").strip() == after.strip():
909
+ cc[field] = before
910
+ return content, True
911
+ return content, False
912
+ if where == "body":
913
+ if after in content:
914
+ return content.replace(after, before, 1), True
915
+ return content, False
916
+ return content, False
917
+
918
+
919
+ # 生效条件:x 经 _as_cg 后,对 read_jsonl(_log_path(cg)) 中 action=="crosscheck"、batch 为 None 或等于参数 batch、且 entry_ids 为假值不做 id 过滤(为真值时仅取 entry_id 在集合中的)的记录逐条处理:node 缺失或已处理则跳过,索引无该 node 或 cg._read 得 fm 为 None 或 crypto.is_encrypted(content) 为真时 skipped_drift 加一,write_id 双方非空且不等时 conflict 加一,否则撤销 claims_conditioned、在当前「验证方式」行非空且等于 rec 的 verification_value 时撤销该行、再按 fm_before 还原,reverted 为空则 conflict 加一,非空则写回节点、追加 crosscheck_rollback 日志、reverted 与 entry_ids 加一,最终 reverted 非零时 cg.rebuild_index(),返回 rep;
920
+ def rollback(x, batch=None, entry_ids=None, actor=None) -> dict:
921
+ """按留痕反向应用:撤销核对写入(当前值 ≠ 写入值时跳过,计入 conflict)。"""
922
+ cg = _as_cg(x)
923
+ want = set(entry_ids) if entry_ids else None
924
+ done = set()
925
+ rep = {"root": cg.root, "dry_run": False, "action": "crosscheck_rollback",
926
+ "batch": batch, "actor": actor, "planned": 0, "reverted": 0,
927
+ "skipped_drift": 0, "conflict": 0, "entry_ids": []}
928
+ recs = [r for r in (read_jsonl(_log_path(cg)) or [])
929
+ if r.get("action") == "crosscheck"
930
+ and (batch is None or r.get("batch") == batch)
931
+ and (want is None or r.get("entry_id") in want)]
932
+ rep["planned"] = len(recs)
933
+ for rec in recs:
934
+ nid = rec.get("node")
935
+ if not nid or nid in done:
936
+ continue
937
+ e = cg.index["nodes"].get(nid)
938
+ if not e:
939
+ rep["skipped_drift"] += 1
940
+ continue
941
+ fm, content = cg._read(e)
942
+ if fm is None or crypto.is_encrypted(content):
943
+ rep["skipped_drift"] += 1
944
+ continue
945
+ # 写入现场校验:write_id 一致才回滚(防「写入后又被改过」被误撤)
946
+ wid = (fm.get("verification_evidence") or {}).get("write_id")
947
+ if wid and rec.get("write_id") and wid != rec.get("write_id"):
948
+ rep["conflict"] += 1
949
+ continue
950
+ # 1) 撤销条件化改写(先于验证方式行,避免行被覆盖影响定位)
951
+ reverted = []
952
+ for c in rec.get("claims_conditioned") or []:
953
+ content, ok = _rewind_claim(fm, content, c)
954
+ if ok:
955
+ reverted.append(c.get("field") or c.get("where"))
956
+ # 2) 撤销「验证方式」行
957
+ lb = rec.get("verification_line_before") or {}
958
+ cur_line = _ccg_field(content, "验证方式")
959
+ if cur_line.strip() and cur_line.strip() == str(
960
+ rec.get("verification_value") or "").strip():
961
+ content = (_upsert_ccg_line(content, "验证方式", lb.get("value") or "")
962
+ if lb.get("had") else _remove_ccg_line(content, "验证方式"))
963
+ reverted.append("验证方式")
964
+ # 3) 还原 frontmatter 现场
965
+ for key, box in (rec.get("fm_before") or {}).items():
966
+ _reattach(fm, content, box, key)
967
+ if not reverted:
968
+ rep["conflict"] += 1
969
+ continue
970
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
971
+ durable=True)
972
+ append_jsonl(_log_path(cg), {
973
+ "action": "crosscheck_rollback", "ts": time.time(),
974
+ "batch": rec.get("batch"), "actor": actor,
975
+ "entry_id": rec.get("entry_id"), "node": nid,
976
+ "reverted": reverted, "content_hash_after": _sha(content)})
977
+ rep["reverted"] += 1
978
+ rep["entry_ids"].append(rec.get("entry_id"))
979
+ done.add(nid)
980
+ if rep["reverted"]:
981
+ cg.rebuild_index()
982
+ return rep
983
+
984
+
985
+ # 生效条件:遍历 _log_path(cg) 的记录时,action 为真值只留 rec.get("action")==action 的行、batch 为真值只留 rec.get("batch")==batch 的行;limit 非 None 且 limit>=0 时按 recs[-limit:] 截取(limit 为 0 时 [-0:] 即整表不被削减),否则保留全部;返回 {'root','total','returned','records'}。
986
+ def history(x, limit=100, action=None, batch=None) -> dict:
987
+ cg = _as_cg(x)
988
+ recs = []
989
+ for rec in read_jsonl(_log_path(cg)) or []:
990
+ if action and rec.get("action") != action:
991
+ continue
992
+ if batch and rec.get("batch") != batch:
993
+ continue
994
+ recs.append(rec)
995
+ total = len(recs)
996
+ if limit is not None and limit >= 0:
997
+ recs = recs[-limit:]
998
+ return {"root": cg.root, "total": total, "returned": len(recs),
999
+ "records": recs}
1000
+
1001
+
1002
+ # ---- 权限与 CLI ----------------------------------------------------------
1003
+
1004
+ # 生效条件:principal 为 None 时返回 False;否则仅当 principal.expired() 为假、principal.can_write 为真、且 principal.allows_layer("knowledge") 为真时返回 True,期间任一步抛 Exception 亦返回 False。
1005
+ def can_write_knowledge(principal) -> bool:
1006
+ """落 knowledge 层必须持有可写该层的令牌(designer 派生);否则 fail-closed。"""
1007
+ if principal is None:
1008
+ return False
1009
+ try:
1010
+ if principal.expired() or not principal.can_write:
1011
+ return False
1012
+ return bool(principal.allows_layer("knowledge"))
1013
+ except Exception: # noqa: BLE001
1014
+ return False
1015
+
1016
+
1017
+ # 生效条件:path 为假值(空串/None)返回 None;path 不存在则 raise SystemExit;已存在且读取文本 strip 后为空串返回 [],非空时整段 json.loads 成功即返回该值,抛 ValueError 时按行解析(跳过空行与 "//" 开头行)返回行列表。
1018
+ def _load_verdicts(path: str):
1019
+ if not path:
1020
+ return None
1021
+ if not os.path.exists(path):
1022
+ raise SystemExit(f"裁决文件不存在:{path}")
1023
+ with open(path, "r", encoding="utf-8") as fh:
1024
+ text = fh.read().strip()
1025
+ if not text:
1026
+ return []
1027
+ try:
1028
+ return json.loads(text)
1029
+ except ValueError:
1030
+ rows = []
1031
+ for line in text.splitlines():
1032
+ line = line.strip()
1033
+ if not line or line.startswith("//"):
1034
+ continue
1035
+ rows.append(json.loads(line))
1036
+ return rows
1037
+
1038
+
1039
+ # 生效条件:argv(为 None 时由 argparse 读 sys.argv)解析后按 --action 分派——worklist 调 build_worklist,history 调 history(--limit 默认 None,为 None 时传 100),rollback 在 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 rollback,crosscheck 在 --apply 为真且 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 crosscheck;--token(默认 os.environ.get("MDCG_TOKEN") or "")为真值时先 tokens.verify_token 校验、失败抛 SystemExit;最后打印 rep 并返回 0;
1040
+ def _cli(argv=None) -> int:
1041
+ ap = argparse.ArgumentParser(
1042
+ prog="python -m md_cg.crosscheck",
1043
+ description="kp_ 批量核对管线(工单/核对/回滚/留痕)")
1044
+ ap.add_argument("--root", default=os.environ.get("MDCG_ROOT") or ".")
1045
+ ap.add_argument("--token", default=os.environ.get("MDCG_TOKEN") or "")
1046
+ ap.add_argument("--token-file", default=None)
1047
+ ap.add_argument("--action", default="worklist",
1048
+ choices=("worklist", "crosscheck", "rollback", "history"))
1049
+ ap.add_argument("--verdicts", default="", help="子代理裁决 JSON/JSONL 路径")
1050
+ ap.add_argument("--apply", action="store_true", help="真正写盘(默认 dry-run)")
1051
+ ap.add_argument("--batch", default=CROSSCHECK_BATCH)
1052
+ ap.add_argument("--limit", type=int, default=None)
1053
+ ap.add_argument("--layer", default=None)
1054
+ ap.add_argument("--prefix", default=None, help="按 id 前缀收窄(真实库用 kp_)")
1055
+ ap.add_argument("--ids", default="", help="逗号分隔节点 id")
1056
+ ap.add_argument("--no-verify", action="store_true", help="允许无验证单元(不建议)")
1057
+ ap.add_argument("--allow-self-verify", action="store_true")
1058
+ ap.add_argument("--reflect-actor", default=None)
1059
+ ap.add_argument("--verify-actor", default=None)
1060
+ args = ap.parse_args(argv)
1061
+
1062
+ from . import tokens
1063
+ principal = None
1064
+ if args.token:
1065
+ try:
1066
+ principal = tokens.verify_token(args.token, path=args.token_file)
1067
+ except tokens.TokenError as exc:
1068
+ raise SystemExit(f"令牌校验失败:{exc}")
1069
+ actor = getattr(principal, "actor", None)
1070
+ ids = [s.strip() for s in args.ids.split(",") if s.strip()] or None
1071
+
1072
+ if args.action == "worklist":
1073
+ rep = build_worklist(args.root, layer=args.layer, limit=args.limit,
1074
+ ids=ids, prefix=args.prefix)
1075
+ elif args.action == "history":
1076
+ rep = history(args.root, limit=args.limit if args.limit is not None else 100,
1077
+ batch=args.batch)
1078
+ elif args.action == "rollback":
1079
+ if not can_write_knowledge(principal):
1080
+ raise SystemExit("权限不足:回滚需要可写 knowledge 层的令牌")
1081
+ rep = rollback(args.root, batch=args.batch, actor=actor)
1082
+ else:
1083
+ if args.apply and not can_write_knowledge(principal):
1084
+ raise SystemExit("权限不足:落库需要可写 knowledge 层的令牌(designer 派生)")
1085
+ rep = crosscheck(args.root, layer=args.layer, limit=args.limit, ids=ids,
1086
+ prefix=args.prefix,
1087
+ verdicts=_load_verdicts(args.verdicts), apply=args.apply,
1088
+ batch=args.batch,
1089
+ require_verify=not args.no_verify,
1090
+ allow_self_verify=args.allow_self_verify,
1091
+ reflect_actor=args.reflect_actor,
1092
+ verify_actor=args.verify_actor, actor=actor)
1093
+ print(json.dumps(rep, ensure_ascii=False, indent=2))
1094
+ return 0
1095
+
1096
+
1097
+ if __name__ == "__main__": # pragma: no cover
1098
1098
  sys.exit(_cli())