@furongjun1999/dsh-memory 0.4.11 → 0.5.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (580) hide show
  1. package/README.md +552 -465
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +143 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
  15. package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
  16. package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
  17. package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
  18. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  19. package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
  20. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  21. package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
  22. package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
  23. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
  24. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
  25. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
  26. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
  27. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
  28. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
  29. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
  30. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
  31. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
  32. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
  33. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
  34. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
  35. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
  36. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
  37. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
  38. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
  39. package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
  40. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  41. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  42. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  43. package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
  44. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  45. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  46. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  47. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  48. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +519 -503
  49. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  50. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  51. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  52. package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
  53. package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
  54. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  55. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  56. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  57. package/docs/mdcg/release_v0.4.11.md +49 -0
  58. package/docs/mdcg/release_v0.4.5.md +55 -55
  59. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  60. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  61. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  62. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  63. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  64. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  65. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  66. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  67. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  68. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  69. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  70. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  71. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  72. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  73. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  74. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  75. package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
  76. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  77. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  78. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  79. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  80. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  81. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  82. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  83. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  84. package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
  85. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  86. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  87. package/dsh/README.md +82 -82
  88. package/dsh/cordis.yml.example +139 -139
  89. package/dsh/update-lingshu.bat +11 -11
  90. package/lib/bridge.d.ts +9 -0
  91. package/lib/bridge.js +35 -0
  92. package/lib/hooks.js +36 -2
  93. package/lib/index.js +7 -1
  94. package/lib/lib/roleplay_web.js +116 -29
  95. package/lib/lib/token_store.d.ts +7 -1
  96. package/lib/lib/token_store.js +12 -3
  97. package/md_cg/__init__.py +7 -7
  98. package/md_cg/audit.py +379 -368
  99. package/md_cg/autonomy.py +287 -287
  100. package/md_cg/backfill.py +1328 -1327
  101. package/md_cg/backfill_bigdomain.py +34 -34
  102. package/md_cg/backfill_bucket_zh.py +35 -0
  103. package/md_cg/bench6_arms.py +410 -410
  104. package/md_cg/bench6_common.py +230 -230
  105. package/md_cg/bench6_competitors.py +212 -212
  106. package/md_cg/bench_axis_domain.py +257 -257
  107. package/md_cg/bench_blind_comp.py +308 -308
  108. package/md_cg/bench_e2e_judge.py +532 -0
  109. package/md_cg/bench_e2e_locomo_qa.py +368 -0
  110. package/md_cg/bench_e2e_qa.py +256 -0
  111. package/md_cg/bench_en_atoms_public.py +230 -230
  112. package/md_cg/bench_governance.py +348 -348
  113. package/md_cg/bench_lme_zh.py +410 -410
  114. package/md_cg/bench_locomo.py +121 -121
  115. package/md_cg/bench_locomo_zh.py +450 -450
  116. package/md_cg/bench_locomo_zh_public.py +147 -147
  117. package/md_cg/bench_longmem.py +112 -112
  118. package/md_cg/bench_membench.py +632 -632
  119. package/md_cg/bench_p0.py +149 -149
  120. package/md_cg/bench_progressive.py +287 -287
  121. package/md_cg/bench_role_views.py +238 -238
  122. package/md_cg/bench_task_ab.py +243 -243
  123. package/md_cg/bench_task_ab_llm.py +408 -408
  124. package/md_cg/bench_unified_en.py +204 -204
  125. package/md_cg/bench_zh_mad.py +601 -601
  126. package/md_cg/blindspot_tickets.py +123 -123
  127. package/md_cg/branches.py +301 -285
  128. package/md_cg/build_postings.py +73 -73
  129. package/md_cg/ccgc.py +1006 -948
  130. package/md_cg/census.py +132 -132
  131. package/md_cg/chain.py +315 -300
  132. package/md_cg/codeindex.py +531 -531
  133. package/md_cg/coldverify.py +292 -292
  134. package/md_cg/comment_gate.py +337 -337
  135. package/md_cg/cond_compose.py +190 -190
  136. package/md_cg/cond_facts.py +154 -154
  137. package/md_cg/cond_template.json +106 -106
  138. package/md_cg/condition_anchor.py +142 -142
  139. package/md_cg/conformance.py +726 -726
  140. package/md_cg/consistency.py +717 -717
  141. package/md_cg/consolidate.py +1537 -1439
  142. package/md_cg/corpus.py +110 -110
  143. package/md_cg/crosscheck.py +1098 -1097
  144. package/md_cg/crypto.py +3 -1
  145. package/md_cg/d_meta.py +310 -310
  146. package/md_cg/datapath.py +78 -18
  147. package/md_cg/docindex.py +473 -473
  148. package/md_cg/eval_common.py +575 -575
  149. package/md_cg/evidence.py +4 -2
  150. package/md_cg/evolution.py +477 -477
  151. package/md_cg/export.py +222 -220
  152. package/md_cg/forgetting.py +581 -581
  153. package/md_cg/fsutil.py +377 -329
  154. package/md_cg/hotcache.py +48 -7
  155. package/md_cg/hyperedge.py +251 -251
  156. package/md_cg/identity.py +390 -390
  157. package/md_cg/insight.py +500 -500
  158. package/md_cg/interop.py +338 -0
  159. package/md_cg/judgment_manifest.py +177 -0
  160. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  161. package/md_cg/lexicon/build_standard_en.py +171 -171
  162. package/md_cg/lexicon/expand_en_zh.py +211 -211
  163. package/md_cg/lifecycle.py +272 -272
  164. package/md_cg/linkref.py +280 -280
  165. package/md_cg/links.py +140 -107
  166. package/md_cg/mcp_server.py +403 -55
  167. package/md_cg/md_whitebox.py +345 -345
  168. package/md_cg/mdcg.py +783 -233
  169. package/md_cg/mdcos.py +500 -70
  170. package/md_cg/metacognition.py +591 -591
  171. package/md_cg/migrate.py +119 -119
  172. package/md_cg/migrate_aeis.py +221 -221
  173. package/md_cg/migrate_roleplay.py +293 -293
  174. package/md_cg/migrate_wisdom_graph.py +360 -360
  175. package/md_cg/mreview/__init__.py +25 -25
  176. package/md_cg/mreview/__main__.py +110 -110
  177. package/md_cg/mreview/bundle.py +178 -178
  178. package/md_cg/mreview/candidates.py +262 -262
  179. package/md_cg/mreview/govern.py +694 -693
  180. package/md_cg/mreview/locate.py +939 -939
  181. package/md_cg/mreview/pipeline.py +728 -728
  182. package/md_cg/mreview/rules/duplication.json +21 -21
  183. package/md_cg/mreview/rules/field_coverage.json +54 -54
  184. package/md_cg/mreview/rules/source_license.json +21 -21
  185. package/md_cg/mreview/rules/template_flow.json +21 -21
  186. package/md_cg/mreview/ruleset.py +252 -252
  187. package/md_cg/nodefile.py +575 -575
  188. package/md_cg/pooling.py +484 -472
  189. package/md_cg/postings.py +300 -298
  190. package/md_cg/predict.py +1100 -1100
  191. package/md_cg/progressive.py +123 -123
  192. package/md_cg/protect.py +272 -272
  193. package/md_cg/protocol/md_cg_gate.proto +33 -33
  194. package/md_cg/protocol.py +372 -372
  195. package/md_cg/provenance.py +582 -582
  196. package/md_cg/reach.py +453 -453
  197. package/md_cg/readcache.py +143 -0
  198. package/md_cg/reconcile.py +228 -0
  199. package/md_cg/refindex.py +833 -833
  200. package/md_cg/refine.py +604 -604
  201. package/md_cg/review_cli.py +170 -0
  202. package/md_cg/roleviews.py +89 -89
  203. package/md_cg/routing.py +393 -365
  204. package/md_cg/run_tests.py +211 -0
  205. package/md_cg/scrub.py +13 -3
  206. package/md_cg/security.py +128 -18
  207. package/md_cg/self_state.py +1029 -1029
  208. package/md_cg/selfreport.py +152 -151
  209. package/md_cg/semantic/__init__.py +10 -10
  210. package/md_cg/semantic/canonical.py +122 -122
  211. package/md_cg/semantic/en_normalizer.py +364 -364
  212. package/md_cg/semantic/en_zh_map.json +28694 -0
  213. package/md_cg/semantic/export_en_zh_map.py +64 -0
  214. package/md_cg/semantic/unify.py +45 -0
  215. package/md_cg/semantic/zh_en_atoms.py +139 -139
  216. package/md_cg/signer.py +7 -4
  217. package/md_cg/sources.py +816 -582
  218. package/md_cg/statushdr.py +179 -179
  219. package/md_cg/stg.py +54 -37
  220. package/md_cg/subgraph.py +729 -729
  221. package/md_cg/sustain.py +35 -5
  222. package/md_cg/tasks.py +470 -470
  223. package/md_cg/test_access_hints.py +147 -0
  224. package/md_cg/test_action_derive.py +203 -203
  225. package/md_cg/test_audit_rotate.py +270 -270
  226. package/md_cg/test_autonomy.py +143 -143
  227. package/md_cg/test_bench_governance.py +102 -102
  228. package/md_cg/test_blindspot_tickets.py +166 -166
  229. package/md_cg/test_branch_discard_tombstone.py +136 -0
  230. package/md_cg/test_branches.py +13 -3
  231. package/md_cg/test_ccg_perturb.py +184 -184
  232. package/md_cg/test_ccgc.py +433 -433
  233. package/md_cg/test_census_prune.py +81 -81
  234. package/md_cg/test_chain_read_isolate.py +168 -0
  235. package/md_cg/test_cond_compose_anchors.py +76 -76
  236. package/md_cg/test_cond_match.py +165 -165
  237. package/md_cg/test_condition_anchor.py +81 -81
  238. package/md_cg/test_d_meta.py +412 -412
  239. package/md_cg/test_datapath_device_name.py +203 -0
  240. package/md_cg/test_datapath_root.py +199 -199
  241. package/md_cg/test_emit_negtail_cache.py +156 -0
  242. package/md_cg/test_en_pipeline.py +22 -2
  243. package/md_cg/test_gain_gate.py +212 -212
  244. package/md_cg/test_govern_directread.py +421 -0
  245. package/md_cg/test_health_scale.py +173 -173
  246. package/md_cg/test_hive_ingest.py +285 -0
  247. package/md_cg/test_hot_cold.py +215 -215
  248. package/md_cg/test_hyperedge.py +245 -245
  249. package/md_cg/test_i26_empty_first_write.py +116 -0
  250. package/md_cg/test_i27_e041_identity.py +128 -0
  251. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  252. package/md_cg/test_i32_hotcache_env_key.py +218 -0
  253. package/md_cg/test_identity_attribution.py +96 -15
  254. package/md_cg/test_index_durability.py +17 -3
  255. package/md_cg/test_interop.py +95 -0
  256. package/md_cg/test_interop_judgment.py +228 -0
  257. package/md_cg/test_issue39_utf8_stdio.py +273 -0
  258. package/md_cg/test_lifecycle.py +309 -309
  259. package/md_cg/test_linkref.py +306 -306
  260. package/md_cg/test_links_concurrent_write.py +188 -0
  261. package/md_cg/test_lock.py +43 -43
  262. package/md_cg/test_md_access_parity.py +255 -255
  263. package/md_cg/test_md_writepath.py +345 -345
  264. package/md_cg/test_mdstore_search_parity.py +160 -0
  265. package/md_cg/test_merge_upsert.py +168 -0
  266. package/md_cg/test_mr_m2.py +587 -587
  267. package/md_cg/test_mr_m3.py +710 -710
  268. package/md_cg/test_mr_m4.py +485 -485
  269. package/md_cg/test_n123_derive_expiry_chain.py +205 -0
  270. package/md_cg/test_n130_verify_falsified_protect.py +185 -0
  271. package/md_cg/test_n131_merge_gate.py +205 -0
  272. package/md_cg/test_p0.py +250 -250
  273. package/md_cg/test_p1.py +316 -316
  274. package/md_cg/test_p10_identity.py +173 -173
  275. package/md_cg/test_p11_consistency.py +233 -233
  276. package/md_cg/test_p12_metacognition.py +212 -212
  277. package/md_cg/test_p13_encryption.py +241 -241
  278. package/md_cg/test_p14_sustain.py +249 -249
  279. package/md_cg/test_p15_scrub.py +280 -280
  280. package/md_cg/test_p16_self_state.py +301 -301
  281. package/md_cg/test_p17_predict.py +354 -354
  282. package/md_cg/test_p18_whitebox.py +171 -171
  283. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  284. package/md_cg/test_p1x_ref_root.py +160 -0
  285. package/md_cg/test_p20_evolution.py +315 -315
  286. package/md_cg/test_p21_tokens.py +293 -270
  287. package/md_cg/test_p22_theory.py +175 -175
  288. package/md_cg/test_p23_links.py +311 -311
  289. package/md_cg/test_p24_evidence.py +227 -227
  290. package/md_cg/test_p25_weights.py +156 -156
  291. package/md_cg/test_p26_refindex.py +416 -416
  292. package/md_cg/test_p27_docindex.py +16 -7
  293. package/md_cg/test_p28_refcheck.py +305 -305
  294. package/md_cg/test_p29_session_ingest_export.py +354 -333
  295. package/md_cg/test_p2_mcp.py +3 -0
  296. package/md_cg/test_p3.py +11 -2
  297. package/md_cg/test_p30_maintain.py +330 -330
  298. package/md_cg/test_p31_insight.py +534 -534
  299. package/md_cg/test_p32_backfill.py +7 -1
  300. package/md_cg/test_p33_ccg_wiring.py +293 -293
  301. package/md_cg/test_p34_crosscheck.py +331 -331
  302. package/md_cg/test_p35_conditioned_claim.py +252 -252
  303. package/md_cg/test_p36_kp_align.py +230 -230
  304. package/md_cg/test_p37_condition_space.py +248 -248
  305. package/md_cg/test_p38_concurrent_flush.py +102 -0
  306. package/md_cg/test_p38_contextualize.py +273 -273
  307. package/md_cg/test_p39_verify_flow.py +153 -0
  308. package/md_cg/test_p39_vision_evidence.py +369 -369
  309. package/md_cg/test_p40_refine_worklist.py +241 -241
  310. package/md_cg/test_p41_evolve_patrol.py +224 -224
  311. package/md_cg/test_p42_provenance.py +269 -269
  312. package/md_cg/test_p43_pooling.py +412 -398
  313. package/md_cg/test_p44_md_whitebox.py +231 -231
  314. package/md_cg/test_p45_session_identity.py +219 -219
  315. package/md_cg/test_p46_unit_scope.py +272 -272
  316. package/md_cg/test_p47_session_view.py +316 -0
  317. package/md_cg/test_p4_fuzzy.py +223 -223
  318. package/md_cg/test_p5_semantic.py +226 -226
  319. package/md_cg/test_p6_consolidate.py +440 -387
  320. package/md_cg/test_p7_goals_recent.py +202 -202
  321. package/md_cg/test_p8_subgraph_chain.py +200 -200
  322. package/md_cg/test_p9_forget_protect.py +231 -231
  323. package/md_cg/test_predict_beta.py +135 -135
  324. package/md_cg/test_preflight_failclosed.py +100 -100
  325. package/md_cg/test_progressive.py +146 -146
  326. package/md_cg/test_propose_tail_index.py +157 -0
  327. package/md_cg/test_protocol.py +243 -243
  328. package/md_cg/test_reach.py +378 -378
  329. package/md_cg/test_reach_keys.py +201 -201
  330. package/md_cg/test_read_clip.py +141 -141
  331. package/md_cg/test_read_scope_b27.py +277 -0
  332. package/md_cg/test_readcache_default_on.py +168 -0
  333. package/md_cg/test_readcache_precise_inval.py +270 -0
  334. package/md_cg/test_readcache_prodpath.py +203 -0
  335. package/md_cg/test_reconcile_v0.py +294 -0
  336. package/md_cg/test_retr_gates_prodpath.py +140 -0
  337. package/md_cg/test_retr_s1.py +344 -340
  338. package/md_cg/test_retr_s1b.py +276 -209
  339. package/md_cg/test_retr_s3.py +194 -194
  340. package/md_cg/test_retr_s4.py +163 -163
  341. package/md_cg/test_retr_s5.py +200 -200
  342. package/md_cg/test_retr_s6.py +157 -157
  343. package/md_cg/test_retr_s7.py +392 -384
  344. package/md_cg/test_retr_s8_time.py +369 -316
  345. package/md_cg/test_retr_s9_edges.py +286 -286
  346. package/md_cg/test_retr_s9_entity_ctx.py +9 -3
  347. package/md_cg/test_retr_score_once.py +208 -0
  348. package/md_cg/test_review_conformance.py +367 -367
  349. package/md_cg/test_review_onepass.py +170 -0
  350. package/md_cg/test_role_views.py +354 -354
  351. package/md_cg/test_rrf_graph_seed_cache.py +154 -0
  352. package/md_cg/test_security_audit.py +155 -0
  353. package/md_cg/test_security_audit_b26.py +161 -0
  354. package/md_cg/test_security_audit_v21.py +250 -0
  355. package/md_cg/test_sem_noise.py +242 -242
  356. package/md_cg/test_semantic_canonical.py +16 -2
  357. package/md_cg/test_session_isolation.py +168 -0
  358. package/md_cg/test_snapshot_autoclose.py +187 -0
  359. package/md_cg/test_subproc_encoding.py +192 -192
  360. package/md_cg/test_sustain_mutual.py +153 -153
  361. package/md_cg/test_tail_watermark_race.py +208 -0
  362. package/md_cg/test_tasks.py +409 -409
  363. package/md_cg/test_tenant_env_override_warn.py +139 -0
  364. package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
  365. package/md_cg/test_tool_face.py +189 -189
  366. package/md_cg/test_transfer.py +180 -180
  367. package/md_cg/test_trust.py +361 -361
  368. package/md_cg/test_twophase.py +286 -286
  369. package/md_cg/test_v14_fixes.py +38 -20
  370. package/md_cg/test_validity_filter.py +280 -280
  371. package/md_cg/test_verify_answer.py +138 -138
  372. package/md_cg/test_verify_dirty_reconcile.py +157 -0
  373. package/md_cg/test_wisdom_md_store.py +292 -292
  374. package/md_cg/test_writelimit.py +197 -197
  375. package/md_cg/test_writepipe.py +214 -214
  376. package/md_cg/theory.py +6 -3
  377. package/md_cg/tokens.py +85 -14
  378. package/md_cg/tool_face.py +260 -260
  379. package/md_cg/trust.py +986 -950
  380. package/md_cg/twophase.py +231 -231
  381. package/md_cg/units.py +668 -667
  382. package/md_cg/vision_evidence.py +667 -666
  383. package/md_cg/weights.py +624 -624
  384. package/md_cg/whitebox.py +527 -527
  385. package/md_cg/whitebox_kb/__init__.py +37 -37
  386. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  387. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  388. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  389. package/md_cg/whitebox_kb/engine.py +310 -310
  390. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  391. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  392. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  393. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  394. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  395. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  396. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  397. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  398. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  399. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  400. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  401. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  402. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  403. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  404. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  405. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  406. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  407. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  408. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  409. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  410. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  411. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  412. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  413. package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
  414. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  415. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  416. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  417. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  418. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  419. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  420. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  421. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  422. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  423. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  424. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  425. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  426. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  427. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  428. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  429. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  430. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  431. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  432. package/md_cg/writelimit.py +356 -356
  433. package/md_cg/writepipe.py +20 -8
  434. package/package.json +101 -96
  435. package/skills/plugin.json +54 -54
  436. package/skills/skills/designer-perspective/SKILL.md +158 -158
  437. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  438. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  439. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  440. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  441. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  442. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  443. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  444. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  445. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  446. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  447. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  488. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  489. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  490. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  491. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  492. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  493. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  494. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  495. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  496. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  497. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  498. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  499. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  500. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  501. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  502. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  503. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  504. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  505. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  506. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  507. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  508. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  509. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  510. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  511. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  512. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  513. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  514. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  515. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  516. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  517. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  518. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  519. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  520. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  521. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  522. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  523. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  524. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  525. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  526. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  527. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  528. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  529. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  530. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  531. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  532. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  533. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  534. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  535. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  536. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  537. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  538. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  539. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  540. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  541. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  542. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  543. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  544. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  545. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  546. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  547. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  548. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  549. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  550. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  551. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  552. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  553. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  554. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  555. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  556. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  557. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  558. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  559. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  560. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  561. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  562. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  563. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  564. package/skills/skills/lingshu-net/SKILL.md +48 -48
  565. package/skills/skills/lingshu-os/SKILL.md +64 -64
  566. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  567. package/src/bridge.ts +33 -0
  568. package/src/hooks.ts +38 -2
  569. package/src/index.ts +526 -518
  570. package/src/lib/datapath.ts +326 -326
  571. package/src/lib/mdcg_client.ts +413 -413
  572. package/src/lib/mutual.ts +428 -428
  573. package/src/lib/prompt_safety.ts +62 -62
  574. package/src/lib/python_path.ts +71 -71
  575. package/src/lib/roleplay_web.ts +116 -29
  576. package/src/lib/token_store.ts +13 -3
  577. package/src/tools.ts +212 -212
  578. package/zcode/AGENTS.md +11 -3
  579. package/zcode/README.md +41 -41
  580. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
@@ -1,1098 +1,1099 @@
1
- # -*- coding: utf-8 -*-
2
- """批量核对(P34/P35):工单 → 反思候选 → 白箱闸门 → 验证否决 → 来源执照 → 落库。
3
-
4
- 设计约束(与计划一致):
5
-
6
- * **不新增 MCP op**:本模块是「模块 + `_cli`」形态(同 `backfill.py`),子代理经
7
- `python -m md_cg.crosscheck …` 调用;令牌决定权限边界。
8
- * **防自证**:反思单元(`reflect`)与验证单元(`verify`)必须为**不同执行者**;
9
- 验证单元**只能否决、不能新增**候选(越界字段在白箱闸门直接丢弃)。
10
- * **来源执照**:文科(`humanities`)只接受来源一致性档(`textbook`/`public_kb`);
11
- 理科(`science`)只接受可复现档(`compiler`/`test`/`measurement`/`formal_proof`/`data`);
12
- 赛道未定(`undetermined`)或无来源一律不写,保持 DEFER 并登记待补。
13
- * **诚实边界**:占位空壳节点(`骨架锚点`/`内容待填充`)不进接线,转待填充工单;
14
- 拿不出证据的节点绝不写 `verification_basis`。
15
- * **留痕可回滚**:写入前记录 `_crosscheck.jsonl`,`rollback` 仅在「当前值 == 写入值」
16
- 时撤销,否则计入 conflict 跳过。
17
-
18
- `_cli` 之外的所有函数都是纯逻辑(无网络、无第三方依赖),便于离线回归。
19
- """
20
-
21
- from __future__ import annotations
22
-
23
- import argparse
24
- import json
25
- import os
26
- import re
27
- import sys
28
- import time
29
- from collections import OrderedDict
30
-
31
- from . import crypto, nodefile
32
- from .backfill import (BASIS_ENUM_DEFAULT, BASIS_TEXT, INTERNAL_LAYERS,
33
- SKIP_LAYERS, SKIP_TAGS, _as_cg, _as_text, _comment,
34
- _ensure_comment, _entry_id, _readable_guard,
35
- _remove_ccg_line, _sha, derive_fields)
36
- from .consolidate import _has_ccg_line, _upsert_ccg_line
37
- from .fsutil import append_jsonl, read_jsonl
38
- from .mdcos import _ccg_field
39
-
40
- # ---- 常量 ----------------------------------------------------------------
41
-
42
- CROSSCHECK_LOG = "_crosscheck.jsonl"
43
- CROSSCHECK_BATCH = "crosscheck"
44
-
45
- REFLECT_UNIT = "reflect"
46
- VERIFY_UNIT = "verify"
47
-
48
- # 可写字段(本管线只动这一处,杜绝越界面)
49
- WRITABLE_FIELDS = ("验证方式",)
50
- FIELD_NORMALIZE = {
51
- "验证方式": "验证方式",
52
- "verification_basis": "验证方式",
53
- "验证": "验证方式",
54
- "verification": "验证方式",
55
- }
56
-
57
- # 赛道 → 来源执照策略
58
- SOURCE_POLICY = {"science": "reproducible", "humanities": "consistency"}
59
-
60
- # 学科关键词(赛道判定;两栖词不入表 → undetermined,宁缺勿猜)
61
- SCIENCE_SUBJECT_HINTS = (
62
- "数学", "物理", "化学", "生物", "科学", "信息技术", "通用技术", "计算机",
63
- )
64
- HUMANITIES_SUBJECT_HINTS = (
65
- "语文", "历史", "政治", "道德与法治", "思想政治", "思想品德", "英语",
66
- "文学", "哲学", "艺术", "音乐", "美术",
67
- )
68
-
69
- # 基底枚举 → 可读标签(条件化表述引用)
70
- BASIS_LABEL = {
71
- "compiler": "编译器/静态检查",
72
- "test": "单元测试",
73
- "measurement": "实测数据",
74
- "formal_proof": "形式化证明",
75
- "data": "数据统计",
76
- "textbook": "人教版教材",
77
- "public_kb": "公开知识库",
78
- "other": "人工评审",
79
- }
80
-
81
- CLAIM_FIELDS = ("功能名", "生效条件", "子功能", "执行")
82
- B_CLAIM = "B_valuation"
83
- A_CLAIM = "A_fact"
84
-
85
- # B 型(评价性断言)标记词:只收「明显是价值判断/修饰」的表达,避免把事实误判。
86
- VALUATION_MARKERS = (
87
- "结晶", "瑰宝", "杰作", "卓越", "杰出", "伟大", "不朽", "巅峰", "典范",
88
- "精华", "珍品", "璀璨", "辉煌", "丰碑", "博大精深", "源远流长",
89
- "不可估量", "无与伦比", "举足轻重", "首屈一指", "独树一帜", "别具一格",
90
- "最优秀", "极富", "令人叹为观止", "不可磨灭", "辉煌成就", "灿烂", "崇高",
91
- "非凡",
92
- )
93
- VALUATION_PATTERNS = (
94
- re.compile(r"被誉为|被称[之为]|堪称|不愧[为是]"),
95
- re.compile(r"是[^,。;\n]{0,24}的(结晶|瑰宝|杰作|典范|精华|骄傲|象征|丰碑)"),
96
- )
97
-
98
- CONDITION_MARK = "〔来源限定〕"
99
-
100
-
101
- # ---- 赛道与来源执照 ------------------------------------------------------
102
-
103
- # 生效条件:给定 fm,若显式 track/discipline_type 命中枚举则返回对应赛道;否则用相关元数据与正文 CCG 字段匹配提示词,返回 humanities/science/undetermined。
104
- def classify_track(fm: dict, content: str = "") -> str:
105
- """判定节点赛道:`humanities` / `science` / `undetermined`。
106
-
107
- 只看**已声明**的元数据(显式字段 > 学科标记),不扫正文散文,避免误判。
108
- """
109
- fm = fm or {}
110
- explicit = str(fm.get("track") or fm.get("discipline_type") or "").strip().lower()
111
- if explicit in ("humanities", "文科", "arts"):
112
- return "humanities"
113
- if explicit in ("science", "理科", "stem"):
114
- return "science"
115
- st = fm.get("state_attributes")
116
- name = _as_text(st.get("name")) if isinstance(st, dict) else ""
117
- parts = [
118
- name,
119
- _as_text(fm.get("title")),
120
- _as_text(fm.get("discipline")),
121
- _as_text(fm.get("subject")),
122
- " ".join(str(t) for t in (fm.get("tags") or [])),
123
- _ccg_field(content, "功能名"),
124
- _ccg_field(content, "执行"),
125
- _ccg_field(content, "子功能"),
126
- ]
127
- text = " ".join(p for p in parts if p)
128
- has_sci = any(k in text for k in SCIENCE_SUBJECT_HINTS)
129
- has_hum = any(k in text for k in HUMANITIES_SUBJECT_HINTS)
130
- if has_sci and not has_hum:
131
- return "science"
132
- if has_hum and not has_sci:
133
- return "humanities"
134
- return "undetermined"
135
-
136
-
137
- # 生效条件:给定 track,返回 SOURCE_POLICY 中映射的策略名;未知 track 返回空串。
138
- def source_policy(track: str) -> str:
139
- """赛道 → 来源策略名(空串表示不可判定,应 DEFER)。"""
140
- return SOURCE_POLICY.get(track or "", "")
141
-
142
-
143
- # 生效条件:给定 track,若为 science 返回 REPRODUCIBLE_BASIS,若为 humanities 返回 CONSISTENCY_BASIS,否则返回 ()。
144
- def allowed_basis(track: str) -> tuple:
145
- if track == "science":
146
- return tuple(nodefile.REPRODUCIBLE_BASIS)
147
- if track == "humanities":
148
- return tuple(nodefile.CONSISTENCY_BASIS)
149
- return ()
150
-
151
-
152
- # 生效条件:给定 track 与 basis,当 basis 非空且其字符串形式属于 allowed_basis(track) 时返回 True,否则 False。
153
- def basis_licensed(track: str, basis) -> bool:
154
- """来源执照:理科要可复现证据,文科要来源一致性;赛道未定一律不发放。"""
155
- return bool(basis) and str(basis) in allowed_basis(track)
156
-
157
-
158
- # 生效条件:给定 field,返回 FIELD_NORMALIZE 映射值;未知字段返回空串。
159
- def normalize_field(field) -> str:
160
- return FIELD_NORMALIZE.get(str(field or "").strip(), "")
161
-
162
-
163
- # 生效条件:给定 v,若为 None 返回 [];否则将单值或列表转为去除空白后非空字符串的列表。
164
- def _as_source(v) -> list:
165
- if v is None:
166
- return []
167
- items = list(v) if isinstance(v, (list, tuple)) else [v]
168
- return [str(x).strip() for x in items if str(x).strip()]
169
-
170
-
171
- # ---- B 型识别与条件化改写 ------------------------------------------------
172
-
173
- # 生效条件:给定 text,若含 VALUATION_MARKERS 或匹配 VALUATION_PATTERNS 则返回 B_CLAIM,否则 A_CLAIM。
174
- def claim_type(text) -> str:
175
- """`A_fact`(事实性)或 `B_valuation`(评价性断言)。"""
176
- s = str(text or "")
177
- if not s.strip():
178
- return A_CLAIM
179
- if any(m in s for m in VALUATION_MARKERS):
180
- return B_CLAIM
181
- if any(p.search(s) for p in VALUATION_PATTERNS):
182
- return B_CLAIM
183
- return A_CLAIM
184
-
185
-
186
- # 生效条件:给定 text,返回其去除首尾空白后是否以 CONDITION_MARK 开头。
187
- def is_conditioned(text) -> bool:
188
- return str(text or "").strip().startswith(CONDITION_MARK)
189
-
190
-
191
- # 生效条件:给定 text、label、source,若 text 非空且 label 非空且 source 解析后非空,则返回带 CONDITION_MARK 的来源限定表述;已条件化原样返回;否则 None。
192
- def conditioned_claim(text, label, source):
193
- """把评价性断言改写为**带来源限定的条件表述**;缺来源/标签则返回 `None`(不写)。
194
-
195
- 形态:`〔来源限定〕据<来源标签>(<来源>)的表述:<原文>`——
196
- 原文完整保留(可追溯),前缀显式声明「这是某来源的表述」而非无条件事实。
197
- 已条件化的文本原样返回(幂等)。
198
- """
199
- body = str(text or "").strip()
200
- if not body or not label:
201
- return None
202
- if is_conditioned(body):
203
- return body
204
- src = ";".join(_as_source(source))
205
- if not src:
206
- return None
207
- return f"{CONDITION_MARK}据{label}({src})的表述:{body}"
208
-
209
-
210
- # 生效条件:给定 fm 与 content,提取 CCG 声明字段、comment 值与正文长句,返回断言列表,每项含 text/type/where/field。
211
- def extract_claims(fm: dict, content: str) -> list:
212
- """提取可核对断言:CCG 声明字段 + comment 值 + 正文长句。
213
-
214
- 每条:`{"text", "type", "where", "field"}`;`where` ∈ ccg/comment/body。
215
- 占位标记不成为断言。
216
- """
217
- out, seen = [], set()
218
-
219
- # 生效条件:仅当 str(text or "").strip() 得到的 s 长度 >= 4、s 不在 seen 中、且 nodefile.is_placeholder_text(s) 为假时,把 {text: s, type: claim_type(s), where, field} 追加进 out 并把 s 加入 seen,否则直接返回(field 默认 "")。
220
- def _push(text, where, field=""):
221
- s = str(text or "").strip()
222
- if len(s) < 4 or s in seen or nodefile.is_placeholder_text(s):
223
- return
224
- seen.add(s)
225
- out.append({"text": s, "type": claim_type(s), "where": where,
226
- "field": field})
227
-
228
- for f in CLAIM_FIELDS:
229
- v = _ccg_field(content, f)
230
- if v:
231
- _push(v, "ccg", f)
232
- c = _comment(fm)
233
- for f in CLAIM_FIELDS:
234
- v = c.get(f)
235
- if isinstance(v, list):
236
- for item in v:
237
- _push(item, "comment", f)
238
- elif v:
239
- _push(v, "comment", f)
240
- for line in (content or "").split("\n"):
241
- raw = line.strip()
242
- if not raw:
243
- continue
244
- body = raw.lstrip("#").strip()
245
- if body.split(":", 1)[0].strip() in nodefile.CCG_MARKS:
246
- continue # 声明行已按 CCG 字段处理,不重复断言
247
- for sent in re.split(r"[。!?]", body):
248
- s = sent.strip()
249
- if len(s) >= 8 and not s.startswith(CONDITION_MARK):
250
- _push(s, "body", "")
251
- return out
252
-
253
-
254
- # 生效条件:给定 fm、content、claim、new_text,按 claim.where 定位并在唯一匹配时替换断言返回 (content, True),否则返回 (content, False)。
255
- def _rewrite_claim(fm: dict, content: str, claim: dict, new_text: str):
256
- """节点内定位并替换一条断言 → `(content, ok)`;定位不唯一则 fail-closed 不动。"""
257
- where, field = claim.get("where"), claim.get("field")
258
- before = str(claim.get("text") or "")
259
- if where == "ccg" and field:
260
- if _ccg_field(content, field).strip() == before.strip():
261
- return _upsert_ccg_line(content, field, new_text), True
262
- return content, False
263
- if where == "comment" and field:
264
- c = _comment(fm)
265
- v = c.get(field)
266
- if isinstance(v, list):
267
- if before in v:
268
- c[field] = [new_text if x == before else x for x in v]
269
- return content, True
270
- return content, False
271
- if str(v or "").strip() == before.strip():
272
- c[field] = new_text
273
- return content, True
274
- return content, False
275
- if where == "body":
276
- if before and content.count(before) == 1:
277
- return content.replace(before, new_text), True
278
- return content, False
279
- return content, False
280
-
281
-
282
- # ---- 工单 ----------------------------------------------------------------
283
-
284
- # 生效条件:给定 fm 与 content,返回缺失项列表:verification_basis 无效则加入该名,正文无 "# 验证方式" 行则加入该名。
285
- def _need(fm: dict, content: str) -> list:
286
- need = []
287
- if not nodefile.verification_basis_valid(fm):
288
- need.append("verification_basis")
289
- if not _has_ccg_line(content, "验证方式"):
290
- need.append("验证方式")
291
- return need
292
-
293
-
294
- # 生效条件:给定 nid、e、fm、content,返回含 id、layer、track、claims、need、source_policy 的工单行字典。
295
- def _worklist_row(nid: str, e: dict, fm: dict, content: str) -> dict:
296
- track = classify_track(fm, content)
297
- return {
298
- "id": nid,
299
- "layer": e.get("layer"),
300
- "track": track,
301
- "claims": extract_claims(fm, content),
302
- "need": _need(fm, content),
303
- "source_policy": source_policy(track),
304
- }
305
-
306
-
307
- # 生效条件:给定 fm 与 content,若正文或 comment 中声明的执行字段为占位文本则返回 True;未声明执行时以正文整体占位判定。
308
- def _is_placeholder_shell(fm: dict, content: str) -> bool:
309
- """空壳判定:核心可执行内容未被填充 → 禁止接线(不得把「待填充」固化成事实)。
310
-
311
- 口径(宁漏判不误判):
312
- 1. 已声明 `执行`(正文 `# 执行:` 行优先,其次 `state_attributes.comment.执行`)
313
- 且值为占位标记 → 空壳;
314
- 2. 未声明 `执行` 时,以正文整体是否为空/占位标记为准——无 comment 但正文写实的
315
- `kp_archaeo_*` 类节点因此不被误判为空壳。
316
- """
317
- decl = _ccg_field(content, "执行") or _as_text(_comment(fm).get("执行"))
318
- if decl:
319
- return nodefile.is_placeholder_text(decl)
320
- return nodefile.is_placeholder_text(content)
321
-
322
-
323
- # 生效条件:给定 cg,逐节点按 layer/ids/prefix 过滤后产出状态为 skip(internal/denied/locked/derived/present/placeholder/unreadable 等)或 row 的扫描结果。
324
- def _scan(cg, layer=None, ids=None, prefix=None):
325
- """逐节点产出扫描结果:`{"status", "reason"?, "id", "row"?}`。
326
-
327
- `prefix`:只纳入 id 以该前缀开头的节点(真实库以 `kp_` 收窄到用户知识节点,
328
- 避免 `node_`/`note_`/`imgpart_` 等派生记忆混入工单);**不计数**,与 `layer` 同理。
329
- """
330
- want = set(ids) if ids else None
331
- for nid, e in list((cg.index.get("nodes") or {}).items()):
332
- if want is not None and nid not in want:
333
- continue
334
- if prefix and not str(nid).startswith(prefix):
335
- continue
336
- if layer and e.get("layer") != layer:
337
- continue
338
- if e.get("layer") in INTERNAL_LAYERS:
339
- yield {"status": "skip", "reason": "internal", "id": nid}
340
- continue
341
- if not _readable_guard(cg, e):
342
- yield {"status": "skip", "reason": "denied", "id": nid}
343
- continue
344
- fm, content = cg._read(e)
345
- if fm is None:
346
- yield {"status": "skip", "reason": "unreadable", "id": nid}
347
- continue
348
- if crypto.is_encrypted(content):
349
- yield {"status": "skip", "reason": "locked", "id": nid}
350
- continue
351
- if e.get("layer") in SKIP_LAYERS or any(
352
- t in SKIP_TAGS for t in (fm.get("tags") or [])):
353
- yield {"status": "skip", "reason": "derived", "id": nid}
354
- continue
355
- need = _need(fm, content)
356
- if not need: # 已齐备
357
- yield {"status": "skip", "reason": "present", "id": nid}
358
- continue
359
- ph = []
360
- der = derive_fields(fm, content, placeholder_out=ph)
361
- if _is_placeholder_shell(fm, content): # 空壳:转待填充工单
362
- yield {"status": "skip", "reason": "placeholder", "id": nid,
363
- "placeholder_fields": ph}
364
- continue
365
- yield {"status": "row", "id": nid,
366
- "row": _worklist_row(nid, e, fm, content)}
367
-
368
-
369
- _SKIP_KEY = {"locked": "skipped_locked", "derived": "skipped_derived",
370
- "present": "skipped_present", "denied": "skipped_denied",
371
- "unreadable": "skipped_unreadable", "internal": "skipped_internal"}
372
-
373
-
374
- # 生效条件:给定 x(路径或 MdCGOS),只读扫描并生成缺 verification_basis 或 "# 验证方式" 的节点工单,返回统计 rep。
375
- def build_worklist(x, layer=None, limit=None, ids=None, prefix=None) -> dict:
376
- """生成核对工单(只读):缺 `verification_basis`/`验证方式` 的节点入列。
377
-
378
- `prefix`:按 id 前缀收窄范围(真实库用 `kp_`,与计划交付边界一致);
379
- 内部 `anchor`/`self` 层一律出局(计入 `skipped_internal`)。
380
- """
381
- cg = _as_cg(x)
382
- rep = {"root": cg.root, "dry_run": True, "action": "crosscheck_worklist",
383
- "prefix": prefix, "nodes_scanned": 0, "skipped_locked": 0,
384
- "skipped_derived": 0, "skipped_internal": 0, "skipped_present": 0,
385
- "skipped_denied": 0, "skipped_placeholder": 0,
386
- "skipped_unreadable": 0, "placeholder_ids": [], "undetermined": 0,
387
- "targeted": 0, "items": []}
388
- for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
389
- rep["nodes_scanned"] += 1
390
- if scan["status"] == "skip":
391
- reason = scan["reason"]
392
- if reason == "placeholder":
393
- rep["skipped_placeholder"] += 1
394
- rep["placeholder_ids"].append(scan["id"])
395
- else:
396
- key = _SKIP_KEY.get(reason)
397
- if key:
398
- rep[key] += 1
399
- continue
400
- row = scan["row"]
401
- if row["track"] == "undetermined":
402
- rep["undetermined"] += 1
403
- rep["targeted"] += 1
404
- if limit is None or len(rep["items"]) < limit:
405
- rep["items"].append(row)
406
- rep["planned_ids"] = [r["id"] for r in rep["items"]]
407
- return rep
408
-
409
-
410
- # ---- 子代理接口(提示词 + 解析) -----------------------------------------
411
-
412
- _REFLECT_TEMPLATE = """你是认知图节点的**反思单元**(reflect)。为节点补齐「验证方式」与其验证基底。
413
- 赛道:{track};来源策略:{policy};待补字段:{need}
414
- 节点标题:{title}
415
- 已声明断言:
416
- {claims}
417
- 正文:
418
- {body}
419
-
420
- 只输出 JSON 数组,元素形如:
421
- {{"field":"验证方式","value":"<一句可核对的验证方式声明>","basis":"<基底枚举>","source":["<教材版本+章节 或 公开知识库条目地址>"],"verdict":"accept|defer","reason":"<理由>"}}
422
- 硬约束:
423
- 1. 文科(humanities)basis 只能是 textbook / public_kb;
424
- 2. 理科(science)basis 只能是 compiler / test / measurement / formal_proof / data;
425
- 3. 来源必须可追溯(教材名称+章节,或公开知识库条目地址);拿不出来就把 verdict 置 defer、source 留空;
426
- 4. 只能补 field=验证方式,禁止新增其它字段。"""
427
-
428
- _VERIFY_TEMPLATE = """你是独立**验证单元**(verify)。对下列候选逐条复核:来源是否真实可追溯、基底是否与赛道相容。
429
- 赛道:{track};来源策略:{policy}
430
- 候选(JSON):
431
- {candidates}
432
-
433
- 只输出 JSON 数组,元素形如:
434
- {{"field":"验证方式","value":"<原样回填候选 value>","verdict":"accept|drop|defer","reason":"<理由>"}}
435
- 硬约束:你只能否决(drop)或存疑(defer),**不得新增候选、不得改写 value**。"""
436
-
437
-
438
- # 生效条件:给定 row、fm、content,用 row 的 track/source_policy/need/claims 与 fm 标题、content 前 1200 字符填充反思模板并返回字符串。
439
- def reflect_prompt(row: dict, fm: dict, content: str) -> str:
440
- claims = "\n".join(f"- [{c['type']}] {c['text']}" for c in (row.get("claims") or []))
441
- return _REFLECT_TEMPLATE.format(
442
- track=row.get("track"), policy=row.get("source_policy") or "(未定)",
443
- need="、".join(row.get("need") or []),
444
- title=_as_text(fm.get("title")) or row.get("id"),
445
- claims=claims or "(无)",
446
- body=(content or "")[:1200])
447
-
448
-
449
- # 生效条件:给定 row 与 rows,把候选字段、值、依据、来源序列化为 JSON 并填充验证模板返回字符串。
450
- def verify_prompt(row: dict, rows: list) -> str:
451
- cands = [{"field": r.get("field"), "value": r.get("value"),
452
- "basis": r.get("basis"), "source": r.get("source")} for r in rows]
453
- return _VERIFY_TEMPLATE.format(
454
- track=row.get("track"), policy=row.get("source_policy") or "(未定)",
455
- candidates=json.dumps(cands, ensure_ascii=False))
456
-
457
-
458
- # 生效条件:raw 经 str(raw or "") 得 s 后,want_list 为真时先试 s 首个 "[" 至末个 "]"、再试首个 "{" 至末个 "}"(want_list 假值时只试花括号),区间可被 json.loads 解析且结果为 list 时原样返回该 list;结果为 dict 时按 rows/items/verdicts/candidates/data 顺序取首个 obj.get(key) 为 list 的 obj[key],都不满足则返回 [obj],非 list/dict 或区间缺失、解析抛 ValueError 时继续下一组括号,全部落空(含 raw 为假值使 s 为空串)返回 []。
459
- def _extract_json(raw, want_list=True):
460
- """从模型输出里抽取 JSON(容忍代码围栏与前后废话)。"""
461
- s = str(raw or "")
462
- pairs = ([("[", "]")] if want_list else []) + [("{", "}")]
463
- for op, cl in pairs:
464
- i, j = s.find(op), s.rfind(cl)
465
- if i < 0 or j <= i:
466
- continue
467
- try:
468
- obj = json.loads(s[i:j + 1])
469
- except ValueError:
470
- continue
471
- if isinstance(obj, list):
472
- return obj
473
- if isinstance(obj, dict):
474
- for key in ("rows", "items", "verdicts", "candidates", "data"):
475
- if isinstance(obj.get(key), list):
476
- return obj[key]
477
- return [obj]
478
- return []
479
-
480
-
481
- # 生效条件:给定 item,若为 dict 则规范化 field/value/basis/source/verdict/reason 后返回字典,否则返回 {}。
482
- def _norm_row(item) -> dict:
483
- if not isinstance(item, dict):
484
- return {}
485
- return {
486
- "field": normalize_field(item.get("field")) or str(item.get("field") or "").strip(),
487
- "value": _as_text(item.get("value")),
488
- "basis": str(item.get("basis") or "").strip(),
489
- "source": _as_source(item.get("source")),
490
- "verdict": str(item.get("verdict") or "").strip().lower(),
491
- "reason": str(item.get("reason") or "").strip(),
492
- }
493
-
494
-
495
- # 生效条件:给定 raw,解析 JSON 行并保留 field 为“验证方式”或“verification_basis”(统一为“验证方式”)的行,返回列表。
496
- def parse_reflect_rows(raw) -> list:
497
- out = []
498
- for item in _extract_json(raw, want_list=True):
499
- r = _norm_row(item)
500
- if not r or r["field"] not in ("验证方式", "verification_basis"):
501
- continue
502
- r["field"] = "验证方式"
503
- out.append(r)
504
- return out
505
-
506
-
507
- # 生效条件:遍历 _extract_json(raw, want_list=False)(只认花括号 JSON)的结果,仅当 item 经 _norm_row 后为真且 r["field"] 非空时产出 {field,value,verdict,reason} 四键行,否则跳过(raw 无可解析花括号对象时 out 为空列表)。
508
- def parse_verify_rows(raw) -> list:
509
- out = []
510
- for item in _extract_json(raw, want_list=False):
511
- r = _norm_row(item)
512
- if not r or not r["field"]:
513
- continue
514
- out.append({k: r[k] for k in ("field", "value", "verdict", "reason")})
515
- return out
516
-
517
-
518
- # ---- 白箱闸门与双单元折叠 ------------------------------------------------
519
-
520
- # 生效条件:逐行处理 rows,仅当 normalize_field(r.get("field")) 落在 WRITABLE_FIELDS、basis_licensed(track, r.get("basis")) 为真、r.get("source") 为真、且 r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "") 非空时进入 kept(附 verdict="accept"),否则该行带对应 reason 进入 gated。
521
- def gate_rows(rows: list, track: str) -> tuple:
522
- """零模型白箱闸门:字段越界 / 来源执照不通过 / 无来源 → 一律降级为 defer。
523
-
524
- 返回 `(kept, gated)`;`kept` 只含「执照齐全」的候选,可进验证单元。
525
- """
526
- kept, gated = [], []
527
- for r in rows:
528
- f = normalize_field(r.get("field"))
529
- row = dict(r, field=f)
530
- if f not in WRITABLE_FIELDS:
531
- gated.append(dict(row, reason=f"越界字段:{r.get('field')}"))
532
- continue
533
- if not basis_licensed(track, r.get("basis")):
534
- gated.append(dict(row, reason=f"{track or '未定赛道'} 不接受基底 {r.get('basis') or '(缺)'}"))
535
- continue
536
- if not r.get("source"):
537
- gated.append(dict(row, reason="无来源,不写"))
538
- continue
539
- value = r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "")
540
- if not value:
541
- gated.append(dict(row, reason="无验证方式声明"))
542
- continue
543
- kept.append(dict(row, value=value, verdict="accept"))
544
- return kept, gated
545
-
546
-
547
- # 生效条件:按 (r.get("id"), normalize_field(r.get("field")) or r.get("field"), r.get("value")) 分组后,组内缺 unit==REFLECT_UNIT 或 unit==VERIFY_UNIT 的行时进 deferred,否则 verify 侧出现 verdict=="drop" 即进 dropped(veto 优先),再否则仅当 reflect 与 verify 各存在 verdict=="accept" 时才进 accepted(附 units),其余进 deferred。
548
- def fold_verdicts(rows: list) -> tuple:
549
- """把两单元裁决折叠为可落库结论 → `(accepted, deferred, dropped)`。
550
-
551
- 接受条件:同 `(id, field, value)` 同时存在 reflect-accept 与 verify-accept;
552
- 任一 verify-drop 即否决(veto 优先)。
553
- """
554
- groups = OrderedDict()
555
- for r in rows:
556
- key = (r.get("id"), normalize_field(r.get("field")) or r.get("field"),
557
- r.get("value"))
558
- groups.setdefault(key, []).append(r)
559
- accepted, deferred, dropped = [], [], []
560
- for (nid, field, value), rs in groups.items():
561
- refl = [r for r in rs if r.get("unit") == REFLECT_UNIT]
562
- ver = [r for r in rs if r.get("unit") == VERIFY_UNIT]
563
- base = {"id": nid, "field": field or "验证方式", "value": value,
564
- "basis": next((r.get("basis") for r in refl if r.get("basis")), ""),
565
- "source": next((r.get("source") for r in refl if r.get("source")), []),
566
- "track": next((r.get("track") for r in refl if r.get("track")), "")}
567
- if not refl or not ver:
568
- base["reason"] = "缺" + ("反思裁决" if not refl else "验证裁决")
569
- deferred.append(base)
570
- continue
571
- if any(r.get("verdict") == "drop" for r in ver):
572
- base["reason"] = next((r.get("reason") for r in ver
573
- if r.get("verdict") == "drop"), "验证单元否决")
574
- dropped.append(base)
575
- continue
576
- ra = any(r.get("verdict") == "accept" for r in refl)
577
- va = any(r.get("verdict") == "accept" for r in ver)
578
- if ra and va:
579
- base["units"] = {
580
- "reflect": sorted({str(r.get("actor") or REFLECT_UNIT) for r in refl}),
581
- "verify": sorted({str(r.get("actor") or VERIFY_UNIT) for r in ver}),
582
- }
583
- accepted.append(base)
584
- else:
585
- base["reason"] = f"单元未确认(reflect={ra}, verify={va})"
586
- deferred.append(base)
587
- return accepted, deferred, dropped
588
-
589
-
590
- # 生效条件:当 rows 中 unit==REFLECT_UNIT 与 unit==VERIFY_UNIT 的执行者经 str(x.get("actor") or "") 后存在相同的非空值(空串被 discard)时返回 True,否则返回 False。
591
- def detect_self_verify(rows: list) -> bool:
592
- """同一执行者同时充当反思与验证 = 自证(禁止)。"""
593
- r = {str(x.get("actor") or "") for x in rows if x.get("unit") == REFLECT_UNIT}
594
- v = {str(x.get("actor") or "") for x in rows if x.get("unit") == VERIFY_UNIT}
595
- r.discard("")
596
- v.discard("")
597
- return bool(r & v)
598
-
599
-
600
- # ---- 落库写入 ------------------------------------------------------------
601
-
602
- # 生效条件:verdicts 为 None 时返回 None;verdicts 为 dict 时对每个键值把 (rs or []) 中的 dict 元素收为 {str(nid): [...]};否则遍历 verdicts or [],仅当元素为 dict 且 str(r.get("id") or "") 非空时按该 id 追加到对应列表。
603
- def _norm_verdicts(verdicts):
604
- """外部裁决(子代理落盘)→ `{id: [rows]}`。"""
605
- if verdicts is None:
606
- return None
607
- out = OrderedDict()
608
- if isinstance(verdicts, dict):
609
- for nid, rs in verdicts.items():
610
- out[str(nid)] = [dict(x) for x in (rs or []) if isinstance(x, dict)]
611
- return out
612
- for r in verdicts or []:
613
- if not isinstance(r, dict):
614
- continue
615
- nid = str(r.get("id") or "")
616
- if nid:
617
- out.setdefault(nid, []).append(dict(r))
618
- return out
619
-
620
-
621
- # 生效条件:accepted 非空(取 accepted[0])时,先以 basis=str(a.get("basis") or BASIS_ENUM_DEFAULT) 与 value=str(a.get("value") or BASIS_TEXT.get(basis, "")).strip() 写「验证方式」行与 comment,之后才在 nodefile.verification_basis_valid(fm) 为真时把 basis 换成 fm.get("verification_basis")、否则把该 basis 写入 fm["verification_basis"];condition_claims 为真时仅对 row.get("claims") 中 type==B_CLAIM 且未被 is_conditioned 的条目做条件化改写,返回含 fm_before、content_hash_before 等留痕的 dict。
622
- def _apply_node(cg, nid, e, fm, content, accepted, row, batch, actor,
623
- condition_claims=True):
624
- """把一个节点的已接受结论写入 md,返回留痕记录(含回滚所需现场)。"""
625
- a = accepted[0]
626
- basis = str(a.get("basis") or BASIS_ENUM_DEFAULT)
627
- source = _as_source(a.get("source"))
628
- content_before = content
629
- value = str(a.get("value") or BASIS_TEXT.get(basis, "")).strip()
630
-
631
- vb_before = {"had": "verification_basis" in fm,
632
- "value": fm.get("verification_basis")}
633
- prov_before = {"had": "verification_evidence" in fm,
634
- "value": fm.get("verification_evidence")}
635
- line_before = {"had": _has_ccg_line(content, "验证方式"),
636
- "value": _ccg_field(content, "验证方式")}
637
- comment0 = _comment(fm)
638
- cv_before = {"had": "验证方式" in comment0, "value": comment0.get("验证方式")}
639
-
640
- # 1) 落「验证方式」规范行 + comment + verification_basis(已有合法基底不覆盖)
641
- content = _upsert_ccg_line(content, "验证方式", value)
642
- _ensure_comment(fm)["验证方式"] = value
643
- if nodefile.verification_basis_valid(fm):
644
- basis = str(fm.get("verification_basis"))
645
- else:
646
- fm["verification_basis"] = basis
647
-
648
- # 2) B 型评价断言 → 条件化表述(A 型保持原样;无来源已由闸门挡掉)
649
- conditioned = []
650
- if condition_claims:
651
- label = BASIS_LABEL.get(basis, basis)
652
- for c in row.get("claims") or []:
653
- if c.get("type") != B_CLAIM or is_conditioned(c.get("text")):
654
- continue
655
- new = conditioned_claim(c.get("text"), label, source)
656
- if not new or new == c.get("text"):
657
- continue
658
- nc, ok = _rewrite_claim(fm, content, c, new)
659
- if not ok:
660
- continue
661
- content = nc
662
- conditioned.append({"where": c.get("where"), "field": c.get("field") or "",
663
- "before": c.get("text"), "after": new})
664
-
665
- wid = _sha(f"{nid}|{batch}|{time.time()}")
666
- fm["verification_evidence"] = {
667
- "at": round(time.time(), 3), "batch": batch, "basis": basis,
668
- "source": source, "track": row.get("track"),
669
- "policy": row.get("source_policy"), "write_id": wid,
670
- "units": a.get("units") or {},
671
- "conditioned": len(conditioned),
672
- }
673
-
674
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
675
- durable=True)
676
- return {
677
- "action": "crosscheck", "ts": time.time(), "batch": batch,
678
- "actor": actor, "entry_id": _entry_id(batch, nid), "write_id": wid,
679
- "node": nid, "layer": e.get("layer"), "track": row.get("track"),
680
- "policy": row.get("source_policy"), "basis": basis,
681
- "verification_value": value, "source": source,
682
- "units": a.get("units") or {},
683
- "fm_before": {"verification_basis": vb_before,
684
- "verification_evidence": prov_before,
685
- "comment_verification": cv_before},
686
- "verification_line_before": line_before,
687
- "claims_conditioned": conditioned,
688
- "content_hash_before": _sha(content_before),
689
- "content_hash_after": _sha(content),
690
- }
691
-
692
-
693
- # ---- 主流程 --------------------------------------------------------------
694
-
695
- # 生效条件:reflect_fn(reflect_prompt(scan_row, fm, content)) 经 parse_reflect_rows 得到非空候选时返回 (rrows, vrows),rrows 为空则返回 ([], []);verify_fn 为 None 时 vrows 为空列表,非 None 时由 parse_verify_rows(verify_fn(verify_prompt(scan_row, rrows))) 生成、每行 value 为 c.get("value") or rrows 中同 field 的 value、再回落 ""。
696
- def _rows_for(scan_row, fm, content, reflect_fn, verify_fn, r_actor, v_actor):
697
- """调用两单元子代理,返回合并后的裁决行(reflect + verify)。"""
698
- prompt = reflect_prompt(scan_row, fm, content)
699
- cands = parse_reflect_rows(reflect_fn(prompt))
700
- rrows = [dict(c, id=scan_row["id"], unit=REFLECT_UNIT, actor=r_actor,
701
- track=scan_row["track"]) for c in cands]
702
- if not rrows:
703
- return [], []
704
- vrows = []
705
- if verify_fn is not None:
706
- vp = verify_prompt(scan_row, rrows)
707
- vrows = [dict(c, id=scan_row["id"], unit=VERIFY_UNIT, actor=v_actor,
708
- track=scan_row["track"],
709
- value=c.get("value") or next(
710
- (x.get("value") for x in rrows
711
- if x.get("field") == c.get("field")), ""))
712
- for c in parse_verify_rows(verify_fn(vp))]
713
- return rrows, vrows
714
-
715
-
716
- # 生效条件:对 pre 中每个 r,str(r.get("unit") or REFLECT_UNIT).strip().lower() 等于 VERIFY_UNIT 时进 vrows,否则(含 unit 缺失回落到 REFLECT_UNIT 及任何其他取值)进 rrows,两组行均覆盖 id=nid、unit、track=track。
717
- def _rows_from_verdicts(nid, track, pre):
718
- """从外部裁决中拆出 (reflect, verify) 两组行。"""
719
- rrows, vrows = [], []
720
- for r in pre:
721
- unit = str(r.get("unit") or REFLECT_UNIT).strip().lower()
722
- base = dict(r, id=nid, unit=unit, track=track)
723
- (vrows if unit == VERIFY_UNIT else rrows).append(base)
724
- return rrows, vrows
725
-
726
-
727
- # 生效条件:x 经 _as_cg 解析且 batch = batch or CROSSCHECK_BATCH 后逐节点扫描,裁决来源按 vmap(verdicts 归一化后非 None)→ reflect_fn 非 None → 二者皆无记 no_reflect 三条分支取行;allow_self_verify=False 时同执行者自证记 self_verify_disallowed,再经 gate_rows 闸门与 require_verify 后 fold_verdicts,仅 apply=True 才 _apply_node 写盘并在有写入时 cg.rebuild_index;limit 非 None 且已达标数 >= limit 时用 continue 跳过(非终止)。
728
- def crosscheck(x, layer=None, limit=None, ids=None, reflect_fn=None,
729
- verify_fn=None, verdicts=None, apply=False,
730
- batch=CROSSCHECK_BATCH, actor=None, require_verify=True,
731
- allow_self_verify=False, reflect_actor=None, verify_actor=None,
732
- condition_claims=True, verbose=True, prefix=None) -> dict:
733
- """批量核对主流程:工单 → 反思候选 → 白箱闸门 → 验证否决 → 落库。
734
-
735
- `reflect_fn`/`verify_fn`:可注入的子代理函数(接收提示词、返回 JSON 文本);
736
- `verdicts`:子代理离线产出的裁决行(`[VERDICT_ROW]` 或 `{id: [rows]}`),
737
- 二选一。`apply=True` 才写盘。
738
- """
739
- cg = _as_cg(x)
740
- batch = batch or CROSSCHECK_BATCH
741
- vmap = _norm_verdicts(verdicts)
742
- r_actor = reflect_actor or getattr(reflect_fn, "__name__", "") or REFLECT_UNIT
743
- v_actor = verify_actor or getattr(verify_fn, "__name__", "") or VERIFY_UNIT
744
-
745
- rep = {"root": cg.root, "dry_run": not apply, "action": "crosscheck",
746
- "batch": batch, "actor": actor, "prefix": prefix, "nodes_scanned": 0,
747
- "targeted": 0,
748
- "accepted": 0, "rejected": 0, "deferred": 0, "written": 0,
749
- "claims_conditioned": 0, "skipped_locked": 0, "skipped_derived": 0,
750
- "skipped_internal": 0, "skipped_present": 0, "skipped_denied": 0,
751
- "skipped_unreadable": 0, "skipped_placeholder": 0,
752
- "placeholder_ids": [], "undetermined": 0,
753
- "reasons": {}, "samples": [], "entry_ids": []}
754
-
755
- # 生效条件:无条件执行 rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1(reason 缺键时按 .get 的第二参数 0 起算),返回 None。
756
- def _bump(reason):
757
- rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
758
-
759
- # 生效条件:仅当外层 verbose 为真且 len(rep["samples"]) < 20 时把 {kind, id: nid, detail} 追加进 rep["samples"],否则不追加(已达 20 条即停止采样)。
760
- def _sample(kind, nid, detail=""):
761
- if verbose and len(rep["samples"]) < 20:
762
- rep["samples"].append({"kind": kind, "id": nid, "detail": detail})
763
-
764
- seen_targets = 0
765
- for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
766
- rep["nodes_scanned"] += 1
767
- if scan["status"] == "skip":
768
- reason = scan["reason"]
769
- if reason == "placeholder":
770
- rep["skipped_placeholder"] += 1
771
- rep["placeholder_ids"].append(scan["id"])
772
- else:
773
- key = _SKIP_KEY.get(reason)
774
- if key:
775
- rep[key] += 1
776
- continue
777
- row = scan["row"]
778
- if row["track"] == "undetermined":
779
- rep["undetermined"] += 1
780
- if limit is not None and seen_targets >= limit:
781
- continue
782
- seen_targets += 1
783
- rep["targeted"] += 1
784
- nid = row["id"]
785
- e = cg.index["nodes"].get(nid)
786
- fm, content = cg._read(e) if e else (None, None)
787
- if fm is None or crypto.is_encrypted(content):
788
- rep["skipped_locked"] += 1
789
- continue
790
-
791
- # ---- 取两单元裁决 ----
792
- if vmap is not None:
793
- pre = vmap.get(nid)
794
- if not pre:
795
- rep["deferred"] += 1
796
- _bump("no_verdict")
797
- continue
798
- rrows, vrows = _rows_from_verdicts(nid, row["track"], pre)
799
- elif reflect_fn is not None:
800
- try:
801
- rrows, vrows = _rows_for(row, fm, content, reflect_fn,
802
- verify_fn, r_actor, v_actor)
803
- except Exception as exc: # noqa: BLE001
804
- rep["deferred"] += 1
805
- _bump(f"unit_error:{type(exc).__name__}")
806
- continue
807
- else:
808
- rep["deferred"] += 1
809
- _bump("no_reflect")
810
- continue
811
-
812
- # 自证:同一执行者既反思又验证 → 拒收
813
- if not allow_self_verify and detect_self_verify(rrows + vrows):
814
- rep["deferred"] += 1
815
- _bump("self_verify_disallowed")
816
- _sample("self_verify", nid)
817
- continue
818
-
819
- # ---- 白箱闸门(对反思候选;验证行只做字段归位) ----
820
- kept, gated = gate_rows(rrows, row["track"])
821
- for g in gated:
822
- _bump(f"gate:{g.get('reason')[:24]}")
823
- if not kept:
824
- rep["deferred"] += 1
825
- _bump("no_candidate")
826
- _sample("gated", nid, gated[0].get("reason") if gated else "")
827
- continue
828
- if require_verify and not vrows:
829
- rep["deferred"] += 1
830
- _bump("verify_unavailable")
831
- continue
832
-
833
- rows = kept + vrows
834
- accepted, deferred, dropped = fold_verdicts(rows)
835
- if dropped and not accepted:
836
- rep["rejected"] += 1
837
- _bump("verify_veto")
838
- _sample("veto", nid, dropped[0].get("reason", ""))
839
- continue
840
- if not accepted:
841
- rep["deferred"] += 1
842
- _bump("verdict_deferred")
843
- _sample("deferred", nid, deferred[0].get("reason", "") if deferred else "")
844
- continue
845
-
846
- rep["accepted"] += 1
847
- rep["claims_conditioned"] += sum(
848
- 1 for c in (row.get("claims") or []) if c.get("type") == B_CLAIM)
849
- if apply:
850
- rec = _apply_node(cg, nid, e, fm, content, accepted, row, batch,
851
- actor, condition_claims=condition_claims)
852
- append_jsonl(_log_path(cg), rec)
853
- rep["written"] += 1
854
- rep["entry_ids"].append(rec["entry_id"])
855
- _sample("accepted", nid, accepted[0].get("basis", ""))
856
-
857
- if rep["written"]:
858
- cg.rebuild_index()
859
- return rep
860
-
861
-
862
- # ---- 留痕查询 / 回滚 -----------------------------------------------------
863
-
864
- # 生效条件:无条件返回 os.path.join(cg.root, CROSSCHECK_LOG)(以 cg.root 与常量 CROSSCHECK_LOG 拼接,无分支)。
865
- def _log_path(cg) -> str:
866
- return os.path.join(cg.root, CROSSCHECK_LOG)
867
-
868
-
869
- # 生效条件:box 非 dict 时返回 False;box 为 dict 且 key=="comment_verification" 时按 box.get("had") 为真则把 comment 的「验证方式」设为 box.get("value")、否则删除该键并返回 True;其他 key 时 had 为真赋 fm[key]=value、否则 fm.pop(key, None) 并返回 True。
870
- def _reattach(fm: dict, content: str, box: dict, key: str):
871
- """把 `fm_before[key]` 现场还原到 fm,返回是否发生还原。"""
872
- if not isinstance(box, dict):
873
- return False
874
- had, value = box.get("had"), box.get("value")
875
- if key == "comment_verification":
876
- c = _ensure_comment(fm)
877
- if had:
878
- c["验证方式"] = value
879
- else:
880
- c.pop("验证方式", None)
881
- return True
882
- if had:
883
- fm[key] = value
884
- else:
885
- fm.pop(key, None)
886
- return True
887
-
888
-
889
- # 生效条件:仅当 str(c.get("after") or "") 非空,且分别满足 where=="ccg" 且 field 真值且 _ccg_field(content, field).strip()==after.strip()(用 before 覆盖该行)、where=="comment" 且 field 真值且 comment 该 field 为含 after 的 list 或 str(v or "").strip()==after.strip()(改为 before)、where=="body" 且 after 出现在 content 中(替换首个匹配)时返回 (content, True);其余情形(含 where 为其他值、字段缺失、当前值不等于写入值)返回 (content, False)。
890
- def _rewind_claim(fm: dict, content: str, c: dict):
891
- """撤销一条条件化改写(仅当前值 == 写入值时才动)→ `(content, ok)`。"""
892
- where, field = c.get("where"), c.get("field")
893
- before, after = str(c.get("before") or ""), str(c.get("after") or "")
894
- if not after:
895
- return content, False
896
- if where == "ccg" and field:
897
- if _ccg_field(content, field).strip() == after.strip():
898
- return _upsert_ccg_line(content, field, before), True
899
- return content, False
900
- if where == "comment" and field:
901
- cc = _comment(fm)
902
- v = cc.get(field)
903
- if isinstance(v, list):
904
- if after in v:
905
- cc[field] = [before if x == after else x for x in v]
906
- return content, True
907
- return content, False
908
- if str(v or "").strip() == after.strip():
909
- cc[field] = before
910
- return content, True
911
- return content, False
912
- if where == "body":
913
- if after in content:
914
- return content.replace(after, before, 1), True
915
- return content, False
916
- return content, False
917
-
918
-
919
- # 生效条件:x 经 _as_cg 后,对 read_jsonl(_log_path(cg)) 中 action=="crosscheck"、batch 为 None 或等于参数 batch、且 entry_ids 为假值不做 id 过滤(为真值时仅取 entry_id 在集合中的)的记录逐条处理:node 缺失或已处理则跳过,索引无该 node 或 cg._read 得 fm 为 None 或 crypto.is_encrypted(content) 为真时 skipped_drift 加一,write_id 双方非空且不等时 conflict 加一,否则撤销 claims_conditioned、在当前「验证方式」行非空且等于 rec 的 verification_value 时撤销该行、再按 fm_before 还原,reverted 为空则 conflict 加一,非空则写回节点、追加 crosscheck_rollback 日志、reverted 与 entry_ids 加一,最终 reverted 非零时 cg.rebuild_index(),返回 rep;
920
- def rollback(x, batch=None, entry_ids=None, actor=None) -> dict:
921
- """按留痕反向应用:撤销核对写入(当前值 ≠ 写入值时跳过,计入 conflict)。"""
922
- cg = _as_cg(x)
923
- want = set(entry_ids) if entry_ids else None
924
- done = set()
925
- rep = {"root": cg.root, "dry_run": False, "action": "crosscheck_rollback",
926
- "batch": batch, "actor": actor, "planned": 0, "reverted": 0,
927
- "skipped_drift": 0, "conflict": 0, "entry_ids": []}
928
- recs = [r for r in (read_jsonl(_log_path(cg)) or [])
929
- if r.get("action") == "crosscheck"
930
- and (batch is None or r.get("batch") == batch)
931
- and (want is None or r.get("entry_id") in want)]
932
- rep["planned"] = len(recs)
933
- for rec in recs:
934
- nid = rec.get("node")
935
- if not nid or nid in done:
936
- continue
937
- e = cg.index["nodes"].get(nid)
938
- if not e:
939
- rep["skipped_drift"] += 1
940
- continue
941
- fm, content = cg._read(e)
942
- if fm is None or crypto.is_encrypted(content):
943
- rep["skipped_drift"] += 1
944
- continue
945
- # 写入现场校验:write_id 一致才回滚(防「写入后又被改过」被误撤)
946
- wid = (fm.get("verification_evidence") or {}).get("write_id")
947
- if wid and rec.get("write_id") and wid != rec.get("write_id"):
948
- rep["conflict"] += 1
949
- continue
950
- # 1) 撤销条件化改写(先于验证方式行,避免行被覆盖影响定位)
951
- reverted = []
952
- for c in rec.get("claims_conditioned") or []:
953
- content, ok = _rewind_claim(fm, content, c)
954
- if ok:
955
- reverted.append(c.get("field") or c.get("where"))
956
- # 2) 撤销「验证方式」行
957
- lb = rec.get("verification_line_before") or {}
958
- cur_line = _ccg_field(content, "验证方式")
959
- if cur_line.strip() and cur_line.strip() == str(
960
- rec.get("verification_value") or "").strip():
961
- content = (_upsert_ccg_line(content, "验证方式", lb.get("value") or "")
962
- if lb.get("had") else _remove_ccg_line(content, "验证方式"))
963
- reverted.append("验证方式")
964
- # 3) 还原 frontmatter 现场
965
- for key, box in (rec.get("fm_before") or {}).items():
966
- _reattach(fm, content, box, key)
967
- if not reverted:
968
- rep["conflict"] += 1
969
- continue
970
- cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
971
- durable=True)
972
- append_jsonl(_log_path(cg), {
973
- "action": "crosscheck_rollback", "ts": time.time(),
974
- "batch": rec.get("batch"), "actor": actor,
975
- "entry_id": rec.get("entry_id"), "node": nid,
976
- "reverted": reverted, "content_hash_after": _sha(content)})
977
- rep["reverted"] += 1
978
- rep["entry_ids"].append(rec.get("entry_id"))
979
- done.add(nid)
980
- if rep["reverted"]:
981
- cg.rebuild_index()
982
- return rep
983
-
984
-
985
- # 生效条件:遍历 _log_path(cg) 的记录时,action 为真值只留 rec.get("action")==action 的行、batch 为真值只留 rec.get("batch")==batch 的行;limit 非 None 且 limit>=0 时按 recs[-limit:] 截取(limit 为 0 时 [-0:] 即整表不被削减),否则保留全部;返回 {'root','total','returned','records'}。
986
- def history(x, limit=100, action=None, batch=None) -> dict:
987
- cg = _as_cg(x)
988
- recs = []
989
- for rec in read_jsonl(_log_path(cg)) or []:
990
- if action and rec.get("action") != action:
991
- continue
992
- if batch and rec.get("batch") != batch:
993
- continue
994
- recs.append(rec)
995
- total = len(recs)
996
- if limit is not None and limit >= 0:
997
- recs = recs[-limit:]
998
- return {"root": cg.root, "total": total, "returned": len(recs),
999
- "records": recs}
1000
-
1001
-
1002
- # ---- 权限与 CLI ----------------------------------------------------------
1003
-
1004
- # 生效条件:principal 为 None 时返回 False;否则仅当 principal.expired() 为假、principal.can_write 为真、且 principal.allows_layer("knowledge") 为真时返回 True,期间任一步抛 Exception 亦返回 False。
1005
- def can_write_knowledge(principal) -> bool:
1006
- """落 knowledge 层必须持有可写该层的令牌(designer 派生);否则 fail-closed。"""
1007
- if principal is None:
1008
- return False
1009
- try:
1010
- if principal.expired() or not principal.can_write:
1011
- return False
1012
- return bool(principal.allows_layer("knowledge"))
1013
- except Exception: # noqa: BLE001
1014
- return False
1015
-
1016
-
1017
- # 生效条件:path 为假值(空串/None)返回 None;path 不存在则 raise SystemExit;已存在且读取文本 strip 后为空串返回 [],非空时整段 json.loads 成功即返回该值,抛 ValueError 时按行解析(跳过空行与 "//" 开头行)返回行列表。
1018
- def _load_verdicts(path: str):
1019
- if not path:
1020
- return None
1021
- if not os.path.exists(path):
1022
- raise SystemExit(f"裁决文件不存在:{path}")
1023
- with open(path, "r", encoding="utf-8") as fh:
1024
- text = fh.read().strip()
1025
- if not text:
1026
- return []
1027
- try:
1028
- return json.loads(text)
1029
- except ValueError:
1030
- rows = []
1031
- for line in text.splitlines():
1032
- line = line.strip()
1033
- if not line or line.startswith("//"):
1034
- continue
1035
- rows.append(json.loads(line))
1036
- return rows
1037
-
1038
-
1039
- # 生效条件:argv(为 None 时由 argparse 读 sys.argv)解析后按 --action 分派——worklist 调 build_worklist,history 调 history(--limit 默认 None,为 None 时传 100),rollback 在 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 rollback,crosscheck 在 --apply 为真且 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 crosscheck;--token(默认 os.environ.get("MDCG_TOKEN") or "")为真值时先 tokens.verify_token 校验、失败抛 SystemExit;最后打印 rep 并返回 0;
1040
- def _cli(argv=None) -> int:
1041
- ap = argparse.ArgumentParser(
1042
- prog="python -m md_cg.crosscheck",
1043
- description="kp_ 批量核对管线(工单/核对/回滚/留痕)")
1044
- ap.add_argument("--root", default=os.environ.get("MDCG_ROOT") or ".")
1045
- ap.add_argument("--token", default=os.environ.get("MDCG_TOKEN") or "")
1046
- ap.add_argument("--token-file", default=None)
1047
- ap.add_argument("--action", default="worklist",
1048
- choices=("worklist", "crosscheck", "rollback", "history"))
1049
- ap.add_argument("--verdicts", default="", help="子代理裁决 JSON/JSONL 路径")
1050
- ap.add_argument("--apply", action="store_true", help="真正写盘(默认 dry-run)")
1051
- ap.add_argument("--batch", default=CROSSCHECK_BATCH)
1052
- ap.add_argument("--limit", type=int, default=None)
1053
- ap.add_argument("--layer", default=None)
1054
- ap.add_argument("--prefix", default=None, help="按 id 前缀收窄(真实库用 kp_)")
1055
- ap.add_argument("--ids", default="", help="逗号分隔节点 id")
1056
- ap.add_argument("--no-verify", action="store_true", help="允许无验证单元(不建议)")
1057
- ap.add_argument("--allow-self-verify", action="store_true")
1058
- ap.add_argument("--reflect-actor", default=None)
1059
- ap.add_argument("--verify-actor", default=None)
1060
- args = ap.parse_args(argv)
1061
-
1062
- from . import tokens
1063
- principal = None
1064
- if args.token:
1065
- try:
1066
- principal = tokens.verify_token(args.token, path=args.token_file)
1067
- except tokens.TokenError as exc:
1068
- raise SystemExit(f"令牌校验失败:{exc}")
1069
- actor = getattr(principal, "actor", None)
1070
- ids = [s.strip() for s in args.ids.split(",") if s.strip()] or None
1071
-
1072
- if args.action == "worklist":
1073
- rep = build_worklist(args.root, layer=args.layer, limit=args.limit,
1074
- ids=ids, prefix=args.prefix)
1075
- elif args.action == "history":
1076
- rep = history(args.root, limit=args.limit if args.limit is not None else 100,
1077
- batch=args.batch)
1078
- elif args.action == "rollback":
1079
- if not can_write_knowledge(principal):
1080
- raise SystemExit("权限不足:回滚需要可写 knowledge 层的令牌")
1081
- rep = rollback(args.root, batch=args.batch, actor=actor)
1082
- else:
1083
- if args.apply and not can_write_knowledge(principal):
1084
- raise SystemExit("权限不足:落库需要可写 knowledge 层的令牌(designer 派生)")
1085
- rep = crosscheck(args.root, layer=args.layer, limit=args.limit, ids=ids,
1086
- prefix=args.prefix,
1087
- verdicts=_load_verdicts(args.verdicts), apply=args.apply,
1088
- batch=args.batch,
1089
- require_verify=not args.no_verify,
1090
- allow_self_verify=args.allow_self_verify,
1091
- reflect_actor=args.reflect_actor,
1092
- verify_actor=args.verify_actor, actor=actor)
1093
- print(json.dumps(rep, ensure_ascii=False, indent=2))
1094
- return 0
1095
-
1096
-
1097
- if __name__ == "__main__": # pragma: no cover
1
+ # -*- coding: utf-8 -*-
2
+ """批量核对(P34/P35):工单 → 反思候选 → 白箱闸门 → 验证否决 → 来源执照 → 落库。
3
+
4
+ 设计约束(与计划一致):
5
+
6
+ * **不新增 MCP op**:本模块是「模块 + `_cli`」形态(同 `backfill.py`),子代理经
7
+ `python -m md_cg.crosscheck …` 调用;令牌决定权限边界。
8
+ * **防自证**:反思单元(`reflect`)与验证单元(`verify`)必须为**不同执行者**;
9
+ 验证单元**只能否决、不能新增**候选(越界字段在白箱闸门直接丢弃)。
10
+ * **来源执照**:文科(`humanities`)只接受来源一致性档(`textbook`/`public_kb`);
11
+ 理科(`science`)只接受可复现档(`compiler`/`test`/`measurement`/`formal_proof`/`data`);
12
+ 赛道未定(`undetermined`)或无来源一律不写,保持 DEFER 并登记待补。
13
+ * **诚实边界**:占位空壳节点(`骨架锚点`/`内容待填充`)不进接线,转待填充工单;
14
+ 拿不出证据的节点绝不写 `verification_basis`。
15
+ * **留痕可回滚**:写入前记录 `_crosscheck.jsonl`,`rollback` 仅在「当前值 == 写入值」
16
+ 时撤销,否则计入 conflict 跳过。
17
+
18
+ `_cli` 之外的所有函数都是纯逻辑(无网络、无第三方依赖),便于离线回归。
19
+ """
20
+
21
+ from __future__ import annotations
22
+
23
+ import argparse
24
+ import json
25
+ import os
26
+ import re
27
+ import sys
28
+ import time
29
+ from collections import OrderedDict
30
+
31
+ from . import crypto, nodefile
32
+ from .backfill import (BASIS_ENUM_DEFAULT, BASIS_TEXT, INTERNAL_LAYERS,
33
+ SKIP_LAYERS, SKIP_TAGS, _as_cg, _as_text, _comment,
34
+ _ensure_comment, _entry_id, _readable_guard,
35
+ _remove_ccg_line, _sha, derive_fields)
36
+ from .consolidate import _has_ccg_line, _upsert_ccg_line
37
+ from .fsutil import append_jsonl, read_jsonl
38
+ from .mdcos import _ccg_field
39
+ from .readcache import direct_read
40
+
41
+ # ---- 常量 ----------------------------------------------------------------
42
+
43
+ CROSSCHECK_LOG = "_crosscheck.jsonl"
44
+ CROSSCHECK_BATCH = "crosscheck"
45
+
46
+ REFLECT_UNIT = "reflect"
47
+ VERIFY_UNIT = "verify"
48
+
49
+ # 可写字段(本管线只动这一处,杜绝越界面)
50
+ WRITABLE_FIELDS = ("验证方式",)
51
+ FIELD_NORMALIZE = {
52
+ "验证方式": "验证方式",
53
+ "verification_basis": "验证方式",
54
+ "验证": "验证方式",
55
+ "verification": "验证方式",
56
+ }
57
+
58
+ # 赛道 → 来源执照策略
59
+ SOURCE_POLICY = {"science": "reproducible", "humanities": "consistency"}
60
+
61
+ # 学科关键词(赛道判定;两栖词不入表 → undetermined,宁缺勿猜)
62
+ SCIENCE_SUBJECT_HINTS = (
63
+ "数学", "物理", "化学", "生物", "科学", "信息技术", "通用技术", "计算机",
64
+ )
65
+ HUMANITIES_SUBJECT_HINTS = (
66
+ "语文", "历史", "政治", "道德与法治", "思想政治", "思想品德", "英语",
67
+ "文学", "哲学", "艺术", "音乐", "美术",
68
+ )
69
+
70
+ # 基底枚举 → 可读标签(条件化表述引用)
71
+ BASIS_LABEL = {
72
+ "compiler": "编译器/静态检查",
73
+ "test": "单元测试",
74
+ "measurement": "实测数据",
75
+ "formal_proof": "形式化证明",
76
+ "data": "数据统计",
77
+ "textbook": "人教版教材",
78
+ "public_kb": "公开知识库",
79
+ "other": "人工评审",
80
+ }
81
+
82
+ CLAIM_FIELDS = ("功能名", "生效条件", "子功能", "执行")
83
+ B_CLAIM = "B_valuation"
84
+ A_CLAIM = "A_fact"
85
+
86
+ # B 型(评价性断言)标记词:只收「明显是价值判断/修饰」的表达,避免把事实误判。
87
+ VALUATION_MARKERS = (
88
+ "结晶", "瑰宝", "杰作", "卓越", "杰出", "伟大", "不朽", "巅峰", "典范",
89
+ "精华", "珍品", "璀璨", "辉煌", "丰碑", "博大精深", "源远流长",
90
+ "不可估量", "无与伦比", "举足轻重", "首屈一指", "独树一帜", "别具一格",
91
+ "最优秀", "极富", "令人叹为观止", "不可磨灭", "辉煌成就", "灿烂", "崇高",
92
+ "非凡",
93
+ )
94
+ VALUATION_PATTERNS = (
95
+ re.compile(r"被誉为|被称[之为]|堪称|不愧[为是]"),
96
+ re.compile(r"是[^,。;\n]{0,24}的(结晶|瑰宝|杰作|典范|精华|骄傲|象征|丰碑)"),
97
+ )
98
+
99
+ CONDITION_MARK = "〔来源限定〕"
100
+
101
+
102
+ # ---- 赛道与来源执照 ------------------------------------------------------
103
+
104
+ # 生效条件:给定 fm,若显式 track/discipline_type 命中枚举则返回对应赛道;否则用相关元数据与正文 CCG 字段匹配提示词,返回 humanities/science/undetermined。
105
+ def classify_track(fm: dict, content: str = "") -> str:
106
+ """判定节点赛道:`humanities` / `science` / `undetermined`。
107
+
108
+ 只看**已声明**的元数据(显式字段 > 学科标记),不扫正文散文,避免误判。
109
+ """
110
+ fm = fm or {}
111
+ explicit = str(fm.get("track") or fm.get("discipline_type") or "").strip().lower()
112
+ if explicit in ("humanities", "文科", "arts"):
113
+ return "humanities"
114
+ if explicit in ("science", "理科", "stem"):
115
+ return "science"
116
+ st = fm.get("state_attributes")
117
+ name = _as_text(st.get("name")) if isinstance(st, dict) else ""
118
+ parts = [
119
+ name,
120
+ _as_text(fm.get("title")),
121
+ _as_text(fm.get("discipline")),
122
+ _as_text(fm.get("subject")),
123
+ " ".join(str(t) for t in (fm.get("tags") or [])),
124
+ _ccg_field(content, "功能名"),
125
+ _ccg_field(content, "执行"),
126
+ _ccg_field(content, "子功能"),
127
+ ]
128
+ text = " ".join(p for p in parts if p)
129
+ has_sci = any(k in text for k in SCIENCE_SUBJECT_HINTS)
130
+ has_hum = any(k in text for k in HUMANITIES_SUBJECT_HINTS)
131
+ if has_sci and not has_hum:
132
+ return "science"
133
+ if has_hum and not has_sci:
134
+ return "humanities"
135
+ return "undetermined"
136
+
137
+
138
+ # 生效条件:给定 track,返回 SOURCE_POLICY 中映射的策略名;未知 track 返回空串。
139
+ def source_policy(track: str) -> str:
140
+ """赛道 → 来源策略名(空串表示不可判定,应 DEFER)。"""
141
+ return SOURCE_POLICY.get(track or "", "")
142
+
143
+
144
+ # 生效条件:给定 track,若为 science 返回 REPRODUCIBLE_BASIS,若为 humanities 返回 CONSISTENCY_BASIS,否则返回 ()。
145
+ def allowed_basis(track: str) -> tuple:
146
+ if track == "science":
147
+ return tuple(nodefile.REPRODUCIBLE_BASIS)
148
+ if track == "humanities":
149
+ return tuple(nodefile.CONSISTENCY_BASIS)
150
+ return ()
151
+
152
+
153
+ # 生效条件:给定 track 与 basis,当 basis 非空且其字符串形式属于 allowed_basis(track) 时返回 True,否则 False。
154
+ def basis_licensed(track: str, basis) -> bool:
155
+ """来源执照:理科要可复现证据,文科要来源一致性;赛道未定一律不发放。"""
156
+ return bool(basis) and str(basis) in allowed_basis(track)
157
+
158
+
159
+ # 生效条件:给定 field,返回 FIELD_NORMALIZE 映射值;未知字段返回空串。
160
+ def normalize_field(field) -> str:
161
+ return FIELD_NORMALIZE.get(str(field or "").strip(), "")
162
+
163
+
164
+ # 生效条件:给定 v,若为 None 返回 [];否则将单值或列表转为去除空白后非空字符串的列表。
165
+ def _as_source(v) -> list:
166
+ if v is None:
167
+ return []
168
+ items = list(v) if isinstance(v, (list, tuple)) else [v]
169
+ return [str(x).strip() for x in items if str(x).strip()]
170
+
171
+
172
+ # ---- B 型识别与条件化改写 ------------------------------------------------
173
+
174
+ # 生效条件:给定 text,若含 VALUATION_MARKERS 或匹配 VALUATION_PATTERNS 则返回 B_CLAIM,否则 A_CLAIM。
175
+ def claim_type(text) -> str:
176
+ """`A_fact`(事实性)或 `B_valuation`(评价性断言)。"""
177
+ s = str(text or "")
178
+ if not s.strip():
179
+ return A_CLAIM
180
+ if any(m in s for m in VALUATION_MARKERS):
181
+ return B_CLAIM
182
+ if any(p.search(s) for p in VALUATION_PATTERNS):
183
+ return B_CLAIM
184
+ return A_CLAIM
185
+
186
+
187
+ # 生效条件:给定 text,返回其去除首尾空白后是否以 CONDITION_MARK 开头。
188
+ def is_conditioned(text) -> bool:
189
+ return str(text or "").strip().startswith(CONDITION_MARK)
190
+
191
+
192
+ # 生效条件:给定 text、label、source,若 text 非空且 label 非空且 source 解析后非空,则返回带 CONDITION_MARK 的来源限定表述;已条件化原样返回;否则 None。
193
+ def conditioned_claim(text, label, source):
194
+ """把评价性断言改写为**带来源限定的条件表述**;缺来源/标签则返回 `None`(不写)。
195
+
196
+ 形态:`〔来源限定〕据<来源标签>(<来源>)的表述:<原文>`——
197
+ 原文完整保留(可追溯),前缀显式声明「这是某来源的表述」而非无条件事实。
198
+ 已条件化的文本原样返回(幂等)。
199
+ """
200
+ body = str(text or "").strip()
201
+ if not body or not label:
202
+ return None
203
+ if is_conditioned(body):
204
+ return body
205
+ src = ";".join(_as_source(source))
206
+ if not src:
207
+ return None
208
+ return f"{CONDITION_MARK}据{label}({src})的表述:{body}"
209
+
210
+
211
+ # 生效条件:给定 fm 与 content,提取 CCG 声明字段、comment 值与正文长句,返回断言列表,每项含 text/type/where/field。
212
+ def extract_claims(fm: dict, content: str) -> list:
213
+ """提取可核对断言:CCG 声明字段 + comment 值 + 正文长句。
214
+
215
+ 每条:`{"text", "type", "where", "field"}`;`where` ∈ ccg/comment/body。
216
+ 占位标记不成为断言。
217
+ """
218
+ out, seen = [], set()
219
+
220
+ # 生效条件:仅当 str(text or "").strip() 得到的 s 长度 >= 4、s 不在 seen 中、且 nodefile.is_placeholder_text(s) 为假时,把 {text: s, type: claim_type(s), where, field} 追加进 out 并把 s 加入 seen,否则直接返回(field 默认 "")。
221
+ def _push(text, where, field=""):
222
+ s = str(text or "").strip()
223
+ if len(s) < 4 or s in seen or nodefile.is_placeholder_text(s):
224
+ return
225
+ seen.add(s)
226
+ out.append({"text": s, "type": claim_type(s), "where": where,
227
+ "field": field})
228
+
229
+ for f in CLAIM_FIELDS:
230
+ v = _ccg_field(content, f)
231
+ if v:
232
+ _push(v, "ccg", f)
233
+ c = _comment(fm)
234
+ for f in CLAIM_FIELDS:
235
+ v = c.get(f)
236
+ if isinstance(v, list):
237
+ for item in v:
238
+ _push(item, "comment", f)
239
+ elif v:
240
+ _push(v, "comment", f)
241
+ for line in (content or "").split("\n"):
242
+ raw = line.strip()
243
+ if not raw:
244
+ continue
245
+ body = raw.lstrip("#").strip()
246
+ if body.split(":", 1)[0].strip() in nodefile.CCG_MARKS:
247
+ continue # 声明行已按 CCG 字段处理,不重复断言
248
+ for sent in re.split(r"[。!?]", body):
249
+ s = sent.strip()
250
+ if len(s) >= 8 and not s.startswith(CONDITION_MARK):
251
+ _push(s, "body", "")
252
+ return out
253
+
254
+
255
+ # 生效条件:给定 fm、content、claim、new_text,按 claim.where 定位并在唯一匹配时替换断言返回 (content, True),否则返回 (content, False)。
256
+ def _rewrite_claim(fm: dict, content: str, claim: dict, new_text: str):
257
+ """节点内定位并替换一条断言 → `(content, ok)`;定位不唯一则 fail-closed 不动。"""
258
+ where, field = claim.get("where"), claim.get("field")
259
+ before = str(claim.get("text") or "")
260
+ if where == "ccg" and field:
261
+ if _ccg_field(content, field).strip() == before.strip():
262
+ return _upsert_ccg_line(content, field, new_text), True
263
+ return content, False
264
+ if where == "comment" and field:
265
+ c = _comment(fm)
266
+ v = c.get(field)
267
+ if isinstance(v, list):
268
+ if before in v:
269
+ c[field] = [new_text if x == before else x for x in v]
270
+ return content, True
271
+ return content, False
272
+ if str(v or "").strip() == before.strip():
273
+ c[field] = new_text
274
+ return content, True
275
+ return content, False
276
+ if where == "body":
277
+ if before and content.count(before) == 1:
278
+ return content.replace(before, new_text), True
279
+ return content, False
280
+ return content, False
281
+
282
+
283
+ # ---- 工单 ----------------------------------------------------------------
284
+
285
+ # 生效条件:给定 fm 与 content,返回缺失项列表:verification_basis 无效则加入该名,正文无 "# 验证方式" 行则加入该名。
286
+ def _need(fm: dict, content: str) -> list:
287
+ need = []
288
+ if not nodefile.verification_basis_valid(fm):
289
+ need.append("verification_basis")
290
+ if not _has_ccg_line(content, "验证方式"):
291
+ need.append("验证方式")
292
+ return need
293
+
294
+
295
+ # 生效条件:给定 nid、e、fm、content,返回含 id、layer、track、claims、need、source_policy 的工单行字典。
296
+ def _worklist_row(nid: str, e: dict, fm: dict, content: str) -> dict:
297
+ track = classify_track(fm, content)
298
+ return {
299
+ "id": nid,
300
+ "layer": e.get("layer"),
301
+ "track": track,
302
+ "claims": extract_claims(fm, content),
303
+ "need": _need(fm, content),
304
+ "source_policy": source_policy(track),
305
+ }
306
+
307
+
308
+ # 生效条件:给定 fm 与 content,若正文或 comment 中声明的执行字段为占位文本则返回 True;未声明执行时以正文整体占位判定。
309
+ def _is_placeholder_shell(fm: dict, content: str) -> bool:
310
+ """空壳判定:核心可执行内容未被填充 → 禁止接线(不得把「待填充」固化成事实)。
311
+
312
+ 口径(宁漏判不误判):
313
+ 1. 已声明 `执行`(正文 `# 执行:` 行优先,其次 `state_attributes.comment.执行`)
314
+ 且值为占位标记 → 空壳;
315
+ 2. 未声明 `执行` 时,以正文整体是否为空/占位标记为准——无 comment 但正文写实的
316
+ `kp_archaeo_*` 类节点因此不被误判为空壳。
317
+ """
318
+ decl = _ccg_field(content, "执行") or _as_text(_comment(fm).get("执行"))
319
+ if decl:
320
+ return nodefile.is_placeholder_text(decl)
321
+ return nodefile.is_placeholder_text(content)
322
+
323
+
324
+ # 生效条件:给定 cg,逐节点按 layer/ids/prefix 过滤后产出状态为 skip(internal/denied/locked/derived/present/placeholder/unreadable 等)或 row 的扫描结果。
325
+ def _scan(cg, layer=None, ids=None, prefix=None):
326
+ """逐节点产出扫描结果:`{"status", "reason"?, "id", "row"?}`。
327
+
328
+ `prefix`:只纳入 id 以该前缀开头的节点(真实库以 `kp_` 收窄到用户知识节点,
329
+ 避免 `node_`/`note_`/`imgpart_` 等派生记忆混入工单);**不计数**,与 `layer` 同理。
330
+ """
331
+ want = set(ids) if ids else None
332
+ for nid, e in list((cg.index.get("nodes") or {}).items()):
333
+ if want is not None and nid not in want:
334
+ continue
335
+ if prefix and not str(nid).startswith(prefix):
336
+ continue
337
+ if layer and e.get("layer") != layer:
338
+ continue
339
+ if e.get("layer") in INTERNAL_LAYERS:
340
+ yield {"status": "skip", "reason": "internal", "id": nid}
341
+ continue
342
+ if not _readable_guard(cg, e):
343
+ yield {"status": "skip", "reason": "denied", "id": nid}
344
+ continue
345
+ fm, content = direct_read(cg, e) # 扫描读喂写面:走盘上真值(读缓存口径)
346
+ if fm is None:
347
+ yield {"status": "skip", "reason": "unreadable", "id": nid}
348
+ continue
349
+ if crypto.is_encrypted(content):
350
+ yield {"status": "skip", "reason": "locked", "id": nid}
351
+ continue
352
+ if e.get("layer") in SKIP_LAYERS or any(
353
+ t in SKIP_TAGS for t in (fm.get("tags") or [])):
354
+ yield {"status": "skip", "reason": "derived", "id": nid}
355
+ continue
356
+ need = _need(fm, content)
357
+ if not need: # 已齐备
358
+ yield {"status": "skip", "reason": "present", "id": nid}
359
+ continue
360
+ ph = []
361
+ der = derive_fields(fm, content, placeholder_out=ph)
362
+ if _is_placeholder_shell(fm, content): # 空壳:转待填充工单
363
+ yield {"status": "skip", "reason": "placeholder", "id": nid,
364
+ "placeholder_fields": ph}
365
+ continue
366
+ yield {"status": "row", "id": nid,
367
+ "row": _worklist_row(nid, e, fm, content)}
368
+
369
+
370
+ _SKIP_KEY = {"locked": "skipped_locked", "derived": "skipped_derived",
371
+ "present": "skipped_present", "denied": "skipped_denied",
372
+ "unreadable": "skipped_unreadable", "internal": "skipped_internal"}
373
+
374
+
375
+ # 生效条件:给定 x(路径或 MdCGOS),只读扫描并生成缺 verification_basis 或 "# 验证方式" 的节点工单,返回统计 rep。
376
+ def build_worklist(x, layer=None, limit=None, ids=None, prefix=None) -> dict:
377
+ """生成核对工单(只读):缺 `verification_basis`/`验证方式` 的节点入列。
378
+
379
+ `prefix`:按 id 前缀收窄范围(真实库用 `kp_`,与计划交付边界一致);
380
+ 内部 `anchor`/`self` 层一律出局(计入 `skipped_internal`)。
381
+ """
382
+ cg = _as_cg(x)
383
+ rep = {"root": cg.root, "dry_run": True, "action": "crosscheck_worklist",
384
+ "prefix": prefix, "nodes_scanned": 0, "skipped_locked": 0,
385
+ "skipped_derived": 0, "skipped_internal": 0, "skipped_present": 0,
386
+ "skipped_denied": 0, "skipped_placeholder": 0,
387
+ "skipped_unreadable": 0, "placeholder_ids": [], "undetermined": 0,
388
+ "targeted": 0, "items": []}
389
+ for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
390
+ rep["nodes_scanned"] += 1
391
+ if scan["status"] == "skip":
392
+ reason = scan["reason"]
393
+ if reason == "placeholder":
394
+ rep["skipped_placeholder"] += 1
395
+ rep["placeholder_ids"].append(scan["id"])
396
+ else:
397
+ key = _SKIP_KEY.get(reason)
398
+ if key:
399
+ rep[key] += 1
400
+ continue
401
+ row = scan["row"]
402
+ if row["track"] == "undetermined":
403
+ rep["undetermined"] += 1
404
+ rep["targeted"] += 1
405
+ if limit is None or len(rep["items"]) < limit:
406
+ rep["items"].append(row)
407
+ rep["planned_ids"] = [r["id"] for r in rep["items"]]
408
+ return rep
409
+
410
+
411
+ # ---- 子代理接口(提示词 + 解析) -----------------------------------------
412
+
413
+ _REFLECT_TEMPLATE = """你是认知图节点的**反思单元**(reflect)。为节点补齐「验证方式」与其验证基底。
414
+ 赛道:{track};来源策略:{policy};待补字段:{need}
415
+ 节点标题:{title}
416
+ 已声明断言:
417
+ {claims}
418
+ 正文:
419
+ {body}
420
+
421
+ 只输出 JSON 数组,元素形如:
422
+ {{"field":"验证方式","value":"<一句可核对的验证方式声明>","basis":"<基底枚举>","source":["<教材版本+章节 或 公开知识库条目地址>"],"verdict":"accept|defer","reason":"<理由>"}}
423
+ 硬约束:
424
+ 1. 文科(humanities)basis 只能是 textbook / public_kb;
425
+ 2. 理科(science)basis 只能是 compiler / test / measurement / formal_proof / data;
426
+ 3. 来源必须可追溯(教材名称+章节,或公开知识库条目地址);拿不出来就把 verdict 置 defer、source 留空;
427
+ 4. 只能补 field=验证方式,禁止新增其它字段。"""
428
+
429
+ _VERIFY_TEMPLATE = """你是独立**验证单元**(verify)。对下列候选逐条复核:来源是否真实可追溯、基底是否与赛道相容。
430
+ 赛道:{track};来源策略:{policy}
431
+ 候选(JSON):
432
+ {candidates}
433
+
434
+ 只输出 JSON 数组,元素形如:
435
+ {{"field":"验证方式","value":"<原样回填候选 value>","verdict":"accept|drop|defer","reason":"<理由>"}}
436
+ 硬约束:你只能否决(drop)或存疑(defer),**不得新增候选、不得改写 value**。"""
437
+
438
+
439
+ # 生效条件:给定 row、fm、content,用 row 的 track/source_policy/need/claims 与 fm 标题、content 前 1200 字符填充反思模板并返回字符串。
440
+ def reflect_prompt(row: dict, fm: dict, content: str) -> str:
441
+ claims = "\n".join(f"- [{c['type']}] {c['text']}" for c in (row.get("claims") or []))
442
+ return _REFLECT_TEMPLATE.format(
443
+ track=row.get("track"), policy=row.get("source_policy") or "(未定)",
444
+ need="、".join(row.get("need") or []),
445
+ title=_as_text(fm.get("title")) or row.get("id"),
446
+ claims=claims or "(无)",
447
+ body=(content or "")[:1200])
448
+
449
+
450
+ # 生效条件:给定 row 与 rows,把候选字段、值、依据、来源序列化为 JSON 并填充验证模板返回字符串。
451
+ def verify_prompt(row: dict, rows: list) -> str:
452
+ cands = [{"field": r.get("field"), "value": r.get("value"),
453
+ "basis": r.get("basis"), "source": r.get("source")} for r in rows]
454
+ return _VERIFY_TEMPLATE.format(
455
+ track=row.get("track"), policy=row.get("source_policy") or "(未定)",
456
+ candidates=json.dumps(cands, ensure_ascii=False))
457
+
458
+
459
+ # 生效条件:raw 经 str(raw or "") 得 s 后,want_list 为真时先试 s 首个 "[" 至末个 "]"、再试首个 "{" 至末个 "}"(want_list 假值时只试花括号),区间可被 json.loads 解析且结果为 list 时原样返回该 list;结果为 dict 时按 rows/items/verdicts/candidates/data 顺序取首个 obj.get(key) 为 list 的 obj[key],都不满足则返回 [obj],非 list/dict 或区间缺失、解析抛 ValueError 时继续下一组括号,全部落空(含 raw 为假值使 s 为空串)返回 []。
460
+ def _extract_json(raw, want_list=True):
461
+ """从模型输出里抽取 JSON(容忍代码围栏与前后废话)。"""
462
+ s = str(raw or "")
463
+ pairs = ([("[", "]")] if want_list else []) + [("{", "}")]
464
+ for op, cl in pairs:
465
+ i, j = s.find(op), s.rfind(cl)
466
+ if i < 0 or j <= i:
467
+ continue
468
+ try:
469
+ obj = json.loads(s[i:j + 1])
470
+ except ValueError:
471
+ continue
472
+ if isinstance(obj, list):
473
+ return obj
474
+ if isinstance(obj, dict):
475
+ for key in ("rows", "items", "verdicts", "candidates", "data"):
476
+ if isinstance(obj.get(key), list):
477
+ return obj[key]
478
+ return [obj]
479
+ return []
480
+
481
+
482
+ # 生效条件:给定 item,若为 dict 则规范化 field/value/basis/source/verdict/reason 后返回字典,否则返回 {}。
483
+ def _norm_row(item) -> dict:
484
+ if not isinstance(item, dict):
485
+ return {}
486
+ return {
487
+ "field": normalize_field(item.get("field")) or str(item.get("field") or "").strip(),
488
+ "value": _as_text(item.get("value")),
489
+ "basis": str(item.get("basis") or "").strip(),
490
+ "source": _as_source(item.get("source")),
491
+ "verdict": str(item.get("verdict") or "").strip().lower(),
492
+ "reason": str(item.get("reason") or "").strip(),
493
+ }
494
+
495
+
496
+ # 生效条件:给定 raw,解析 JSON 行并保留 field 为“验证方式”或“verification_basis”(统一为“验证方式”)的行,返回列表。
497
+ def parse_reflect_rows(raw) -> list:
498
+ out = []
499
+ for item in _extract_json(raw, want_list=True):
500
+ r = _norm_row(item)
501
+ if not r or r["field"] not in ("验证方式", "verification_basis"):
502
+ continue
503
+ r["field"] = "验证方式"
504
+ out.append(r)
505
+ return out
506
+
507
+
508
+ # 生效条件:遍历 _extract_json(raw, want_list=False)(只认花括号 JSON)的结果,仅当 item 经 _norm_row 后为真且 r["field"] 非空时产出 {field,value,verdict,reason} 四键行,否则跳过(raw 无可解析花括号对象时 out 为空列表)。
509
+ def parse_verify_rows(raw) -> list:
510
+ out = []
511
+ for item in _extract_json(raw, want_list=False):
512
+ r = _norm_row(item)
513
+ if not r or not r["field"]:
514
+ continue
515
+ out.append({k: r[k] for k in ("field", "value", "verdict", "reason")})
516
+ return out
517
+
518
+
519
+ # ---- 白箱闸门与双单元折叠 ------------------------------------------------
520
+
521
+ # 生效条件:逐行处理 rows,仅当 normalize_field(r.get("field")) 落在 WRITABLE_FIELDS、basis_licensed(track, r.get("basis")) 为真、r.get("source") 为真、且 r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "") 非空时进入 kept(附 verdict="accept"),否则该行带对应 reason 进入 gated。
522
+ def gate_rows(rows: list, track: str) -> tuple:
523
+ """零模型白箱闸门:字段越界 / 来源执照不通过 / 无来源 → 一律降级为 defer。
524
+
525
+ 返回 `(kept, gated)`;`kept` 只含「执照齐全」的候选,可进验证单元。
526
+ """
527
+ kept, gated = [], []
528
+ for r in rows:
529
+ f = normalize_field(r.get("field"))
530
+ row = dict(r, field=f)
531
+ if f not in WRITABLE_FIELDS:
532
+ gated.append(dict(row, reason=f"越界字段:{r.get('field')}"))
533
+ continue
534
+ if not basis_licensed(track, r.get("basis")):
535
+ gated.append(dict(row, reason=f"{track or '未定赛道'} 不接受基底 {r.get('basis') or '(缺)'}"))
536
+ continue
537
+ if not r.get("source"):
538
+ gated.append(dict(row, reason="无来源,不写"))
539
+ continue
540
+ value = r.get("value") or BASIS_TEXT.get(str(r.get("basis")), "")
541
+ if not value:
542
+ gated.append(dict(row, reason="无验证方式声明"))
543
+ continue
544
+ kept.append(dict(row, value=value, verdict="accept"))
545
+ return kept, gated
546
+
547
+
548
+ # 生效条件:按 (r.get("id"), normalize_field(r.get("field")) or r.get("field"), r.get("value")) 分组后,组内缺 unit==REFLECT_UNIT 或 unit==VERIFY_UNIT 的行时进 deferred,否则 verify 侧出现 verdict=="drop" 即进 dropped(veto 优先),再否则仅当 reflect 与 verify 各存在 verdict=="accept" 时才进 accepted(附 units),其余进 deferred。
549
+ def fold_verdicts(rows: list) -> tuple:
550
+ """把两单元裁决折叠为可落库结论 → `(accepted, deferred, dropped)`。
551
+
552
+ 接受条件:同 `(id, field, value)` 同时存在 reflect-accept 与 verify-accept;
553
+ 任一 verify-drop 即否决(veto 优先)。
554
+ """
555
+ groups = OrderedDict()
556
+ for r in rows:
557
+ key = (r.get("id"), normalize_field(r.get("field")) or r.get("field"),
558
+ r.get("value"))
559
+ groups.setdefault(key, []).append(r)
560
+ accepted, deferred, dropped = [], [], []
561
+ for (nid, field, value), rs in groups.items():
562
+ refl = [r for r in rs if r.get("unit") == REFLECT_UNIT]
563
+ ver = [r for r in rs if r.get("unit") == VERIFY_UNIT]
564
+ base = {"id": nid, "field": field or "验证方式", "value": value,
565
+ "basis": next((r.get("basis") for r in refl if r.get("basis")), ""),
566
+ "source": next((r.get("source") for r in refl if r.get("source")), []),
567
+ "track": next((r.get("track") for r in refl if r.get("track")), "")}
568
+ if not refl or not ver:
569
+ base["reason"] = "缺" + ("反思裁决" if not refl else "验证裁决")
570
+ deferred.append(base)
571
+ continue
572
+ if any(r.get("verdict") == "drop" for r in ver):
573
+ base["reason"] = next((r.get("reason") for r in ver
574
+ if r.get("verdict") == "drop"), "验证单元否决")
575
+ dropped.append(base)
576
+ continue
577
+ ra = any(r.get("verdict") == "accept" for r in refl)
578
+ va = any(r.get("verdict") == "accept" for r in ver)
579
+ if ra and va:
580
+ base["units"] = {
581
+ "reflect": sorted({str(r.get("actor") or REFLECT_UNIT) for r in refl}),
582
+ "verify": sorted({str(r.get("actor") or VERIFY_UNIT) for r in ver}),
583
+ }
584
+ accepted.append(base)
585
+ else:
586
+ base["reason"] = f"单元未确认(reflect={ra}, verify={va})"
587
+ deferred.append(base)
588
+ return accepted, deferred, dropped
589
+
590
+
591
+ # 生效条件:当 rows 中 unit==REFLECT_UNIT 与 unit==VERIFY_UNIT 的执行者经 str(x.get("actor") or "") 后存在相同的非空值(空串被 discard)时返回 True,否则返回 False。
592
+ def detect_self_verify(rows: list) -> bool:
593
+ """同一执行者同时充当反思与验证 = 自证(禁止)。"""
594
+ r = {str(x.get("actor") or "") for x in rows if x.get("unit") == REFLECT_UNIT}
595
+ v = {str(x.get("actor") or "") for x in rows if x.get("unit") == VERIFY_UNIT}
596
+ r.discard("")
597
+ v.discard("")
598
+ return bool(r & v)
599
+
600
+
601
+ # ---- 落库写入 ------------------------------------------------------------
602
+
603
+ # 生效条件:verdicts 为 None 时返回 None;verdicts 为 dict 时对每个键值把 (rs or []) 中的 dict 元素收为 {str(nid): [...]};否则遍历 verdicts or [],仅当元素为 dict 且 str(r.get("id") or "") 非空时按该 id 追加到对应列表。
604
+ def _norm_verdicts(verdicts):
605
+ """外部裁决(子代理落盘)→ `{id: [rows]}`。"""
606
+ if verdicts is None:
607
+ return None
608
+ out = OrderedDict()
609
+ if isinstance(verdicts, dict):
610
+ for nid, rs in verdicts.items():
611
+ out[str(nid)] = [dict(x) for x in (rs or []) if isinstance(x, dict)]
612
+ return out
613
+ for r in verdicts or []:
614
+ if not isinstance(r, dict):
615
+ continue
616
+ nid = str(r.get("id") or "")
617
+ if nid:
618
+ out.setdefault(nid, []).append(dict(r))
619
+ return out
620
+
621
+
622
+ # 生效条件:accepted 非空(取 accepted[0])时,先以 basis=str(a.get("basis") or BASIS_ENUM_DEFAULT) 与 value=str(a.get("value") or BASIS_TEXT.get(basis, "")).strip() 写「验证方式」行与 comment,之后才在 nodefile.verification_basis_valid(fm) 为真时把 basis 换成 fm.get("verification_basis")、否则把该 basis 写入 fm["verification_basis"];condition_claims 为真时仅对 row.get("claims") 中 type==B_CLAIM 且未被 is_conditioned 的条目做条件化改写,返回含 fm_before、content_hash_before 等留痕的 dict。
623
+ def _apply_node(cg, nid, e, fm, content, accepted, row, batch, actor,
624
+ condition_claims=True):
625
+ """把一个节点的已接受结论写入 md,返回留痕记录(含回滚所需现场)。"""
626
+ a = accepted[0]
627
+ basis = str(a.get("basis") or BASIS_ENUM_DEFAULT)
628
+ source = _as_source(a.get("source"))
629
+ content_before = content
630
+ value = str(a.get("value") or BASIS_TEXT.get(basis, "")).strip()
631
+
632
+ vb_before = {"had": "verification_basis" in fm,
633
+ "value": fm.get("verification_basis")}
634
+ prov_before = {"had": "verification_evidence" in fm,
635
+ "value": fm.get("verification_evidence")}
636
+ line_before = {"had": _has_ccg_line(content, "验证方式"),
637
+ "value": _ccg_field(content, "验证方式")}
638
+ comment0 = _comment(fm)
639
+ cv_before = {"had": "验证方式" in comment0, "value": comment0.get("验证方式")}
640
+
641
+ # 1) 落「验证方式」规范行 + comment + verification_basis(已有合法基底不覆盖)
642
+ content = _upsert_ccg_line(content, "验证方式", value)
643
+ _ensure_comment(fm)["验证方式"] = value
644
+ if nodefile.verification_basis_valid(fm):
645
+ basis = str(fm.get("verification_basis"))
646
+ else:
647
+ fm["verification_basis"] = basis
648
+
649
+ # 2) B 型评价断言 → 条件化表述(A 型保持原样;无来源已由闸门挡掉)
650
+ conditioned = []
651
+ if condition_claims:
652
+ label = BASIS_LABEL.get(basis, basis)
653
+ for c in row.get("claims") or []:
654
+ if c.get("type") != B_CLAIM or is_conditioned(c.get("text")):
655
+ continue
656
+ new = conditioned_claim(c.get("text"), label, source)
657
+ if not new or new == c.get("text"):
658
+ continue
659
+ nc, ok = _rewrite_claim(fm, content, c, new)
660
+ if not ok:
661
+ continue
662
+ content = nc
663
+ conditioned.append({"where": c.get("where"), "field": c.get("field") or "",
664
+ "before": c.get("text"), "after": new})
665
+
666
+ wid = _sha(f"{nid}|{batch}|{time.time()}")
667
+ fm["verification_evidence"] = {
668
+ "at": round(time.time(), 3), "batch": batch, "basis": basis,
669
+ "source": source, "track": row.get("track"),
670
+ "policy": row.get("source_policy"), "write_id": wid,
671
+ "units": a.get("units") or {},
672
+ "conditioned": len(conditioned),
673
+ }
674
+
675
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
676
+ durable=True)
677
+ return {
678
+ "action": "crosscheck", "ts": time.time(), "batch": batch,
679
+ "actor": actor, "entry_id": _entry_id(batch, nid), "write_id": wid,
680
+ "node": nid, "layer": e.get("layer"), "track": row.get("track"),
681
+ "policy": row.get("source_policy"), "basis": basis,
682
+ "verification_value": value, "source": source,
683
+ "units": a.get("units") or {},
684
+ "fm_before": {"verification_basis": vb_before,
685
+ "verification_evidence": prov_before,
686
+ "comment_verification": cv_before},
687
+ "verification_line_before": line_before,
688
+ "claims_conditioned": conditioned,
689
+ "content_hash_before": _sha(content_before),
690
+ "content_hash_after": _sha(content),
691
+ }
692
+
693
+
694
+ # ---- 主流程 --------------------------------------------------------------
695
+
696
+ # 生效条件:reflect_fn(reflect_prompt(scan_row, fm, content)) 经 parse_reflect_rows 得到非空候选时返回 (rrows, vrows),rrows 为空则返回 ([], []);verify_fn 为 None 时 vrows 为空列表,非 None 时由 parse_verify_rows(verify_fn(verify_prompt(scan_row, rrows))) 生成、每行 value 为 c.get("value") or rrows 中同 field 的 value、再回落 ""。
697
+ def _rows_for(scan_row, fm, content, reflect_fn, verify_fn, r_actor, v_actor):
698
+ """调用两单元子代理,返回合并后的裁决行(reflect + verify)。"""
699
+ prompt = reflect_prompt(scan_row, fm, content)
700
+ cands = parse_reflect_rows(reflect_fn(prompt))
701
+ rrows = [dict(c, id=scan_row["id"], unit=REFLECT_UNIT, actor=r_actor,
702
+ track=scan_row["track"]) for c in cands]
703
+ if not rrows:
704
+ return [], []
705
+ vrows = []
706
+ if verify_fn is not None:
707
+ vp = verify_prompt(scan_row, rrows)
708
+ vrows = [dict(c, id=scan_row["id"], unit=VERIFY_UNIT, actor=v_actor,
709
+ track=scan_row["track"],
710
+ value=c.get("value") or next(
711
+ (x.get("value") for x in rrows
712
+ if x.get("field") == c.get("field")), ""))
713
+ for c in parse_verify_rows(verify_fn(vp))]
714
+ return rrows, vrows
715
+
716
+
717
+ # 生效条件:对 pre 中每个 r,str(r.get("unit") or REFLECT_UNIT).strip().lower() 等于 VERIFY_UNIT 时进 vrows,否则(含 unit 缺失回落到 REFLECT_UNIT 及任何其他取值)进 rrows,两组行均覆盖 id=nid、unit、track=track。
718
+ def _rows_from_verdicts(nid, track, pre):
719
+ """从外部裁决中拆出 (reflect, verify) 两组行。"""
720
+ rrows, vrows = [], []
721
+ for r in pre:
722
+ unit = str(r.get("unit") or REFLECT_UNIT).strip().lower()
723
+ base = dict(r, id=nid, unit=unit, track=track)
724
+ (vrows if unit == VERIFY_UNIT else rrows).append(base)
725
+ return rrows, vrows
726
+
727
+
728
+ # 生效条件:x 经 _as_cg 解析且 batch = batch or CROSSCHECK_BATCH 后逐节点扫描,裁决来源按 vmap(verdicts 归一化后非 None)→ reflect_fn 非 None → 二者皆无记 no_reflect 三条分支取行;allow_self_verify=False 时同执行者自证记 self_verify_disallowed,再经 gate_rows 闸门与 require_verify 后 fold_verdicts,仅 apply=True 才 _apply_node 写盘并在有写入时 cg.rebuild_index;limit 非 None 且已达标数 >= limit 时用 continue 跳过(非终止)。
729
+ def crosscheck(x, layer=None, limit=None, ids=None, reflect_fn=None,
730
+ verify_fn=None, verdicts=None, apply=False,
731
+ batch=CROSSCHECK_BATCH, actor=None, require_verify=True,
732
+ allow_self_verify=False, reflect_actor=None, verify_actor=None,
733
+ condition_claims=True, verbose=True, prefix=None) -> dict:
734
+ """批量核对主流程:工单 → 反思候选 → 白箱闸门 → 验证否决 → 落库。
735
+
736
+ `reflect_fn`/`verify_fn`:可注入的子代理函数(接收提示词、返回 JSON 文本);
737
+ `verdicts`:子代理离线产出的裁决行(`[VERDICT_ROW]` 或 `{id: [rows]}`),
738
+ 二选一。`apply=True` 才写盘。
739
+ """
740
+ cg = _as_cg(x)
741
+ batch = batch or CROSSCHECK_BATCH
742
+ vmap = _norm_verdicts(verdicts)
743
+ r_actor = reflect_actor or getattr(reflect_fn, "__name__", "") or REFLECT_UNIT
744
+ v_actor = verify_actor or getattr(verify_fn, "__name__", "") or VERIFY_UNIT
745
+
746
+ rep = {"root": cg.root, "dry_run": not apply, "action": "crosscheck",
747
+ "batch": batch, "actor": actor, "prefix": prefix, "nodes_scanned": 0,
748
+ "targeted": 0,
749
+ "accepted": 0, "rejected": 0, "deferred": 0, "written": 0,
750
+ "claims_conditioned": 0, "skipped_locked": 0, "skipped_derived": 0,
751
+ "skipped_internal": 0, "skipped_present": 0, "skipped_denied": 0,
752
+ "skipped_unreadable": 0, "skipped_placeholder": 0,
753
+ "placeholder_ids": [], "undetermined": 0,
754
+ "reasons": {}, "samples": [], "entry_ids": []}
755
+
756
+ # 生效条件:无条件执行 rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1(reason 缺键时按 .get 的第二参数 0 起算),返回 None。
757
+ def _bump(reason):
758
+ rep["reasons"][reason] = rep["reasons"].get(reason, 0) + 1
759
+
760
+ # 生效条件:仅当外层 verbose 为真且 len(rep["samples"]) < 20 时把 {kind, id: nid, detail} 追加进 rep["samples"],否则不追加(已达 20 条即停止采样)。
761
+ def _sample(kind, nid, detail=""):
762
+ if verbose and len(rep["samples"]) < 20:
763
+ rep["samples"].append({"kind": kind, "id": nid, "detail": detail})
764
+
765
+ seen_targets = 0
766
+ for scan in _scan(cg, layer=layer, ids=ids, prefix=prefix):
767
+ rep["nodes_scanned"] += 1
768
+ if scan["status"] == "skip":
769
+ reason = scan["reason"]
770
+ if reason == "placeholder":
771
+ rep["skipped_placeholder"] += 1
772
+ rep["placeholder_ids"].append(scan["id"])
773
+ else:
774
+ key = _SKIP_KEY.get(reason)
775
+ if key:
776
+ rep[key] += 1
777
+ continue
778
+ row = scan["row"]
779
+ if row["track"] == "undetermined":
780
+ rep["undetermined"] += 1
781
+ if limit is not None and seen_targets >= limit:
782
+ continue
783
+ seen_targets += 1
784
+ rep["targeted"] += 1
785
+ nid = row["id"]
786
+ e = cg.index["nodes"].get(nid)
787
+ fm, content = direct_read(cg, e) if e else (None, None) # 写前重查走盘上真值
788
+ if fm is None or crypto.is_encrypted(content):
789
+ rep["skipped_locked"] += 1
790
+ continue
791
+
792
+ # ---- 取两单元裁决 ----
793
+ if vmap is not None:
794
+ pre = vmap.get(nid)
795
+ if not pre:
796
+ rep["deferred"] += 1
797
+ _bump("no_verdict")
798
+ continue
799
+ rrows, vrows = _rows_from_verdicts(nid, row["track"], pre)
800
+ elif reflect_fn is not None:
801
+ try:
802
+ rrows, vrows = _rows_for(row, fm, content, reflect_fn,
803
+ verify_fn, r_actor, v_actor)
804
+ except Exception as exc: # noqa: BLE001
805
+ rep["deferred"] += 1
806
+ _bump(f"unit_error:{type(exc).__name__}")
807
+ continue
808
+ else:
809
+ rep["deferred"] += 1
810
+ _bump("no_reflect")
811
+ continue
812
+
813
+ # 自证:同一执行者既反思又验证 → 拒收
814
+ if not allow_self_verify and detect_self_verify(rrows + vrows):
815
+ rep["deferred"] += 1
816
+ _bump("self_verify_disallowed")
817
+ _sample("self_verify", nid)
818
+ continue
819
+
820
+ # ---- 白箱闸门(对反思候选;验证行只做字段归位) ----
821
+ kept, gated = gate_rows(rrows, row["track"])
822
+ for g in gated:
823
+ _bump(f"gate:{g.get('reason')[:24]}")
824
+ if not kept:
825
+ rep["deferred"] += 1
826
+ _bump("no_candidate")
827
+ _sample("gated", nid, gated[0].get("reason") if gated else "")
828
+ continue
829
+ if require_verify and not vrows:
830
+ rep["deferred"] += 1
831
+ _bump("verify_unavailable")
832
+ continue
833
+
834
+ rows = kept + vrows
835
+ accepted, deferred, dropped = fold_verdicts(rows)
836
+ if dropped and not accepted:
837
+ rep["rejected"] += 1
838
+ _bump("verify_veto")
839
+ _sample("veto", nid, dropped[0].get("reason", ""))
840
+ continue
841
+ if not accepted:
842
+ rep["deferred"] += 1
843
+ _bump("verdict_deferred")
844
+ _sample("deferred", nid, deferred[0].get("reason", "") if deferred else "")
845
+ continue
846
+
847
+ rep["accepted"] += 1
848
+ rep["claims_conditioned"] += sum(
849
+ 1 for c in (row.get("claims") or []) if c.get("type") == B_CLAIM)
850
+ if apply:
851
+ rec = _apply_node(cg, nid, e, fm, content, accepted, row, batch,
852
+ actor, condition_claims=condition_claims)
853
+ append_jsonl(_log_path(cg), rec)
854
+ rep["written"] += 1
855
+ rep["entry_ids"].append(rec["entry_id"])
856
+ _sample("accepted", nid, accepted[0].get("basis", ""))
857
+
858
+ if rep["written"]:
859
+ cg.rebuild_index()
860
+ return rep
861
+
862
+
863
+ # ---- 留痕查询 / 回滚 -----------------------------------------------------
864
+
865
+ # 生效条件:无条件返回 os.path.join(cg.root, CROSSCHECK_LOG)(以 cg.root 与常量 CROSSCHECK_LOG 拼接,无分支)。
866
+ def _log_path(cg) -> str:
867
+ return os.path.join(cg.root, CROSSCHECK_LOG)
868
+
869
+
870
+ # 生效条件:box 非 dict 时返回 False;box 为 dict 且 key=="comment_verification" 时按 box.get("had") 为真则把 comment 的「验证方式」设为 box.get("value")、否则删除该键并返回 True;其他 key 时 had 为真赋 fm[key]=value、否则 fm.pop(key, None) 并返回 True。
871
+ def _reattach(fm: dict, content: str, box: dict, key: str):
872
+ """把 `fm_before[key]` 现场还原到 fm,返回是否发生还原。"""
873
+ if not isinstance(box, dict):
874
+ return False
875
+ had, value = box.get("had"), box.get("value")
876
+ if key == "comment_verification":
877
+ c = _ensure_comment(fm)
878
+ if had:
879
+ c["验证方式"] = value
880
+ else:
881
+ c.pop("验证方式", None)
882
+ return True
883
+ if had:
884
+ fm[key] = value
885
+ else:
886
+ fm.pop(key, None)
887
+ return True
888
+
889
+
890
+ # 生效条件:仅当 str(c.get("after") or "") 非空,且分别满足 where=="ccg" 且 field 真值且 _ccg_field(content, field).strip()==after.strip()(用 before 覆盖该行)、where=="comment" 且 field 真值且 comment 该 field 为含 after 的 list 或 str(v or "").strip()==after.strip()(改为 before)、where=="body" 且 after 出现在 content 中(替换首个匹配)时返回 (content, True);其余情形(含 where 为其他值、字段缺失、当前值不等于写入值)返回 (content, False)。
891
+ def _rewind_claim(fm: dict, content: str, c: dict):
892
+ """撤销一条条件化改写(仅当前值 == 写入值时才动)→ `(content, ok)`。"""
893
+ where, field = c.get("where"), c.get("field")
894
+ before, after = str(c.get("before") or ""), str(c.get("after") or "")
895
+ if not after:
896
+ return content, False
897
+ if where == "ccg" and field:
898
+ if _ccg_field(content, field).strip() == after.strip():
899
+ return _upsert_ccg_line(content, field, before), True
900
+ return content, False
901
+ if where == "comment" and field:
902
+ cc = _comment(fm)
903
+ v = cc.get(field)
904
+ if isinstance(v, list):
905
+ if after in v:
906
+ cc[field] = [before if x == after else x for x in v]
907
+ return content, True
908
+ return content, False
909
+ if str(v or "").strip() == after.strip():
910
+ cc[field] = before
911
+ return content, True
912
+ return content, False
913
+ if where == "body":
914
+ if after in content:
915
+ return content.replace(after, before, 1), True
916
+ return content, False
917
+ return content, False
918
+
919
+
920
+ # 生效条件:x 经 _as_cg 后,对 read_jsonl(_log_path(cg)) 中 action=="crosscheck"、batch 为 None 或等于参数 batch、且 entry_ids 为假值不做 id 过滤(为真值时仅取 entry_id 在集合中的)的记录逐条处理:node 缺失或已处理则跳过,索引无该 node 或 cg._read 得 fm 为 None 或 crypto.is_encrypted(content) 为真时 skipped_drift 加一,write_id 双方非空且不等时 conflict 加一,否则撤销 claims_conditioned、在当前「验证方式」行非空且等于 rec 的 verification_value 时撤销该行、再按 fm_before 还原,reverted 为空则 conflict 加一,非空则写回节点、追加 crosscheck_rollback 日志、reverted 与 entry_ids 加一,最终 reverted 非零时 cg.rebuild_index(),返回 rep;
921
+ def rollback(x, batch=None, entry_ids=None, actor=None) -> dict:
922
+ """按留痕反向应用:撤销核对写入(当前值 ≠ 写入值时跳过,计入 conflict)。"""
923
+ cg = _as_cg(x)
924
+ want = set(entry_ids) if entry_ids else None
925
+ done = set()
926
+ rep = {"root": cg.root, "dry_run": False, "action": "crosscheck_rollback",
927
+ "batch": batch, "actor": actor, "planned": 0, "reverted": 0,
928
+ "skipped_drift": 0, "conflict": 0, "entry_ids": []}
929
+ recs = [r for r in (read_jsonl(_log_path(cg)) or [])
930
+ if r.get("action") == "crosscheck"
931
+ and (batch is None or r.get("batch") == batch)
932
+ and (want is None or r.get("entry_id") in want)]
933
+ rep["planned"] = len(recs)
934
+ for rec in recs:
935
+ nid = rec.get("node")
936
+ if not nid or nid in done:
937
+ continue
938
+ e = cg.index["nodes"].get(nid)
939
+ if not e:
940
+ rep["skipped_drift"] += 1
941
+ continue
942
+ fm, content = direct_read(cg, e) # 回滚比对走盘上真值(write_id 防护所验 fm 同源)
943
+ if fm is None or crypto.is_encrypted(content):
944
+ rep["skipped_drift"] += 1
945
+ continue
946
+ # 写入现场校验:write_id 一致才回滚(防「写入后又被改过」被误撤)
947
+ wid = (fm.get("verification_evidence") or {}).get("write_id")
948
+ if wid and rec.get("write_id") and wid != rec.get("write_id"):
949
+ rep["conflict"] += 1
950
+ continue
951
+ # 1) 撤销条件化改写(先于验证方式行,避免行被覆盖影响定位)
952
+ reverted = []
953
+ for c in rec.get("claims_conditioned") or []:
954
+ content, ok = _rewind_claim(fm, content, c)
955
+ if ok:
956
+ reverted.append(c.get("field") or c.get("where"))
957
+ # 2) 撤销「验证方式」行
958
+ lb = rec.get("verification_line_before") or {}
959
+ cur_line = _ccg_field(content, "验证方式")
960
+ if cur_line.strip() and cur_line.strip() == str(
961
+ rec.get("verification_value") or "").strip():
962
+ content = (_upsert_ccg_line(content, "验证方式", lb.get("value") or "")
963
+ if lb.get("had") else _remove_ccg_line(content, "验证方式"))
964
+ reverted.append("验证方式")
965
+ # 3) 还原 frontmatter 现场
966
+ for key, box in (rec.get("fm_before") or {}).items():
967
+ _reattach(fm, content, box, key)
968
+ if not reverted:
969
+ rep["conflict"] += 1
970
+ continue
971
+ cg._write_node(nid, os.path.join(cg.root, e["path"]), fm, content,
972
+ durable=True)
973
+ append_jsonl(_log_path(cg), {
974
+ "action": "crosscheck_rollback", "ts": time.time(),
975
+ "batch": rec.get("batch"), "actor": actor,
976
+ "entry_id": rec.get("entry_id"), "node": nid,
977
+ "reverted": reverted, "content_hash_after": _sha(content)})
978
+ rep["reverted"] += 1
979
+ rep["entry_ids"].append(rec.get("entry_id"))
980
+ done.add(nid)
981
+ if rep["reverted"]:
982
+ cg.rebuild_index()
983
+ return rep
984
+
985
+
986
+ # 生效条件:遍历 _log_path(cg) 的记录时,action 为真值只留 rec.get("action")==action 的行、batch 为真值只留 rec.get("batch")==batch 的行;limit 非 None 且 limit>=0 时按 recs[-limit:] 截取(limit 为 0 时 [-0:] 即整表不被削减),否则保留全部;返回 {'root','total','returned','records'}。
987
+ def history(x, limit=100, action=None, batch=None) -> dict:
988
+ cg = _as_cg(x)
989
+ recs = []
990
+ for rec in read_jsonl(_log_path(cg)) or []:
991
+ if action and rec.get("action") != action:
992
+ continue
993
+ if batch and rec.get("batch") != batch:
994
+ continue
995
+ recs.append(rec)
996
+ total = len(recs)
997
+ if limit is not None and limit >= 0:
998
+ recs = recs[-limit:]
999
+ return {"root": cg.root, "total": total, "returned": len(recs),
1000
+ "records": recs}
1001
+
1002
+
1003
+ # ---- 权限与 CLI ----------------------------------------------------------
1004
+
1005
+ # 生效条件:principal 为 None 时返回 False;否则仅当 principal.expired() 为假、principal.can_write 为真、且 principal.allows_layer("knowledge") 为真时返回 True,期间任一步抛 Exception 亦返回 False。
1006
+ def can_write_knowledge(principal) -> bool:
1007
+ """落 knowledge 层必须持有可写该层的令牌(designer 派生);否则 fail-closed。"""
1008
+ if principal is None:
1009
+ return False
1010
+ try:
1011
+ if principal.expired() or not principal.can_write:
1012
+ return False
1013
+ return bool(principal.allows_layer("knowledge"))
1014
+ except Exception: # noqa: BLE001
1015
+ return False
1016
+
1017
+
1018
+ # 生效条件:path 为假值(空串/None)返回 None;path 不存在则 raise SystemExit;已存在且读取文本 strip 后为空串返回 [],非空时整段 json.loads 成功即返回该值,抛 ValueError 时按行解析(跳过空行与 "//" 开头行)返回行列表。
1019
+ def _load_verdicts(path: str):
1020
+ if not path:
1021
+ return None
1022
+ if not os.path.exists(path):
1023
+ raise SystemExit(f"裁决文件不存在:{path}")
1024
+ with open(path, "r", encoding="utf-8") as fh:
1025
+ text = fh.read().strip()
1026
+ if not text:
1027
+ return []
1028
+ try:
1029
+ return json.loads(text)
1030
+ except ValueError:
1031
+ rows = []
1032
+ for line in text.splitlines():
1033
+ line = line.strip()
1034
+ if not line or line.startswith("//"):
1035
+ continue
1036
+ rows.append(json.loads(line))
1037
+ return rows
1038
+
1039
+
1040
+ # 生效条件:argv(为 None 时由 argparse 读 sys.argv)解析后按 --action 分派——worklist 调 build_worklist,history 调 history(--limit 默认 None,为 None 时传 100),rollback 在 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 rollback,crosscheck 在 --apply 为真且 can_write_knowledge(principal) 为假时抛 SystemExit 否则调 crosscheck;--token(默认 os.environ.get("MDCG_TOKEN") or "")为真值时先 tokens.verify_token 校验、失败抛 SystemExit;最后打印 rep 并返回 0;
1041
+ def _cli(argv=None) -> int:
1042
+ ap = argparse.ArgumentParser(
1043
+ prog="python -m md_cg.crosscheck",
1044
+ description="kp_ 批量核对管线(工单/核对/回滚/留痕)")
1045
+ ap.add_argument("--root", default=os.environ.get("MDCG_ROOT") or ".")
1046
+ ap.add_argument("--token", default=os.environ.get("MDCG_TOKEN") or "")
1047
+ ap.add_argument("--token-file", default=None)
1048
+ ap.add_argument("--action", default="worklist",
1049
+ choices=("worklist", "crosscheck", "rollback", "history"))
1050
+ ap.add_argument("--verdicts", default="", help="子代理裁决 JSON/JSONL 路径")
1051
+ ap.add_argument("--apply", action="store_true", help="真正写盘(默认 dry-run)")
1052
+ ap.add_argument("--batch", default=CROSSCHECK_BATCH)
1053
+ ap.add_argument("--limit", type=int, default=None)
1054
+ ap.add_argument("--layer", default=None)
1055
+ ap.add_argument("--prefix", default=None, help="按 id 前缀收窄(真实库用 kp_)")
1056
+ ap.add_argument("--ids", default="", help="逗号分隔节点 id")
1057
+ ap.add_argument("--no-verify", action="store_true", help="允许无验证单元(不建议)")
1058
+ ap.add_argument("--allow-self-verify", action="store_true")
1059
+ ap.add_argument("--reflect-actor", default=None)
1060
+ ap.add_argument("--verify-actor", default=None)
1061
+ args = ap.parse_args(argv)
1062
+
1063
+ from . import tokens
1064
+ principal = None
1065
+ if args.token:
1066
+ try:
1067
+ principal = tokens.verify_token(args.token, path=args.token_file)
1068
+ except tokens.TokenError as exc:
1069
+ raise SystemExit(f"令牌校验失败:{exc}")
1070
+ actor = getattr(principal, "actor", None)
1071
+ ids = [s.strip() for s in args.ids.split(",") if s.strip()] or None
1072
+
1073
+ if args.action == "worklist":
1074
+ rep = build_worklist(args.root, layer=args.layer, limit=args.limit,
1075
+ ids=ids, prefix=args.prefix)
1076
+ elif args.action == "history":
1077
+ rep = history(args.root, limit=args.limit if args.limit is not None else 100,
1078
+ batch=args.batch)
1079
+ elif args.action == "rollback":
1080
+ if not can_write_knowledge(principal):
1081
+ raise SystemExit("权限不足:回滚需要可写 knowledge 层的令牌")
1082
+ rep = rollback(args.root, batch=args.batch, actor=actor)
1083
+ else:
1084
+ if args.apply and not can_write_knowledge(principal):
1085
+ raise SystemExit("权限不足:落库需要可写 knowledge 层的令牌(designer 派生)")
1086
+ rep = crosscheck(args.root, layer=args.layer, limit=args.limit, ids=ids,
1087
+ prefix=args.prefix,
1088
+ verdicts=_load_verdicts(args.verdicts), apply=args.apply,
1089
+ batch=args.batch,
1090
+ require_verify=not args.no_verify,
1091
+ allow_self_verify=args.allow_self_verify,
1092
+ reflect_actor=args.reflect_actor,
1093
+ verify_actor=args.verify_actor, actor=actor)
1094
+ print(json.dumps(rep, ensure_ascii=False, indent=2))
1095
+ return 0
1096
+
1097
+
1098
+ if __name__ == "__main__": # pragma: no cover
1098
1099
  sys.exit(_cli())