@furongjun1999/dsh-memory 0.4.11 → 0.5.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (504) hide show
  1. package/README.md +16 -16
  2. package/codebuddy/CODEBUDDY.md +11 -3
  3. package/codebuddy/README.md +92 -90
  4. package/codebuddy/mcp.json +27 -27
  5. package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
  6. package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
  7. package/docs/README.md +142 -111
  8. package/docs/discipline/harnesses.yaml +244 -226
  9. package/docs/discipline/templates/full.md.tmpl +61 -61
  10. package/docs/discipline/templates/rules.mdc.tmpl +68 -0
  11. package/docs/discipline/templates/skill.md.tmpl +23 -23
  12. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
  13. package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
  14. package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
  15. package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
  16. package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
  17. package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
  18. package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
  19. package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
  20. package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
  21. package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
  22. package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
  23. package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
  24. package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
  25. package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
  26. package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
  27. package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
  28. package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
  29. package/docs/mdcg/lingshu_tutorial.html +14449 -14449
  30. package/docs/mdcg/release_v0.4.11.md +49 -0
  31. package/docs/mdcg/release_v0.4.5.md +55 -55
  32. package/docs/mdcg/tool_table_v0.3.0.md +117 -117
  33. package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
  34. package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
  35. package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
  36. package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
  37. package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
  38. package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
  39. package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
  40. package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
  41. package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
  42. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
  43. package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
  44. package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
  45. package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
  46. package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
  47. package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
  48. package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
  49. package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
  50. package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  51. package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
  52. package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
  53. package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
  54. package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
  55. package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
  56. package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
  57. package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
  58. package/dsh/README.md +82 -82
  59. package/dsh/cordis.yml.example +139 -139
  60. package/dsh/update-lingshu.bat +11 -11
  61. package/lib/hooks.js +36 -2
  62. package/lib/lib/roleplay_web.js +427 -427
  63. package/md_cg/__init__.py +7 -7
  64. package/md_cg/audit.py +368 -368
  65. package/md_cg/autonomy.py +287 -287
  66. package/md_cg/backfill.py +1327 -1327
  67. package/md_cg/backfill_bigdomain.py +34 -34
  68. package/md_cg/bench6_arms.py +410 -410
  69. package/md_cg/bench6_common.py +230 -230
  70. package/md_cg/bench6_competitors.py +212 -212
  71. package/md_cg/bench_axis_domain.py +257 -257
  72. package/md_cg/bench_blind_comp.py +308 -308
  73. package/md_cg/bench_en_atoms_public.py +230 -230
  74. package/md_cg/bench_governance.py +348 -348
  75. package/md_cg/bench_lme_zh.py +410 -410
  76. package/md_cg/bench_locomo.py +121 -121
  77. package/md_cg/bench_locomo_zh.py +450 -450
  78. package/md_cg/bench_locomo_zh_public.py +147 -147
  79. package/md_cg/bench_longmem.py +112 -112
  80. package/md_cg/bench_membench.py +632 -632
  81. package/md_cg/bench_p0.py +149 -149
  82. package/md_cg/bench_progressive.py +287 -287
  83. package/md_cg/bench_role_views.py +238 -238
  84. package/md_cg/bench_task_ab.py +243 -243
  85. package/md_cg/bench_task_ab_llm.py +408 -408
  86. package/md_cg/bench_unified_en.py +204 -204
  87. package/md_cg/bench_zh_mad.py +601 -601
  88. package/md_cg/blindspot_tickets.py +123 -123
  89. package/md_cg/branches.py +285 -285
  90. package/md_cg/build_postings.py +73 -73
  91. package/md_cg/ccgc.py +1005 -948
  92. package/md_cg/census.py +132 -132
  93. package/md_cg/chain.py +300 -300
  94. package/md_cg/codeindex.py +531 -531
  95. package/md_cg/coldverify.py +292 -292
  96. package/md_cg/comment_gate.py +337 -337
  97. package/md_cg/cond_compose.py +190 -190
  98. package/md_cg/cond_facts.py +154 -154
  99. package/md_cg/cond_template.json +106 -106
  100. package/md_cg/condition_anchor.py +142 -142
  101. package/md_cg/conformance.py +726 -726
  102. package/md_cg/consistency.py +717 -717
  103. package/md_cg/consolidate.py +1536 -1439
  104. package/md_cg/corpus.py +110 -110
  105. package/md_cg/crosscheck.py +1097 -1097
  106. package/md_cg/crypto.py +437 -437
  107. package/md_cg/d_meta.py +310 -310
  108. package/md_cg/datapath.py +334 -334
  109. package/md_cg/docindex.py +473 -473
  110. package/md_cg/eval_common.py +575 -575
  111. package/md_cg/evidence.py +580 -580
  112. package/md_cg/evolution.py +477 -477
  113. package/md_cg/export.py +220 -220
  114. package/md_cg/forgetting.py +581 -581
  115. package/md_cg/fsutil.py +329 -329
  116. package/md_cg/hotcache.py +238 -214
  117. package/md_cg/hyperedge.py +251 -251
  118. package/md_cg/identity.py +390 -390
  119. package/md_cg/insight.py +500 -500
  120. package/md_cg/interop.py +199 -0
  121. package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
  122. package/md_cg/lexicon/build_standard_en.py +171 -171
  123. package/md_cg/lexicon/expand_en_zh.py +211 -211
  124. package/md_cg/lifecycle.py +272 -272
  125. package/md_cg/linkref.py +280 -280
  126. package/md_cg/links.py +622 -622
  127. package/md_cg/mcp_server.py +129 -32
  128. package/md_cg/md_whitebox.py +345 -345
  129. package/md_cg/mdcg.py +176 -117
  130. package/md_cg/mdcos.py +79 -13
  131. package/md_cg/metacognition.py +591 -591
  132. package/md_cg/migrate.py +119 -119
  133. package/md_cg/migrate_aeis.py +221 -221
  134. package/md_cg/migrate_roleplay.py +293 -293
  135. package/md_cg/migrate_wisdom_graph.py +360 -360
  136. package/md_cg/mreview/__init__.py +25 -25
  137. package/md_cg/mreview/__main__.py +110 -110
  138. package/md_cg/mreview/bundle.py +178 -178
  139. package/md_cg/mreview/candidates.py +262 -262
  140. package/md_cg/mreview/govern.py +693 -693
  141. package/md_cg/mreview/locate.py +939 -939
  142. package/md_cg/mreview/pipeline.py +728 -728
  143. package/md_cg/mreview/rules/duplication.json +21 -21
  144. package/md_cg/mreview/rules/field_coverage.json +54 -54
  145. package/md_cg/mreview/rules/source_license.json +21 -21
  146. package/md_cg/mreview/rules/template_flow.json +21 -21
  147. package/md_cg/mreview/ruleset.py +252 -252
  148. package/md_cg/nodefile.py +575 -575
  149. package/md_cg/pooling.py +484 -472
  150. package/md_cg/postings.py +298 -298
  151. package/md_cg/predict.py +1100 -1100
  152. package/md_cg/progressive.py +123 -123
  153. package/md_cg/protect.py +272 -272
  154. package/md_cg/protocol/md_cg_gate.proto +33 -33
  155. package/md_cg/protocol.py +372 -372
  156. package/md_cg/provenance.py +582 -582
  157. package/md_cg/reach.py +453 -453
  158. package/md_cg/readcache.py +85 -0
  159. package/md_cg/refindex.py +833 -833
  160. package/md_cg/refine.py +604 -604
  161. package/md_cg/roleviews.py +89 -89
  162. package/md_cg/routing.py +365 -365
  163. package/md_cg/scrub.py +852 -852
  164. package/md_cg/security.py +274 -274
  165. package/md_cg/self_state.py +1029 -1029
  166. package/md_cg/selfreport.py +151 -151
  167. package/md_cg/semantic/__init__.py +10 -10
  168. package/md_cg/semantic/canonical.py +122 -122
  169. package/md_cg/semantic/en_normalizer.py +364 -364
  170. package/md_cg/semantic/en_zh_map.json +28694 -0
  171. package/md_cg/semantic/export_en_zh_map.py +64 -0
  172. package/md_cg/semantic/unify.py +45 -0
  173. package/md_cg/semantic/zh_en_atoms.py +139 -139
  174. package/md_cg/signer.py +562 -562
  175. package/md_cg/sources.py +815 -582
  176. package/md_cg/statushdr.py +179 -179
  177. package/md_cg/stg.py +48 -37
  178. package/md_cg/subgraph.py +729 -729
  179. package/md_cg/sustain.py +1138 -1138
  180. package/md_cg/tasks.py +470 -470
  181. package/md_cg/test_action_derive.py +203 -203
  182. package/md_cg/test_audit_rotate.py +270 -270
  183. package/md_cg/test_autonomy.py +143 -143
  184. package/md_cg/test_bench_governance.py +102 -102
  185. package/md_cg/test_blindspot_tickets.py +166 -166
  186. package/md_cg/test_branches.py +249 -249
  187. package/md_cg/test_ccg_perturb.py +184 -184
  188. package/md_cg/test_ccgc.py +433 -433
  189. package/md_cg/test_census_prune.py +81 -81
  190. package/md_cg/test_cond_compose_anchors.py +76 -76
  191. package/md_cg/test_cond_match.py +165 -165
  192. package/md_cg/test_condition_anchor.py +81 -81
  193. package/md_cg/test_d_meta.py +412 -412
  194. package/md_cg/test_datapath_root.py +199 -199
  195. package/md_cg/test_en_pipeline.py +166 -166
  196. package/md_cg/test_gain_gate.py +212 -212
  197. package/md_cg/test_health_scale.py +173 -173
  198. package/md_cg/test_hive_ingest.py +285 -0
  199. package/md_cg/test_hot_cold.py +215 -215
  200. package/md_cg/test_hyperedge.py +245 -245
  201. package/md_cg/test_i26_empty_first_write.py +116 -0
  202. package/md_cg/test_i27_e041_identity.py +128 -0
  203. package/md_cg/test_i28_hotcache_prodpath.py +122 -0
  204. package/md_cg/test_identity_attribution.py +147 -147
  205. package/md_cg/test_index_durability.py +224 -224
  206. package/md_cg/test_interop.py +93 -0
  207. package/md_cg/test_lifecycle.py +309 -309
  208. package/md_cg/test_linkref.py +306 -306
  209. package/md_cg/test_lock.py +43 -43
  210. package/md_cg/test_md_access_parity.py +255 -255
  211. package/md_cg/test_md_writepath.py +345 -345
  212. package/md_cg/test_mdstore_search_parity.py +160 -0
  213. package/md_cg/test_mr_m2.py +587 -587
  214. package/md_cg/test_mr_m3.py +710 -710
  215. package/md_cg/test_mr_m4.py +485 -485
  216. package/md_cg/test_p0.py +250 -250
  217. package/md_cg/test_p1.py +316 -316
  218. package/md_cg/test_p10_identity.py +173 -173
  219. package/md_cg/test_p11_consistency.py +233 -233
  220. package/md_cg/test_p12_metacognition.py +212 -212
  221. package/md_cg/test_p13_encryption.py +241 -241
  222. package/md_cg/test_p14_sustain.py +249 -249
  223. package/md_cg/test_p15_scrub.py +280 -280
  224. package/md_cg/test_p16_self_state.py +301 -301
  225. package/md_cg/test_p17_predict.py +354 -354
  226. package/md_cg/test_p18_whitebox.py +171 -171
  227. package/md_cg/test_p19_migrate_roleplay.py +149 -149
  228. package/md_cg/test_p20_evolution.py +315 -315
  229. package/md_cg/test_p21_tokens.py +293 -270
  230. package/md_cg/test_p22_theory.py +175 -175
  231. package/md_cg/test_p23_links.py +311 -311
  232. package/md_cg/test_p24_evidence.py +227 -227
  233. package/md_cg/test_p25_weights.py +156 -156
  234. package/md_cg/test_p26_refindex.py +416 -416
  235. package/md_cg/test_p27_docindex.py +765 -765
  236. package/md_cg/test_p28_refcheck.py +305 -305
  237. package/md_cg/test_p29_session_ingest_export.py +354 -333
  238. package/md_cg/test_p3.py +11 -2
  239. package/md_cg/test_p30_maintain.py +330 -330
  240. package/md_cg/test_p31_insight.py +534 -534
  241. package/md_cg/test_p32_backfill.py +298 -298
  242. package/md_cg/test_p33_ccg_wiring.py +293 -293
  243. package/md_cg/test_p34_crosscheck.py +331 -331
  244. package/md_cg/test_p35_conditioned_claim.py +252 -252
  245. package/md_cg/test_p36_kp_align.py +230 -230
  246. package/md_cg/test_p37_condition_space.py +248 -248
  247. package/md_cg/test_p38_concurrent_flush.py +102 -0
  248. package/md_cg/test_p38_contextualize.py +273 -273
  249. package/md_cg/test_p39_verify_flow.py +113 -0
  250. package/md_cg/test_p39_vision_evidence.py +369 -369
  251. package/md_cg/test_p40_refine_worklist.py +241 -241
  252. package/md_cg/test_p41_evolve_patrol.py +224 -224
  253. package/md_cg/test_p42_provenance.py +269 -269
  254. package/md_cg/test_p43_pooling.py +412 -398
  255. package/md_cg/test_p44_md_whitebox.py +231 -231
  256. package/md_cg/test_p45_session_identity.py +219 -219
  257. package/md_cg/test_p46_unit_scope.py +272 -272
  258. package/md_cg/test_p47_session_view.py +281 -0
  259. package/md_cg/test_p4_fuzzy.py +223 -223
  260. package/md_cg/test_p5_semantic.py +226 -226
  261. package/md_cg/test_p6_consolidate.py +440 -387
  262. package/md_cg/test_p7_goals_recent.py +202 -202
  263. package/md_cg/test_p8_subgraph_chain.py +200 -200
  264. package/md_cg/test_p9_forget_protect.py +231 -231
  265. package/md_cg/test_predict_beta.py +135 -135
  266. package/md_cg/test_preflight_failclosed.py +100 -100
  267. package/md_cg/test_progressive.py +146 -146
  268. package/md_cg/test_protocol.py +243 -243
  269. package/md_cg/test_reach.py +378 -378
  270. package/md_cg/test_reach_keys.py +201 -201
  271. package/md_cg/test_read_clip.py +141 -141
  272. package/md_cg/test_readcache_prodpath.py +155 -0
  273. package/md_cg/test_retr_gates_prodpath.py +140 -0
  274. package/md_cg/test_retr_s1.py +340 -340
  275. package/md_cg/test_retr_s1b.py +209 -209
  276. package/md_cg/test_retr_s3.py +194 -194
  277. package/md_cg/test_retr_s4.py +163 -163
  278. package/md_cg/test_retr_s5.py +200 -200
  279. package/md_cg/test_retr_s6.py +157 -157
  280. package/md_cg/test_retr_s7.py +384 -384
  281. package/md_cg/test_retr_s8_time.py +369 -316
  282. package/md_cg/test_retr_s9_edges.py +286 -286
  283. package/md_cg/test_retr_s9_entity_ctx.py +175 -175
  284. package/md_cg/test_review_conformance.py +367 -367
  285. package/md_cg/test_role_views.py +354 -354
  286. package/md_cg/test_sem_noise.py +242 -242
  287. package/md_cg/test_semantic_canonical.py +241 -241
  288. package/md_cg/test_subproc_encoding.py +192 -192
  289. package/md_cg/test_sustain_mutual.py +153 -153
  290. package/md_cg/test_tasks.py +409 -409
  291. package/md_cg/test_tool_face.py +189 -189
  292. package/md_cg/test_transfer.py +180 -180
  293. package/md_cg/test_trust.py +361 -361
  294. package/md_cg/test_twophase.py +286 -286
  295. package/md_cg/test_v14_fixes.py +397 -397
  296. package/md_cg/test_validity_filter.py +280 -280
  297. package/md_cg/test_verify_answer.py +138 -138
  298. package/md_cg/test_wisdom_md_store.py +292 -292
  299. package/md_cg/test_writelimit.py +197 -197
  300. package/md_cg/test_writepipe.py +214 -214
  301. package/md_cg/theory.py +273 -273
  302. package/md_cg/tokens.py +677 -663
  303. package/md_cg/tool_face.py +260 -260
  304. package/md_cg/trust.py +986 -950
  305. package/md_cg/twophase.py +231 -231
  306. package/md_cg/units.py +667 -667
  307. package/md_cg/vision_evidence.py +666 -666
  308. package/md_cg/weights.py +624 -624
  309. package/md_cg/whitebox.py +527 -527
  310. package/md_cg/whitebox_kb/__init__.py +37 -37
  311. package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
  312. package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
  313. package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
  314. package/md_cg/whitebox_kb/engine.py +310 -310
  315. package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
  316. package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
  317. package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
  318. package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
  319. package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
  320. package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
  321. package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
  322. package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
  323. package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
  324. package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
  325. package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
  326. package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
  327. package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
  328. package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
  329. package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
  330. package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
  331. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
  332. package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
  333. package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
  334. package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
  335. package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
  336. package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
  337. package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
  338. package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
  339. package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
  340. package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
  341. package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
  342. package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
  343. package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
  344. package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
  345. package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
  346. package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
  347. package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
  348. package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
  349. package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
  350. package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
  351. package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
  352. package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
  353. package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
  354. package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
  355. package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
  356. package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
  357. package/md_cg/writelimit.py +356 -356
  358. package/md_cg/writepipe.py +550 -542
  359. package/package.json +97 -96
  360. package/skills/plugin.json +54 -54
  361. package/skills/skills/designer-perspective/SKILL.md +158 -158
  362. package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
  363. package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
  364. package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
  365. package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
  366. package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
  367. package/skills/skills/designer-perspective/scripts/designer.py +545 -545
  368. package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
  369. package/skills/skills/designer-perspective/tests/selftest.py +61 -61
  370. package/skills/skills/lingshu-browser/SKILL.md +60 -60
  371. package/skills/skills/lingshu-compiler/SKILL.md +56 -56
  372. package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
  373. package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
  374. package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
  375. package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
  376. package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
  377. package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
  378. package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
  379. package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
  380. package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
  381. package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
  382. package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
  383. package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
  384. package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
  385. package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
  386. package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
  387. package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
  388. package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
  389. package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
  390. package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
  391. package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
  392. package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
  393. package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
  394. package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
  395. package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
  396. package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
  397. package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
  398. package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
  399. package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
  400. package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
  401. package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
  402. package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
  403. package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
  404. package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
  405. package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
  406. package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
  407. package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
  408. package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
  409. package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
  410. package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
  411. package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
  412. package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
  413. package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
  414. package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
  415. package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
  416. package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
  417. package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
  418. package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
  419. package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
  420. package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
  421. package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
  422. package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
  423. package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
  424. package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
  425. package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
  426. package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
  427. package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
  428. package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
  429. package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
  430. package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
  431. package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
  432. package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
  433. package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
  434. package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
  435. package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
  436. package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
  437. package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
  438. package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
  439. package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
  440. package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
  441. package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
  442. package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
  443. package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
  444. package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
  445. package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
  446. package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
  447. package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
  448. package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
  449. package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
  450. package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
  451. package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
  452. package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
  453. package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
  454. package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
  455. package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
  456. package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
  457. package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
  458. package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
  459. package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
  460. package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
  461. package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
  462. package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
  463. package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
  464. package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
  465. package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
  466. package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
  467. package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
  468. package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
  469. package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
  470. package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
  471. package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
  472. package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
  473. package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
  474. package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
  475. package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
  476. package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
  477. package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
  478. package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
  479. package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
  480. package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
  481. package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
  482. package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
  483. package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
  484. package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
  485. package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
  486. package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
  487. package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
  488. package/skills/skills/lingshu-graph/SKILL.md +63 -63
  489. package/skills/skills/lingshu-net/SKILL.md +48 -48
  490. package/skills/skills/lingshu-os/SKILL.md +64 -64
  491. package/skills/skills/lingshu-pylang/SKILL.md +71 -71
  492. package/src/bridge.ts +401 -401
  493. package/src/hooks.ts +38 -2
  494. package/src/lib/datapath.ts +326 -326
  495. package/src/lib/mdcg_client.ts +413 -413
  496. package/src/lib/mutual.ts +428 -428
  497. package/src/lib/prompt_safety.ts +62 -62
  498. package/src/lib/python_path.ts +71 -71
  499. package/src/lib/roleplay_web.ts +932 -932
  500. package/src/lib/token_store.ts +192 -192
  501. package/src/tools.ts +212 -212
  502. package/zcode/AGENTS.md +11 -3
  503. package/zcode/README.md +41 -41
  504. /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
@@ -1,409 +1,409 @@
1
- # -*- coding: utf-8 -*-
2
- """任务级 A/B 真实端到端评测(先验陷阱版):无记忆 vs 有记忆
3
-
4
- 为什么要重做成「先验陷阱」:
5
- 上一版用 12 个教科书级报错(缺依赖、权限、超时…)。实测发现:这类题的正确答案
6
- 就在模型先验里——9B 小模型无记忆也有 94.4%,强模型 98.6%。也就是说**题目本身
7
- 低于模型能力下限,天花板效应把记忆的价值全遮住了**。用户判断成立。
8
-
9
- 本版的设计原则(用户提出):
10
- 「所有的知识都有隐形的条件。只有具有完整的记忆,才能发现哪些是当前不适用的知识。」
11
- 因此每个用例都是一个**先验陷阱**:
12
- · trap = 通用最佳实践 / 教科书答案(模型会自信地选它)
13
- · fix = 本项目规范下的正确做法(与最佳实践相反,只有记忆里的隐性条件能推出)
14
- · decoy = 明显错误的做法
15
- 无记忆臂只能靠先验,会掉进 trap;有记忆臂能读到「项目规范 + 历史事故」这条隐性
16
- 条件,从而发现最佳实践在当前项目里**不适用**,改选 fix。
17
-
18
- 唯一变量 = 记忆库;其余全同(同一模型、temperature=0、同一系统提示)。
19
-
20
- 防作弊设计:
21
- 1) 措辞不同:记忆写「项目规范/原因」,候选写「具体动作」,模型须自行映射,不能照抄。
22
- 2) 多排列去位置偏差:每例跑 3 种**循环排列**(rotations),使每个候选在 3 个位置上
23
- 各出现一次——只有「按内容选」才能 3/3 全中;纯位置偏差只能 1/3。位置分布单列。
24
-
25
- 指标(以 (用例 × 排列) 为单位):
26
- content_acc 按内容选对的比例(主指标)
27
- trap_rate 选中「最佳实践陷阱」的比例(无记忆臂应显著 >0)
28
- invalid_rate 输出无法解析为候选编号的比例(不剔除,如实计入)
29
- pos_dist 选中位置分布(1/2/3),偏离均匀 = 位置偏差
30
- avg_tokens 每次调用总 token(记忆的代价)
31
- recall_hit_rate 记忆臂中 recall 确实召回该项目规范的比例
32
-
33
- 诚实边界:
34
- · 5 用例 × 3 排列 = 15 次/臂,样本很小,只作方向性验证。
35
- · 陷阱是否真的骗到模型,由 trap_rate 实测决定;骗不到就如实报告「该题不成立」。
36
- · 记忆能召回 ≠ 模型会用;hit 与 content_acc 分开统计正是为暴露这个差异。
37
-
38
- 跑法:
39
- python -m md_cg.bench_task_ab_llm # deepseek-v4-flash(默认)
40
- python -m md_cg.bench_task_ab_llm --local # 追加本地小模型
41
- python -m md_cg.bench_task_ab_llm --perms 6 # 全 6 种排列
42
- """
43
- from __future__ import annotations
44
-
45
- import argparse
46
- import itertools
47
- import json
48
- import os
49
- import random
50
- import re
51
- import shutil
52
- import sys
53
- import tempfile
54
- import time
55
- import urllib.error
56
- import urllib.request
57
- from concurrent.futures import ThreadPoolExecutor
58
-
59
- from md_cg.mdcos import MdCGOS
60
-
61
- # ---- 系统提示(两臂完全相同)---------------------------------------------
62
- SYS_PROMPT = (
63
- "你是本项目的资深工程师。用户给出一个工程场景和若干候选做法,"
64
- "你选一个最符合本项目既定规范的做法。\n"
65
- "只输出一行 JSON,不要解释、不要代码围栏,格式:\n"
66
- '{"pick": 2, "reason": "本项目规范要求..."}'
67
- )
68
-
69
- # ---- 5 个先验陷阱用例 ------------------------------------------------------
70
- # task :场景(模型据此决策)
71
- # lesson :写入记忆的「项目规范 + 历史事故」(隐性条件,候选里不出现)
72
- # query :召回用检索词
73
- # fix/trap/decoy:候选做法;trap 是通用最佳实践(模型先验),fix 才是本项目正解
74
- CASES = [
75
- {
76
- "id": "p01",
77
- "task": "API 的参数校验失败了,应该怎么给客户端返回?",
78
- "query": "API 参数校验失败 业务错误 返回 HTTP 状态码",
79
- "lesson": "本项目 API 契约:所有业务错误(含参数校验失败)必须返回 HTTP 200,"
80
- "在 body 的 code 字段带业务错误码,禁止用 HTTP 4xx/5xx。"
81
- "原因:老旧移动端 SDK 把任何非 200 都当成网络故障无限重试。",
82
- "trap": "返回 HTTP 400,body 里写清楚校验失败详情",
83
- "fix": "返回 HTTP 200,body 的 code 字段带业务错误码",
84
- "decoy": "返回 HTTP 500,让客户端稍后重试",
85
- },
86
- {
87
- "id": "p02",
88
- "task": "API 响应里要返回一个时间戳,用什么格式?",
89
- "query": "API 时间戳 格式 ISO 毫秒 时区",
90
- "lesson": "本项目 API 契约:所有时间戳统一返回 Unix 毫秒整数(int64),"
91
- "禁止 ISO 8601 字符串和本地时间字符串。"
92
- "原因:跨语言客户端解析不一致,历史事故是时区偏移导致订单时间错乱 8 小时。",
93
- "trap": "返回 ISO 8601 字符串,如 2026-09-09T12:00:00Z",
94
- "fix": "返回 Unix 毫秒整数(int64)",
95
- "decoy": "返回本地时间字符串,如 '2026/9/9 12:00'",
96
- },
97
- {
98
- "id": "p03",
99
- "task": "校验请求参数 x 必须是整数,怎么写这个判断?",
100
- "query": "校验 参数 整数 布尔 bool isinstance 精确类型",
101
- "lesson": "本项目校验规范:需要精确类型时必须用 type(x) is int,禁止 isinstance。"
102
- "原因:bool 是 int 的子类,isinstance(True, int) 为真,"
103
- "历史事故是布尔值被当成整数接受,写进了金额字段。",
104
- "trap": "用 isinstance(x, int) 判断",
105
- "fix": "用 type(x) is int 判断",
106
- "decoy": "用 x == int 判断",
107
- },
108
- {
109
- "id": "p04",
110
- "task": "从配置字典 cfg 里读一个可选字段的值,怎么写?",
111
- "query": "配置 字典 可选字段 缺失 None get 默认值",
112
- "lesson": "本项目配置规范:可选字段必须用 'key' in cfg 判断后再取值,禁止 .get()。"
113
- "原因:.get() 把「显式 None」和「未配置」混为一谈,"
114
- "历史事故是默认值覆盖了用户显式设置的空值。",
115
- "trap": "用 cfg.get('key', default) 取值",
116
- "fix": "用 'key' in cfg 判断后再取值",
117
- "decoy": "用 cfg['key'] 直接索引",
118
- },
119
- {
120
- "id": "p05",
121
- "task": "线上要记录一条警告信息(比如缓存 miss),用哪个日志级别?",
122
- "query": "日志 级别 警告 缓存 miss warning info 告警",
123
- "lesson": "本项目日志规范:warning 及以上级别会触发 PagerDuty 告警;"
124
- "非故障类的警告必须用 info 级别并在消息里加 [WARN] 前缀,禁止直接用 warning。"
125
- "原因:历史事故是缓存 miss 的 warning 半夜触发全组告警。",
126
- "trap": "用 logger.warning(...) 记录",
127
- "fix": "用 logger.info(...) 并在消息里加 [WARN] 前缀",
128
- "decoy": "用 logger.error(...) 记录",
129
- },
130
- ]
131
-
132
- for _c in CASES:
133
- _c["marker"] = _c["lesson"][:12]
134
-
135
-
136
- # ---- LLM 客户端(OpenAI 兼容,零第三方依赖)-------------------------------
137
- # 生效条件:给定 model/base/key/messages 即构造 JSON POST 到 base.rstrip("/")+"/chat/completions",返回 (choices[0].message 的 content 或缺失/假值回落 ""、data.get("usage") 或缺失/假值回落 {}、耗时 dt),timeout/max_tokens/temperature 仅作为请求参数传入。
138
- def llm_chat(model, base, key, messages, timeout=180, max_tokens=200,
139
- temperature=0.0):
140
- payload = json.dumps({
141
- "model": model,
142
- "messages": messages,
143
- "temperature": temperature,
144
- "max_tokens": max_tokens,
145
- }).encode("utf-8")
146
- req = urllib.request.Request(
147
- base.rstrip("/") + "/chat/completions", data=payload,
148
- headers={"Content-Type": "application/json",
149
- "Authorization": f"Bearer {key}"})
150
- t0 = time.time()
151
- with urllib.request.urlopen(req, timeout=timeout) as resp:
152
- data = json.loads(resp.read().decode("utf-8"))
153
- dt = time.time() - t0
154
- content = data["choices"][0]["message"].get("content") or ""
155
- return content, (data.get("usage") or {}), dt
156
-
157
-
158
- # 生效条件:raw(None/空串视为 "")中首个 "{" 至末个 "}" 的子串能解析出 pick,或全文匹配到单个数字 1-9,且该编号落在 1..n 内时返回该编号(pick 为数字字符串先转 int),否则返回 None。
159
- def _parse_pick(raw, n):
160
- """从模型输出里抠出候选编号(1..n);解析失败返回 None。"""
161
- s = raw or ""
162
- i, j = s.find("{"), s.rfind("}")
163
- if i >= 0 and j > i:
164
- try:
165
- obj = json.loads(s[i:j + 1])
166
- p = obj.get("pick")
167
- if isinstance(p, str) and p.strip().isdigit():
168
- p = int(p.strip())
169
- if isinstance(p, (int, float)) and 1 <= int(p) <= n:
170
- return int(p)
171
- except ValueError:
172
- pass
173
- m = re.search(r"\b([1-9])\b", s)
174
- if m and 1 <= int(m.group(1)) <= n:
175
- return int(m.group(1))
176
- return None
177
-
178
-
179
- # 生效条件:以 case["id"] 播种打乱 [(fix,case["fix"]),(trap,case["trap"]),(decoy,case["decoy"])] 后取三种循环排列,n<=3 返回其前 n 个,n>3 返回 6 种全排列的前 n 个。
180
- def _orders(case, n):
181
- """候选顺序。默认取 3 种循环排列(每个候选在 3 个位置各出现一次,天然去位置偏差);
182
- n>3 时退回全 6 种排列。基准顺序由 case id 播种打乱,避免跨用例雷同。"""
183
- base = [("fix", case["fix"]), ("trap", case["trap"]),
184
- ("decoy", case["decoy"])]
185
- random.Random(case["id"]).shuffle(base)
186
- rots = [tuple(base[i:] + base[:i]) for i in range(3)]
187
- if n <= 3:
188
- return rots[:n]
189
- return list(itertools.permutations(base))[:n]
190
-
191
-
192
- # 生效条件:case 含 task 且 opts 为 (kind, txt) 序列时,返回含场景、编号候选与选择指令的提示文本,memory 非空时在开头插入记忆段。
193
- def _user_prompt(case, opts, memory=None):
194
- lines = []
195
- if memory:
196
- lines.append("【本项目规范 / 历史经验(来自记忆库,请自行判断相关性)】")
197
- lines.append(memory.strip())
198
- lines.append("")
199
- lines.append("场景:")
200
- lines.append(case["task"])
201
- lines.append("")
202
- lines.append("候选做法:")
203
- for idx, (_kind, txt) in enumerate(opts, 1):
204
- lines.append(f"{idx}. {txt}")
205
- lines.append("")
206
- lines.append("请选择最符合本项目规范的做法。")
207
- return "\n".join(lines)
208
-
209
-
210
- # 生效条件:cg 存在时把模块常量 CASES 逐条以 c["task"] 为 error、c["lesson"] 为 fix 组成列表传给 cg.mine_fix_pairs 并原样返回其结果。
211
- def _build_memory(cg):
212
- """Phase A:把 5 条「项目规范」写进记忆(真实 mine_fix_pairs)。"""
213
- return cg.mine_fix_pairs(
214
- [{"error": c["task"], "fix": c["lesson"]} for c in CASES])
215
-
216
-
217
- # 生效条件:以 "本项目规范:"+case["query"] 且 budget_tokens=budget、k=10 调用 cg.recall,取 pack 各项 content(缺失/假值为 "")拼接,返回其前 1500 字符与 case["marker"] 是否出现在未截断拼接文本中。
218
- def _recall_memory(cg, case, budget):
219
- res = cg.recall("本项目规范:" + case["query"],
220
- budget_tokens=budget, k=10)
221
- text = " ".join((p.get("content") or "") for p in res.get("pack") or [])
222
- return text[:1500], (case["marker"] in text)
223
-
224
-
225
- # 生效条件:在 cfg["max_turns"] 轮内以 cfg["model"]/cfg["base"]/cfg["key"] 调 llm_chat(timeout=cfg["timeout"]、max_tokens=cfg["max_tokens"])并累加 usage["total_tokens"],pick 等于 correct_pos 或为 None 时中止(否则在仍有剩余轮次时追加 assistant/user 消息重选),最终 pick 为 None 则 kind="invalid"、否则 kind=opts[pick-1][0]。
226
- def _run_one(cfg, case, opts, correct_pos, memory):
227
- messages = [{"role": "system", "content": SYS_PROMPT},
228
- {"role": "user", "content": _user_prompt(case, opts, memory)}]
229
- tokens, turns, pick, raw, dt = 0, 0, None, "", 0.0
230
- while turns < cfg["max_turns"]:
231
- turns += 1
232
- raw, usage, dt = llm_chat(cfg["model"], cfg["base"], cfg["key"],
233
- messages, timeout=cfg["timeout"],
234
- max_tokens=cfg["max_tokens"])
235
- tokens += int(usage.get("total_tokens") or 0)
236
- pick = _parse_pick(raw, len(opts))
237
- if pick == correct_pos or pick is None:
238
- break
239
- if turns < cfg["max_turns"]:
240
- messages.append({"role": "assistant", "content": raw})
241
- messages.append({"role": "user",
242
- "content": f"第 {pick} 个做法执行后问题依旧。"
243
- f"请在剩余候选中重新选择,仍然只输出 JSON。"})
244
- kind = "invalid" if pick is None else opts[pick - 1][0]
245
- return {"pick": pick, "kind": kind, "correct": pick == correct_pos,
246
- "turns": turns, "tokens": tokens, "latency": dt}
247
-
248
-
249
- # 生效条件:cfg["use_mem"] 为真时对每个 case 以 cfg["budget"] 先串行召回写入 mems、否则存 (None, False),再对每 case × _orders(case, n_perms) 的排列 × ("none","mem") 两臂建任务交 ThreadPoolExecutor(max_workers=workers) 执行,返回按 arm 分组的 {"none": [...], "mem": [...]} 行表。
250
- def _run_model(label, cfg, cases, cg, n_perms, workers):
251
- # 记忆召回先串行算好(MdCGOS 非线程安全),LLM 调用再并发。
252
- mems = {}
253
- for case in cases:
254
- mems[case["id"]] = (_recall_memory(cg, case, cfg["budget"])
255
- if cfg["use_mem"] else (None, False))
256
-
257
- tasks = []
258
- for case in cases:
259
- for oi, opts in enumerate(_orders(case, n_perms)):
260
- correct_pos = next(i for i, (k, _t) in enumerate(opts, 1)
261
- if k == "fix")
262
- for arm in ("none", "mem"):
263
- tasks.append((case, oi, opts, correct_pos, arm))
264
-
265
- # 生效条件:t 解包为 (case, oi, opts, correct_pos, arm),arm=="mem" 时 memory/hit 取闭包 mems[case["id"]]、否则为 (None, False),_run_one 抛异常时以 kind="error"、pick=None 的占位行替代,随后补上 case/arm/hit 并打印该行后返回 r。
266
- def work(t):
267
- case, oi, opts, correct_pos, arm = t
268
- memory, hit = mems[case["id"]] if arm == "mem" else (None, False)
269
- try:
270
- r = _run_one(cfg, case, opts, correct_pos, memory)
271
- except Exception as exc: # noqa: BLE001
272
- r = {"pick": None, "kind": "error", "correct": False,
273
- "turns": 0, "tokens": 0, "latency": 0.0,
274
- "raw": f"{type(exc).__name__}: {exc}"}
275
- r.update({"case": case["id"], "arm": arm, "hit": hit})
276
- print(f" [{label}] {case['id']} {arm:<4} rot={oi} "
277
- f"-> pick={r['pick']} {r['kind']:<6} tok={r['tokens']:>4} "
278
- f"({r['latency']:.1f}s)", flush=True)
279
- return r
280
-
281
- rows = {"none": [], "mem": []}
282
- with ThreadPoolExecutor(max_workers=workers) as ex:
283
- for r in ex.map(work, tasks):
284
- rows[r["arm"]].append(r)
285
- return rows
286
-
287
-
288
- # 生效条件:n 为真值时返回 f"{100.0*x/n:.1f}%",n 为假值(0)时返回 "-"。
289
- def _pct(x, n):
290
- return f"{100.0 * x / n:.1f}%" if n else "-"
291
-
292
-
293
- # 生效条件:以 rows["none"] 的行数 n 汇总 none/mem 两臂的 correct、kind=="trap"、kind∈("invalid","error")、tokens、latency、hit 与 pick∈(1,2,3) 的分布并打印(n_perms 仅用于表头文字),返回该 agg 字典。
294
- def _report(label, rows, n_perms):
295
- n = len(rows["none"])
296
- print(f"\n{'-' * 78}")
297
- print(f"模型:{label} 样本 {n} 次(用例 × 排列)· 每例 {n_perms} 种排列")
298
- print(f"{'指标':<18}{'arm_none':>14}{'arm_mem':>14}{'差值':>16}")
299
- print("-" * 78)
300
- agg = {}
301
- for arm in ("none", "mem"):
302
- rs = rows[arm]
303
- agg[arm] = {
304
- "acc": sum(1 for r in rs if r["correct"]),
305
- "trap": sum(1 for r in rs if r["kind"] == "trap"),
306
- "invalid": sum(1 for r in rs if r["kind"] in ("invalid", "error")),
307
- "tokens": sum(r["tokens"] for r in rs),
308
- "latency": sum(r["latency"] for r in rs),
309
- "hit": sum(1 for r in rs if r["hit"]),
310
- "pos": [sum(1 for r in rs if r["pick"] == p) for p in (1, 2, 3)],
311
- }
312
- a, b = agg["none"], agg["mem"]
313
- d_acc = f"+{100.0 * (b['acc'] - a['acc']) / n:.1f}pp"
314
- d_trap = f"{100.0 * (b['trap'] - a['trap']) / n:.1f}pp"
315
- d_inv = f"{100.0 * (b['invalid'] - a['invalid']) / n:.1f}pp"
316
- d_tok = f"{(b['tokens'] - a['tokens']) / n:+.1f}"
317
- d_lat = f"{(b['latency'] - a['latency']) / n:+.2f}"
318
- print(f"{'content_acc':<18}{_pct(a['acc'], n):>14}{_pct(b['acc'], n):>14}"
319
- f"{d_acc:>16}")
320
- print(f"{'trap_rate':<18}{_pct(a['trap'], n):>14}{_pct(b['trap'], n):>14}"
321
- f"{d_trap:>16}")
322
- print(f"{'invalid_rate':<18}{_pct(a['invalid'], n):>14}"
323
- f"{_pct(b['invalid'], n):>14}{d_inv:>16}")
324
- print(f"{'avg_tokens':<18}{a['tokens'] / n:>14.1f}{b['tokens'] / n:>14.1f}"
325
- f"{d_tok:>16}")
326
- print(f"{'avg_latency_s':<18}{a['latency'] / n:>14.2f}"
327
- f"{b['latency'] / n:>14.2f}{d_lat:>16}")
328
- print(f"{'recall_hit_rate':<18}{'-':>14}{_pct(b['hit'], n):>14}{'-':>16}")
329
- print(f"{'pos_dist(1/2/3)':<18}"
330
- f"{'/'.join(str(x) for x in a['pos']):>14}"
331
- f"{'/'.join(str(x) for x in b['pos']):>14}{'—':>16}")
332
-
333
- print("\n逐用例 content_acc(按内容选对次数 / 该例排列数):")
334
- for case in CASES:
335
- if not any(r["case"] == case["id"] for r in rows["none"]):
336
- continue
337
- an = [r for r in rows["none"] if r["case"] == case["id"]]
338
- am = [r for r in rows["mem"] if r["case"] == case["id"]]
339
- na = sum(1 for r in an if r["correct"])
340
- ma = sum(1 for r in am if r["correct"])
341
- hits = sum(1 for r in am if r["hit"])
342
- print(f" {case['id']} none {na}/{len(an)} mem {ma}/{len(am)} "
343
- f"hit {hits}/{len(am)} {case['task'][:38]}")
344
- return agg
345
-
346
-
347
- # 生效条件:argv 为 None 时改读 sys.argv;若 --key 与 DEEPSEEK_API_KEY 均为空则返回 2,否则按 --only/--cases 从 CASES 取用例跑完双臂后返回 0。
348
- def main(argv=None):
349
- ap = argparse.ArgumentParser(
350
- description="先验陷阱 A/B:无记忆 vs 有记忆(真实 LLM)")
351
- ap.add_argument("--model", default="deepseek-v4.1-flash-expires-on-0910")
352
- ap.add_argument("--base", default="https://api.deepseek.com/v1")
353
- ap.add_argument("--key", default="")
354
- ap.add_argument("--local", action="store_true",
355
- help="追加本地小模型做对照")
356
- ap.add_argument("--local-model", default="smegmma-deluxe-9b-v1")
357
- ap.add_argument("--local-base", default="http://localhost:1234/v1")
358
- ap.add_argument("--cases", type=int, default=0, help="只用前 N 个用例")
359
- ap.add_argument("--only", default="", help="只用指定 id,逗号分隔,如 p01,p02")
360
- ap.add_argument("--perms", type=int, default=3, help="每例候选排列数(默认 3)")
361
- ap.add_argument("--budget", type=int, default=3000, help="召回 token 预算")
362
- ap.add_argument("--max-tokens", type=int, default=2000)
363
- ap.add_argument("--timeout", type=int, default=180)
364
- ap.add_argument("--workers", type=int, default=4, help="并发调用数")
365
- ap.add_argument("--max-turns", type=int, default=1,
366
- help="最大轮数(1=单发;>1 则在选错后反馈重选)")
367
- a = ap.parse_args(argv)
368
-
369
- cases = CASES
370
- if a.only:
371
- want = {x.strip() for x in a.only.split(",") if x.strip()}
372
- cases = [c for c in cases if c["id"] in want]
373
- if a.cases:
374
- cases = cases[:a.cases]
375
- n_perms = max(1, min(6, a.perms))
376
-
377
- key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
378
- if not key:
379
- print("[error] 需要 DEEPSEEK_API_KEY 或 --key")
380
- return 2
381
-
382
- root = tempfile.mkdtemp(prefix="mdcg_ab_prior_")
383
- try:
384
- cg = MdCGOS(os.path.join(root, "mem"))
385
- mined = _build_memory(cg)
386
- print(f"Phase A 写入记忆:{len(mined['pairs'])} 组"
387
- f"(知识 {len(mined['knowledge_ids'])} + 负记忆 {len(mined['rejected_ids'])})")
388
- print(f"用例 {len(cases)} × 排列 {n_perms} × 2 臂 = "
389
- f"{len(cases) * n_perms * 2} 次调用")
390
-
391
- targets = [(a.model, a.model, a.base, key)]
392
- if a.local:
393
- targets.append(("local:" + a.local_model, a.local_model,
394
- a.local_base, "lm-studio"))
395
-
396
- for label, model, base, mkey in targets:
397
- cfg = {"model": model, "base": base, "key": mkey,
398
- "budget": a.budget, "timeout": a.timeout,
399
- "max_tokens": a.max_tokens, "max_turns": a.max_turns,
400
- "use_mem": True}
401
- rows = _run_model(label, cfg, cases, cg, n_perms, a.workers)
402
- _report(label, rows, n_perms)
403
- finally:
404
- shutil.rmtree(root, ignore_errors=True)
405
- return 0
406
-
407
-
408
- if __name__ == "__main__":
1
+ # -*- coding: utf-8 -*-
2
+ """任务级 A/B 真实端到端评测(先验陷阱版):无记忆 vs 有记忆
3
+
4
+ 为什么要重做成「先验陷阱」:
5
+ 上一版用 12 个教科书级报错(缺依赖、权限、超时…)。实测发现:这类题的正确答案
6
+ 就在模型先验里——9B 小模型无记忆也有 94.4%,强模型 98.6%。也就是说**题目本身
7
+ 低于模型能力下限,天花板效应把记忆的价值全遮住了**。用户判断成立。
8
+
9
+ 本版的设计原则(用户提出):
10
+ 「所有的知识都有隐形的条件。只有具有完整的记忆,才能发现哪些是当前不适用的知识。」
11
+ 因此每个用例都是一个**先验陷阱**:
12
+ · trap = 通用最佳实践 / 教科书答案(模型会自信地选它)
13
+ · fix = 本项目规范下的正确做法(与最佳实践相反,只有记忆里的隐性条件能推出)
14
+ · decoy = 明显错误的做法
15
+ 无记忆臂只能靠先验,会掉进 trap;有记忆臂能读到「项目规范 + 历史事故」这条隐性
16
+ 条件,从而发现最佳实践在当前项目里**不适用**,改选 fix。
17
+
18
+ 唯一变量 = 记忆库;其余全同(同一模型、temperature=0、同一系统提示)。
19
+
20
+ 防作弊设计:
21
+ 1) 措辞不同:记忆写「项目规范/原因」,候选写「具体动作」,模型须自行映射,不能照抄。
22
+ 2) 多排列去位置偏差:每例跑 3 种**循环排列**(rotations),使每个候选在 3 个位置上
23
+ 各出现一次——只有「按内容选」才能 3/3 全中;纯位置偏差只能 1/3。位置分布单列。
24
+
25
+ 指标(以 (用例 × 排列) 为单位):
26
+ content_acc 按内容选对的比例(主指标)
27
+ trap_rate 选中「最佳实践陷阱」的比例(无记忆臂应显著 >0)
28
+ invalid_rate 输出无法解析为候选编号的比例(不剔除,如实计入)
29
+ pos_dist 选中位置分布(1/2/3),偏离均匀 = 位置偏差
30
+ avg_tokens 每次调用总 token(记忆的代价)
31
+ recall_hit_rate 记忆臂中 recall 确实召回该项目规范的比例
32
+
33
+ 诚实边界:
34
+ · 5 用例 × 3 排列 = 15 次/臂,样本很小,只作方向性验证。
35
+ · 陷阱是否真的骗到模型,由 trap_rate 实测决定;骗不到就如实报告「该题不成立」。
36
+ · 记忆能召回 ≠ 模型会用;hit 与 content_acc 分开统计正是为暴露这个差异。
37
+
38
+ 跑法:
39
+ python -m md_cg.bench_task_ab_llm # deepseek-v4-flash(默认)
40
+ python -m md_cg.bench_task_ab_llm --local # 追加本地小模型
41
+ python -m md_cg.bench_task_ab_llm --perms 6 # 全 6 种排列
42
+ """
43
+ from __future__ import annotations
44
+
45
+ import argparse
46
+ import itertools
47
+ import json
48
+ import os
49
+ import random
50
+ import re
51
+ import shutil
52
+ import sys
53
+ import tempfile
54
+ import time
55
+ import urllib.error
56
+ import urllib.request
57
+ from concurrent.futures import ThreadPoolExecutor
58
+
59
+ from md_cg.mdcos import MdCGOS
60
+
61
+ # ---- 系统提示(两臂完全相同)---------------------------------------------
62
+ SYS_PROMPT = (
63
+ "你是本项目的资深工程师。用户给出一个工程场景和若干候选做法,"
64
+ "你选一个最符合本项目既定规范的做法。\n"
65
+ "只输出一行 JSON,不要解释、不要代码围栏,格式:\n"
66
+ '{"pick": 2, "reason": "本项目规范要求..."}'
67
+ )
68
+
69
+ # ---- 5 个先验陷阱用例 ------------------------------------------------------
70
+ # task :场景(模型据此决策)
71
+ # lesson :写入记忆的「项目规范 + 历史事故」(隐性条件,候选里不出现)
72
+ # query :召回用检索词
73
+ # fix/trap/decoy:候选做法;trap 是通用最佳实践(模型先验),fix 才是本项目正解
74
+ CASES = [
75
+ {
76
+ "id": "p01",
77
+ "task": "API 的参数校验失败了,应该怎么给客户端返回?",
78
+ "query": "API 参数校验失败 业务错误 返回 HTTP 状态码",
79
+ "lesson": "本项目 API 契约:所有业务错误(含参数校验失败)必须返回 HTTP 200,"
80
+ "在 body 的 code 字段带业务错误码,禁止用 HTTP 4xx/5xx。"
81
+ "原因:老旧移动端 SDK 把任何非 200 都当成网络故障无限重试。",
82
+ "trap": "返回 HTTP 400,body 里写清楚校验失败详情",
83
+ "fix": "返回 HTTP 200,body 的 code 字段带业务错误码",
84
+ "decoy": "返回 HTTP 500,让客户端稍后重试",
85
+ },
86
+ {
87
+ "id": "p02",
88
+ "task": "API 响应里要返回一个时间戳,用什么格式?",
89
+ "query": "API 时间戳 格式 ISO 毫秒 时区",
90
+ "lesson": "本项目 API 契约:所有时间戳统一返回 Unix 毫秒整数(int64),"
91
+ "禁止 ISO 8601 字符串和本地时间字符串。"
92
+ "原因:跨语言客户端解析不一致,历史事故是时区偏移导致订单时间错乱 8 小时。",
93
+ "trap": "返回 ISO 8601 字符串,如 2026-09-09T12:00:00Z",
94
+ "fix": "返回 Unix 毫秒整数(int64)",
95
+ "decoy": "返回本地时间字符串,如 '2026/9/9 12:00'",
96
+ },
97
+ {
98
+ "id": "p03",
99
+ "task": "校验请求参数 x 必须是整数,怎么写这个判断?",
100
+ "query": "校验 参数 整数 布尔 bool isinstance 精确类型",
101
+ "lesson": "本项目校验规范:需要精确类型时必须用 type(x) is int,禁止 isinstance。"
102
+ "原因:bool 是 int 的子类,isinstance(True, int) 为真,"
103
+ "历史事故是布尔值被当成整数接受,写进了金额字段。",
104
+ "trap": "用 isinstance(x, int) 判断",
105
+ "fix": "用 type(x) is int 判断",
106
+ "decoy": "用 x == int 判断",
107
+ },
108
+ {
109
+ "id": "p04",
110
+ "task": "从配置字典 cfg 里读一个可选字段的值,怎么写?",
111
+ "query": "配置 字典 可选字段 缺失 None get 默认值",
112
+ "lesson": "本项目配置规范:可选字段必须用 'key' in cfg 判断后再取值,禁止 .get()。"
113
+ "原因:.get() 把「显式 None」和「未配置」混为一谈,"
114
+ "历史事故是默认值覆盖了用户显式设置的空值。",
115
+ "trap": "用 cfg.get('key', default) 取值",
116
+ "fix": "用 'key' in cfg 判断后再取值",
117
+ "decoy": "用 cfg['key'] 直接索引",
118
+ },
119
+ {
120
+ "id": "p05",
121
+ "task": "线上要记录一条警告信息(比如缓存 miss),用哪个日志级别?",
122
+ "query": "日志 级别 警告 缓存 miss warning info 告警",
123
+ "lesson": "本项目日志规范:warning 及以上级别会触发 PagerDuty 告警;"
124
+ "非故障类的警告必须用 info 级别并在消息里加 [WARN] 前缀,禁止直接用 warning。"
125
+ "原因:历史事故是缓存 miss 的 warning 半夜触发全组告警。",
126
+ "trap": "用 logger.warning(...) 记录",
127
+ "fix": "用 logger.info(...) 并在消息里加 [WARN] 前缀",
128
+ "decoy": "用 logger.error(...) 记录",
129
+ },
130
+ ]
131
+
132
+ for _c in CASES:
133
+ _c["marker"] = _c["lesson"][:12]
134
+
135
+
136
+ # ---- LLM 客户端(OpenAI 兼容,零第三方依赖)-------------------------------
137
+ # 生效条件:给定 model/base/key/messages 即构造 JSON POST 到 base.rstrip("/")+"/chat/completions",返回 (choices[0].message 的 content 或缺失/假值回落 ""、data.get("usage") 或缺失/假值回落 {}、耗时 dt),timeout/max_tokens/temperature 仅作为请求参数传入。
138
+ def llm_chat(model, base, key, messages, timeout=180, max_tokens=200,
139
+ temperature=0.0):
140
+ payload = json.dumps({
141
+ "model": model,
142
+ "messages": messages,
143
+ "temperature": temperature,
144
+ "max_tokens": max_tokens,
145
+ }).encode("utf-8")
146
+ req = urllib.request.Request(
147
+ base.rstrip("/") + "/chat/completions", data=payload,
148
+ headers={"Content-Type": "application/json",
149
+ "Authorization": f"Bearer {key}"})
150
+ t0 = time.time()
151
+ with urllib.request.urlopen(req, timeout=timeout) as resp:
152
+ data = json.loads(resp.read().decode("utf-8"))
153
+ dt = time.time() - t0
154
+ content = data["choices"][0]["message"].get("content") or ""
155
+ return content, (data.get("usage") or {}), dt
156
+
157
+
158
+ # 生效条件:raw(None/空串视为 "")中首个 "{" 至末个 "}" 的子串能解析出 pick,或全文匹配到单个数字 1-9,且该编号落在 1..n 内时返回该编号(pick 为数字字符串先转 int),否则返回 None。
159
+ def _parse_pick(raw, n):
160
+ """从模型输出里抠出候选编号(1..n);解析失败返回 None。"""
161
+ s = raw or ""
162
+ i, j = s.find("{"), s.rfind("}")
163
+ if i >= 0 and j > i:
164
+ try:
165
+ obj = json.loads(s[i:j + 1])
166
+ p = obj.get("pick")
167
+ if isinstance(p, str) and p.strip().isdigit():
168
+ p = int(p.strip())
169
+ if isinstance(p, (int, float)) and 1 <= int(p) <= n:
170
+ return int(p)
171
+ except ValueError:
172
+ pass
173
+ m = re.search(r"\b([1-9])\b", s)
174
+ if m and 1 <= int(m.group(1)) <= n:
175
+ return int(m.group(1))
176
+ return None
177
+
178
+
179
+ # 生效条件:以 case["id"] 播种打乱 [(fix,case["fix"]),(trap,case["trap"]),(decoy,case["decoy"])] 后取三种循环排列,n<=3 返回其前 n 个,n>3 返回 6 种全排列的前 n 个。
180
+ def _orders(case, n):
181
+ """候选顺序。默认取 3 种循环排列(每个候选在 3 个位置各出现一次,天然去位置偏差);
182
+ n>3 时退回全 6 种排列。基准顺序由 case id 播种打乱,避免跨用例雷同。"""
183
+ base = [("fix", case["fix"]), ("trap", case["trap"]),
184
+ ("decoy", case["decoy"])]
185
+ random.Random(case["id"]).shuffle(base)
186
+ rots = [tuple(base[i:] + base[:i]) for i in range(3)]
187
+ if n <= 3:
188
+ return rots[:n]
189
+ return list(itertools.permutations(base))[:n]
190
+
191
+
192
+ # 生效条件:case 含 task 且 opts 为 (kind, txt) 序列时,返回含场景、编号候选与选择指令的提示文本,memory 非空时在开头插入记忆段。
193
+ def _user_prompt(case, opts, memory=None):
194
+ lines = []
195
+ if memory:
196
+ lines.append("【本项目规范 / 历史经验(来自记忆库,请自行判断相关性)】")
197
+ lines.append(memory.strip())
198
+ lines.append("")
199
+ lines.append("场景:")
200
+ lines.append(case["task"])
201
+ lines.append("")
202
+ lines.append("候选做法:")
203
+ for idx, (_kind, txt) in enumerate(opts, 1):
204
+ lines.append(f"{idx}. {txt}")
205
+ lines.append("")
206
+ lines.append("请选择最符合本项目规范的做法。")
207
+ return "\n".join(lines)
208
+
209
+
210
+ # 生效条件:cg 存在时把模块常量 CASES 逐条以 c["task"] 为 error、c["lesson"] 为 fix 组成列表传给 cg.mine_fix_pairs 并原样返回其结果。
211
+ def _build_memory(cg):
212
+ """Phase A:把 5 条「项目规范」写进记忆(真实 mine_fix_pairs)。"""
213
+ return cg.mine_fix_pairs(
214
+ [{"error": c["task"], "fix": c["lesson"]} for c in CASES])
215
+
216
+
217
+ # 生效条件:以 "本项目规范:"+case["query"] 且 budget_tokens=budget、k=10 调用 cg.recall,取 pack 各项 content(缺失/假值为 "")拼接,返回其前 1500 字符与 case["marker"] 是否出现在未截断拼接文本中。
218
+ def _recall_memory(cg, case, budget):
219
+ res = cg.recall("本项目规范:" + case["query"],
220
+ budget_tokens=budget, k=10)
221
+ text = " ".join((p.get("content") or "") for p in res.get("pack") or [])
222
+ return text[:1500], (case["marker"] in text)
223
+
224
+
225
+ # 生效条件:在 cfg["max_turns"] 轮内以 cfg["model"]/cfg["base"]/cfg["key"] 调 llm_chat(timeout=cfg["timeout"]、max_tokens=cfg["max_tokens"])并累加 usage["total_tokens"],pick 等于 correct_pos 或为 None 时中止(否则在仍有剩余轮次时追加 assistant/user 消息重选),最终 pick 为 None 则 kind="invalid"、否则 kind=opts[pick-1][0]。
226
+ def _run_one(cfg, case, opts, correct_pos, memory):
227
+ messages = [{"role": "system", "content": SYS_PROMPT},
228
+ {"role": "user", "content": _user_prompt(case, opts, memory)}]
229
+ tokens, turns, pick, raw, dt = 0, 0, None, "", 0.0
230
+ while turns < cfg["max_turns"]:
231
+ turns += 1
232
+ raw, usage, dt = llm_chat(cfg["model"], cfg["base"], cfg["key"],
233
+ messages, timeout=cfg["timeout"],
234
+ max_tokens=cfg["max_tokens"])
235
+ tokens += int(usage.get("total_tokens") or 0)
236
+ pick = _parse_pick(raw, len(opts))
237
+ if pick == correct_pos or pick is None:
238
+ break
239
+ if turns < cfg["max_turns"]:
240
+ messages.append({"role": "assistant", "content": raw})
241
+ messages.append({"role": "user",
242
+ "content": f"第 {pick} 个做法执行后问题依旧。"
243
+ f"请在剩余候选中重新选择,仍然只输出 JSON。"})
244
+ kind = "invalid" if pick is None else opts[pick - 1][0]
245
+ return {"pick": pick, "kind": kind, "correct": pick == correct_pos,
246
+ "turns": turns, "tokens": tokens, "latency": dt}
247
+
248
+
249
+ # 生效条件:cfg["use_mem"] 为真时对每个 case 以 cfg["budget"] 先串行召回写入 mems、否则存 (None, False),再对每 case × _orders(case, n_perms) 的排列 × ("none","mem") 两臂建任务交 ThreadPoolExecutor(max_workers=workers) 执行,返回按 arm 分组的 {"none": [...], "mem": [...]} 行表。
250
+ def _run_model(label, cfg, cases, cg, n_perms, workers):
251
+ # 记忆召回先串行算好(MdCGOS 非线程安全),LLM 调用再并发。
252
+ mems = {}
253
+ for case in cases:
254
+ mems[case["id"]] = (_recall_memory(cg, case, cfg["budget"])
255
+ if cfg["use_mem"] else (None, False))
256
+
257
+ tasks = []
258
+ for case in cases:
259
+ for oi, opts in enumerate(_orders(case, n_perms)):
260
+ correct_pos = next(i for i, (k, _t) in enumerate(opts, 1)
261
+ if k == "fix")
262
+ for arm in ("none", "mem"):
263
+ tasks.append((case, oi, opts, correct_pos, arm))
264
+
265
+ # 生效条件:t 解包为 (case, oi, opts, correct_pos, arm),arm=="mem" 时 memory/hit 取闭包 mems[case["id"]]、否则为 (None, False),_run_one 抛异常时以 kind="error"、pick=None 的占位行替代,随后补上 case/arm/hit 并打印该行后返回 r。
266
+ def work(t):
267
+ case, oi, opts, correct_pos, arm = t
268
+ memory, hit = mems[case["id"]] if arm == "mem" else (None, False)
269
+ try:
270
+ r = _run_one(cfg, case, opts, correct_pos, memory)
271
+ except Exception as exc: # noqa: BLE001
272
+ r = {"pick": None, "kind": "error", "correct": False,
273
+ "turns": 0, "tokens": 0, "latency": 0.0,
274
+ "raw": f"{type(exc).__name__}: {exc}"}
275
+ r.update({"case": case["id"], "arm": arm, "hit": hit})
276
+ print(f" [{label}] {case['id']} {arm:<4} rot={oi} "
277
+ f"-> pick={r['pick']} {r['kind']:<6} tok={r['tokens']:>4} "
278
+ f"({r['latency']:.1f}s)", flush=True)
279
+ return r
280
+
281
+ rows = {"none": [], "mem": []}
282
+ with ThreadPoolExecutor(max_workers=workers) as ex:
283
+ for r in ex.map(work, tasks):
284
+ rows[r["arm"]].append(r)
285
+ return rows
286
+
287
+
288
+ # 生效条件:n 为真值时返回 f"{100.0*x/n:.1f}%",n 为假值(0)时返回 "-"。
289
+ def _pct(x, n):
290
+ return f"{100.0 * x / n:.1f}%" if n else "-"
291
+
292
+
293
+ # 生效条件:以 rows["none"] 的行数 n 汇总 none/mem 两臂的 correct、kind=="trap"、kind∈("invalid","error")、tokens、latency、hit 与 pick∈(1,2,3) 的分布并打印(n_perms 仅用于表头文字),返回该 agg 字典。
294
+ def _report(label, rows, n_perms):
295
+ n = len(rows["none"])
296
+ print(f"\n{'-' * 78}")
297
+ print(f"模型:{label} 样本 {n} 次(用例 × 排列)· 每例 {n_perms} 种排列")
298
+ print(f"{'指标':<18}{'arm_none':>14}{'arm_mem':>14}{'差值':>16}")
299
+ print("-" * 78)
300
+ agg = {}
301
+ for arm in ("none", "mem"):
302
+ rs = rows[arm]
303
+ agg[arm] = {
304
+ "acc": sum(1 for r in rs if r["correct"]),
305
+ "trap": sum(1 for r in rs if r["kind"] == "trap"),
306
+ "invalid": sum(1 for r in rs if r["kind"] in ("invalid", "error")),
307
+ "tokens": sum(r["tokens"] for r in rs),
308
+ "latency": sum(r["latency"] for r in rs),
309
+ "hit": sum(1 for r in rs if r["hit"]),
310
+ "pos": [sum(1 for r in rs if r["pick"] == p) for p in (1, 2, 3)],
311
+ }
312
+ a, b = agg["none"], agg["mem"]
313
+ d_acc = f"+{100.0 * (b['acc'] - a['acc']) / n:.1f}pp"
314
+ d_trap = f"{100.0 * (b['trap'] - a['trap']) / n:.1f}pp"
315
+ d_inv = f"{100.0 * (b['invalid'] - a['invalid']) / n:.1f}pp"
316
+ d_tok = f"{(b['tokens'] - a['tokens']) / n:+.1f}"
317
+ d_lat = f"{(b['latency'] - a['latency']) / n:+.2f}"
318
+ print(f"{'content_acc':<18}{_pct(a['acc'], n):>14}{_pct(b['acc'], n):>14}"
319
+ f"{d_acc:>16}")
320
+ print(f"{'trap_rate':<18}{_pct(a['trap'], n):>14}{_pct(b['trap'], n):>14}"
321
+ f"{d_trap:>16}")
322
+ print(f"{'invalid_rate':<18}{_pct(a['invalid'], n):>14}"
323
+ f"{_pct(b['invalid'], n):>14}{d_inv:>16}")
324
+ print(f"{'avg_tokens':<18}{a['tokens'] / n:>14.1f}{b['tokens'] / n:>14.1f}"
325
+ f"{d_tok:>16}")
326
+ print(f"{'avg_latency_s':<18}{a['latency'] / n:>14.2f}"
327
+ f"{b['latency'] / n:>14.2f}{d_lat:>16}")
328
+ print(f"{'recall_hit_rate':<18}{'-':>14}{_pct(b['hit'], n):>14}{'-':>16}")
329
+ print(f"{'pos_dist(1/2/3)':<18}"
330
+ f"{'/'.join(str(x) for x in a['pos']):>14}"
331
+ f"{'/'.join(str(x) for x in b['pos']):>14}{'—':>16}")
332
+
333
+ print("\n逐用例 content_acc(按内容选对次数 / 该例排列数):")
334
+ for case in CASES:
335
+ if not any(r["case"] == case["id"] for r in rows["none"]):
336
+ continue
337
+ an = [r for r in rows["none"] if r["case"] == case["id"]]
338
+ am = [r for r in rows["mem"] if r["case"] == case["id"]]
339
+ na = sum(1 for r in an if r["correct"])
340
+ ma = sum(1 for r in am if r["correct"])
341
+ hits = sum(1 for r in am if r["hit"])
342
+ print(f" {case['id']} none {na}/{len(an)} mem {ma}/{len(am)} "
343
+ f"hit {hits}/{len(am)} {case['task'][:38]}")
344
+ return agg
345
+
346
+
347
+ # 生效条件:argv 为 None 时改读 sys.argv;若 --key 与 DEEPSEEK_API_KEY 均为空则返回 2,否则按 --only/--cases 从 CASES 取用例跑完双臂后返回 0。
348
+ def main(argv=None):
349
+ ap = argparse.ArgumentParser(
350
+ description="先验陷阱 A/B:无记忆 vs 有记忆(真实 LLM)")
351
+ ap.add_argument("--model", default="deepseek-v4.1-flash-expires-on-0910")
352
+ ap.add_argument("--base", default="https://api.deepseek.com/v1")
353
+ ap.add_argument("--key", default="")
354
+ ap.add_argument("--local", action="store_true",
355
+ help="追加本地小模型做对照")
356
+ ap.add_argument("--local-model", default="smegmma-deluxe-9b-v1")
357
+ ap.add_argument("--local-base", default="http://localhost:1234/v1")
358
+ ap.add_argument("--cases", type=int, default=0, help="只用前 N 个用例")
359
+ ap.add_argument("--only", default="", help="只用指定 id,逗号分隔,如 p01,p02")
360
+ ap.add_argument("--perms", type=int, default=3, help="每例候选排列数(默认 3)")
361
+ ap.add_argument("--budget", type=int, default=3000, help="召回 token 预算")
362
+ ap.add_argument("--max-tokens", type=int, default=2000)
363
+ ap.add_argument("--timeout", type=int, default=180)
364
+ ap.add_argument("--workers", type=int, default=4, help="并发调用数")
365
+ ap.add_argument("--max-turns", type=int, default=1,
366
+ help="最大轮数(1=单发;>1 则在选错后反馈重选)")
367
+ a = ap.parse_args(argv)
368
+
369
+ cases = CASES
370
+ if a.only:
371
+ want = {x.strip() for x in a.only.split(",") if x.strip()}
372
+ cases = [c for c in cases if c["id"] in want]
373
+ if a.cases:
374
+ cases = cases[:a.cases]
375
+ n_perms = max(1, min(6, a.perms))
376
+
377
+ key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
378
+ if not key:
379
+ print("[error] 需要 DEEPSEEK_API_KEY 或 --key")
380
+ return 2
381
+
382
+ root = tempfile.mkdtemp(prefix="mdcg_ab_prior_")
383
+ try:
384
+ cg = MdCGOS(os.path.join(root, "mem"))
385
+ mined = _build_memory(cg)
386
+ print(f"Phase A 写入记忆:{len(mined['pairs'])} 组"
387
+ f"(知识 {len(mined['knowledge_ids'])} + 负记忆 {len(mined['rejected_ids'])})")
388
+ print(f"用例 {len(cases)} × 排列 {n_perms} × 2 臂 = "
389
+ f"{len(cases) * n_perms * 2} 次调用")
390
+
391
+ targets = [(a.model, a.model, a.base, key)]
392
+ if a.local:
393
+ targets.append(("local:" + a.local_model, a.local_model,
394
+ a.local_base, "lm-studio"))
395
+
396
+ for label, model, base, mkey in targets:
397
+ cfg = {"model": model, "base": base, "key": mkey,
398
+ "budget": a.budget, "timeout": a.timeout,
399
+ "max_tokens": a.max_tokens, "max_turns": a.max_turns,
400
+ "use_mem": True}
401
+ rows = _run_model(label, cfg, cases, cg, n_perms, a.workers)
402
+ _report(label, rows, n_perms)
403
+ finally:
404
+ shutil.rmtree(root, ignore_errors=True)
405
+ return 0
406
+
407
+
408
+ if __name__ == "__main__":
409
409
  sys.exit(main())