@furongjun1999/dsh-memory 0.4.11 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -16
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +142 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/hooks.js +36 -2
- package/lib/lib/roleplay_web.js +427 -427
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +368 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1327 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +285 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1005 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +300 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1536 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1097 -1097
- package/md_cg/crypto.py +437 -437
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +334 -334
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +580 -580
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +220 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +329 -329
- package/md_cg/hotcache.py +238 -214
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +199 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +622 -622
- package/md_cg/mcp_server.py +129 -32
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +176 -117
- package/md_cg/mdcos.py +79 -13
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +693 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +298 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +85 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +365 -365
- package/md_cg/scrub.py +852 -852
- package/md_cg/security.py +274 -274
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +151 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +562 -562
- package/md_cg/sources.py +815 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +48 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +1138 -1138
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branches.py +249 -249
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_en_pipeline.py +166 -166
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_identity_attribution.py +147 -147
- package/md_cg/test_index_durability.py +224 -224
- package/md_cg/test_interop.py +93 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +765 -765
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +298 -298
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +113 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +281 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_readcache_prodpath.py +155 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +340 -340
- package/md_cg/test_retr_s1b.py +209 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +384 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +175 -175
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +241 -241
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +397 -397
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +273 -273
- package/md_cg/tokens.py +677 -663
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +667 -667
- package/md_cg/vision_evidence.py +666 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +550 -542
- package/package.json +97 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +401 -401
- package/src/hooks.ts +38 -2
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +932 -932
- package/src/lib/token_store.ts +192 -192
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
|
@@ -1,409 +1,409 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""任务级 A/B 真实端到端评测(先验陷阱版):无记忆 vs 有记忆
|
|
3
|
-
|
|
4
|
-
为什么要重做成「先验陷阱」:
|
|
5
|
-
上一版用 12 个教科书级报错(缺依赖、权限、超时…)。实测发现:这类题的正确答案
|
|
6
|
-
就在模型先验里——9B 小模型无记忆也有 94.4%,强模型 98.6%。也就是说**题目本身
|
|
7
|
-
低于模型能力下限,天花板效应把记忆的价值全遮住了**。用户判断成立。
|
|
8
|
-
|
|
9
|
-
本版的设计原则(用户提出):
|
|
10
|
-
「所有的知识都有隐形的条件。只有具有完整的记忆,才能发现哪些是当前不适用的知识。」
|
|
11
|
-
因此每个用例都是一个**先验陷阱**:
|
|
12
|
-
· trap = 通用最佳实践 / 教科书答案(模型会自信地选它)
|
|
13
|
-
· fix = 本项目规范下的正确做法(与最佳实践相反,只有记忆里的隐性条件能推出)
|
|
14
|
-
· decoy = 明显错误的做法
|
|
15
|
-
无记忆臂只能靠先验,会掉进 trap;有记忆臂能读到「项目规范 + 历史事故」这条隐性
|
|
16
|
-
条件,从而发现最佳实践在当前项目里**不适用**,改选 fix。
|
|
17
|
-
|
|
18
|
-
唯一变量 = 记忆库;其余全同(同一模型、temperature=0、同一系统提示)。
|
|
19
|
-
|
|
20
|
-
防作弊设计:
|
|
21
|
-
1) 措辞不同:记忆写「项目规范/原因」,候选写「具体动作」,模型须自行映射,不能照抄。
|
|
22
|
-
2) 多排列去位置偏差:每例跑 3 种**循环排列**(rotations),使每个候选在 3 个位置上
|
|
23
|
-
各出现一次——只有「按内容选」才能 3/3 全中;纯位置偏差只能 1/3。位置分布单列。
|
|
24
|
-
|
|
25
|
-
指标(以 (用例 × 排列) 为单位):
|
|
26
|
-
content_acc 按内容选对的比例(主指标)
|
|
27
|
-
trap_rate 选中「最佳实践陷阱」的比例(无记忆臂应显著 >0)
|
|
28
|
-
invalid_rate 输出无法解析为候选编号的比例(不剔除,如实计入)
|
|
29
|
-
pos_dist 选中位置分布(1/2/3),偏离均匀 = 位置偏差
|
|
30
|
-
avg_tokens 每次调用总 token(记忆的代价)
|
|
31
|
-
recall_hit_rate 记忆臂中 recall 确实召回该项目规范的比例
|
|
32
|
-
|
|
33
|
-
诚实边界:
|
|
34
|
-
· 5 用例 × 3 排列 = 15 次/臂,样本很小,只作方向性验证。
|
|
35
|
-
· 陷阱是否真的骗到模型,由 trap_rate 实测决定;骗不到就如实报告「该题不成立」。
|
|
36
|
-
· 记忆能召回 ≠ 模型会用;hit 与 content_acc 分开统计正是为暴露这个差异。
|
|
37
|
-
|
|
38
|
-
跑法:
|
|
39
|
-
python -m md_cg.bench_task_ab_llm # deepseek-v4-flash(默认)
|
|
40
|
-
python -m md_cg.bench_task_ab_llm --local # 追加本地小模型
|
|
41
|
-
python -m md_cg.bench_task_ab_llm --perms 6 # 全 6 种排列
|
|
42
|
-
"""
|
|
43
|
-
from __future__ import annotations
|
|
44
|
-
|
|
45
|
-
import argparse
|
|
46
|
-
import itertools
|
|
47
|
-
import json
|
|
48
|
-
import os
|
|
49
|
-
import random
|
|
50
|
-
import re
|
|
51
|
-
import shutil
|
|
52
|
-
import sys
|
|
53
|
-
import tempfile
|
|
54
|
-
import time
|
|
55
|
-
import urllib.error
|
|
56
|
-
import urllib.request
|
|
57
|
-
from concurrent.futures import ThreadPoolExecutor
|
|
58
|
-
|
|
59
|
-
from md_cg.mdcos import MdCGOS
|
|
60
|
-
|
|
61
|
-
# ---- 系统提示(两臂完全相同)---------------------------------------------
|
|
62
|
-
SYS_PROMPT = (
|
|
63
|
-
"你是本项目的资深工程师。用户给出一个工程场景和若干候选做法,"
|
|
64
|
-
"你选一个最符合本项目既定规范的做法。\n"
|
|
65
|
-
"只输出一行 JSON,不要解释、不要代码围栏,格式:\n"
|
|
66
|
-
'{"pick": 2, "reason": "本项目规范要求..."}'
|
|
67
|
-
)
|
|
68
|
-
|
|
69
|
-
# ---- 5 个先验陷阱用例 ------------------------------------------------------
|
|
70
|
-
# task :场景(模型据此决策)
|
|
71
|
-
# lesson :写入记忆的「项目规范 + 历史事故」(隐性条件,候选里不出现)
|
|
72
|
-
# query :召回用检索词
|
|
73
|
-
# fix/trap/decoy:候选做法;trap 是通用最佳实践(模型先验),fix 才是本项目正解
|
|
74
|
-
CASES = [
|
|
75
|
-
{
|
|
76
|
-
"id": "p01",
|
|
77
|
-
"task": "API 的参数校验失败了,应该怎么给客户端返回?",
|
|
78
|
-
"query": "API 参数校验失败 业务错误 返回 HTTP 状态码",
|
|
79
|
-
"lesson": "本项目 API 契约:所有业务错误(含参数校验失败)必须返回 HTTP 200,"
|
|
80
|
-
"在 body 的 code 字段带业务错误码,禁止用 HTTP 4xx/5xx。"
|
|
81
|
-
"原因:老旧移动端 SDK 把任何非 200 都当成网络故障无限重试。",
|
|
82
|
-
"trap": "返回 HTTP 400,body 里写清楚校验失败详情",
|
|
83
|
-
"fix": "返回 HTTP 200,body 的 code 字段带业务错误码",
|
|
84
|
-
"decoy": "返回 HTTP 500,让客户端稍后重试",
|
|
85
|
-
},
|
|
86
|
-
{
|
|
87
|
-
"id": "p02",
|
|
88
|
-
"task": "API 响应里要返回一个时间戳,用什么格式?",
|
|
89
|
-
"query": "API 时间戳 格式 ISO 毫秒 时区",
|
|
90
|
-
"lesson": "本项目 API 契约:所有时间戳统一返回 Unix 毫秒整数(int64),"
|
|
91
|
-
"禁止 ISO 8601 字符串和本地时间字符串。"
|
|
92
|
-
"原因:跨语言客户端解析不一致,历史事故是时区偏移导致订单时间错乱 8 小时。",
|
|
93
|
-
"trap": "返回 ISO 8601 字符串,如 2026-09-09T12:00:00Z",
|
|
94
|
-
"fix": "返回 Unix 毫秒整数(int64)",
|
|
95
|
-
"decoy": "返回本地时间字符串,如 '2026/9/9 12:00'",
|
|
96
|
-
},
|
|
97
|
-
{
|
|
98
|
-
"id": "p03",
|
|
99
|
-
"task": "校验请求参数 x 必须是整数,怎么写这个判断?",
|
|
100
|
-
"query": "校验 参数 整数 布尔 bool isinstance 精确类型",
|
|
101
|
-
"lesson": "本项目校验规范:需要精确类型时必须用 type(x) is int,禁止 isinstance。"
|
|
102
|
-
"原因:bool 是 int 的子类,isinstance(True, int) 为真,"
|
|
103
|
-
"历史事故是布尔值被当成整数接受,写进了金额字段。",
|
|
104
|
-
"trap": "用 isinstance(x, int) 判断",
|
|
105
|
-
"fix": "用 type(x) is int 判断",
|
|
106
|
-
"decoy": "用 x == int 判断",
|
|
107
|
-
},
|
|
108
|
-
{
|
|
109
|
-
"id": "p04",
|
|
110
|
-
"task": "从配置字典 cfg 里读一个可选字段的值,怎么写?",
|
|
111
|
-
"query": "配置 字典 可选字段 缺失 None get 默认值",
|
|
112
|
-
"lesson": "本项目配置规范:可选字段必须用 'key' in cfg 判断后再取值,禁止 .get()。"
|
|
113
|
-
"原因:.get() 把「显式 None」和「未配置」混为一谈,"
|
|
114
|
-
"历史事故是默认值覆盖了用户显式设置的空值。",
|
|
115
|
-
"trap": "用 cfg.get('key', default) 取值",
|
|
116
|
-
"fix": "用 'key' in cfg 判断后再取值",
|
|
117
|
-
"decoy": "用 cfg['key'] 直接索引",
|
|
118
|
-
},
|
|
119
|
-
{
|
|
120
|
-
"id": "p05",
|
|
121
|
-
"task": "线上要记录一条警告信息(比如缓存 miss),用哪个日志级别?",
|
|
122
|
-
"query": "日志 级别 警告 缓存 miss warning info 告警",
|
|
123
|
-
"lesson": "本项目日志规范:warning 及以上级别会触发 PagerDuty 告警;"
|
|
124
|
-
"非故障类的警告必须用 info 级别并在消息里加 [WARN] 前缀,禁止直接用 warning。"
|
|
125
|
-
"原因:历史事故是缓存 miss 的 warning 半夜触发全组告警。",
|
|
126
|
-
"trap": "用 logger.warning(...) 记录",
|
|
127
|
-
"fix": "用 logger.info(...) 并在消息里加 [WARN] 前缀",
|
|
128
|
-
"decoy": "用 logger.error(...) 记录",
|
|
129
|
-
},
|
|
130
|
-
]
|
|
131
|
-
|
|
132
|
-
for _c in CASES:
|
|
133
|
-
_c["marker"] = _c["lesson"][:12]
|
|
134
|
-
|
|
135
|
-
|
|
136
|
-
# ---- LLM 客户端(OpenAI 兼容,零第三方依赖)-------------------------------
|
|
137
|
-
# 生效条件:给定 model/base/key/messages 即构造 JSON POST 到 base.rstrip("/")+"/chat/completions",返回 (choices[0].message 的 content 或缺失/假值回落 ""、data.get("usage") 或缺失/假值回落 {}、耗时 dt),timeout/max_tokens/temperature 仅作为请求参数传入。
|
|
138
|
-
def llm_chat(model, base, key, messages, timeout=180, max_tokens=200,
|
|
139
|
-
temperature=0.0):
|
|
140
|
-
payload = json.dumps({
|
|
141
|
-
"model": model,
|
|
142
|
-
"messages": messages,
|
|
143
|
-
"temperature": temperature,
|
|
144
|
-
"max_tokens": max_tokens,
|
|
145
|
-
}).encode("utf-8")
|
|
146
|
-
req = urllib.request.Request(
|
|
147
|
-
base.rstrip("/") + "/chat/completions", data=payload,
|
|
148
|
-
headers={"Content-Type": "application/json",
|
|
149
|
-
"Authorization": f"Bearer {key}"})
|
|
150
|
-
t0 = time.time()
|
|
151
|
-
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
152
|
-
data = json.loads(resp.read().decode("utf-8"))
|
|
153
|
-
dt = time.time() - t0
|
|
154
|
-
content = data["choices"][0]["message"].get("content") or ""
|
|
155
|
-
return content, (data.get("usage") or {}), dt
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
# 生效条件:raw(None/空串视为 "")中首个 "{" 至末个 "}" 的子串能解析出 pick,或全文匹配到单个数字 1-9,且该编号落在 1..n 内时返回该编号(pick 为数字字符串先转 int),否则返回 None。
|
|
159
|
-
def _parse_pick(raw, n):
|
|
160
|
-
"""从模型输出里抠出候选编号(1..n);解析失败返回 None。"""
|
|
161
|
-
s = raw or ""
|
|
162
|
-
i, j = s.find("{"), s.rfind("}")
|
|
163
|
-
if i >= 0 and j > i:
|
|
164
|
-
try:
|
|
165
|
-
obj = json.loads(s[i:j + 1])
|
|
166
|
-
p = obj.get("pick")
|
|
167
|
-
if isinstance(p, str) and p.strip().isdigit():
|
|
168
|
-
p = int(p.strip())
|
|
169
|
-
if isinstance(p, (int, float)) and 1 <= int(p) <= n:
|
|
170
|
-
return int(p)
|
|
171
|
-
except ValueError:
|
|
172
|
-
pass
|
|
173
|
-
m = re.search(r"\b([1-9])\b", s)
|
|
174
|
-
if m and 1 <= int(m.group(1)) <= n:
|
|
175
|
-
return int(m.group(1))
|
|
176
|
-
return None
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
# 生效条件:以 case["id"] 播种打乱 [(fix,case["fix"]),(trap,case["trap"]),(decoy,case["decoy"])] 后取三种循环排列,n<=3 返回其前 n 个,n>3 返回 6 种全排列的前 n 个。
|
|
180
|
-
def _orders(case, n):
|
|
181
|
-
"""候选顺序。默认取 3 种循环排列(每个候选在 3 个位置各出现一次,天然去位置偏差);
|
|
182
|
-
n>3 时退回全 6 种排列。基准顺序由 case id 播种打乱,避免跨用例雷同。"""
|
|
183
|
-
base = [("fix", case["fix"]), ("trap", case["trap"]),
|
|
184
|
-
("decoy", case["decoy"])]
|
|
185
|
-
random.Random(case["id"]).shuffle(base)
|
|
186
|
-
rots = [tuple(base[i:] + base[:i]) for i in range(3)]
|
|
187
|
-
if n <= 3:
|
|
188
|
-
return rots[:n]
|
|
189
|
-
return list(itertools.permutations(base))[:n]
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
# 生效条件:case 含 task 且 opts 为 (kind, txt) 序列时,返回含场景、编号候选与选择指令的提示文本,memory 非空时在开头插入记忆段。
|
|
193
|
-
def _user_prompt(case, opts, memory=None):
|
|
194
|
-
lines = []
|
|
195
|
-
if memory:
|
|
196
|
-
lines.append("【本项目规范 / 历史经验(来自记忆库,请自行判断相关性)】")
|
|
197
|
-
lines.append(memory.strip())
|
|
198
|
-
lines.append("")
|
|
199
|
-
lines.append("场景:")
|
|
200
|
-
lines.append(case["task"])
|
|
201
|
-
lines.append("")
|
|
202
|
-
lines.append("候选做法:")
|
|
203
|
-
for idx, (_kind, txt) in enumerate(opts, 1):
|
|
204
|
-
lines.append(f"{idx}. {txt}")
|
|
205
|
-
lines.append("")
|
|
206
|
-
lines.append("请选择最符合本项目规范的做法。")
|
|
207
|
-
return "\n".join(lines)
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
# 生效条件:cg 存在时把模块常量 CASES 逐条以 c["task"] 为 error、c["lesson"] 为 fix 组成列表传给 cg.mine_fix_pairs 并原样返回其结果。
|
|
211
|
-
def _build_memory(cg):
|
|
212
|
-
"""Phase A:把 5 条「项目规范」写进记忆(真实 mine_fix_pairs)。"""
|
|
213
|
-
return cg.mine_fix_pairs(
|
|
214
|
-
[{"error": c["task"], "fix": c["lesson"]} for c in CASES])
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
# 生效条件:以 "本项目规范:"+case["query"] 且 budget_tokens=budget、k=10 调用 cg.recall,取 pack 各项 content(缺失/假值为 "")拼接,返回其前 1500 字符与 case["marker"] 是否出现在未截断拼接文本中。
|
|
218
|
-
def _recall_memory(cg, case, budget):
|
|
219
|
-
res = cg.recall("本项目规范:" + case["query"],
|
|
220
|
-
budget_tokens=budget, k=10)
|
|
221
|
-
text = " ".join((p.get("content") or "") for p in res.get("pack") or [])
|
|
222
|
-
return text[:1500], (case["marker"] in text)
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
# 生效条件:在 cfg["max_turns"] 轮内以 cfg["model"]/cfg["base"]/cfg["key"] 调 llm_chat(timeout=cfg["timeout"]、max_tokens=cfg["max_tokens"])并累加 usage["total_tokens"],pick 等于 correct_pos 或为 None 时中止(否则在仍有剩余轮次时追加 assistant/user 消息重选),最终 pick 为 None 则 kind="invalid"、否则 kind=opts[pick-1][0]。
|
|
226
|
-
def _run_one(cfg, case, opts, correct_pos, memory):
|
|
227
|
-
messages = [{"role": "system", "content": SYS_PROMPT},
|
|
228
|
-
{"role": "user", "content": _user_prompt(case, opts, memory)}]
|
|
229
|
-
tokens, turns, pick, raw, dt = 0, 0, None, "", 0.0
|
|
230
|
-
while turns < cfg["max_turns"]:
|
|
231
|
-
turns += 1
|
|
232
|
-
raw, usage, dt = llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
233
|
-
messages, timeout=cfg["timeout"],
|
|
234
|
-
max_tokens=cfg["max_tokens"])
|
|
235
|
-
tokens += int(usage.get("total_tokens") or 0)
|
|
236
|
-
pick = _parse_pick(raw, len(opts))
|
|
237
|
-
if pick == correct_pos or pick is None:
|
|
238
|
-
break
|
|
239
|
-
if turns < cfg["max_turns"]:
|
|
240
|
-
messages.append({"role": "assistant", "content": raw})
|
|
241
|
-
messages.append({"role": "user",
|
|
242
|
-
"content": f"第 {pick} 个做法执行后问题依旧。"
|
|
243
|
-
f"请在剩余候选中重新选择,仍然只输出 JSON。"})
|
|
244
|
-
kind = "invalid" if pick is None else opts[pick - 1][0]
|
|
245
|
-
return {"pick": pick, "kind": kind, "correct": pick == correct_pos,
|
|
246
|
-
"turns": turns, "tokens": tokens, "latency": dt}
|
|
247
|
-
|
|
248
|
-
|
|
249
|
-
# 生效条件:cfg["use_mem"] 为真时对每个 case 以 cfg["budget"] 先串行召回写入 mems、否则存 (None, False),再对每 case × _orders(case, n_perms) 的排列 × ("none","mem") 两臂建任务交 ThreadPoolExecutor(max_workers=workers) 执行,返回按 arm 分组的 {"none": [...], "mem": [...]} 行表。
|
|
250
|
-
def _run_model(label, cfg, cases, cg, n_perms, workers):
|
|
251
|
-
# 记忆召回先串行算好(MdCGOS 非线程安全),LLM 调用再并发。
|
|
252
|
-
mems = {}
|
|
253
|
-
for case in cases:
|
|
254
|
-
mems[case["id"]] = (_recall_memory(cg, case, cfg["budget"])
|
|
255
|
-
if cfg["use_mem"] else (None, False))
|
|
256
|
-
|
|
257
|
-
tasks = []
|
|
258
|
-
for case in cases:
|
|
259
|
-
for oi, opts in enumerate(_orders(case, n_perms)):
|
|
260
|
-
correct_pos = next(i for i, (k, _t) in enumerate(opts, 1)
|
|
261
|
-
if k == "fix")
|
|
262
|
-
for arm in ("none", "mem"):
|
|
263
|
-
tasks.append((case, oi, opts, correct_pos, arm))
|
|
264
|
-
|
|
265
|
-
# 生效条件:t 解包为 (case, oi, opts, correct_pos, arm),arm=="mem" 时 memory/hit 取闭包 mems[case["id"]]、否则为 (None, False),_run_one 抛异常时以 kind="error"、pick=None 的占位行替代,随后补上 case/arm/hit 并打印该行后返回 r。
|
|
266
|
-
def work(t):
|
|
267
|
-
case, oi, opts, correct_pos, arm = t
|
|
268
|
-
memory, hit = mems[case["id"]] if arm == "mem" else (None, False)
|
|
269
|
-
try:
|
|
270
|
-
r = _run_one(cfg, case, opts, correct_pos, memory)
|
|
271
|
-
except Exception as exc: # noqa: BLE001
|
|
272
|
-
r = {"pick": None, "kind": "error", "correct": False,
|
|
273
|
-
"turns": 0, "tokens": 0, "latency": 0.0,
|
|
274
|
-
"raw": f"{type(exc).__name__}: {exc}"}
|
|
275
|
-
r.update({"case": case["id"], "arm": arm, "hit": hit})
|
|
276
|
-
print(f" [{label}] {case['id']} {arm:<4} rot={oi} "
|
|
277
|
-
f"-> pick={r['pick']} {r['kind']:<6} tok={r['tokens']:>4} "
|
|
278
|
-
f"({r['latency']:.1f}s)", flush=True)
|
|
279
|
-
return r
|
|
280
|
-
|
|
281
|
-
rows = {"none": [], "mem": []}
|
|
282
|
-
with ThreadPoolExecutor(max_workers=workers) as ex:
|
|
283
|
-
for r in ex.map(work, tasks):
|
|
284
|
-
rows[r["arm"]].append(r)
|
|
285
|
-
return rows
|
|
286
|
-
|
|
287
|
-
|
|
288
|
-
# 生效条件:n 为真值时返回 f"{100.0*x/n:.1f}%",n 为假值(0)时返回 "-"。
|
|
289
|
-
def _pct(x, n):
|
|
290
|
-
return f"{100.0 * x / n:.1f}%" if n else "-"
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
# 生效条件:以 rows["none"] 的行数 n 汇总 none/mem 两臂的 correct、kind=="trap"、kind∈("invalid","error")、tokens、latency、hit 与 pick∈(1,2,3) 的分布并打印(n_perms 仅用于表头文字),返回该 agg 字典。
|
|
294
|
-
def _report(label, rows, n_perms):
|
|
295
|
-
n = len(rows["none"])
|
|
296
|
-
print(f"\n{'-' * 78}")
|
|
297
|
-
print(f"模型:{label} 样本 {n} 次(用例 × 排列)· 每例 {n_perms} 种排列")
|
|
298
|
-
print(f"{'指标':<18}{'arm_none':>14}{'arm_mem':>14}{'差值':>16}")
|
|
299
|
-
print("-" * 78)
|
|
300
|
-
agg = {}
|
|
301
|
-
for arm in ("none", "mem"):
|
|
302
|
-
rs = rows[arm]
|
|
303
|
-
agg[arm] = {
|
|
304
|
-
"acc": sum(1 for r in rs if r["correct"]),
|
|
305
|
-
"trap": sum(1 for r in rs if r["kind"] == "trap"),
|
|
306
|
-
"invalid": sum(1 for r in rs if r["kind"] in ("invalid", "error")),
|
|
307
|
-
"tokens": sum(r["tokens"] for r in rs),
|
|
308
|
-
"latency": sum(r["latency"] for r in rs),
|
|
309
|
-
"hit": sum(1 for r in rs if r["hit"]),
|
|
310
|
-
"pos": [sum(1 for r in rs if r["pick"] == p) for p in (1, 2, 3)],
|
|
311
|
-
}
|
|
312
|
-
a, b = agg["none"], agg["mem"]
|
|
313
|
-
d_acc = f"+{100.0 * (b['acc'] - a['acc']) / n:.1f}pp"
|
|
314
|
-
d_trap = f"{100.0 * (b['trap'] - a['trap']) / n:.1f}pp"
|
|
315
|
-
d_inv = f"{100.0 * (b['invalid'] - a['invalid']) / n:.1f}pp"
|
|
316
|
-
d_tok = f"{(b['tokens'] - a['tokens']) / n:+.1f}"
|
|
317
|
-
d_lat = f"{(b['latency'] - a['latency']) / n:+.2f}"
|
|
318
|
-
print(f"{'content_acc':<18}{_pct(a['acc'], n):>14}{_pct(b['acc'], n):>14}"
|
|
319
|
-
f"{d_acc:>16}")
|
|
320
|
-
print(f"{'trap_rate':<18}{_pct(a['trap'], n):>14}{_pct(b['trap'], n):>14}"
|
|
321
|
-
f"{d_trap:>16}")
|
|
322
|
-
print(f"{'invalid_rate':<18}{_pct(a['invalid'], n):>14}"
|
|
323
|
-
f"{_pct(b['invalid'], n):>14}{d_inv:>16}")
|
|
324
|
-
print(f"{'avg_tokens':<18}{a['tokens'] / n:>14.1f}{b['tokens'] / n:>14.1f}"
|
|
325
|
-
f"{d_tok:>16}")
|
|
326
|
-
print(f"{'avg_latency_s':<18}{a['latency'] / n:>14.2f}"
|
|
327
|
-
f"{b['latency'] / n:>14.2f}{d_lat:>16}")
|
|
328
|
-
print(f"{'recall_hit_rate':<18}{'-':>14}{_pct(b['hit'], n):>14}{'-':>16}")
|
|
329
|
-
print(f"{'pos_dist(1/2/3)':<18}"
|
|
330
|
-
f"{'/'.join(str(x) for x in a['pos']):>14}"
|
|
331
|
-
f"{'/'.join(str(x) for x in b['pos']):>14}{'—':>16}")
|
|
332
|
-
|
|
333
|
-
print("\n逐用例 content_acc(按内容选对次数 / 该例排列数):")
|
|
334
|
-
for case in CASES:
|
|
335
|
-
if not any(r["case"] == case["id"] for r in rows["none"]):
|
|
336
|
-
continue
|
|
337
|
-
an = [r for r in rows["none"] if r["case"] == case["id"]]
|
|
338
|
-
am = [r for r in rows["mem"] if r["case"] == case["id"]]
|
|
339
|
-
na = sum(1 for r in an if r["correct"])
|
|
340
|
-
ma = sum(1 for r in am if r["correct"])
|
|
341
|
-
hits = sum(1 for r in am if r["hit"])
|
|
342
|
-
print(f" {case['id']} none {na}/{len(an)} mem {ma}/{len(am)} "
|
|
343
|
-
f"hit {hits}/{len(am)} {case['task'][:38]}")
|
|
344
|
-
return agg
|
|
345
|
-
|
|
346
|
-
|
|
347
|
-
# 生效条件:argv 为 None 时改读 sys.argv;若 --key 与 DEEPSEEK_API_KEY 均为空则返回 2,否则按 --only/--cases 从 CASES 取用例跑完双臂后返回 0。
|
|
348
|
-
def main(argv=None):
|
|
349
|
-
ap = argparse.ArgumentParser(
|
|
350
|
-
description="先验陷阱 A/B:无记忆 vs 有记忆(真实 LLM)")
|
|
351
|
-
ap.add_argument("--model", default="deepseek-v4.1-flash-expires-on-0910")
|
|
352
|
-
ap.add_argument("--base", default="https://api.deepseek.com/v1")
|
|
353
|
-
ap.add_argument("--key", default="")
|
|
354
|
-
ap.add_argument("--local", action="store_true",
|
|
355
|
-
help="追加本地小模型做对照")
|
|
356
|
-
ap.add_argument("--local-model", default="smegmma-deluxe-9b-v1")
|
|
357
|
-
ap.add_argument("--local-base", default="http://localhost:1234/v1")
|
|
358
|
-
ap.add_argument("--cases", type=int, default=0, help="只用前 N 个用例")
|
|
359
|
-
ap.add_argument("--only", default="", help="只用指定 id,逗号分隔,如 p01,p02")
|
|
360
|
-
ap.add_argument("--perms", type=int, default=3, help="每例候选排列数(默认 3)")
|
|
361
|
-
ap.add_argument("--budget", type=int, default=3000, help="召回 token 预算")
|
|
362
|
-
ap.add_argument("--max-tokens", type=int, default=2000)
|
|
363
|
-
ap.add_argument("--timeout", type=int, default=180)
|
|
364
|
-
ap.add_argument("--workers", type=int, default=4, help="并发调用数")
|
|
365
|
-
ap.add_argument("--max-turns", type=int, default=1,
|
|
366
|
-
help="最大轮数(1=单发;>1 则在选错后反馈重选)")
|
|
367
|
-
a = ap.parse_args(argv)
|
|
368
|
-
|
|
369
|
-
cases = CASES
|
|
370
|
-
if a.only:
|
|
371
|
-
want = {x.strip() for x in a.only.split(",") if x.strip()}
|
|
372
|
-
cases = [c for c in cases if c["id"] in want]
|
|
373
|
-
if a.cases:
|
|
374
|
-
cases = cases[:a.cases]
|
|
375
|
-
n_perms = max(1, min(6, a.perms))
|
|
376
|
-
|
|
377
|
-
key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
|
|
378
|
-
if not key:
|
|
379
|
-
print("[error] 需要 DEEPSEEK_API_KEY 或 --key")
|
|
380
|
-
return 2
|
|
381
|
-
|
|
382
|
-
root = tempfile.mkdtemp(prefix="mdcg_ab_prior_")
|
|
383
|
-
try:
|
|
384
|
-
cg = MdCGOS(os.path.join(root, "mem"))
|
|
385
|
-
mined = _build_memory(cg)
|
|
386
|
-
print(f"Phase A 写入记忆:{len(mined['pairs'])} 组"
|
|
387
|
-
f"(知识 {len(mined['knowledge_ids'])} + 负记忆 {len(mined['rejected_ids'])})")
|
|
388
|
-
print(f"用例 {len(cases)} × 排列 {n_perms} × 2 臂 = "
|
|
389
|
-
f"{len(cases) * n_perms * 2} 次调用")
|
|
390
|
-
|
|
391
|
-
targets = [(a.model, a.model, a.base, key)]
|
|
392
|
-
if a.local:
|
|
393
|
-
targets.append(("local:" + a.local_model, a.local_model,
|
|
394
|
-
a.local_base, "lm-studio"))
|
|
395
|
-
|
|
396
|
-
for label, model, base, mkey in targets:
|
|
397
|
-
cfg = {"model": model, "base": base, "key": mkey,
|
|
398
|
-
"budget": a.budget, "timeout": a.timeout,
|
|
399
|
-
"max_tokens": a.max_tokens, "max_turns": a.max_turns,
|
|
400
|
-
"use_mem": True}
|
|
401
|
-
rows = _run_model(label, cfg, cases, cg, n_perms, a.workers)
|
|
402
|
-
_report(label, rows, n_perms)
|
|
403
|
-
finally:
|
|
404
|
-
shutil.rmtree(root, ignore_errors=True)
|
|
405
|
-
return 0
|
|
406
|
-
|
|
407
|
-
|
|
408
|
-
if __name__ == "__main__":
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""任务级 A/B 真实端到端评测(先验陷阱版):无记忆 vs 有记忆
|
|
3
|
+
|
|
4
|
+
为什么要重做成「先验陷阱」:
|
|
5
|
+
上一版用 12 个教科书级报错(缺依赖、权限、超时…)。实测发现:这类题的正确答案
|
|
6
|
+
就在模型先验里——9B 小模型无记忆也有 94.4%,强模型 98.6%。也就是说**题目本身
|
|
7
|
+
低于模型能力下限,天花板效应把记忆的价值全遮住了**。用户判断成立。
|
|
8
|
+
|
|
9
|
+
本版的设计原则(用户提出):
|
|
10
|
+
「所有的知识都有隐形的条件。只有具有完整的记忆,才能发现哪些是当前不适用的知识。」
|
|
11
|
+
因此每个用例都是一个**先验陷阱**:
|
|
12
|
+
· trap = 通用最佳实践 / 教科书答案(模型会自信地选它)
|
|
13
|
+
· fix = 本项目规范下的正确做法(与最佳实践相反,只有记忆里的隐性条件能推出)
|
|
14
|
+
· decoy = 明显错误的做法
|
|
15
|
+
无记忆臂只能靠先验,会掉进 trap;有记忆臂能读到「项目规范 + 历史事故」这条隐性
|
|
16
|
+
条件,从而发现最佳实践在当前项目里**不适用**,改选 fix。
|
|
17
|
+
|
|
18
|
+
唯一变量 = 记忆库;其余全同(同一模型、temperature=0、同一系统提示)。
|
|
19
|
+
|
|
20
|
+
防作弊设计:
|
|
21
|
+
1) 措辞不同:记忆写「项目规范/原因」,候选写「具体动作」,模型须自行映射,不能照抄。
|
|
22
|
+
2) 多排列去位置偏差:每例跑 3 种**循环排列**(rotations),使每个候选在 3 个位置上
|
|
23
|
+
各出现一次——只有「按内容选」才能 3/3 全中;纯位置偏差只能 1/3。位置分布单列。
|
|
24
|
+
|
|
25
|
+
指标(以 (用例 × 排列) 为单位):
|
|
26
|
+
content_acc 按内容选对的比例(主指标)
|
|
27
|
+
trap_rate 选中「最佳实践陷阱」的比例(无记忆臂应显著 >0)
|
|
28
|
+
invalid_rate 输出无法解析为候选编号的比例(不剔除,如实计入)
|
|
29
|
+
pos_dist 选中位置分布(1/2/3),偏离均匀 = 位置偏差
|
|
30
|
+
avg_tokens 每次调用总 token(记忆的代价)
|
|
31
|
+
recall_hit_rate 记忆臂中 recall 确实召回该项目规范的比例
|
|
32
|
+
|
|
33
|
+
诚实边界:
|
|
34
|
+
· 5 用例 × 3 排列 = 15 次/臂,样本很小,只作方向性验证。
|
|
35
|
+
· 陷阱是否真的骗到模型,由 trap_rate 实测决定;骗不到就如实报告「该题不成立」。
|
|
36
|
+
· 记忆能召回 ≠ 模型会用;hit 与 content_acc 分开统计正是为暴露这个差异。
|
|
37
|
+
|
|
38
|
+
跑法:
|
|
39
|
+
python -m md_cg.bench_task_ab_llm # deepseek-v4-flash(默认)
|
|
40
|
+
python -m md_cg.bench_task_ab_llm --local # 追加本地小模型
|
|
41
|
+
python -m md_cg.bench_task_ab_llm --perms 6 # 全 6 种排列
|
|
42
|
+
"""
|
|
43
|
+
from __future__ import annotations
|
|
44
|
+
|
|
45
|
+
import argparse
|
|
46
|
+
import itertools
|
|
47
|
+
import json
|
|
48
|
+
import os
|
|
49
|
+
import random
|
|
50
|
+
import re
|
|
51
|
+
import shutil
|
|
52
|
+
import sys
|
|
53
|
+
import tempfile
|
|
54
|
+
import time
|
|
55
|
+
import urllib.error
|
|
56
|
+
import urllib.request
|
|
57
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
58
|
+
|
|
59
|
+
from md_cg.mdcos import MdCGOS
|
|
60
|
+
|
|
61
|
+
# ---- 系统提示(两臂完全相同)---------------------------------------------
|
|
62
|
+
SYS_PROMPT = (
|
|
63
|
+
"你是本项目的资深工程师。用户给出一个工程场景和若干候选做法,"
|
|
64
|
+
"你选一个最符合本项目既定规范的做法。\n"
|
|
65
|
+
"只输出一行 JSON,不要解释、不要代码围栏,格式:\n"
|
|
66
|
+
'{"pick": 2, "reason": "本项目规范要求..."}'
|
|
67
|
+
)
|
|
68
|
+
|
|
69
|
+
# ---- 5 个先验陷阱用例 ------------------------------------------------------
|
|
70
|
+
# task :场景(模型据此决策)
|
|
71
|
+
# lesson :写入记忆的「项目规范 + 历史事故」(隐性条件,候选里不出现)
|
|
72
|
+
# query :召回用检索词
|
|
73
|
+
# fix/trap/decoy:候选做法;trap 是通用最佳实践(模型先验),fix 才是本项目正解
|
|
74
|
+
CASES = [
|
|
75
|
+
{
|
|
76
|
+
"id": "p01",
|
|
77
|
+
"task": "API 的参数校验失败了,应该怎么给客户端返回?",
|
|
78
|
+
"query": "API 参数校验失败 业务错误 返回 HTTP 状态码",
|
|
79
|
+
"lesson": "本项目 API 契约:所有业务错误(含参数校验失败)必须返回 HTTP 200,"
|
|
80
|
+
"在 body 的 code 字段带业务错误码,禁止用 HTTP 4xx/5xx。"
|
|
81
|
+
"原因:老旧移动端 SDK 把任何非 200 都当成网络故障无限重试。",
|
|
82
|
+
"trap": "返回 HTTP 400,body 里写清楚校验失败详情",
|
|
83
|
+
"fix": "返回 HTTP 200,body 的 code 字段带业务错误码",
|
|
84
|
+
"decoy": "返回 HTTP 500,让客户端稍后重试",
|
|
85
|
+
},
|
|
86
|
+
{
|
|
87
|
+
"id": "p02",
|
|
88
|
+
"task": "API 响应里要返回一个时间戳,用什么格式?",
|
|
89
|
+
"query": "API 时间戳 格式 ISO 毫秒 时区",
|
|
90
|
+
"lesson": "本项目 API 契约:所有时间戳统一返回 Unix 毫秒整数(int64),"
|
|
91
|
+
"禁止 ISO 8601 字符串和本地时间字符串。"
|
|
92
|
+
"原因:跨语言客户端解析不一致,历史事故是时区偏移导致订单时间错乱 8 小时。",
|
|
93
|
+
"trap": "返回 ISO 8601 字符串,如 2026-09-09T12:00:00Z",
|
|
94
|
+
"fix": "返回 Unix 毫秒整数(int64)",
|
|
95
|
+
"decoy": "返回本地时间字符串,如 '2026/9/9 12:00'",
|
|
96
|
+
},
|
|
97
|
+
{
|
|
98
|
+
"id": "p03",
|
|
99
|
+
"task": "校验请求参数 x 必须是整数,怎么写这个判断?",
|
|
100
|
+
"query": "校验 参数 整数 布尔 bool isinstance 精确类型",
|
|
101
|
+
"lesson": "本项目校验规范:需要精确类型时必须用 type(x) is int,禁止 isinstance。"
|
|
102
|
+
"原因:bool 是 int 的子类,isinstance(True, int) 为真,"
|
|
103
|
+
"历史事故是布尔值被当成整数接受,写进了金额字段。",
|
|
104
|
+
"trap": "用 isinstance(x, int) 判断",
|
|
105
|
+
"fix": "用 type(x) is int 判断",
|
|
106
|
+
"decoy": "用 x == int 判断",
|
|
107
|
+
},
|
|
108
|
+
{
|
|
109
|
+
"id": "p04",
|
|
110
|
+
"task": "从配置字典 cfg 里读一个可选字段的值,怎么写?",
|
|
111
|
+
"query": "配置 字典 可选字段 缺失 None get 默认值",
|
|
112
|
+
"lesson": "本项目配置规范:可选字段必须用 'key' in cfg 判断后再取值,禁止 .get()。"
|
|
113
|
+
"原因:.get() 把「显式 None」和「未配置」混为一谈,"
|
|
114
|
+
"历史事故是默认值覆盖了用户显式设置的空值。",
|
|
115
|
+
"trap": "用 cfg.get('key', default) 取值",
|
|
116
|
+
"fix": "用 'key' in cfg 判断后再取值",
|
|
117
|
+
"decoy": "用 cfg['key'] 直接索引",
|
|
118
|
+
},
|
|
119
|
+
{
|
|
120
|
+
"id": "p05",
|
|
121
|
+
"task": "线上要记录一条警告信息(比如缓存 miss),用哪个日志级别?",
|
|
122
|
+
"query": "日志 级别 警告 缓存 miss warning info 告警",
|
|
123
|
+
"lesson": "本项目日志规范:warning 及以上级别会触发 PagerDuty 告警;"
|
|
124
|
+
"非故障类的警告必须用 info 级别并在消息里加 [WARN] 前缀,禁止直接用 warning。"
|
|
125
|
+
"原因:历史事故是缓存 miss 的 warning 半夜触发全组告警。",
|
|
126
|
+
"trap": "用 logger.warning(...) 记录",
|
|
127
|
+
"fix": "用 logger.info(...) 并在消息里加 [WARN] 前缀",
|
|
128
|
+
"decoy": "用 logger.error(...) 记录",
|
|
129
|
+
},
|
|
130
|
+
]
|
|
131
|
+
|
|
132
|
+
for _c in CASES:
|
|
133
|
+
_c["marker"] = _c["lesson"][:12]
|
|
134
|
+
|
|
135
|
+
|
|
136
|
+
# ---- LLM 客户端(OpenAI 兼容,零第三方依赖)-------------------------------
|
|
137
|
+
# 生效条件:给定 model/base/key/messages 即构造 JSON POST 到 base.rstrip("/")+"/chat/completions",返回 (choices[0].message 的 content 或缺失/假值回落 ""、data.get("usage") 或缺失/假值回落 {}、耗时 dt),timeout/max_tokens/temperature 仅作为请求参数传入。
|
|
138
|
+
def llm_chat(model, base, key, messages, timeout=180, max_tokens=200,
|
|
139
|
+
temperature=0.0):
|
|
140
|
+
payload = json.dumps({
|
|
141
|
+
"model": model,
|
|
142
|
+
"messages": messages,
|
|
143
|
+
"temperature": temperature,
|
|
144
|
+
"max_tokens": max_tokens,
|
|
145
|
+
}).encode("utf-8")
|
|
146
|
+
req = urllib.request.Request(
|
|
147
|
+
base.rstrip("/") + "/chat/completions", data=payload,
|
|
148
|
+
headers={"Content-Type": "application/json",
|
|
149
|
+
"Authorization": f"Bearer {key}"})
|
|
150
|
+
t0 = time.time()
|
|
151
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
152
|
+
data = json.loads(resp.read().decode("utf-8"))
|
|
153
|
+
dt = time.time() - t0
|
|
154
|
+
content = data["choices"][0]["message"].get("content") or ""
|
|
155
|
+
return content, (data.get("usage") or {}), dt
|
|
156
|
+
|
|
157
|
+
|
|
158
|
+
# 生效条件:raw(None/空串视为 "")中首个 "{" 至末个 "}" 的子串能解析出 pick,或全文匹配到单个数字 1-9,且该编号落在 1..n 内时返回该编号(pick 为数字字符串先转 int),否则返回 None。
|
|
159
|
+
def _parse_pick(raw, n):
|
|
160
|
+
"""从模型输出里抠出候选编号(1..n);解析失败返回 None。"""
|
|
161
|
+
s = raw or ""
|
|
162
|
+
i, j = s.find("{"), s.rfind("}")
|
|
163
|
+
if i >= 0 and j > i:
|
|
164
|
+
try:
|
|
165
|
+
obj = json.loads(s[i:j + 1])
|
|
166
|
+
p = obj.get("pick")
|
|
167
|
+
if isinstance(p, str) and p.strip().isdigit():
|
|
168
|
+
p = int(p.strip())
|
|
169
|
+
if isinstance(p, (int, float)) and 1 <= int(p) <= n:
|
|
170
|
+
return int(p)
|
|
171
|
+
except ValueError:
|
|
172
|
+
pass
|
|
173
|
+
m = re.search(r"\b([1-9])\b", s)
|
|
174
|
+
if m and 1 <= int(m.group(1)) <= n:
|
|
175
|
+
return int(m.group(1))
|
|
176
|
+
return None
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
# 生效条件:以 case["id"] 播种打乱 [(fix,case["fix"]),(trap,case["trap"]),(decoy,case["decoy"])] 后取三种循环排列,n<=3 返回其前 n 个,n>3 返回 6 种全排列的前 n 个。
|
|
180
|
+
def _orders(case, n):
|
|
181
|
+
"""候选顺序。默认取 3 种循环排列(每个候选在 3 个位置各出现一次,天然去位置偏差);
|
|
182
|
+
n>3 时退回全 6 种排列。基准顺序由 case id 播种打乱,避免跨用例雷同。"""
|
|
183
|
+
base = [("fix", case["fix"]), ("trap", case["trap"]),
|
|
184
|
+
("decoy", case["decoy"])]
|
|
185
|
+
random.Random(case["id"]).shuffle(base)
|
|
186
|
+
rots = [tuple(base[i:] + base[:i]) for i in range(3)]
|
|
187
|
+
if n <= 3:
|
|
188
|
+
return rots[:n]
|
|
189
|
+
return list(itertools.permutations(base))[:n]
|
|
190
|
+
|
|
191
|
+
|
|
192
|
+
# 生效条件:case 含 task 且 opts 为 (kind, txt) 序列时,返回含场景、编号候选与选择指令的提示文本,memory 非空时在开头插入记忆段。
|
|
193
|
+
def _user_prompt(case, opts, memory=None):
|
|
194
|
+
lines = []
|
|
195
|
+
if memory:
|
|
196
|
+
lines.append("【本项目规范 / 历史经验(来自记忆库,请自行判断相关性)】")
|
|
197
|
+
lines.append(memory.strip())
|
|
198
|
+
lines.append("")
|
|
199
|
+
lines.append("场景:")
|
|
200
|
+
lines.append(case["task"])
|
|
201
|
+
lines.append("")
|
|
202
|
+
lines.append("候选做法:")
|
|
203
|
+
for idx, (_kind, txt) in enumerate(opts, 1):
|
|
204
|
+
lines.append(f"{idx}. {txt}")
|
|
205
|
+
lines.append("")
|
|
206
|
+
lines.append("请选择最符合本项目规范的做法。")
|
|
207
|
+
return "\n".join(lines)
|
|
208
|
+
|
|
209
|
+
|
|
210
|
+
# 生效条件:cg 存在时把模块常量 CASES 逐条以 c["task"] 为 error、c["lesson"] 为 fix 组成列表传给 cg.mine_fix_pairs 并原样返回其结果。
|
|
211
|
+
def _build_memory(cg):
|
|
212
|
+
"""Phase A:把 5 条「项目规范」写进记忆(真实 mine_fix_pairs)。"""
|
|
213
|
+
return cg.mine_fix_pairs(
|
|
214
|
+
[{"error": c["task"], "fix": c["lesson"]} for c in CASES])
|
|
215
|
+
|
|
216
|
+
|
|
217
|
+
# 生效条件:以 "本项目规范:"+case["query"] 且 budget_tokens=budget、k=10 调用 cg.recall,取 pack 各项 content(缺失/假值为 "")拼接,返回其前 1500 字符与 case["marker"] 是否出现在未截断拼接文本中。
|
|
218
|
+
def _recall_memory(cg, case, budget):
|
|
219
|
+
res = cg.recall("本项目规范:" + case["query"],
|
|
220
|
+
budget_tokens=budget, k=10)
|
|
221
|
+
text = " ".join((p.get("content") or "") for p in res.get("pack") or [])
|
|
222
|
+
return text[:1500], (case["marker"] in text)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
# 生效条件:在 cfg["max_turns"] 轮内以 cfg["model"]/cfg["base"]/cfg["key"] 调 llm_chat(timeout=cfg["timeout"]、max_tokens=cfg["max_tokens"])并累加 usage["total_tokens"],pick 等于 correct_pos 或为 None 时中止(否则在仍有剩余轮次时追加 assistant/user 消息重选),最终 pick 为 None 则 kind="invalid"、否则 kind=opts[pick-1][0]。
|
|
226
|
+
def _run_one(cfg, case, opts, correct_pos, memory):
|
|
227
|
+
messages = [{"role": "system", "content": SYS_PROMPT},
|
|
228
|
+
{"role": "user", "content": _user_prompt(case, opts, memory)}]
|
|
229
|
+
tokens, turns, pick, raw, dt = 0, 0, None, "", 0.0
|
|
230
|
+
while turns < cfg["max_turns"]:
|
|
231
|
+
turns += 1
|
|
232
|
+
raw, usage, dt = llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
233
|
+
messages, timeout=cfg["timeout"],
|
|
234
|
+
max_tokens=cfg["max_tokens"])
|
|
235
|
+
tokens += int(usage.get("total_tokens") or 0)
|
|
236
|
+
pick = _parse_pick(raw, len(opts))
|
|
237
|
+
if pick == correct_pos or pick is None:
|
|
238
|
+
break
|
|
239
|
+
if turns < cfg["max_turns"]:
|
|
240
|
+
messages.append({"role": "assistant", "content": raw})
|
|
241
|
+
messages.append({"role": "user",
|
|
242
|
+
"content": f"第 {pick} 个做法执行后问题依旧。"
|
|
243
|
+
f"请在剩余候选中重新选择,仍然只输出 JSON。"})
|
|
244
|
+
kind = "invalid" if pick is None else opts[pick - 1][0]
|
|
245
|
+
return {"pick": pick, "kind": kind, "correct": pick == correct_pos,
|
|
246
|
+
"turns": turns, "tokens": tokens, "latency": dt}
|
|
247
|
+
|
|
248
|
+
|
|
249
|
+
# 生效条件:cfg["use_mem"] 为真时对每个 case 以 cfg["budget"] 先串行召回写入 mems、否则存 (None, False),再对每 case × _orders(case, n_perms) 的排列 × ("none","mem") 两臂建任务交 ThreadPoolExecutor(max_workers=workers) 执行,返回按 arm 分组的 {"none": [...], "mem": [...]} 行表。
|
|
250
|
+
def _run_model(label, cfg, cases, cg, n_perms, workers):
|
|
251
|
+
# 记忆召回先串行算好(MdCGOS 非线程安全),LLM 调用再并发。
|
|
252
|
+
mems = {}
|
|
253
|
+
for case in cases:
|
|
254
|
+
mems[case["id"]] = (_recall_memory(cg, case, cfg["budget"])
|
|
255
|
+
if cfg["use_mem"] else (None, False))
|
|
256
|
+
|
|
257
|
+
tasks = []
|
|
258
|
+
for case in cases:
|
|
259
|
+
for oi, opts in enumerate(_orders(case, n_perms)):
|
|
260
|
+
correct_pos = next(i for i, (k, _t) in enumerate(opts, 1)
|
|
261
|
+
if k == "fix")
|
|
262
|
+
for arm in ("none", "mem"):
|
|
263
|
+
tasks.append((case, oi, opts, correct_pos, arm))
|
|
264
|
+
|
|
265
|
+
# 生效条件:t 解包为 (case, oi, opts, correct_pos, arm),arm=="mem" 时 memory/hit 取闭包 mems[case["id"]]、否则为 (None, False),_run_one 抛异常时以 kind="error"、pick=None 的占位行替代,随后补上 case/arm/hit 并打印该行后返回 r。
|
|
266
|
+
def work(t):
|
|
267
|
+
case, oi, opts, correct_pos, arm = t
|
|
268
|
+
memory, hit = mems[case["id"]] if arm == "mem" else (None, False)
|
|
269
|
+
try:
|
|
270
|
+
r = _run_one(cfg, case, opts, correct_pos, memory)
|
|
271
|
+
except Exception as exc: # noqa: BLE001
|
|
272
|
+
r = {"pick": None, "kind": "error", "correct": False,
|
|
273
|
+
"turns": 0, "tokens": 0, "latency": 0.0,
|
|
274
|
+
"raw": f"{type(exc).__name__}: {exc}"}
|
|
275
|
+
r.update({"case": case["id"], "arm": arm, "hit": hit})
|
|
276
|
+
print(f" [{label}] {case['id']} {arm:<4} rot={oi} "
|
|
277
|
+
f"-> pick={r['pick']} {r['kind']:<6} tok={r['tokens']:>4} "
|
|
278
|
+
f"({r['latency']:.1f}s)", flush=True)
|
|
279
|
+
return r
|
|
280
|
+
|
|
281
|
+
rows = {"none": [], "mem": []}
|
|
282
|
+
with ThreadPoolExecutor(max_workers=workers) as ex:
|
|
283
|
+
for r in ex.map(work, tasks):
|
|
284
|
+
rows[r["arm"]].append(r)
|
|
285
|
+
return rows
|
|
286
|
+
|
|
287
|
+
|
|
288
|
+
# 生效条件:n 为真值时返回 f"{100.0*x/n:.1f}%",n 为假值(0)时返回 "-"。
|
|
289
|
+
def _pct(x, n):
|
|
290
|
+
return f"{100.0 * x / n:.1f}%" if n else "-"
|
|
291
|
+
|
|
292
|
+
|
|
293
|
+
# 生效条件:以 rows["none"] 的行数 n 汇总 none/mem 两臂的 correct、kind=="trap"、kind∈("invalid","error")、tokens、latency、hit 与 pick∈(1,2,3) 的分布并打印(n_perms 仅用于表头文字),返回该 agg 字典。
|
|
294
|
+
def _report(label, rows, n_perms):
|
|
295
|
+
n = len(rows["none"])
|
|
296
|
+
print(f"\n{'-' * 78}")
|
|
297
|
+
print(f"模型:{label} 样本 {n} 次(用例 × 排列)· 每例 {n_perms} 种排列")
|
|
298
|
+
print(f"{'指标':<18}{'arm_none':>14}{'arm_mem':>14}{'差值':>16}")
|
|
299
|
+
print("-" * 78)
|
|
300
|
+
agg = {}
|
|
301
|
+
for arm in ("none", "mem"):
|
|
302
|
+
rs = rows[arm]
|
|
303
|
+
agg[arm] = {
|
|
304
|
+
"acc": sum(1 for r in rs if r["correct"]),
|
|
305
|
+
"trap": sum(1 for r in rs if r["kind"] == "trap"),
|
|
306
|
+
"invalid": sum(1 for r in rs if r["kind"] in ("invalid", "error")),
|
|
307
|
+
"tokens": sum(r["tokens"] for r in rs),
|
|
308
|
+
"latency": sum(r["latency"] for r in rs),
|
|
309
|
+
"hit": sum(1 for r in rs if r["hit"]),
|
|
310
|
+
"pos": [sum(1 for r in rs if r["pick"] == p) for p in (1, 2, 3)],
|
|
311
|
+
}
|
|
312
|
+
a, b = agg["none"], agg["mem"]
|
|
313
|
+
d_acc = f"+{100.0 * (b['acc'] - a['acc']) / n:.1f}pp"
|
|
314
|
+
d_trap = f"{100.0 * (b['trap'] - a['trap']) / n:.1f}pp"
|
|
315
|
+
d_inv = f"{100.0 * (b['invalid'] - a['invalid']) / n:.1f}pp"
|
|
316
|
+
d_tok = f"{(b['tokens'] - a['tokens']) / n:+.1f}"
|
|
317
|
+
d_lat = f"{(b['latency'] - a['latency']) / n:+.2f}"
|
|
318
|
+
print(f"{'content_acc':<18}{_pct(a['acc'], n):>14}{_pct(b['acc'], n):>14}"
|
|
319
|
+
f"{d_acc:>16}")
|
|
320
|
+
print(f"{'trap_rate':<18}{_pct(a['trap'], n):>14}{_pct(b['trap'], n):>14}"
|
|
321
|
+
f"{d_trap:>16}")
|
|
322
|
+
print(f"{'invalid_rate':<18}{_pct(a['invalid'], n):>14}"
|
|
323
|
+
f"{_pct(b['invalid'], n):>14}{d_inv:>16}")
|
|
324
|
+
print(f"{'avg_tokens':<18}{a['tokens'] / n:>14.1f}{b['tokens'] / n:>14.1f}"
|
|
325
|
+
f"{d_tok:>16}")
|
|
326
|
+
print(f"{'avg_latency_s':<18}{a['latency'] / n:>14.2f}"
|
|
327
|
+
f"{b['latency'] / n:>14.2f}{d_lat:>16}")
|
|
328
|
+
print(f"{'recall_hit_rate':<18}{'-':>14}{_pct(b['hit'], n):>14}{'-':>16}")
|
|
329
|
+
print(f"{'pos_dist(1/2/3)':<18}"
|
|
330
|
+
f"{'/'.join(str(x) for x in a['pos']):>14}"
|
|
331
|
+
f"{'/'.join(str(x) for x in b['pos']):>14}{'—':>16}")
|
|
332
|
+
|
|
333
|
+
print("\n逐用例 content_acc(按内容选对次数 / 该例排列数):")
|
|
334
|
+
for case in CASES:
|
|
335
|
+
if not any(r["case"] == case["id"] for r in rows["none"]):
|
|
336
|
+
continue
|
|
337
|
+
an = [r for r in rows["none"] if r["case"] == case["id"]]
|
|
338
|
+
am = [r for r in rows["mem"] if r["case"] == case["id"]]
|
|
339
|
+
na = sum(1 for r in an if r["correct"])
|
|
340
|
+
ma = sum(1 for r in am if r["correct"])
|
|
341
|
+
hits = sum(1 for r in am if r["hit"])
|
|
342
|
+
print(f" {case['id']} none {na}/{len(an)} mem {ma}/{len(am)} "
|
|
343
|
+
f"hit {hits}/{len(am)} {case['task'][:38]}")
|
|
344
|
+
return agg
|
|
345
|
+
|
|
346
|
+
|
|
347
|
+
# 生效条件:argv 为 None 时改读 sys.argv;若 --key 与 DEEPSEEK_API_KEY 均为空则返回 2,否则按 --only/--cases 从 CASES 取用例跑完双臂后返回 0。
|
|
348
|
+
def main(argv=None):
|
|
349
|
+
ap = argparse.ArgumentParser(
|
|
350
|
+
description="先验陷阱 A/B:无记忆 vs 有记忆(真实 LLM)")
|
|
351
|
+
ap.add_argument("--model", default="deepseek-v4.1-flash-expires-on-0910")
|
|
352
|
+
ap.add_argument("--base", default="https://api.deepseek.com/v1")
|
|
353
|
+
ap.add_argument("--key", default="")
|
|
354
|
+
ap.add_argument("--local", action="store_true",
|
|
355
|
+
help="追加本地小模型做对照")
|
|
356
|
+
ap.add_argument("--local-model", default="smegmma-deluxe-9b-v1")
|
|
357
|
+
ap.add_argument("--local-base", default="http://localhost:1234/v1")
|
|
358
|
+
ap.add_argument("--cases", type=int, default=0, help="只用前 N 个用例")
|
|
359
|
+
ap.add_argument("--only", default="", help="只用指定 id,逗号分隔,如 p01,p02")
|
|
360
|
+
ap.add_argument("--perms", type=int, default=3, help="每例候选排列数(默认 3)")
|
|
361
|
+
ap.add_argument("--budget", type=int, default=3000, help="召回 token 预算")
|
|
362
|
+
ap.add_argument("--max-tokens", type=int, default=2000)
|
|
363
|
+
ap.add_argument("--timeout", type=int, default=180)
|
|
364
|
+
ap.add_argument("--workers", type=int, default=4, help="并发调用数")
|
|
365
|
+
ap.add_argument("--max-turns", type=int, default=1,
|
|
366
|
+
help="最大轮数(1=单发;>1 则在选错后反馈重选)")
|
|
367
|
+
a = ap.parse_args(argv)
|
|
368
|
+
|
|
369
|
+
cases = CASES
|
|
370
|
+
if a.only:
|
|
371
|
+
want = {x.strip() for x in a.only.split(",") if x.strip()}
|
|
372
|
+
cases = [c for c in cases if c["id"] in want]
|
|
373
|
+
if a.cases:
|
|
374
|
+
cases = cases[:a.cases]
|
|
375
|
+
n_perms = max(1, min(6, a.perms))
|
|
376
|
+
|
|
377
|
+
key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
|
|
378
|
+
if not key:
|
|
379
|
+
print("[error] 需要 DEEPSEEK_API_KEY 或 --key")
|
|
380
|
+
return 2
|
|
381
|
+
|
|
382
|
+
root = tempfile.mkdtemp(prefix="mdcg_ab_prior_")
|
|
383
|
+
try:
|
|
384
|
+
cg = MdCGOS(os.path.join(root, "mem"))
|
|
385
|
+
mined = _build_memory(cg)
|
|
386
|
+
print(f"Phase A 写入记忆:{len(mined['pairs'])} 组"
|
|
387
|
+
f"(知识 {len(mined['knowledge_ids'])} + 负记忆 {len(mined['rejected_ids'])})")
|
|
388
|
+
print(f"用例 {len(cases)} × 排列 {n_perms} × 2 臂 = "
|
|
389
|
+
f"{len(cases) * n_perms * 2} 次调用")
|
|
390
|
+
|
|
391
|
+
targets = [(a.model, a.model, a.base, key)]
|
|
392
|
+
if a.local:
|
|
393
|
+
targets.append(("local:" + a.local_model, a.local_model,
|
|
394
|
+
a.local_base, "lm-studio"))
|
|
395
|
+
|
|
396
|
+
for label, model, base, mkey in targets:
|
|
397
|
+
cfg = {"model": model, "base": base, "key": mkey,
|
|
398
|
+
"budget": a.budget, "timeout": a.timeout,
|
|
399
|
+
"max_tokens": a.max_tokens, "max_turns": a.max_turns,
|
|
400
|
+
"use_mem": True}
|
|
401
|
+
rows = _run_model(label, cfg, cases, cg, n_perms, a.workers)
|
|
402
|
+
_report(label, rows, n_perms)
|
|
403
|
+
finally:
|
|
404
|
+
shutil.rmtree(root, ignore_errors=True)
|
|
405
|
+
return 0
|
|
406
|
+
|
|
407
|
+
|
|
408
|
+
if __name__ == "__main__":
|
|
409
409
|
sys.exit(main())
|