@furongjun1999/dsh-memory 0.4.11 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +143 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +519 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/hooks.js +36 -2
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +116 -29
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +379 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1328 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +301 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1006 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +315 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1537 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1098 -1097
- package/md_cg/crypto.py +3 -1
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +78 -18
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +4 -2
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +222 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +377 -329
- package/md_cg/hotcache.py +48 -7
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +338 -0
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +140 -107
- package/md_cg/mcp_server.py +403 -55
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +783 -233
- package/md_cg/mdcos.py +500 -70
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +694 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +300 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +143 -0
- package/md_cg/reconcile.py +228 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/review_cli.py +170 -0
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +393 -365
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +13 -3
- package/md_cg/security.py +128 -18
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +152 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +7 -4
- package/md_cg/sources.py +816 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +54 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +35 -5
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +13 -3
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +22 -2
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +96 -15
- package/md_cg/test_index_durability.py +17 -3
- package/md_cg/test_interop.py +95 -0
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +16 -7
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +7 -1
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +153 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +316 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +203 -0
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +344 -340
- package/md_cg/test_retr_s1b.py +276 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +392 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +9 -3
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +16 -2
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +38 -20
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +6 -3
- package/md_cg/tokens.py +85 -14
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +668 -667
- package/md_cg/vision_evidence.py +667 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +20 -8
- package/package.json +101 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +33 -0
- package/src/hooks.ts +38 -2
- package/src/index.ts +526 -518
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +116 -29
- package/src/lib/token_store.ts +13 -3
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""LoCoMo QA 同口径对照评测:完整对话做记忆 · 自然问句做查询(2026-09-23)
|
|
3
|
+
|
|
4
|
+
目的(使用者口径):与 Mem0/Letta/Zep 等同设定直接比数字——**对话做记忆、自然问句做
|
|
5
|
+
查询、LLM reader + LLM judge**,不再以「摘要卡/关键词串」口径差异回避对照。
|
|
6
|
+
|
|
7
|
+
与同行的对齐面:
|
|
8
|
+
· 记忆面:LoCoMo 全量 5882 turn 完整对话(mteb/LoCoMo BEIR corpus,id 与仓内
|
|
9
|
+
evidence_turns 逐字对齐已验证)→ LLM 译为中文(人名保留)→ 逐 turn 写入灵枢
|
|
10
|
+
(对话原文,不做摘要加工——写入侧加工差异正是各记忆系统的被测对象本身)
|
|
11
|
+
· 查询面:上游英文自然问句 → LLM 译为中文自然问句
|
|
12
|
+
· 答题面:reader 只据注入记忆作答(不足答「无法确定」)→ judge 对上游 gold
|
|
13
|
+
语义判等(correct/incorrect/refused;adversarial gold 空=正确行为拒答)
|
|
14
|
+
· 两臂:
|
|
15
|
+
retrieval 灵枢检索注入(cg.search 阶梯路 top-10,与生产主形态同路)
|
|
16
|
+
full_context 整段对话全量注入(对标 Mem0 论文 Table 2 的 Full-context 72.90)
|
|
17
|
+
|
|
18
|
+
标尺(Mem0 论文 arXiv:2504.19413 Table 2,agent=GPT-4o-mini):
|
|
19
|
+
Full-context 72.90 · Mem0ᵍ 68.44 · Mem0 66.88 · Zep 65.99 · LangMem 58.10 ·
|
|
20
|
+
OpenAI memory 52.90 · A-Mem 48.38;Letta Filesystem 复测 74.0(letta.com 2025-08)。
|
|
21
|
+
本评测 reader/judge=deepseek-chat(与同行 GPT-4o-mini 不同档,绝对分含模型差,
|
|
22
|
+
对照时声明;同标尺内两臂之差是检索质量的净效应,与模型无关)。
|
|
23
|
+
|
|
24
|
+
跑法:
|
|
25
|
+
python -X utf8 -m md_cg.bench_e2e_locomo_qa --translate # 翻译 5882 turn + 500 问句
|
|
26
|
+
python -X utf8 -m md_cg.bench_e2e_locomo_qa # 全量 500 题 × 2 臂
|
|
27
|
+
python -X utf8 -m md_cg.bench_e2e_locomo_qa --quick # 冒烟 3 题(须先完成翻译)
|
|
28
|
+
环境:DEEPSEEK_API_KEY;建议 MDCG_UNIFY_QUERY=0(中文问句含英文名时批次15归一有
|
|
29
|
+
形态缺陷,见 docs/eval/端到端干扰池评测_v1.1 §4)。
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import argparse
|
|
34
|
+
import hashlib
|
|
35
|
+
import json
|
|
36
|
+
import os
|
|
37
|
+
import re
|
|
38
|
+
import sys
|
|
39
|
+
import time
|
|
40
|
+
import urllib.request
|
|
41
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
42
|
+
|
|
43
|
+
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
44
|
+
if HERE not in sys.path:
|
|
45
|
+
sys.path.insert(0, HERE)
|
|
46
|
+
|
|
47
|
+
from md_cg.bench_e2e_judge import DEFAULT_ROOT, llm_chat # noqa: E402
|
|
48
|
+
from md_cg.bench_e2e_qa import ( # noqa: E402
|
|
49
|
+
JUDGE_SYS, READER_SYS, _cache_call, parse_json_field)
|
|
50
|
+
|
|
51
|
+
UP = os.path.join(DEFAULT_ROOT, "upstream")
|
|
52
|
+
CORPUS_PQ = os.path.join(UP, "corpus.parquet")
|
|
53
|
+
ANSWERS = os.path.join(UP, "answers_map.json")
|
|
54
|
+
ZH_TURNS = os.path.join(UP, "zh_turns.json")
|
|
55
|
+
ZH_QUERIES = os.path.join(UP, "zh_queries.json")
|
|
56
|
+
T_CACHE = os.path.join(UP, "translate_cache")
|
|
57
|
+
|
|
58
|
+
TRANS_TURN_SYS = (
|
|
59
|
+
"你是对话翻译器。把英文对话 turn 逐条译成中文:忠实原义、不压缩不增删;"
|
|
60
|
+
"人名与专有名词保留英文原文(如 Caroline / Ed Sheeran);说话人前缀保留"
|
|
61
|
+
"(格式「名字: 译文」)。\n只输出一行 JSON:{\"x1\": \"名字: 译文\", ...}"
|
|
62
|
+
)
|
|
63
|
+
TRANS_Q_SYS = (
|
|
64
|
+
"你是问题翻译器。把英文问句译成自然的中文问句:语气自然、不逐词直译;"
|
|
65
|
+
"人名与专有名词保留英文原文。\n只输出一行 JSON:{\"x1\": \"中文问句\", ...}"
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def load_corpus():
|
|
70
|
+
import pyarrow.parquet as pq
|
|
71
|
+
t = pq.read_table(CORPUS_PQ)
|
|
72
|
+
ids = t.column("id").to_pylist()
|
|
73
|
+
txts = t.column("text").to_pylist()
|
|
74
|
+
return ids, txts
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _batch(lst, n):
|
|
78
|
+
for i in range(0, len(lst), n):
|
|
79
|
+
yield lst[i:i + n]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def translate_all(key, model_cfg, items, batch_n, sys_prompt, out_path,
|
|
83
|
+
max_tokens=4000):
|
|
84
|
+
"""items=[(k, en_text)] → {k: zh},分片缓存、断点续跑。"""
|
|
85
|
+
os.makedirs(T_CACHE, exist_ok=True)
|
|
86
|
+
out = {}
|
|
87
|
+
if os.path.isfile(out_path):
|
|
88
|
+
out = json.load(open(out_path, encoding="utf-8"))
|
|
89
|
+
todo = [(k, t) for k, t in items if k not in out]
|
|
90
|
+
print(f"[translate:{key}] 已有 {len(out)} / 待译 {len(todo)}"
|
|
91
|
+
f" · engine={model_cfg['model']}@{model_cfg['base']}")
|
|
92
|
+
jobs = list(enumerate(_batch(todo, batch_n)))
|
|
93
|
+
|
|
94
|
+
def call_one(items, mt):
|
|
95
|
+
# 行前缀从 x1 起(与系统提示示例 {"x1": …} 一致):模型对 x0 起头
|
|
96
|
+
# 的批量会整体偏移成 x1 起,导致键错位解析失败(214 条顽固缺口的根因)
|
|
97
|
+
mapping = {f"x{i + 1}": k for i, (k, _t) in enumerate(items)}
|
|
98
|
+
user = "\n".join(f"x{i + 1}. {t}" for i, (_k, t) in enumerate(items))
|
|
99
|
+
try:
|
|
100
|
+
raw, _u, _d = llm_chat(
|
|
101
|
+
model_cfg["model"], model_cfg["base"], model_cfg["key"],
|
|
102
|
+
[{"role": "system", "content": sys_prompt},
|
|
103
|
+
{"role": "user", "content": user}],
|
|
104
|
+
timeout=model_cfg["timeout"], max_tokens=mt,
|
|
105
|
+
extra_payload=model_cfg.get("extra"))
|
|
106
|
+
except Exception: # noqa: BLE001
|
|
107
|
+
return {}
|
|
108
|
+
s = re.sub(r"<think>.*?</think>", "", str(raw), flags=re.S)
|
|
109
|
+
obj = {}
|
|
110
|
+
_i, _j = s.find("{"), s.rfind("}")
|
|
111
|
+
if _i >= 0 and _j > _i:
|
|
112
|
+
try:
|
|
113
|
+
obj = json.loads(s[_i:_j + 1])
|
|
114
|
+
except ValueError:
|
|
115
|
+
obj = {}
|
|
116
|
+
res = {}
|
|
117
|
+
for tag, zh in obj.items():
|
|
118
|
+
if tag in mapping and isinstance(zh, str) and zh.strip():
|
|
119
|
+
res[mapping[tag]] = zh.strip()
|
|
120
|
+
return res
|
|
121
|
+
|
|
122
|
+
def work(item):
|
|
123
|
+
idx, job = item
|
|
124
|
+
# 缓存按**内容寻址**(分片键集 hash):早前按 todo 重排 idx 命名,
|
|
125
|
+
# 续传轮 idx 语义漂移会命中**旧内容的分片**(返回已在 out 的旧键,
|
|
126
|
+
# n_ok 虚涨而总数停滞——180 条顽固缺口的第二重根因)。内容寻址
|
|
127
|
+
# 跨轮稳定,同内容分片天然命中、不同内容互不误撞。
|
|
128
|
+
chash = hashlib.md5(",".join(k for k, _ in job)
|
|
129
|
+
.encode("utf-8")).hexdigest()[:12]
|
|
130
|
+
cpath = os.path.join(T_CACHE, f"{key}_{chash}.json")
|
|
131
|
+
if os.path.isfile(cpath):
|
|
132
|
+
return json.load(open(cpath, encoding="utf-8"))
|
|
133
|
+
res = call_one(job, max_tokens)
|
|
134
|
+
if len(res) < len(job):
|
|
135
|
+
# 分片级失败兜底:逐条重译(长 turn 挤爆批量输出导致 JSON 截断
|
|
136
|
+
# 的形态,单条请求给足生成空间即可收敛)
|
|
137
|
+
for one in job:
|
|
138
|
+
if one[0] not in res:
|
|
139
|
+
res.update(call_one([one], 1000))
|
|
140
|
+
if len(res) == len(job):
|
|
141
|
+
json.dump(res, open(cpath, "w", encoding="utf-8"),
|
|
142
|
+
ensure_ascii=False)
|
|
143
|
+
return res
|
|
144
|
+
|
|
145
|
+
n_ok = 0
|
|
146
|
+
with ThreadPoolExecutor(max_workers=model_cfg["workers"]) as ex:
|
|
147
|
+
for res in ex.map(work, jobs):
|
|
148
|
+
out.update(res)
|
|
149
|
+
n_ok += len(res)
|
|
150
|
+
json.dump(out, open(out_path, "w", encoding="utf-8"), ensure_ascii=False)
|
|
151
|
+
print(f"[translate:{key}] 本轮完成 {n_ok},总 {len(out)}/{len(items)}")
|
|
152
|
+
return out
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def build_dialog_pool(ids, zh_turns, root_dir):
|
|
156
|
+
"""5882 中文 turn 逐条写入灵枢(对话原文做记忆,无摘要加工)。"""
|
|
157
|
+
root = os.path.join(root_dir, "pools", "dialog_zh")
|
|
158
|
+
if os.path.isfile(os.path.join(root, "_manifest.json")):
|
|
159
|
+
try:
|
|
160
|
+
m = json.load(open(os.path.join(root, "_manifest.json"),
|
|
161
|
+
encoding="utf-8"))
|
|
162
|
+
if m == {"turns": len(ids)}:
|
|
163
|
+
from md_cg.mdcos import MdCGOS
|
|
164
|
+
return MdCGOS(os.path.join(root, "mem"))
|
|
165
|
+
except Exception:
|
|
166
|
+
pass
|
|
167
|
+
import shutil
|
|
168
|
+
from md_cg.mdcos import MdCGOS
|
|
169
|
+
if os.path.isdir(root):
|
|
170
|
+
shutil.rmtree(root)
|
|
171
|
+
os.makedirs(root, exist_ok=True)
|
|
172
|
+
cg = MdCGOS(os.path.join(root, "mem"))
|
|
173
|
+
for nid in ids:
|
|
174
|
+
cg.add(nid, zh_turns.get(nid) or "", layer="knowledge",
|
|
175
|
+
verification_basis="data")
|
|
176
|
+
json.dump({"turns": len(ids)},
|
|
177
|
+
open(os.path.join(root, "_manifest.json"), "w", encoding="utf-8"))
|
|
178
|
+
return cg
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
READER_USER = "问题:{q}\n\n候选记忆:\n{cards}"
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def run_one(cfg, cache_dir, qid, qtype, zh_q, en_q, gold, cards_text, arm):
|
|
185
|
+
row = {"qid": qid, "qtype": qtype, "arm": arm, "gold": gold}
|
|
186
|
+
|
|
187
|
+
def call_reader():
|
|
188
|
+
return llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
189
|
+
[{"role": "system", "content": READER_SYS},
|
|
190
|
+
{"role": "user", "content": READER_USER.format(
|
|
191
|
+
q=zh_q, cards=cards_text)}],
|
|
192
|
+
timeout=cfg["timeout"], max_tokens=300)
|
|
193
|
+
|
|
194
|
+
def call_judge():
|
|
195
|
+
return llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
196
|
+
[{"role": "system", "content": JUDGE_SYS},
|
|
197
|
+
{"role": "user", "content":
|
|
198
|
+
f"问题:{en_q}\ngold:{gold or ''}\n"
|
|
199
|
+
f"模型回答:{row['answer']}"}],
|
|
200
|
+
timeout=cfg["timeout"], max_tokens=200)
|
|
201
|
+
|
|
202
|
+
try:
|
|
203
|
+
pkey = f"{cfg['model']}|{arm}|{qid}|{cards_text[:200]}"
|
|
204
|
+
raw, _t1, _d1, _c1 = _cache_call(cache_dir, "reader",
|
|
205
|
+
hashlib.md5(
|
|
206
|
+
pkey.encode()).hexdigest(),
|
|
207
|
+
call_reader)
|
|
208
|
+
a = parse_json_field(raw, "answer")
|
|
209
|
+
row["answer"] = a if a is not None else (raw or "").strip()[:80]
|
|
210
|
+
jkey = f"{cfg['model']}|{en_q}|{gold}|{row['answer']}"
|
|
211
|
+
raw2, _t2, _d2, _c2 = _cache_call(
|
|
212
|
+
cache_dir, "judge", hashlib.md5(jkey.encode()).hexdigest(),
|
|
213
|
+
call_judge)
|
|
214
|
+
v = parse_json_field(raw2, "verdict")
|
|
215
|
+
row["verdict"] = v if v in ("correct", "incorrect", "refused") else None
|
|
216
|
+
if row["verdict"] is None:
|
|
217
|
+
row["error"] = f"judge 不可解析 {(raw2 or '')[:60]}"
|
|
218
|
+
except Exception as exc: # noqa: BLE001
|
|
219
|
+
row["error"] = f"{type(exc).__name__}: {exc}"
|
|
220
|
+
row.setdefault("answer", "")
|
|
221
|
+
row["verdict"] = None
|
|
222
|
+
return row
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def report(rows, title):
|
|
226
|
+
ok = [r for r in rows if r.get("verdict")]
|
|
227
|
+
n = max(len(ok), 1)
|
|
228
|
+
acc = sum(1 for r in ok if r["verdict"] == "correct")
|
|
229
|
+
out = {"n": len(rows), "judged": len(ok), "acc": acc,
|
|
230
|
+
"acc_pct": 100.0 * acc / n, "refused": sum(
|
|
231
|
+
1 for r in ok if r["verdict"] == "refused")}
|
|
232
|
+
by = {}
|
|
233
|
+
for qt in ("single_hop", "multi_hop", "temporal_reasoning",
|
|
234
|
+
"adversarial", "open_domain"):
|
|
235
|
+
rs = [r for r in ok if r["qtype"] == qt]
|
|
236
|
+
if rs:
|
|
237
|
+
c = sum(1 for r in rs if r["verdict"] == "correct")
|
|
238
|
+
by[qt] = f"{100.0*c/len(rs):.1f}%({c}/{len(rs)})"
|
|
239
|
+
print(f" {title:<14} QA准确率={out['acc_pct']:5.1f}%"
|
|
240
|
+
f"({acc}/{len(ok)}) 拒答={out['refused']} 分题型: {by}")
|
|
241
|
+
return out
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def main(argv=None):
|
|
245
|
+
ap = argparse.ArgumentParser(description="LoCoMo QA 同口径对照(中文全对话)")
|
|
246
|
+
ap.add_argument("--data-root", default=DEFAULT_ROOT)
|
|
247
|
+
ap.add_argument("--n", type=int, default=0, help="题数(0=全量500)")
|
|
248
|
+
ap.add_argument("--model", default="deepseek-chat")
|
|
249
|
+
ap.add_argument("--base-url", default="https://api.deepseek.com/v1")
|
|
250
|
+
ap.add_argument("--key", default="")
|
|
251
|
+
ap.add_argument("--workers", type=int, default=6)
|
|
252
|
+
ap.add_argument("--timeout", type=int, default=180)
|
|
253
|
+
ap.add_argument("--k", type=int, default=10)
|
|
254
|
+
ap.add_argument("--translate", action="store_true",
|
|
255
|
+
help="只做翻译阶段(5882 turn + 500 问句)")
|
|
256
|
+
ap.add_argument("--lm-studio", action="store_true",
|
|
257
|
+
help="翻译走 LM Studio 本地模型(localhost:1234/v1,"
|
|
258
|
+
"自动探测模型名;reader/judge 仍用 --model)")
|
|
259
|
+
ap.add_argument("--batch-turns", type=int, default=8,
|
|
260
|
+
help="翻译每请求 turn 数(LM Studio 建议 5)")
|
|
261
|
+
ap.add_argument("--batch-q", type=int, default=10,
|
|
262
|
+
help="问句翻译每请求数(LM Studio 建议 6)")
|
|
263
|
+
ap.add_argument("--quick", action="store_true")
|
|
264
|
+
a = ap.parse_args(argv)
|
|
265
|
+
if a.quick:
|
|
266
|
+
a.n = 3
|
|
267
|
+
key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
|
|
268
|
+
if not key:
|
|
269
|
+
print("[error] 需要 DEEPSEEK_API_KEY 或 --key")
|
|
270
|
+
return 2
|
|
271
|
+
cfg = {"model": a.model, "base": a.base_url, "key": key,
|
|
272
|
+
"timeout": a.timeout, "workers": a.workers}
|
|
273
|
+
print(f"[locomo_qa] 归一层开关 MDCG_UNIFY_QUERY="
|
|
274
|
+
f"{os.environ.get('MDCG_UNIFY_QUERY', '1')}(建议 0,见模块头注)")
|
|
275
|
+
|
|
276
|
+
ids, txts = load_corpus()
|
|
277
|
+
answers = json.load(open(ANSWERS, encoding="utf-8"))
|
|
278
|
+
ours = [json.loads(l) for l in open(os.path.join(
|
|
279
|
+
HERE, "data", "benchmarks", "locomo-zh-500", "questions500.jsonl"),
|
|
280
|
+
encoding="utf-8") if l.strip()]
|
|
281
|
+
qs = ours[:a.n] if a.n else ours
|
|
282
|
+
|
|
283
|
+
if a.translate or not (os.path.isfile(ZH_TURNS)
|
|
284
|
+
and os.path.isfile(ZH_QUERIES)):
|
|
285
|
+
tcfg = cfg
|
|
286
|
+
if a.lm_studio:
|
|
287
|
+
with urllib.request.urlopen(
|
|
288
|
+
"http://localhost:1234/v1/models", timeout=8) as r:
|
|
289
|
+
mids = [m["id"] for m in
|
|
290
|
+
json.loads(r.read().decode("utf-8"))["data"]
|
|
291
|
+
if "embed" not in m["id"].lower()]
|
|
292
|
+
if not mids:
|
|
293
|
+
print("[error] LM Studio 无已加载 LLM(仅有 embedding 模型)")
|
|
294
|
+
return 2
|
|
295
|
+
tcfg = {"model": mids[0], "base": "http://localhost:1234/v1",
|
|
296
|
+
"key": "lm-studio", "timeout": 600,
|
|
297
|
+
"workers": min(a.workers, 3),
|
|
298
|
+
# 顶层 reasoning_effort=none 是实测唯一能压住思考的传法
|
|
299
|
+
# (chat_template_kwargs/enable_thinking 均无效,思考仍吃满
|
|
300
|
+
# max_tokens;矩阵实验 2026-09-23:none → 0 reasoning 4s/批)
|
|
301
|
+
"extra": {"reasoning_effort": "none"}}
|
|
302
|
+
print(f"[lm-studio] 使用本地模型 {mids[0]}(并发 {tcfg['workers']})")
|
|
303
|
+
items_t = list(zip(ids, txts))
|
|
304
|
+
zh_turns = translate_all("turn", tcfg, items_t, a.batch_turns,
|
|
305
|
+
TRANS_TURN_SYS, ZH_TURNS,
|
|
306
|
+
max_tokens=2000 if a.lm_studio else 4000)
|
|
307
|
+
items_q = [(q["qid"], answers[q["qid"]]["en_q"]) for q in ours
|
|
308
|
+
if q["qid"] in answers]
|
|
309
|
+
zh_qs = translate_all("q", tcfg, items_q, a.batch_q,
|
|
310
|
+
TRANS_Q_SYS, ZH_QUERIES,
|
|
311
|
+
max_tokens=2000 if a.lm_studio else 4000)
|
|
312
|
+
miss_t = len(ids) - len(zh_turns)
|
|
313
|
+
miss_q = len(items_q) - len(zh_qs)
|
|
314
|
+
print(f"[translate] 缺口 turn={miss_t} q={miss_q}"
|
|
315
|
+
f"(缺口>0 可重跑 --translate 续传)")
|
|
316
|
+
if a.translate:
|
|
317
|
+
return 0 if (miss_t == 0 and miss_q == 0) else 1
|
|
318
|
+
|
|
319
|
+
zh_turns = json.load(open(ZH_TURNS, encoding="utf-8"))
|
|
320
|
+
zh_qs = json.load(open(ZH_QUERIES, encoding="utf-8"))
|
|
321
|
+
cache_dir = os.path.join(a.data_root, "lq_cache")
|
|
322
|
+
os.makedirs(cache_dir, exist_ok=True)
|
|
323
|
+
from md_cg.mdcos import MdCGOS
|
|
324
|
+
cg = build_dialog_pool(ids, zh_turns, a.data_root)
|
|
325
|
+
# scene → 全对话中文(full-context 臂用)
|
|
326
|
+
scene_turns = {}
|
|
327
|
+
for nid in ids:
|
|
328
|
+
scene_turns.setdefault(nid.split("_session")[0], []).append(nid)
|
|
329
|
+
|
|
330
|
+
tasks = []
|
|
331
|
+
for q in qs:
|
|
332
|
+
qid = q["qid"]
|
|
333
|
+
if qid not in answers or qid not in zh_qs:
|
|
334
|
+
continue
|
|
335
|
+
gold = answers[qid]["answer"] or ""
|
|
336
|
+
zh_q = zh_qs[qid]
|
|
337
|
+
en_q = answers[qid]["en_q"]
|
|
338
|
+
# 臂1:灵枢检索注入(生产主形态阶梯路;judge 关——对话原文无 CCG 面)
|
|
339
|
+
res, _m = cg.search(zh_q, k=a.k, judge=False, record=False)
|
|
340
|
+
cards = "\n\n".join(f"【{i}】{zh_turns.get(r[0]['id'], '')}"
|
|
341
|
+
for i, r in enumerate(res, 1))
|
|
342
|
+
tasks.append((qid, q.get("qtype"), zh_q, en_q, gold, cards,
|
|
343
|
+
"retrieval"))
|
|
344
|
+
# 臂2:full-context(该 scene 全对话)
|
|
345
|
+
scene = qid.split("_q_")[0]
|
|
346
|
+
full = "\n".join(zh_turns.get(t, "") for t in scene_turns[scene])
|
|
347
|
+
tasks.append((qid, q.get("qtype"), zh_q, en_q, gold, full[:110000],
|
|
348
|
+
"full_context"))
|
|
349
|
+
t0 = time.time()
|
|
350
|
+
with ThreadPoolExecutor(max_workers=a.workers) as ex:
|
|
351
|
+
rows = list(ex.map(lambda t: run_one(cfg, cache_dir, *t), tasks))
|
|
352
|
+
print(f"[locomo_qa] {len(tasks)} 次 reader+judge 耗时 {time.time()-t0:.0f}s")
|
|
353
|
+
errs = sum(1 for r in rows if r.get("error"))
|
|
354
|
+
print(f"判分失败 {errs} 行\n")
|
|
355
|
+
summary = {"meta": {"model": a.model, "n_q": len(qs), "k": a.k,
|
|
356
|
+
"unify": os.environ.get("MDCG_UNIFY_QUERY", "1")}}
|
|
357
|
+
for arm in ("retrieval", "full_context"):
|
|
358
|
+
summary[arm] = report([r for r in rows if r["arm"] == arm], arm)
|
|
359
|
+
out = os.path.join(a.data_root, "results",
|
|
360
|
+
f"locomo_qa_{time.strftime('%Y%m%d_%H%M%S')}.json")
|
|
361
|
+
json.dump({"summary": summary, "rows": rows},
|
|
362
|
+
open(out, "w", encoding="utf-8"), ensure_ascii=False, indent=1)
|
|
363
|
+
print(f"\n[locomo_qa] 明细 → {out}")
|
|
364
|
+
return 0
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
if __name__ == "__main__":
|
|
368
|
+
sys.exit(main())
|
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""端到端 QA 评测:pinpoint / answerability / decision-making(2026-09-23 · 外部建议)
|
|
3
|
+
|
|
4
|
+
评测问题(外部建议口径:hit@k 是上界,端到端要答对题):
|
|
5
|
+
pinpoint 干扰浓度升高(level 0 → 4),reader 从候选记忆中抽取正确事实的
|
|
6
|
+
QA 准确率退化多少?
|
|
7
|
+
answerability adversarial 题(上游设计 gold 为空、正确行为=拒答):reader 会不会
|
|
8
|
+
被干扰节点骗去编答案?
|
|
9
|
+
decision-making 裁决层是否改变下游行为——同一 reader,候选分别来自
|
|
10
|
+
arm_base(裸融合 top10)与 arm_firewall(证据防火墙重排 top10),
|
|
11
|
+
QA 准确率有无差异?
|
|
12
|
+
|
|
13
|
+
管线(全部真实 LLM,deepseek-chat,temperature=0,带缓存):
|
|
14
|
+
检索(与 bench_e2e_judge 同池同采样同臂)→ reader 只据候选卡作答
|
|
15
|
+
(不足则答「无法确定」)→ judge 对 gold 语义判等(correct/incorrect/refused)。
|
|
16
|
+
gold 答案来自上游 LoCoMo 原始标注(mteb/LoCoMo 1976 题英文题面 ↔
|
|
17
|
+
snap-research/locomo10.json 逐题文本精确匹配,500/500 覆盖;
|
|
18
|
+
113 题 adversarial 的 gold 为空 = 拒答语义)。
|
|
19
|
+
|
|
20
|
+
诚实边界:
|
|
21
|
+
· reader 同时拿中文关键词串与上游英文原题(真实用户问题面),记忆卡为中文,
|
|
22
|
+
judge 对英文 gold 做跨语语义判等——这是双语口径,非纯单语评测。
|
|
23
|
+
· judge 亦为 LLM(LLM-as-a-judge),其判分自身有方差;refused 由 judge 判
|
|
24
|
+
reader 文本,边界例(半猜测半拒答)按 judge 裁决计入,不人工复核(v1)。
|
|
25
|
+
· level 2 不入 QA 层(0 vs 4 已给退化区间,节省调用)。
|
|
26
|
+
|
|
27
|
+
跑法:
|
|
28
|
+
python -X utf8 -m md_cg.bench_e2e_qa --quick # 冒烟 3 题
|
|
29
|
+
python -X utf8 -m md_cg.bench_e2e_qa # 全量 120 题 × 4 组
|
|
30
|
+
依赖:DEEPSEEK_API_KEY;answers_map 由上游对齐生成(见 --answers 默认路径)。
|
|
31
|
+
"""
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import argparse
|
|
35
|
+
import hashlib
|
|
36
|
+
import json
|
|
37
|
+
import os
|
|
38
|
+
import sys
|
|
39
|
+
import time
|
|
40
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
41
|
+
|
|
42
|
+
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
43
|
+
if HERE not in sys.path:
|
|
44
|
+
sys.path.insert(0, HERE)
|
|
45
|
+
|
|
46
|
+
from md_cg.bench_e2e_judge import ( # noqa: E402
|
|
47
|
+
DEFAULT_ROOT, load_src, speakers_by_scene, build_pool, run_det_arms,
|
|
48
|
+
llm_chat)
|
|
49
|
+
|
|
50
|
+
READER_SYS = (
|
|
51
|
+
"你是记忆问答器。只根据给定的记忆卡回答问题,不要用记忆卡以外的知识。"
|
|
52
|
+
"记忆卡不足以确定答案时,回答「无法确定」。答案尽量简短(不超过20字),"
|
|
53
|
+
"不要解释。\n只输出一行 JSON:{\"answer\": \"…\"}"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
JUDGE_SYS = (
|
|
57
|
+
"你是 QA 评分器。给定问题、参考答案(gold)与模型回答,判一个结论:\n"
|
|
58
|
+
"- correct:回答与 gold 语义等价(中英跨语判等;日期/人名/实体须一致,"
|
|
59
|
+
"表述差异可容忍)\n"
|
|
60
|
+
"- incorrect:回答了但语义与 gold 不等价(含被错误记忆误导的答案)\n"
|
|
61
|
+
"- refused:回答表示无法确定/不知道/记忆未提及\n"
|
|
62
|
+
"gold 为空字符串时(对抗题):回答「无法确定」= correct,给出任何具体"
|
|
63
|
+
"答案 = incorrect。\n只输出一行 JSON:{\"verdict\": \"correct\"}"
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _cache_call(cache_dir, kind, payload_key, fn):
|
|
68
|
+
"""LLM 调用缓存:payload_key 相同直接回放(qa 层可复现性与省钱)。"""
|
|
69
|
+
hexk = hashlib.md5(payload_key.encode("utf-8")).hexdigest()
|
|
70
|
+
path = os.path.join(cache_dir, f"{kind}_{hexk}.json")
|
|
71
|
+
if os.path.isfile(path):
|
|
72
|
+
d = json.load(open(path, encoding="utf-8"))
|
|
73
|
+
return d["out"], d.get("tokens", 0), 0.0, True
|
|
74
|
+
out, usage, dt = fn()
|
|
75
|
+
json.dump({"out": out, "tokens": int(usage.get("total_tokens") or 0)},
|
|
76
|
+
open(path, "w", encoding="utf-8"), ensure_ascii=False)
|
|
77
|
+
return out, int(usage.get("total_tokens") or 0), dt, False
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def parse_json_field(raw, field):
|
|
81
|
+
s = raw or ""
|
|
82
|
+
i, j = s.find("{"), s.rfind("}")
|
|
83
|
+
if i >= 0 and j > i:
|
|
84
|
+
try:
|
|
85
|
+
return json.loads(s[i:j + 1]).get(field)
|
|
86
|
+
except ValueError:
|
|
87
|
+
pass
|
|
88
|
+
return None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def run_qa_one(cfg, cache_dir, q, ans, cand_nodes, arm, level):
|
|
92
|
+
"""一题一组:reader 作答 → judge 判分。返回行(异常行 error 字段)。"""
|
|
93
|
+
cards = "\n\n".join(
|
|
94
|
+
f"【{i}】{str(nd.get('content') or '')[:400]}"
|
|
95
|
+
for i, nd in enumerate(cand_nodes, 1))
|
|
96
|
+
r_user = (f"问题(中文关键词):{q['question']}\n"
|
|
97
|
+
f"问题(英文原题):{ans.get('en_q') or ''}\n\n候选记忆卡:\n{cards}")
|
|
98
|
+
tag = f"{q['qid']}|{arm}|L{level}"
|
|
99
|
+
|
|
100
|
+
def call_reader():
|
|
101
|
+
return llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
102
|
+
[{"role": "system", "content": READER_SYS},
|
|
103
|
+
{"role": "user", "content": r_user}],
|
|
104
|
+
timeout=cfg["timeout"], max_tokens=300)
|
|
105
|
+
|
|
106
|
+
def call_judge():
|
|
107
|
+
return llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
108
|
+
[{"role": "system", "content": JUDGE_SYS},
|
|
109
|
+
{"role": "user", "content":
|
|
110
|
+
f"问题:{ans.get('en_q') or q['question']}\n"
|
|
111
|
+
f"gold:{ans.get('answer') or ''}\n"
|
|
112
|
+
f"模型回答:{row['answer']}"}],
|
|
113
|
+
timeout=cfg["timeout"], max_tokens=200)
|
|
114
|
+
|
|
115
|
+
row = {"qid": q["qid"], "qtype": q.get("qtype"), "arm": arm,
|
|
116
|
+
"level": level, "gold": ans.get("answer") or ""}
|
|
117
|
+
try:
|
|
118
|
+
raw, tok_r, dt_r, _c = _cache_call(
|
|
119
|
+
cache_dir, "reader", cfg["model"] + "|" + r_user, call_reader)
|
|
120
|
+
ans_txt = parse_json_field(raw, "answer")
|
|
121
|
+
if ans_txt is None:
|
|
122
|
+
ans_txt = (raw or "").strip()[:80]
|
|
123
|
+
row["answer"] = ans_txt
|
|
124
|
+
raw2, tok_j, dt_j, _c2 = _cache_call(
|
|
125
|
+
cache_dir, "judge",
|
|
126
|
+
cfg["model"] + "|" + row["gold"] + "|" + ans_txt + "|"
|
|
127
|
+
+ (ans.get("en_q") or ""), call_judge)
|
|
128
|
+
verdict = parse_json_field(raw2, "verdict")
|
|
129
|
+
row["verdict"] = verdict if verdict in ("correct", "incorrect",
|
|
130
|
+
"refused") else None
|
|
131
|
+
row["tokens"] = tok_r + tok_j
|
|
132
|
+
row["latency"] = dt_r + dt_j
|
|
133
|
+
if row["verdict"] is None:
|
|
134
|
+
row["error"] = f"judge 不可解析:{(raw2 or '')[:80]}"
|
|
135
|
+
except Exception as exc: # noqa: BLE001
|
|
136
|
+
row["error"] = f"{type(exc).__name__}: {exc}"
|
|
137
|
+
row.setdefault("answer", "")
|
|
138
|
+
row["verdict"] = None
|
|
139
|
+
row["tokens"] = row.get("tokens", 0)
|
|
140
|
+
row["latency"] = row.get("latency", 0.0)
|
|
141
|
+
row["_tag"] = tag
|
|
142
|
+
return row
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def qa_report(rows):
|
|
146
|
+
n = len(rows)
|
|
147
|
+
ok = [r for r in rows if r.get("verdict")]
|
|
148
|
+
adv = [r for r in ok if r.get("qtype") == "adversarial"]
|
|
149
|
+
norm = [r for r in ok if r.get("qtype") != "adversarial"]
|
|
150
|
+
c = lambda rs, v: sum(1 for r in rs if r["verdict"] == v) # noqa: E731
|
|
151
|
+
st = {
|
|
152
|
+
"n": n, "judged": len(ok), "errors": n - len(ok),
|
|
153
|
+
"acc": c(ok, "correct") / max(len(ok), 1),
|
|
154
|
+
"incorrect": c(ok, "incorrect"), "refused": c(ok, "refused"),
|
|
155
|
+
"norm_n": len(norm),
|
|
156
|
+
"norm_acc": c(norm, "correct") / max(len(norm), 1),
|
|
157
|
+
"norm_refused": c(norm, "refused"),
|
|
158
|
+
"adv_n": len(adv),
|
|
159
|
+
"adv_correct_refusal": c(adv, "correct"),
|
|
160
|
+
"adv_fooled": c(adv, "incorrect"),
|
|
161
|
+
"tokens": sum(r.get("tokens", 0) for r in rows),
|
|
162
|
+
}
|
|
163
|
+
return st
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def main(argv=None):
|
|
167
|
+
ap = argparse.ArgumentParser(description="端到端 QA:pinpoint/answerability")
|
|
168
|
+
ap.add_argument("--data-root", default=DEFAULT_ROOT)
|
|
169
|
+
ap.add_argument("--answers",
|
|
170
|
+
default=os.path.join(DEFAULT_ROOT, "upstream",
|
|
171
|
+
"answers_map.json"))
|
|
172
|
+
ap.add_argument("--n", type=int, default=120)
|
|
173
|
+
ap.add_argument("--seed", type=int, default=7)
|
|
174
|
+
ap.add_argument("--levels", default="0,4")
|
|
175
|
+
ap.add_argument("--model", default="deepseek-chat")
|
|
176
|
+
ap.add_argument("--base-url", default="https://api.deepseek.com/v1")
|
|
177
|
+
ap.add_argument("--key", default="")
|
|
178
|
+
ap.add_argument("--workers", type=int, default=6)
|
|
179
|
+
ap.add_argument("--timeout", type=int, default=120)
|
|
180
|
+
ap.add_argument("--k", type=int, default=10)
|
|
181
|
+
ap.add_argument("--quick", action="store_true")
|
|
182
|
+
a = ap.parse_args(argv)
|
|
183
|
+
if a.quick:
|
|
184
|
+
a.n = 3
|
|
185
|
+
key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
|
|
186
|
+
if not key:
|
|
187
|
+
print("[error] 需要 DEEPSEEK_API_KEY 或 --key")
|
|
188
|
+
return 2
|
|
189
|
+
if not os.path.isfile(a.answers):
|
|
190
|
+
print(f"[error] 找不到答案映射 {a.answers}(先跑上游对齐生成)")
|
|
191
|
+
return 2
|
|
192
|
+
|
|
193
|
+
levels = [int(x) for x in a.levels.split(",") if x.strip() != ""]
|
|
194
|
+
cache_dir = os.path.join(a.data_root, "qa_cache")
|
|
195
|
+
os.makedirs(cache_dir, exist_ok=True)
|
|
196
|
+
res_dir = os.path.join(a.data_root, "results")
|
|
197
|
+
os.makedirs(res_dir, exist_ok=True)
|
|
198
|
+
|
|
199
|
+
import random
|
|
200
|
+
corpus, questions = load_src()
|
|
201
|
+
corpus_map = {c["id"]: c for c in corpus}
|
|
202
|
+
spk = speakers_by_scene(corpus)
|
|
203
|
+
answers = json.load(open(a.answers, encoding="utf-8"))
|
|
204
|
+
sample = sorted(random.Random(a.seed).sample(
|
|
205
|
+
questions, min(a.n, len(questions))), key=lambda q: q["qid"])
|
|
206
|
+
print(f"[e2e_qa] 采样 {len(sample)} 题(seed={a.seed})× levels={levels}"
|
|
207
|
+
f" × 候选臂 base/firewall × {a.model}")
|
|
208
|
+
|
|
209
|
+
cfg = {"model": a.model, "base": a.base_url, "key": key,
|
|
210
|
+
"timeout": a.timeout}
|
|
211
|
+
all_rows, summary = [], {"meta": {"seed": a.seed, "n": len(sample),
|
|
212
|
+
"model": a.model, "levels": levels}}
|
|
213
|
+
for lv in levels:
|
|
214
|
+
cg, n_synth = build_pool(lv, corpus, questions, corpus_map, spk,
|
|
215
|
+
a.data_root)
|
|
216
|
+
det = []
|
|
217
|
+
for q in sample:
|
|
218
|
+
r = run_det_arms(cg, q, k=a.k)
|
|
219
|
+
r["q"] = q
|
|
220
|
+
det.append(r)
|
|
221
|
+
tasks = []
|
|
222
|
+
for r in det:
|
|
223
|
+
for arm in ("base", "firewall"):
|
|
224
|
+
nodes = r["base_nodes"][:a.k] if arm == "base" \
|
|
225
|
+
else r["fw_nodes"][:a.k]
|
|
226
|
+
tasks.append((r["q"], answers.get(r["q"]["qid"], {}),
|
|
227
|
+
nodes, arm, lv))
|
|
228
|
+
t0 = time.time()
|
|
229
|
+
with ThreadPoolExecutor(max_workers=a.workers) as ex:
|
|
230
|
+
rows = list(ex.map(lambda t: run_qa_one(cfg, cache_dir, *t),
|
|
231
|
+
tasks))
|
|
232
|
+
all_rows.extend(rows)
|
|
233
|
+
print(f"\n===== level {lv}(池={len(corpus)+n_synth})"
|
|
234
|
+
f" · QA 耗时 {time.time()-t0:.1f}s =====")
|
|
235
|
+
print(f"{'臂':<10}{'QA准确率':>9}{'常规准确率':>10}{'常规拒答':>8}"
|
|
236
|
+
f"{'对抗正确拒答':>12}{'对抗被骗':>8}{'判分失败':>8}")
|
|
237
|
+
for arm in ("base", "firewall"):
|
|
238
|
+
st = qa_report([r for r in rows if r["arm"] == arm])
|
|
239
|
+
print(f"{arm:<10}{st['acc']*100:>8.1f}%{st['norm_acc']*100:>9.1f}%"
|
|
240
|
+
f"{st['norm_refused']:>7d}/{st['norm_n']}"
|
|
241
|
+
f"{st['adv_correct_refusal']:>10d}/{st['adv_n']}"
|
|
242
|
+
f"{st['adv_fooled']:>7d}"
|
|
243
|
+
f"{st['errors']:>7d}")
|
|
244
|
+
summary[f"L{lv}_{arm}"] = st
|
|
245
|
+
out = os.path.join(res_dir,
|
|
246
|
+
f"qa_{time.strftime('%Y%m%d_%H%M%S')}.json")
|
|
247
|
+
json.dump({"summary": summary,
|
|
248
|
+
"rows": [{k: v for k, v in r.items() if k != "q"}
|
|
249
|
+
for r in all_rows]},
|
|
250
|
+
open(out, "w", encoding="utf-8"), ensure_ascii=False, indent=1)
|
|
251
|
+
print(f"\n[e2e_qa] 完成。明细 → {out}")
|
|
252
|
+
return 0
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
if __name__ == "__main__":
|
|
256
|
+
sys.exit(main())
|