@furongjun1999/dsh-memory 0.4.11 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +143 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +519 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/hooks.js +36 -2
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +116 -29
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +379 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1328 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +301 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1006 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +315 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1537 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1098 -1097
- package/md_cg/crypto.py +3 -1
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +78 -18
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +4 -2
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +222 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +377 -329
- package/md_cg/hotcache.py +48 -7
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +338 -0
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +140 -107
- package/md_cg/mcp_server.py +403 -55
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +783 -233
- package/md_cg/mdcos.py +500 -70
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +694 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +300 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +143 -0
- package/md_cg/reconcile.py +228 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/review_cli.py +170 -0
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +393 -365
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +13 -3
- package/md_cg/security.py +128 -18
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +152 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +7 -4
- package/md_cg/sources.py +816 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +54 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +35 -5
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +13 -3
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +22 -2
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +96 -15
- package/md_cg/test_index_durability.py +17 -3
- package/md_cg/test_interop.py +95 -0
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +16 -7
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +7 -1
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +153 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +316 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +203 -0
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +344 -340
- package/md_cg/test_retr_s1b.py +276 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +392 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +9 -3
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +16 -2
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +38 -20
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +6 -3
- package/md_cg/tokens.py +85 -14
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +668 -667
- package/md_cg/vision_evidence.py +667 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +20 -8
- package/package.json +101 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +33 -0
- package/src/hooks.ts +38 -2
- package/src/index.ts +526 -518
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +116 -29
- package/src/lib/token_store.ts +13 -3
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/bench_lme_zh.py
CHANGED
|
@@ -1,411 +1,411 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""LongMemEval-S 中文结构层小样本探针(默认 20 题 gold + 180 干扰 = 池 200)。
|
|
3
|
-
|
|
4
|
-
被测假设(用户提出):写入时把英文 turn「翻译为结构化中文」,让中文查询也能
|
|
5
|
-
零-LLM 检索;同时保留英文原文供词法路,避免「翻译即断链」。
|
|
6
|
-
|
|
7
|
-
先验(eval_common.calibrate_turn 注释):v1 五要素模板是**同质模板词**,
|
|
8
|
-
抽样 3.3% < legacy 16% —— 同质模板对词法是净稀释。故中文层设计红线:
|
|
9
|
-
· 只放**特异词**(人名/日期/实体/意图/主题),禁「用户/会话/记录/关于」等同质词;
|
|
10
|
-
· 极短(≤60 汉字),不加任何固定槽位前缀(前缀=全体同质词=纯稀释);
|
|
11
|
-
· 英文原文一字不动,中文层是**附加**而非替换。
|
|
12
|
-
|
|
13
|
-
四臂(只变 写入内容 / 查询语言,其余全同):
|
|
14
|
-
A legacy 裸文本 + 英文原问
|
|
15
|
-
B jaccard 裸文本 + 英文原问 ← 已修底座
|
|
16
|
-
C jaccard 裸文本 + 中文层(20条) + 英文原问 ← 测「稀释」
|
|
17
|
-
D jaccard 裸文本 + 中文层(20条) + 中文问 ← 测「增益」
|
|
18
|
-
|
|
19
|
-
池 200 条 = 20 条 gold(20 题各取 evidence_turns[0])+ 180 条随机干扰。
|
|
20
|
-
随机基线 hit@1 = 20/200 = 10%(报告须以此为参照,不能只看 0 个百分点)。
|
|
21
|
-
|
|
22
|
-
跑法:
|
|
23
|
-
python -m md_cg.bench_lme_zh --build # 采样池 + 调 LLM 生成中文层(落缓存)
|
|
24
|
-
python -m md_cg.bench_lme_zh # 四臂评测
|
|
25
|
-
python -m md_cg.bench_lme_zh --n-q 20 --n-distract 180
|
|
26
|
-
"""
|
|
27
|
-
from __future__ import annotations
|
|
28
|
-
|
|
29
|
-
import argparse
|
|
30
|
-
import json
|
|
31
|
-
import os
|
|
32
|
-
import random
|
|
33
|
-
import re
|
|
34
|
-
import shutil
|
|
35
|
-
import sys
|
|
36
|
-
import time
|
|
37
|
-
from concurrent.futures import ThreadPoolExecutor
|
|
38
|
-
|
|
39
|
-
from md_cg import eval_common as ec
|
|
40
|
-
from md_cg.bench_longmem import GROUPS
|
|
41
|
-
|
|
42
|
-
MODEL = os.environ.get("ZH_PROBE_MODEL") or "deepseek-flash"
|
|
43
|
-
BASE = "https://api.deepseek.com/v1"
|
|
44
|
-
DIR = os.path.join(ec.EXT, "zh_probe")
|
|
45
|
-
POOL = os.path.join(DIR, "pool.jsonl")
|
|
46
|
-
CACHE = os.path.join(DIR, "cache.json")
|
|
47
|
-
# 中文层 / 中文查询词的**真源文件**:由会话模型直接产出并落盘,脚本只读。
|
|
48
|
-
MAN_ZH = os.path.join(DIR, "manual_zh.json")
|
|
49
|
-
MAN_Q = os.path.join(DIR, "manual_q.json")
|
|
50
|
-
ROOT_A = os.path.join(ec.HERE, "_md_cg_eval_zhprobe_a")
|
|
51
|
-
# 版本化 root:写入加工改版必须换名。build_eval_cg 的幂等按**节点数**判断
|
|
52
|
-
# (have >= n_rows),同节点数的旧加工库会被静默复用——首轮 C 臂就会因此
|
|
53
|
-
# 拿到旧的「整句翻译」中文层,实验白跑。
|
|
54
|
-
ROOT_C = os.path.join(ec.HERE, "_md_cg_eval_zhprobe_c2_bag")
|
|
55
|
-
# 「中文层做唯一索引、原文只做载荷」架构的验证库:正文只放中文层,英文原文不入库。
|
|
56
|
-
ROOT_E = os.path.join(ec.HERE, "_md_cg_eval_zhprobe_e_zhonly")
|
|
57
|
-
ROOT_F = os.path.join(ec.HERE, "_md_cg_eval_zhprobe_f_zhpool")
|
|
58
|
-
POOL20 = os.path.join(DIR, "pool20.jsonl") # 只含 20 条 gold(隔离语言隔离假象)
|
|
59
|
-
|
|
60
|
-
# 探针实测(2026-09-11)三则:
|
|
61
|
-
# 1) deepseek-flash / deepseek-v4-pro **均为推理模型**(简单任务 reasoning_tokens
|
|
62
|
-
# 就占 36/38)。**长 prompt 让推理失控**:「6 条硬约束+示例」版在
|
|
63
|
-
# max_tokens=2000 时 completion 全烧在 reasoning 上、content 返回空串
|
|
64
|
-
# (finish_reason=length)→ prompt 必须极短,额度须 ≥1200。
|
|
65
|
-
# 2) 首版 prompt「译成中文检索词」被模型理解成**整句翻译**(均 89.9 字,且带
|
|
66
|
-
# 「用户/我/的」等同质词)→ 实测 C 15% < B 20%,即**净稀释**。本轮改真词袋。
|
|
67
|
-
# 3) 写入侧若是词袋,查询侧必须**同样词袋化**(自然语言长句与词袋的 bigram 交集
|
|
68
|
-
# 很小)→ 否则两侧不同源,「翻译即断链」的变体。
|
|
69
|
-
# 另:首轮 zh_question 只给 max_tokens=120 → 推理烧空 content → 静默 fallback
|
|
70
|
-
# 英文原问,导致 D 臂逐项等于 C 臂(假结论)。额度不足即为该类静默失效的根源。
|
|
71
|
-
ZH_PROMPT = """从下面英文里抽出中文关键词:只输出名词与动词的中文词,逗号分隔,一行。不要句子、不要「我/你/的/了/是/在」这类虚词、不要解释。英文:
|
|
72
|
-
"""
|
|
73
|
-
|
|
74
|
-
ZH_Q_PROMPT = """从下面英文问题里抽出中文关键词:只输出名词与动词的中文词,逗号分隔,一行。不要句子、不要虚词、不要解释。问题:"""
|
|
75
|
-
|
|
76
|
-
|
|
77
|
-
# ------------------------------------------------------------------ LLM
|
|
78
|
-
# 生效条件:环境变量 DEEPSEEK_API_KEY 为真值(缺省或空串即抛 RuntimeError「需要 DEEPSEEK_API_KEY」)且 range(retries) 至少迭代一次(retries≥1)时,用 messages、max_tokens 调 llm_chat 并返回 (content, usage);全部重试失败或 retries=0(last 保持 None)时抛 RuntimeError「LLM 调用失败:…」。
|
|
79
|
-
def _llm(messages, max_tokens=400, retries=3):
|
|
80
|
-
from md_cg.bench_task_ab_llm import llm_chat
|
|
81
|
-
key = os.environ.get("DEEPSEEK_API_KEY")
|
|
82
|
-
if not key:
|
|
83
|
-
raise RuntimeError("需要 DEEPSEEK_API_KEY")
|
|
84
|
-
last = None
|
|
85
|
-
for i in range(retries):
|
|
86
|
-
try:
|
|
87
|
-
content, usage, dt = llm_chat(MODEL, BASE, key, messages,
|
|
88
|
-
timeout=120, max_tokens=max_tokens,
|
|
89
|
-
temperature=0.0)
|
|
90
|
-
return content, usage
|
|
91
|
-
except Exception as exc: # noqa: BLE001
|
|
92
|
-
last = exc
|
|
93
|
-
time.sleep(1.5 * (i + 1))
|
|
94
|
-
raise RuntimeError(f"LLM 调用失败:{type(last).__name__}: {last}")
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
# 条件空间字段名:中文标签 → eval_common 口径
|
|
98
|
-
_COND_KEYS = {"观测位置": "observation_position", "观测工具": "observation_tool",
|
|
99
|
-
"时间窗口": "time_window", "存在约束": "existence_constraint"}
|
|
100
|
-
_DEFAULT_COND = {"observation_position": "会话陈述", "observation_tool": "会话记录",
|
|
101
|
-
"existence_constraint": "公开"}
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
# 生效条件:p 经 os.path.isfile(p) 判定为普通文件时按 utf-8 打开并返回 json.load(f);os.path.isfile(p) 为假(含目录、不存在路径、空串)时返回 {}。
|
|
105
|
-
def _load_json(p):
|
|
106
|
-
if os.path.isfile(p):
|
|
107
|
-
with open(p, encoding="utf-8") as f:
|
|
108
|
-
return json.load(f)
|
|
109
|
-
return {}
|
|
110
|
-
|
|
111
|
-
|
|
112
|
-
# 生效条件:无参调用即返回 (_load_json(MAN_ZH), _load_json(MAN_Q)),每个元素在模块级常量 MAN_ZH / MAN_Q 指向普通文件时为解析出的 JSON、否则为 {}。
|
|
113
|
-
def _load_manual():
|
|
114
|
-
"""读入会话模型直接产出的中文层 / 中文查询词(零 API 依赖)。
|
|
115
|
-
|
|
116
|
-
为什么不由本脚本调 LLM:探针实测(2026-09-11)deepseek-flash 与
|
|
117
|
-
deepseek-v4-pro **都是推理模型**,该任务上 completion 全烧在 reasoning
|
|
118
|
-
上、content 返回空串(首轮 0/20;次轮 13/20 且形态是整句翻译)。故改为
|
|
119
|
-
会话模型本人产出并落盘、脚本只读文件——顺带消除「额度不足 → 静默
|
|
120
|
-
fallback 成英文原问」这类失效模式(次轮 D 臂逐项等于 C 臂即由此而来)。
|
|
121
|
-
"""
|
|
122
|
-
return _load_json(MAN_ZH), _load_json(MAN_Q)
|
|
123
|
-
|
|
124
|
-
|
|
125
|
-
# 生效条件:z 为真值且 re.search(r"条件=([^;;]+)") 命中,并按 [||] 切分后至少有一段以 ":" / ":" 分为两段且首段 strip 后在模块级常量 _COND_KEYS 中时,返回非空映射 dict;z 为假值、正则未命中或映射为空时返回 None。
|
|
126
|
-
def _cond_of(z):
|
|
127
|
-
"""从中文层的「条件=观测位置:…|观测工具:…|…」抽条件空间 dict。"""
|
|
128
|
-
m = re.search(r"条件=([^;;]+)", z or "")
|
|
129
|
-
if not m:
|
|
130
|
-
return None
|
|
131
|
-
out = {}
|
|
132
|
-
for part in re.split(r"[||]", m.group(1)):
|
|
133
|
-
kv = re.split(r"[::]", part, 1)
|
|
134
|
-
if len(kv) == 2 and kv[0].strip() in _COND_KEYS:
|
|
135
|
-
out[_COND_KEYS[kv[0].strip()]] = kv[1].strip()
|
|
136
|
-
return out or None
|
|
137
|
-
|
|
138
|
-
|
|
139
|
-
# 生效条件:模块级常量 CACHE 经 os.path.isfile(CACHE) 判定为普通文件时按 utf-8 打开并返回 json.load(f);否则返回 {}。
|
|
140
|
-
def _load_cache():
|
|
141
|
-
if os.path.isfile(CACHE):
|
|
142
|
-
with open(CACHE, encoding="utf-8") as f:
|
|
143
|
-
return json.load(f)
|
|
144
|
-
return {}
|
|
145
|
-
|
|
146
|
-
|
|
147
|
-
# 生效条件:以 c 为输入即 os.makedirs(DIR, exist_ok=True) 后按 utf-8 把 c 以 ensure_ascii=False、indent=1 写入模块级常量 CACHE,无前置校验、无返回值。
|
|
148
|
-
def _save_cache(c):
|
|
149
|
-
os.makedirs(DIR, exist_ok=True)
|
|
150
|
-
with open(CACHE, "w", encoding="utf-8") as f:
|
|
151
|
-
json.dump(c, f, ensure_ascii=False, indent=1)
|
|
152
|
-
|
|
153
|
-
|
|
154
|
-
# 生效条件:cache 命中 "L2:"+turn["id"] 时直接返回该缓存值;否则用 ZH_PROMPT 拼 json.dumps({"date": turn.get("date"), "text": turn.get("text")})(键缺失回落 null)调 _llm,依次取 max_tokens=1200、2600,对 content or "" 去代码块与标签后 strip(" ,,、|"),首个非空即 break,把结果写回 cache 并返回(两次皆空则缓存并返回 "")。
|
|
155
|
-
def zh_layer(turn, cache):
|
|
156
|
-
"""一条 turn → 中文词袋。缓存 key 带版本号:prompt 改版必须失效旧缓存。
|
|
157
|
-
|
|
158
|
-
不传 speaker:角色标签(user/assistant)是**全体同质词**,翻成「用户」只会
|
|
159
|
-
稀释词法信号(首轮实测即如此)。date 保留——它是特异词。
|
|
160
|
-
"""
|
|
161
|
-
key = "L2:" + turn["id"]
|
|
162
|
-
if key in cache:
|
|
163
|
-
return cache[key]
|
|
164
|
-
msg = [{"role": "user", "content":
|
|
165
|
-
ZH_PROMPT + json.dumps({"date": turn.get("date"),
|
|
166
|
-
"text": turn.get("text")},
|
|
167
|
-
ensure_ascii=False)}]
|
|
168
|
-
s = ""
|
|
169
|
-
for mt in (1200, 2600): # 推理模型额度被 reasoning 吃光 → 空 content,故加额重试
|
|
170
|
-
content, _u = _llm(msg, max_tokens=mt)
|
|
171
|
-
s = content or ""
|
|
172
|
-
s = re.sub(r"```.*?```", " ", s, flags=re.S)
|
|
173
|
-
s = re.sub(r"(人物|日期|实体|意图|主题)\s*[::|]", " ", s)
|
|
174
|
-
s = " ".join(s.split()).strip(" ,,、|")
|
|
175
|
-
if s:
|
|
176
|
-
break
|
|
177
|
-
cache[key] = s
|
|
178
|
-
return s
|
|
179
|
-
|
|
180
|
-
|
|
181
|
-
# 生效条件:cache 命中 "Q2:"+q["qid"] 时直接返回该缓存值;否则以 ZH_Q_PROMPT+q["question"] 调 _llm(max_tokens=1200),去代码块后 strip(" ,,、") 写入 cache[key],结果为空串时抛 RuntimeError「查询侧关键词为空…」,非空时返回 cache[key]。
|
|
182
|
-
def zh_question(q, cache):
|
|
183
|
-
"""英文问题 → 中文关键词(与写入侧同源;额度须 ≥1200,否则推理烧空 content)。"""
|
|
184
|
-
key = "Q2:" + q["qid"]
|
|
185
|
-
if key in cache:
|
|
186
|
-
return cache[key]
|
|
187
|
-
content, _u = _llm([{"role": "user",
|
|
188
|
-
"content": ZH_Q_PROMPT + q["question"]}], max_tokens=1200)
|
|
189
|
-
s = re.sub(r"```.*?```", " ", content or "", flags=re.S)
|
|
190
|
-
cache[key] = " ".join(s.split()).strip(" ,,、")
|
|
191
|
-
if not cache[key]:
|
|
192
|
-
raise RuntimeError("查询侧关键词为空(额度被 reasoning 吃光)")
|
|
193
|
-
return cache[key]
|
|
194
|
-
|
|
195
|
-
|
|
196
|
-
# ------------------------------------------------------------------ 池
|
|
197
|
-
# 生效条件:模块级常量 POOL 为普通文件时直接返回其中非空行的 json.loads 列表(此时不读 n_q、n_distract、seed);否则以 random.Random(seed)、每组 max(1, n_q//len(GROUPS)) 条(取 q["evidence_turns"][0] 为 gold)采样后追加 rng.sample(rest, min(n_distract, len(rest))) 干扰项,写出 POOL 与 questions.json 并返回 pool。
|
|
198
|
-
def build_pool(n_q, n_distract, seed=7):
|
|
199
|
-
"""采样:n_q 题(分组覆盖)各取 evidence_turns[0] 为 gold,再取干扰。"""
|
|
200
|
-
if os.path.isfile(POOL):
|
|
201
|
-
with open(POOL, encoding="utf-8") as f:
|
|
202
|
-
return [json.loads(l) for l in f if l.strip()]
|
|
203
|
-
rows = list(ec.iter_jsonl(ec.LM_H))
|
|
204
|
-
byid = {r["id"]: r for r in rows}
|
|
205
|
-
qs = ec.load_questions("lm")
|
|
206
|
-
rng = random.Random(seed)
|
|
207
|
-
per = max(1, n_q // len(GROUPS))
|
|
208
|
-
picked, gold_ids, chosen = [], set(), []
|
|
209
|
-
for gname in GROUPS:
|
|
210
|
-
qtypes = GROUPS[gname]
|
|
211
|
-
cand = [q for q in qs if q["qtype"] in qtypes and q.get("evidence_turns")]
|
|
212
|
-
rng.shuffle(cand)
|
|
213
|
-
for q in cand:
|
|
214
|
-
if len([p for p in picked if p["_g"] == gname]) >= per:
|
|
215
|
-
break
|
|
216
|
-
tid = q["evidence_turns"][0]
|
|
217
|
-
if tid not in byid or tid in gold_ids:
|
|
218
|
-
continue
|
|
219
|
-
gold_ids.add(tid)
|
|
220
|
-
picked.append({**byid[tid], "_g": gname, "_gold": True})
|
|
221
|
-
chosen.append(q)
|
|
222
|
-
pool = list(picked)
|
|
223
|
-
rest = [r for r in rows if r["id"] not in gold_ids]
|
|
224
|
-
pool += rng.sample(rest, min(n_distract, len(rest)))
|
|
225
|
-
os.makedirs(DIR, exist_ok=True)
|
|
226
|
-
with open(POOL, "w", encoding="utf-8") as f:
|
|
227
|
-
for r in pool:
|
|
228
|
-
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
229
|
-
with open(os.path.join(DIR, "questions.json"), "w", encoding="utf-8") as f:
|
|
230
|
-
json.dump(chosen, f, ensure_ascii=False, indent=1)
|
|
231
|
-
return pool
|
|
232
|
-
|
|
233
|
-
|
|
234
|
-
# 生效条件:DIR 目录下的 questions.json 是 UTF-8 合法 JSON 时,返回 json.load 得到的对象。
|
|
235
|
-
def load_questions_zh():
|
|
236
|
-
with open(os.path.join(DIR, "questions.json"), encoding="utf-8") as f:
|
|
237
|
-
return json.load(f)
|
|
238
|
-
|
|
239
|
-
|
|
240
|
-
# ------------------------------------------------------------------ 建库 / 评测
|
|
241
|
-
# 生效条件:以 ec.build_eval_cg(None, root, corpus or POOL, ec.lm_turn_text, "lmezh", calib_of=calib if zh_of 为真值 else None, rebuild=rebuild) 执行并返回其结果(形参 pool 在该调用中未被使用,实际用 corpus or POOL;zh_only 只在内层 calib 中生效,不传给 build_eval_cg)。
|
|
242
|
-
def build(root, pool, zh_of=None, rebuild=False, zh_only=False, corpus=None):
|
|
243
|
-
"""zh_of: turn_id → 中文层;None = 裸文本库。
|
|
244
|
-
|
|
245
|
-
zh_only=True → **纯中文索引**:正文只放中文层,英文原文不入库(供「中文检索、
|
|
246
|
-
原文只做载荷回填」架构验证)。无中文层的行退化为裸英文——本轮 180 条干扰项
|
|
247
|
-
未译,故 E 臂存在「语言隔离」:中文查询几乎不可能命中英文干扰项,其 hit 是
|
|
248
|
-
上界而非全译库真值。故另设 F 臂(池只含 20 条 gold、正文全中文)隔离该假象,
|
|
249
|
-
单测中文层自身的可区分性。
|
|
250
|
-
"""
|
|
251
|
-
# 生效条件:作为 build 的内层函数闭包使用 zh_of/zh_only,ctx 未被使用;zh_of 为真值且 zh_of.get(r["id"]) 为真值时 cs = _cond_of(z) or cs,否则 cs = dict(_DEFAULT_COND);zh_only 为真值时返回 (z or ec.lm_turn_text(r), [], cs),否则返回 (ec.lm_turn_text(r) 在 z 为真值时再拼 "\n"+z, [], cs)。
|
|
252
|
-
def calib(r, ctx=None):
|
|
253
|
-
z = zh_of.get(r["id"]) if zh_of else None
|
|
254
|
-
cs = dict(_DEFAULT_COND)
|
|
255
|
-
if z:
|
|
256
|
-
cs = _cond_of(z) or cs # 条件空间逐条来自中文层,而非全局写死
|
|
257
|
-
if zh_only:
|
|
258
|
-
return (z or ec.lm_turn_text(r)), [], cs
|
|
259
|
-
body = ec.lm_turn_text(r)
|
|
260
|
-
if z:
|
|
261
|
-
body = body + "\n" + z
|
|
262
|
-
return body, [], cs
|
|
263
|
-
return ec.build_eval_cg(None, root, corpus or POOL, ec.lm_turn_text, "lmezh",
|
|
264
|
-
calib_of=calib if zh_of else None, rebuild=rebuild)
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
# 生效条件:给定 cg、questions 与 qtext_of 时,对每题用 qtext_of(q) 替换 question 后经 ec.evaluate_group 评测,并按 k 汇总为结果。
|
|
268
|
-
def eval_arm(cg, questions, qtext_of, k=5):
|
|
269
|
-
rows = []
|
|
270
|
-
for q in questions:
|
|
271
|
-
qq = dict(q)
|
|
272
|
-
qq["question"] = qtext_of(q)
|
|
273
|
-
rows += ec.evaluate_group(cg, [qq], k=k, paths=ec.PATHS, verbose=False)
|
|
274
|
-
return ec.summarize(rows, k=k)
|
|
275
|
-
|
|
276
|
-
|
|
277
|
-
# 生效条件:argv 由 argparse 解析(--n-q/--n-distract/--k/--build/--rebuild/--workers 等);--build 只产池即返回,否则需外部数据在指定路径就位、缺失即抛异常不静默降级;
|
|
278
|
-
def main():
|
|
279
|
-
ap = argparse.ArgumentParser()
|
|
280
|
-
ap.add_argument("--n-q", type=int, default=20)
|
|
281
|
-
ap.add_argument("--n-distract", type=int, default=180)
|
|
282
|
-
ap.add_argument("--k", type=int, default=5)
|
|
283
|
-
ap.add_argument("--build", action="store_true", help="只生成池与中文层")
|
|
284
|
-
ap.add_argument("--rebuild", action="store_true")
|
|
285
|
-
ap.add_argument("--workers", type=int, default=6)
|
|
286
|
-
args = ap.parse_args()
|
|
287
|
-
|
|
288
|
-
pool = build_pool(args.n_q, args.n_distract)
|
|
289
|
-
golds = [r for r in pool if r.get("_gold")]
|
|
290
|
-
qs = load_questions_zh()
|
|
291
|
-
print(f"池 {len(pool)} 条(gold {len(golds)} / 干扰 "
|
|
292
|
-
f"{len(pool) - len(golds)}),题 {len(qs)} 道,随机基线 hit@1="
|
|
293
|
-
f"{len(golds) / len(pool):.1%}")
|
|
294
|
-
|
|
295
|
-
cache = _load_cache()
|
|
296
|
-
man_zh, man_q = _load_manual()
|
|
297
|
-
for tid, v in man_zh.items(): # 会话模型产出 → 直接进缓存,零 API
|
|
298
|
-
cache["L2:" + tid] = v
|
|
299
|
-
for qid, v in man_q.items():
|
|
300
|
-
cache["Q2:" + qid] = v
|
|
301
|
-
print(f"中文层:本地 {len(man_zh)} 条(查询词本地 {len(man_q)} 条),"
|
|
302
|
-
f"缓存合计 {len(cache)} 条,模型 {MODEL}(本轮不调用)")
|
|
303
|
-
|
|
304
|
-
# 生效条件:对 r 调 zh_layer(r, cache)(cache 为闭包变量)成功时返回 (r["id"], 中文层, None);zh_layer 抛任何异常时返回 (r["id"], "", f"{type(exc).__name__}: {exc}")。
|
|
305
|
-
def work(r):
|
|
306
|
-
try:
|
|
307
|
-
return r["id"], zh_layer(r, cache), None
|
|
308
|
-
except Exception as exc: # noqa: BLE001
|
|
309
|
-
return r["id"], "", f"{type(exc).__name__}: {exc}"
|
|
310
|
-
|
|
311
|
-
zh = {}
|
|
312
|
-
with ThreadPoolExecutor(max_workers=args.workers) as ex:
|
|
313
|
-
for i, (tid, s, err) in enumerate(ex.map(work, golds), 1):
|
|
314
|
-
if err:
|
|
315
|
-
print(f" [{i}/{len(golds)}] {tid} 失败:{err}")
|
|
316
|
-
else:
|
|
317
|
-
zh[tid] = s
|
|
318
|
-
_save_cache(cache)
|
|
319
|
-
okzh = [v for v in zh.values() if v]
|
|
320
|
-
print(f"中文层成功 {len(okzh)}/{len(golds)},平均长度 "
|
|
321
|
-
f"{sum(len(v) for v in okzh) / max(1, len(okzh)):.1f} 字")
|
|
322
|
-
for t in okzh[:5]:
|
|
323
|
-
print(" ·", t[:110])
|
|
324
|
-
|
|
325
|
-
if args.build:
|
|
326
|
-
print("(--build 只生成,不评测)")
|
|
327
|
-
return
|
|
328
|
-
if not okzh:
|
|
329
|
-
print("中文层为空,终止")
|
|
330
|
-
return 1
|
|
331
|
-
|
|
332
|
-
qzh = {}
|
|
333
|
-
for q in qs:
|
|
334
|
-
try:
|
|
335
|
-
qzh[q["qid"]] = zh_question(q, cache)
|
|
336
|
-
except Exception as exc: # noqa: BLE001
|
|
337
|
-
print(f" 问题翻译失败 {q['qid']}:{exc}")
|
|
338
|
-
qzh[q["qid"]] = q["question"]
|
|
339
|
-
_save_cache(cache)
|
|
340
|
-
|
|
341
|
-
en = lambda q: q["question"]
|
|
342
|
-
zhq = lambda q: qzh.get(q["qid"]) or q["question"]
|
|
343
|
-
|
|
344
|
-
# 只含 gold 的 20 条池:E 臂的「语言隔离」假象在其中被消除(全部是中文文档)
|
|
345
|
-
with open(POOL20, "w", encoding="utf-8") as f:
|
|
346
|
-
for r in golds:
|
|
347
|
-
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
348
|
-
|
|
349
|
-
import md_cg.mdcg as m
|
|
350
|
-
results = {}
|
|
351
|
-
arms = (
|
|
352
|
-
("A legacy 裸+英问", ROOT_A, POOL, False, False, en, "legacy"),
|
|
353
|
-
("B jaccard 裸+英问", ROOT_A, POOL, False, False, en, "jaccard"),
|
|
354
|
-
("C jaccard 中+英问", ROOT_C, POOL, True, False, en, "jaccard"),
|
|
355
|
-
("D jaccard 中+中问", ROOT_C, POOL, True, False, zhq, "jaccard"),
|
|
356
|
-
("E 纯中索引+中问", ROOT_E, POOL, True, True, zhq, "jaccard"),
|
|
357
|
-
("F 纯中池+中问", ROOT_F, POOL20, True, True, zhq, "jaccard"),
|
|
358
|
-
("G 纯中池+英问", ROOT_F, POOL20, True, True, en, "jaccard"),
|
|
359
|
-
)
|
|
360
|
-
for tag, root, corpus, usezh, zonly, qf, mode in arms:
|
|
361
|
-
m.SCORE_MODE = mode
|
|
362
|
-
cg = build(root, pool, zh_of=zh if usezh else None, rebuild=args.rebuild,
|
|
363
|
-
zh_only=zonly, corpus=corpus)
|
|
364
|
-
ec.install_read_cache(cg)
|
|
365
|
-
results[tag] = eval_arm(cg, qs, qf, k=args.k)
|
|
366
|
-
print(f" {tag}: hit@1 {results[tag]['hit@1']:.1%} "
|
|
367
|
-
f"hit@{args.k} {results[tag][f'hit@{args.k}']:.1%} "
|
|
368
|
-
f"MRR {results[tag]['mrr']:.3f}")
|
|
369
|
-
|
|
370
|
-
# 命中后回填原始英文原文:原文不进索引、只做载荷 → 验证「中文检索→英文返回」链路
|
|
371
|
-
m.SCORE_MODE = "jaccard"
|
|
372
|
-
cg_e = build(ROOT_E, pool, zh_of=zh, corpus=POOL, zh_only=True)
|
|
373
|
-
ec.install_read_cache(cg_e)
|
|
374
|
-
payload = {r["id"]: ec.lm_turn_text(r) for r in pool}
|
|
375
|
-
line = results["E 纯中索引+中问"]["score_p10"]
|
|
376
|
-
print(f"\n回填验证(中文命中 → 确认匹配 → 返回英文原文;原文不在索引内):"
|
|
377
|
-
f"\n 匹配确认线 = 正例 Top-1 分 p10 = {line:.4f}"
|
|
378
|
-
f"(低于线判未匹配,不回填原文)")
|
|
379
|
-
conf_n = conf_hit = 0
|
|
380
|
-
for q in qs:
|
|
381
|
-
res, _meta = ec.run_query(cg_e, zhq(q), k=1, paths=ec.PATHS)
|
|
382
|
-
sc = res[0][1] if res else 0.0
|
|
383
|
-
if sc >= line:
|
|
384
|
-
conf_n += 1
|
|
385
|
-
if ec.first_evidence_rank(res, set(q.get("evidence_turns") or [])):
|
|
386
|
-
conf_hit += 1
|
|
387
|
-
print(f" 确认匹配 {conf_n}/{len(qs)} 题,其中真命中 {conf_hit} → "
|
|
388
|
-
f"确认后精度 {conf_hit / max(1, conf_n):.1%}")
|
|
389
|
-
for q in qs[:2]:
|
|
390
|
-
res, _meta = ec.run_query(cg_e, zhq(q), k=2, paths=ec.PATHS)
|
|
391
|
-
print(f" 中文查询: {zhq(q)}")
|
|
392
|
-
for it in res[:2]:
|
|
393
|
-
nid, sc = it[0]["id"], it[1]
|
|
394
|
-
verdict = "确认匹配→回填" if sc >= line else "未确认→不回填"
|
|
395
|
-
print(f" · {nid} 分={sc:.4f} [{verdict}] 载荷原文: "
|
|
396
|
-
f"{payload.get(nid, '')[:70]}")
|
|
397
|
-
|
|
398
|
-
ec.print_table("LongMemEval-S 中文层探针(%d 题 / 池 %d)"
|
|
399
|
-
% (len(qs), len(pool)), results, k=args.k)
|
|
400
|
-
ec.save_result("zh_probe.json", {
|
|
401
|
-
"dataset": "LongMemEval-S slice", "model": MODEL,
|
|
402
|
-
"pool": len(pool), "n_q": len(qs), "baseline_hit1": len(golds) / len(pool),
|
|
403
|
-
"arms": results,
|
|
404
|
-
"zh_layer_samples": okzh[:20],
|
|
405
|
-
"questions_zh": qzh,
|
|
406
|
-
})
|
|
407
|
-
return 0
|
|
408
|
-
|
|
409
|
-
|
|
410
|
-
if __name__ == "__main__":
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""LongMemEval-S 中文结构层小样本探针(默认 20 题 gold + 180 干扰 = 池 200)。
|
|
3
|
+
|
|
4
|
+
被测假设(用户提出):写入时把英文 turn「翻译为结构化中文」,让中文查询也能
|
|
5
|
+
零-LLM 检索;同时保留英文原文供词法路,避免「翻译即断链」。
|
|
6
|
+
|
|
7
|
+
先验(eval_common.calibrate_turn 注释):v1 五要素模板是**同质模板词**,
|
|
8
|
+
抽样 3.3% < legacy 16% —— 同质模板对词法是净稀释。故中文层设计红线:
|
|
9
|
+
· 只放**特异词**(人名/日期/实体/意图/主题),禁「用户/会话/记录/关于」等同质词;
|
|
10
|
+
· 极短(≤60 汉字),不加任何固定槽位前缀(前缀=全体同质词=纯稀释);
|
|
11
|
+
· 英文原文一字不动,中文层是**附加**而非替换。
|
|
12
|
+
|
|
13
|
+
四臂(只变 写入内容 / 查询语言,其余全同):
|
|
14
|
+
A legacy 裸文本 + 英文原问
|
|
15
|
+
B jaccard 裸文本 + 英文原问 ← 已修底座
|
|
16
|
+
C jaccard 裸文本 + 中文层(20条) + 英文原问 ← 测「稀释」
|
|
17
|
+
D jaccard 裸文本 + 中文层(20条) + 中文问 ← 测「增益」
|
|
18
|
+
|
|
19
|
+
池 200 条 = 20 条 gold(20 题各取 evidence_turns[0])+ 180 条随机干扰。
|
|
20
|
+
随机基线 hit@1 = 20/200 = 10%(报告须以此为参照,不能只看 0 个百分点)。
|
|
21
|
+
|
|
22
|
+
跑法:
|
|
23
|
+
python -m md_cg.bench_lme_zh --build # 采样池 + 调 LLM 生成中文层(落缓存)
|
|
24
|
+
python -m md_cg.bench_lme_zh # 四臂评测
|
|
25
|
+
python -m md_cg.bench_lme_zh --n-q 20 --n-distract 180
|
|
26
|
+
"""
|
|
27
|
+
from __future__ import annotations
|
|
28
|
+
|
|
29
|
+
import argparse
|
|
30
|
+
import json
|
|
31
|
+
import os
|
|
32
|
+
import random
|
|
33
|
+
import re
|
|
34
|
+
import shutil
|
|
35
|
+
import sys
|
|
36
|
+
import time
|
|
37
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
38
|
+
|
|
39
|
+
from md_cg import eval_common as ec
|
|
40
|
+
from md_cg.bench_longmem import GROUPS
|
|
41
|
+
|
|
42
|
+
MODEL = os.environ.get("ZH_PROBE_MODEL") or "deepseek-flash"
|
|
43
|
+
BASE = "https://api.deepseek.com/v1"
|
|
44
|
+
DIR = os.path.join(ec.EXT, "zh_probe")
|
|
45
|
+
POOL = os.path.join(DIR, "pool.jsonl")
|
|
46
|
+
CACHE = os.path.join(DIR, "cache.json")
|
|
47
|
+
# 中文层 / 中文查询词的**真源文件**:由会话模型直接产出并落盘,脚本只读。
|
|
48
|
+
MAN_ZH = os.path.join(DIR, "manual_zh.json")
|
|
49
|
+
MAN_Q = os.path.join(DIR, "manual_q.json")
|
|
50
|
+
ROOT_A = os.path.join(ec.HERE, "_md_cg_eval_zhprobe_a")
|
|
51
|
+
# 版本化 root:写入加工改版必须换名。build_eval_cg 的幂等按**节点数**判断
|
|
52
|
+
# (have >= n_rows),同节点数的旧加工库会被静默复用——首轮 C 臂就会因此
|
|
53
|
+
# 拿到旧的「整句翻译」中文层,实验白跑。
|
|
54
|
+
ROOT_C = os.path.join(ec.HERE, "_md_cg_eval_zhprobe_c2_bag")
|
|
55
|
+
# 「中文层做唯一索引、原文只做载荷」架构的验证库:正文只放中文层,英文原文不入库。
|
|
56
|
+
ROOT_E = os.path.join(ec.HERE, "_md_cg_eval_zhprobe_e_zhonly")
|
|
57
|
+
ROOT_F = os.path.join(ec.HERE, "_md_cg_eval_zhprobe_f_zhpool")
|
|
58
|
+
POOL20 = os.path.join(DIR, "pool20.jsonl") # 只含 20 条 gold(隔离语言隔离假象)
|
|
59
|
+
|
|
60
|
+
# 探针实测(2026-09-11)三则:
|
|
61
|
+
# 1) deepseek-flash / deepseek-v4-pro **均为推理模型**(简单任务 reasoning_tokens
|
|
62
|
+
# 就占 36/38)。**长 prompt 让推理失控**:「6 条硬约束+示例」版在
|
|
63
|
+
# max_tokens=2000 时 completion 全烧在 reasoning 上、content 返回空串
|
|
64
|
+
# (finish_reason=length)→ prompt 必须极短,额度须 ≥1200。
|
|
65
|
+
# 2) 首版 prompt「译成中文检索词」被模型理解成**整句翻译**(均 89.9 字,且带
|
|
66
|
+
# 「用户/我/的」等同质词)→ 实测 C 15% < B 20%,即**净稀释**。本轮改真词袋。
|
|
67
|
+
# 3) 写入侧若是词袋,查询侧必须**同样词袋化**(自然语言长句与词袋的 bigram 交集
|
|
68
|
+
# 很小)→ 否则两侧不同源,「翻译即断链」的变体。
|
|
69
|
+
# 另:首轮 zh_question 只给 max_tokens=120 → 推理烧空 content → 静默 fallback
|
|
70
|
+
# 英文原问,导致 D 臂逐项等于 C 臂(假结论)。额度不足即为该类静默失效的根源。
|
|
71
|
+
ZH_PROMPT = """从下面英文里抽出中文关键词:只输出名词与动词的中文词,逗号分隔,一行。不要句子、不要「我/你/的/了/是/在」这类虚词、不要解释。英文:
|
|
72
|
+
"""
|
|
73
|
+
|
|
74
|
+
ZH_Q_PROMPT = """从下面英文问题里抽出中文关键词:只输出名词与动词的中文词,逗号分隔,一行。不要句子、不要虚词、不要解释。问题:"""
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
# ------------------------------------------------------------------ LLM
|
|
78
|
+
# 生效条件:环境变量 DEEPSEEK_API_KEY 为真值(缺省或空串即抛 RuntimeError「需要 DEEPSEEK_API_KEY」)且 range(retries) 至少迭代一次(retries≥1)时,用 messages、max_tokens 调 llm_chat 并返回 (content, usage);全部重试失败或 retries=0(last 保持 None)时抛 RuntimeError「LLM 调用失败:…」。
|
|
79
|
+
def _llm(messages, max_tokens=400, retries=3):
|
|
80
|
+
from md_cg.bench_task_ab_llm import llm_chat
|
|
81
|
+
key = os.environ.get("DEEPSEEK_API_KEY")
|
|
82
|
+
if not key:
|
|
83
|
+
raise RuntimeError("需要 DEEPSEEK_API_KEY")
|
|
84
|
+
last = None
|
|
85
|
+
for i in range(retries):
|
|
86
|
+
try:
|
|
87
|
+
content, usage, dt = llm_chat(MODEL, BASE, key, messages,
|
|
88
|
+
timeout=120, max_tokens=max_tokens,
|
|
89
|
+
temperature=0.0)
|
|
90
|
+
return content, usage
|
|
91
|
+
except Exception as exc: # noqa: BLE001
|
|
92
|
+
last = exc
|
|
93
|
+
time.sleep(1.5 * (i + 1))
|
|
94
|
+
raise RuntimeError(f"LLM 调用失败:{type(last).__name__}: {last}")
|
|
95
|
+
|
|
96
|
+
|
|
97
|
+
# 条件空间字段名:中文标签 → eval_common 口径
|
|
98
|
+
_COND_KEYS = {"观测位置": "observation_position", "观测工具": "observation_tool",
|
|
99
|
+
"时间窗口": "time_window", "存在约束": "existence_constraint"}
|
|
100
|
+
_DEFAULT_COND = {"observation_position": "会话陈述", "observation_tool": "会话记录",
|
|
101
|
+
"existence_constraint": "公开"}
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
# 生效条件:p 经 os.path.isfile(p) 判定为普通文件时按 utf-8 打开并返回 json.load(f);os.path.isfile(p) 为假(含目录、不存在路径、空串)时返回 {}。
|
|
105
|
+
def _load_json(p):
|
|
106
|
+
if os.path.isfile(p):
|
|
107
|
+
with open(p, encoding="utf-8") as f:
|
|
108
|
+
return json.load(f)
|
|
109
|
+
return {}
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
# 生效条件:无参调用即返回 (_load_json(MAN_ZH), _load_json(MAN_Q)),每个元素在模块级常量 MAN_ZH / MAN_Q 指向普通文件时为解析出的 JSON、否则为 {}。
|
|
113
|
+
def _load_manual():
|
|
114
|
+
"""读入会话模型直接产出的中文层 / 中文查询词(零 API 依赖)。
|
|
115
|
+
|
|
116
|
+
为什么不由本脚本调 LLM:探针实测(2026-09-11)deepseek-flash 与
|
|
117
|
+
deepseek-v4-pro **都是推理模型**,该任务上 completion 全烧在 reasoning
|
|
118
|
+
上、content 返回空串(首轮 0/20;次轮 13/20 且形态是整句翻译)。故改为
|
|
119
|
+
会话模型本人产出并落盘、脚本只读文件——顺带消除「额度不足 → 静默
|
|
120
|
+
fallback 成英文原问」这类失效模式(次轮 D 臂逐项等于 C 臂即由此而来)。
|
|
121
|
+
"""
|
|
122
|
+
return _load_json(MAN_ZH), _load_json(MAN_Q)
|
|
123
|
+
|
|
124
|
+
|
|
125
|
+
# 生效条件:z 为真值且 re.search(r"条件=([^;;]+)") 命中,并按 [||] 切分后至少有一段以 ":" / ":" 分为两段且首段 strip 后在模块级常量 _COND_KEYS 中时,返回非空映射 dict;z 为假值、正则未命中或映射为空时返回 None。
|
|
126
|
+
def _cond_of(z):
|
|
127
|
+
"""从中文层的「条件=观测位置:…|观测工具:…|…」抽条件空间 dict。"""
|
|
128
|
+
m = re.search(r"条件=([^;;]+)", z or "")
|
|
129
|
+
if not m:
|
|
130
|
+
return None
|
|
131
|
+
out = {}
|
|
132
|
+
for part in re.split(r"[||]", m.group(1)):
|
|
133
|
+
kv = re.split(r"[::]", part, 1)
|
|
134
|
+
if len(kv) == 2 and kv[0].strip() in _COND_KEYS:
|
|
135
|
+
out[_COND_KEYS[kv[0].strip()]] = kv[1].strip()
|
|
136
|
+
return out or None
|
|
137
|
+
|
|
138
|
+
|
|
139
|
+
# 生效条件:模块级常量 CACHE 经 os.path.isfile(CACHE) 判定为普通文件时按 utf-8 打开并返回 json.load(f);否则返回 {}。
|
|
140
|
+
def _load_cache():
|
|
141
|
+
if os.path.isfile(CACHE):
|
|
142
|
+
with open(CACHE, encoding="utf-8") as f:
|
|
143
|
+
return json.load(f)
|
|
144
|
+
return {}
|
|
145
|
+
|
|
146
|
+
|
|
147
|
+
# 生效条件:以 c 为输入即 os.makedirs(DIR, exist_ok=True) 后按 utf-8 把 c 以 ensure_ascii=False、indent=1 写入模块级常量 CACHE,无前置校验、无返回值。
|
|
148
|
+
def _save_cache(c):
|
|
149
|
+
os.makedirs(DIR, exist_ok=True)
|
|
150
|
+
with open(CACHE, "w", encoding="utf-8") as f:
|
|
151
|
+
json.dump(c, f, ensure_ascii=False, indent=1)
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
# 生效条件:cache 命中 "L2:"+turn["id"] 时直接返回该缓存值;否则用 ZH_PROMPT 拼 json.dumps({"date": turn.get("date"), "text": turn.get("text")})(键缺失回落 null)调 _llm,依次取 max_tokens=1200、2600,对 content or "" 去代码块与标签后 strip(" ,,、|"),首个非空即 break,把结果写回 cache 并返回(两次皆空则缓存并返回 "")。
|
|
155
|
+
def zh_layer(turn, cache):
|
|
156
|
+
"""一条 turn → 中文词袋。缓存 key 带版本号:prompt 改版必须失效旧缓存。
|
|
157
|
+
|
|
158
|
+
不传 speaker:角色标签(user/assistant)是**全体同质词**,翻成「用户」只会
|
|
159
|
+
稀释词法信号(首轮实测即如此)。date 保留——它是特异词。
|
|
160
|
+
"""
|
|
161
|
+
key = "L2:" + turn["id"]
|
|
162
|
+
if key in cache:
|
|
163
|
+
return cache[key]
|
|
164
|
+
msg = [{"role": "user", "content":
|
|
165
|
+
ZH_PROMPT + json.dumps({"date": turn.get("date"),
|
|
166
|
+
"text": turn.get("text")},
|
|
167
|
+
ensure_ascii=False)}]
|
|
168
|
+
s = ""
|
|
169
|
+
for mt in (1200, 2600): # 推理模型额度被 reasoning 吃光 → 空 content,故加额重试
|
|
170
|
+
content, _u = _llm(msg, max_tokens=mt)
|
|
171
|
+
s = content or ""
|
|
172
|
+
s = re.sub(r"```.*?```", " ", s, flags=re.S)
|
|
173
|
+
s = re.sub(r"(人物|日期|实体|意图|主题)\s*[::|]", " ", s)
|
|
174
|
+
s = " ".join(s.split()).strip(" ,,、|")
|
|
175
|
+
if s:
|
|
176
|
+
break
|
|
177
|
+
cache[key] = s
|
|
178
|
+
return s
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
# 生效条件:cache 命中 "Q2:"+q["qid"] 时直接返回该缓存值;否则以 ZH_Q_PROMPT+q["question"] 调 _llm(max_tokens=1200),去代码块后 strip(" ,,、") 写入 cache[key],结果为空串时抛 RuntimeError「查询侧关键词为空…」,非空时返回 cache[key]。
|
|
182
|
+
def zh_question(q, cache):
|
|
183
|
+
"""英文问题 → 中文关键词(与写入侧同源;额度须 ≥1200,否则推理烧空 content)。"""
|
|
184
|
+
key = "Q2:" + q["qid"]
|
|
185
|
+
if key in cache:
|
|
186
|
+
return cache[key]
|
|
187
|
+
content, _u = _llm([{"role": "user",
|
|
188
|
+
"content": ZH_Q_PROMPT + q["question"]}], max_tokens=1200)
|
|
189
|
+
s = re.sub(r"```.*?```", " ", content or "", flags=re.S)
|
|
190
|
+
cache[key] = " ".join(s.split()).strip(" ,,、")
|
|
191
|
+
if not cache[key]:
|
|
192
|
+
raise RuntimeError("查询侧关键词为空(额度被 reasoning 吃光)")
|
|
193
|
+
return cache[key]
|
|
194
|
+
|
|
195
|
+
|
|
196
|
+
# ------------------------------------------------------------------ 池
|
|
197
|
+
# 生效条件:模块级常量 POOL 为普通文件时直接返回其中非空行的 json.loads 列表(此时不读 n_q、n_distract、seed);否则以 random.Random(seed)、每组 max(1, n_q//len(GROUPS)) 条(取 q["evidence_turns"][0] 为 gold)采样后追加 rng.sample(rest, min(n_distract, len(rest))) 干扰项,写出 POOL 与 questions.json 并返回 pool。
|
|
198
|
+
def build_pool(n_q, n_distract, seed=7):
|
|
199
|
+
"""采样:n_q 题(分组覆盖)各取 evidence_turns[0] 为 gold,再取干扰。"""
|
|
200
|
+
if os.path.isfile(POOL):
|
|
201
|
+
with open(POOL, encoding="utf-8") as f:
|
|
202
|
+
return [json.loads(l) for l in f if l.strip()]
|
|
203
|
+
rows = list(ec.iter_jsonl(ec.LM_H))
|
|
204
|
+
byid = {r["id"]: r for r in rows}
|
|
205
|
+
qs = ec.load_questions("lm")
|
|
206
|
+
rng = random.Random(seed)
|
|
207
|
+
per = max(1, n_q // len(GROUPS))
|
|
208
|
+
picked, gold_ids, chosen = [], set(), []
|
|
209
|
+
for gname in GROUPS:
|
|
210
|
+
qtypes = GROUPS[gname]
|
|
211
|
+
cand = [q for q in qs if q["qtype"] in qtypes and q.get("evidence_turns")]
|
|
212
|
+
rng.shuffle(cand)
|
|
213
|
+
for q in cand:
|
|
214
|
+
if len([p for p in picked if p["_g"] == gname]) >= per:
|
|
215
|
+
break
|
|
216
|
+
tid = q["evidence_turns"][0]
|
|
217
|
+
if tid not in byid or tid in gold_ids:
|
|
218
|
+
continue
|
|
219
|
+
gold_ids.add(tid)
|
|
220
|
+
picked.append({**byid[tid], "_g": gname, "_gold": True})
|
|
221
|
+
chosen.append(q)
|
|
222
|
+
pool = list(picked)
|
|
223
|
+
rest = [r for r in rows if r["id"] not in gold_ids]
|
|
224
|
+
pool += rng.sample(rest, min(n_distract, len(rest)))
|
|
225
|
+
os.makedirs(DIR, exist_ok=True)
|
|
226
|
+
with open(POOL, "w", encoding="utf-8") as f:
|
|
227
|
+
for r in pool:
|
|
228
|
+
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
229
|
+
with open(os.path.join(DIR, "questions.json"), "w", encoding="utf-8") as f:
|
|
230
|
+
json.dump(chosen, f, ensure_ascii=False, indent=1)
|
|
231
|
+
return pool
|
|
232
|
+
|
|
233
|
+
|
|
234
|
+
# 生效条件:DIR 目录下的 questions.json 是 UTF-8 合法 JSON 时,返回 json.load 得到的对象。
|
|
235
|
+
def load_questions_zh():
|
|
236
|
+
with open(os.path.join(DIR, "questions.json"), encoding="utf-8") as f:
|
|
237
|
+
return json.load(f)
|
|
238
|
+
|
|
239
|
+
|
|
240
|
+
# ------------------------------------------------------------------ 建库 / 评测
|
|
241
|
+
# 生效条件:以 ec.build_eval_cg(None, root, corpus or POOL, ec.lm_turn_text, "lmezh", calib_of=calib if zh_of 为真值 else None, rebuild=rebuild) 执行并返回其结果(形参 pool 在该调用中未被使用,实际用 corpus or POOL;zh_only 只在内层 calib 中生效,不传给 build_eval_cg)。
|
|
242
|
+
def build(root, pool, zh_of=None, rebuild=False, zh_only=False, corpus=None):
|
|
243
|
+
"""zh_of: turn_id → 中文层;None = 裸文本库。
|
|
244
|
+
|
|
245
|
+
zh_only=True → **纯中文索引**:正文只放中文层,英文原文不入库(供「中文检索、
|
|
246
|
+
原文只做载荷回填」架构验证)。无中文层的行退化为裸英文——本轮 180 条干扰项
|
|
247
|
+
未译,故 E 臂存在「语言隔离」:中文查询几乎不可能命中英文干扰项,其 hit 是
|
|
248
|
+
上界而非全译库真值。故另设 F 臂(池只含 20 条 gold、正文全中文)隔离该假象,
|
|
249
|
+
单测中文层自身的可区分性。
|
|
250
|
+
"""
|
|
251
|
+
# 生效条件:作为 build 的内层函数闭包使用 zh_of/zh_only,ctx 未被使用;zh_of 为真值且 zh_of.get(r["id"]) 为真值时 cs = _cond_of(z) or cs,否则 cs = dict(_DEFAULT_COND);zh_only 为真值时返回 (z or ec.lm_turn_text(r), [], cs),否则返回 (ec.lm_turn_text(r) 在 z 为真值时再拼 "\n"+z, [], cs)。
|
|
252
|
+
def calib(r, ctx=None):
|
|
253
|
+
z = zh_of.get(r["id"]) if zh_of else None
|
|
254
|
+
cs = dict(_DEFAULT_COND)
|
|
255
|
+
if z:
|
|
256
|
+
cs = _cond_of(z) or cs # 条件空间逐条来自中文层,而非全局写死
|
|
257
|
+
if zh_only:
|
|
258
|
+
return (z or ec.lm_turn_text(r)), [], cs
|
|
259
|
+
body = ec.lm_turn_text(r)
|
|
260
|
+
if z:
|
|
261
|
+
body = body + "\n" + z
|
|
262
|
+
return body, [], cs
|
|
263
|
+
return ec.build_eval_cg(None, root, corpus or POOL, ec.lm_turn_text, "lmezh",
|
|
264
|
+
calib_of=calib if zh_of else None, rebuild=rebuild)
|
|
265
|
+
|
|
266
|
+
|
|
267
|
+
# 生效条件:给定 cg、questions 与 qtext_of 时,对每题用 qtext_of(q) 替换 question 后经 ec.evaluate_group 评测,并按 k 汇总为结果。
|
|
268
|
+
def eval_arm(cg, questions, qtext_of, k=5):
|
|
269
|
+
rows = []
|
|
270
|
+
for q in questions:
|
|
271
|
+
qq = dict(q)
|
|
272
|
+
qq["question"] = qtext_of(q)
|
|
273
|
+
rows += ec.evaluate_group(cg, [qq], k=k, paths=ec.PATHS, verbose=False)
|
|
274
|
+
return ec.summarize(rows, k=k)
|
|
275
|
+
|
|
276
|
+
|
|
277
|
+
# 生效条件:argv 由 argparse 解析(--n-q/--n-distract/--k/--build/--rebuild/--workers 等);--build 只产池即返回,否则需外部数据在指定路径就位、缺失即抛异常不静默降级;
|
|
278
|
+
def main():
|
|
279
|
+
ap = argparse.ArgumentParser()
|
|
280
|
+
ap.add_argument("--n-q", type=int, default=20)
|
|
281
|
+
ap.add_argument("--n-distract", type=int, default=180)
|
|
282
|
+
ap.add_argument("--k", type=int, default=5)
|
|
283
|
+
ap.add_argument("--build", action="store_true", help="只生成池与中文层")
|
|
284
|
+
ap.add_argument("--rebuild", action="store_true")
|
|
285
|
+
ap.add_argument("--workers", type=int, default=6)
|
|
286
|
+
args = ap.parse_args()
|
|
287
|
+
|
|
288
|
+
pool = build_pool(args.n_q, args.n_distract)
|
|
289
|
+
golds = [r for r in pool if r.get("_gold")]
|
|
290
|
+
qs = load_questions_zh()
|
|
291
|
+
print(f"池 {len(pool)} 条(gold {len(golds)} / 干扰 "
|
|
292
|
+
f"{len(pool) - len(golds)}),题 {len(qs)} 道,随机基线 hit@1="
|
|
293
|
+
f"{len(golds) / len(pool):.1%}")
|
|
294
|
+
|
|
295
|
+
cache = _load_cache()
|
|
296
|
+
man_zh, man_q = _load_manual()
|
|
297
|
+
for tid, v in man_zh.items(): # 会话模型产出 → 直接进缓存,零 API
|
|
298
|
+
cache["L2:" + tid] = v
|
|
299
|
+
for qid, v in man_q.items():
|
|
300
|
+
cache["Q2:" + qid] = v
|
|
301
|
+
print(f"中文层:本地 {len(man_zh)} 条(查询词本地 {len(man_q)} 条),"
|
|
302
|
+
f"缓存合计 {len(cache)} 条,模型 {MODEL}(本轮不调用)")
|
|
303
|
+
|
|
304
|
+
# 生效条件:对 r 调 zh_layer(r, cache)(cache 为闭包变量)成功时返回 (r["id"], 中文层, None);zh_layer 抛任何异常时返回 (r["id"], "", f"{type(exc).__name__}: {exc}")。
|
|
305
|
+
def work(r):
|
|
306
|
+
try:
|
|
307
|
+
return r["id"], zh_layer(r, cache), None
|
|
308
|
+
except Exception as exc: # noqa: BLE001
|
|
309
|
+
return r["id"], "", f"{type(exc).__name__}: {exc}"
|
|
310
|
+
|
|
311
|
+
zh = {}
|
|
312
|
+
with ThreadPoolExecutor(max_workers=args.workers) as ex:
|
|
313
|
+
for i, (tid, s, err) in enumerate(ex.map(work, golds), 1):
|
|
314
|
+
if err:
|
|
315
|
+
print(f" [{i}/{len(golds)}] {tid} 失败:{err}")
|
|
316
|
+
else:
|
|
317
|
+
zh[tid] = s
|
|
318
|
+
_save_cache(cache)
|
|
319
|
+
okzh = [v for v in zh.values() if v]
|
|
320
|
+
print(f"中文层成功 {len(okzh)}/{len(golds)},平均长度 "
|
|
321
|
+
f"{sum(len(v) for v in okzh) / max(1, len(okzh)):.1f} 字")
|
|
322
|
+
for t in okzh[:5]:
|
|
323
|
+
print(" ·", t[:110])
|
|
324
|
+
|
|
325
|
+
if args.build:
|
|
326
|
+
print("(--build 只生成,不评测)")
|
|
327
|
+
return
|
|
328
|
+
if not okzh:
|
|
329
|
+
print("中文层为空,终止")
|
|
330
|
+
return 1
|
|
331
|
+
|
|
332
|
+
qzh = {}
|
|
333
|
+
for q in qs:
|
|
334
|
+
try:
|
|
335
|
+
qzh[q["qid"]] = zh_question(q, cache)
|
|
336
|
+
except Exception as exc: # noqa: BLE001
|
|
337
|
+
print(f" 问题翻译失败 {q['qid']}:{exc}")
|
|
338
|
+
qzh[q["qid"]] = q["question"]
|
|
339
|
+
_save_cache(cache)
|
|
340
|
+
|
|
341
|
+
en = lambda q: q["question"]
|
|
342
|
+
zhq = lambda q: qzh.get(q["qid"]) or q["question"]
|
|
343
|
+
|
|
344
|
+
# 只含 gold 的 20 条池:E 臂的「语言隔离」假象在其中被消除(全部是中文文档)
|
|
345
|
+
with open(POOL20, "w", encoding="utf-8") as f:
|
|
346
|
+
for r in golds:
|
|
347
|
+
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
348
|
+
|
|
349
|
+
import md_cg.mdcg as m
|
|
350
|
+
results = {}
|
|
351
|
+
arms = (
|
|
352
|
+
("A legacy 裸+英问", ROOT_A, POOL, False, False, en, "legacy"),
|
|
353
|
+
("B jaccard 裸+英问", ROOT_A, POOL, False, False, en, "jaccard"),
|
|
354
|
+
("C jaccard 中+英问", ROOT_C, POOL, True, False, en, "jaccard"),
|
|
355
|
+
("D jaccard 中+中问", ROOT_C, POOL, True, False, zhq, "jaccard"),
|
|
356
|
+
("E 纯中索引+中问", ROOT_E, POOL, True, True, zhq, "jaccard"),
|
|
357
|
+
("F 纯中池+中问", ROOT_F, POOL20, True, True, zhq, "jaccard"),
|
|
358
|
+
("G 纯中池+英问", ROOT_F, POOL20, True, True, en, "jaccard"),
|
|
359
|
+
)
|
|
360
|
+
for tag, root, corpus, usezh, zonly, qf, mode in arms:
|
|
361
|
+
m.SCORE_MODE = mode
|
|
362
|
+
cg = build(root, pool, zh_of=zh if usezh else None, rebuild=args.rebuild,
|
|
363
|
+
zh_only=zonly, corpus=corpus)
|
|
364
|
+
ec.install_read_cache(cg)
|
|
365
|
+
results[tag] = eval_arm(cg, qs, qf, k=args.k)
|
|
366
|
+
print(f" {tag}: hit@1 {results[tag]['hit@1']:.1%} "
|
|
367
|
+
f"hit@{args.k} {results[tag][f'hit@{args.k}']:.1%} "
|
|
368
|
+
f"MRR {results[tag]['mrr']:.3f}")
|
|
369
|
+
|
|
370
|
+
# 命中后回填原始英文原文:原文不进索引、只做载荷 → 验证「中文检索→英文返回」链路
|
|
371
|
+
m.SCORE_MODE = "jaccard"
|
|
372
|
+
cg_e = build(ROOT_E, pool, zh_of=zh, corpus=POOL, zh_only=True)
|
|
373
|
+
ec.install_read_cache(cg_e)
|
|
374
|
+
payload = {r["id"]: ec.lm_turn_text(r) for r in pool}
|
|
375
|
+
line = results["E 纯中索引+中问"]["score_p10"]
|
|
376
|
+
print(f"\n回填验证(中文命中 → 确认匹配 → 返回英文原文;原文不在索引内):"
|
|
377
|
+
f"\n 匹配确认线 = 正例 Top-1 分 p10 = {line:.4f}"
|
|
378
|
+
f"(低于线判未匹配,不回填原文)")
|
|
379
|
+
conf_n = conf_hit = 0
|
|
380
|
+
for q in qs:
|
|
381
|
+
res, _meta = ec.run_query(cg_e, zhq(q), k=1, paths=ec.PATHS)
|
|
382
|
+
sc = res[0][1] if res else 0.0
|
|
383
|
+
if sc >= line:
|
|
384
|
+
conf_n += 1
|
|
385
|
+
if ec.first_evidence_rank(res, set(q.get("evidence_turns") or [])):
|
|
386
|
+
conf_hit += 1
|
|
387
|
+
print(f" 确认匹配 {conf_n}/{len(qs)} 题,其中真命中 {conf_hit} → "
|
|
388
|
+
f"确认后精度 {conf_hit / max(1, conf_n):.1%}")
|
|
389
|
+
for q in qs[:2]:
|
|
390
|
+
res, _meta = ec.run_query(cg_e, zhq(q), k=2, paths=ec.PATHS)
|
|
391
|
+
print(f" 中文查询: {zhq(q)}")
|
|
392
|
+
for it in res[:2]:
|
|
393
|
+
nid, sc = it[0]["id"], it[1]
|
|
394
|
+
verdict = "确认匹配→回填" if sc >= line else "未确认→不回填"
|
|
395
|
+
print(f" · {nid} 分={sc:.4f} [{verdict}] 载荷原文: "
|
|
396
|
+
f"{payload.get(nid, '')[:70]}")
|
|
397
|
+
|
|
398
|
+
ec.print_table("LongMemEval-S 中文层探针(%d 题 / 池 %d)"
|
|
399
|
+
% (len(qs), len(pool)), results, k=args.k)
|
|
400
|
+
ec.save_result("zh_probe.json", {
|
|
401
|
+
"dataset": "LongMemEval-S slice", "model": MODEL,
|
|
402
|
+
"pool": len(pool), "n_q": len(qs), "baseline_hit1": len(golds) / len(pool),
|
|
403
|
+
"arms": results,
|
|
404
|
+
"zh_layer_samples": okzh[:20],
|
|
405
|
+
"questions_zh": qzh,
|
|
406
|
+
})
|
|
407
|
+
return 0
|
|
408
|
+
|
|
409
|
+
|
|
410
|
+
if __name__ == "__main__":
|
|
411
411
|
sys.exit(main())
|