@furongjun1999/dsh-memory 0.4.11 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -16
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +142 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/hooks.js +36 -2
- package/lib/lib/roleplay_web.js +427 -427
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +368 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1327 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +285 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1005 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +300 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1536 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1097 -1097
- package/md_cg/crypto.py +437 -437
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +334 -334
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +580 -580
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +220 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +329 -329
- package/md_cg/hotcache.py +238 -214
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +199 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +622 -622
- package/md_cg/mcp_server.py +129 -32
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +176 -117
- package/md_cg/mdcos.py +79 -13
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +693 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +298 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +85 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +365 -365
- package/md_cg/scrub.py +852 -852
- package/md_cg/security.py +274 -274
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +151 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +562 -562
- package/md_cg/sources.py +815 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +48 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +1138 -1138
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branches.py +249 -249
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_en_pipeline.py +166 -166
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_identity_attribution.py +147 -147
- package/md_cg/test_index_durability.py +224 -224
- package/md_cg/test_interop.py +93 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +765 -765
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +298 -298
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +113 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +281 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_readcache_prodpath.py +155 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +340 -340
- package/md_cg/test_retr_s1b.py +209 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +384 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +175 -175
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +241 -241
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +397 -397
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +273 -273
- package/md_cg/tokens.py +677 -663
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +667 -667
- package/md_cg/vision_evidence.py +666 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +550 -542
- package/package.json +97 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +401 -401
- package/src/hooks.ts +38 -2
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +932 -932
- package/src/lib/token_store.ts +192 -192
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/bench_locomo_zh.py
CHANGED
|
@@ -1,451 +1,451 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""LoCoMo 千题规模中文层四项能力探针(zh_mad 的 LoCoMo 扩展)。
|
|
3
|
-
|
|
4
|
-
与 `bench_zh_mad.py` 的关系:后者是 20 条 gold turn 的**方向性**探针(池 20 条,
|
|
5
|
-
±1 题 = ±5pp)。本脚本把同一套写入侧四轴消融搬到 **LoCoMo 全证据覆盖**规模:
|
|
6
|
-
|
|
7
|
-
轴 1 实体规范化 —— 全库 df 聚合出的 canonical 词 + 人物/地点/时间 → tags(entity 路)
|
|
8
|
-
轴 2 指代消解 —— **仅当本条自身无实体**时承接**前一条自身**的 canonical
|
|
9
|
-
轴 3 意图抽象 —— 摘要 → 规范意图类目;类目 + 动词原形 → tags,类目 → 正文子功能行
|
|
10
|
-
轴 4 关系图遍历 —— 人物/事件/时间/身份/地点/条件 多维关系 → edges(graph 路)
|
|
11
|
-
|
|
12
|
-
本脚本**复用** `bench_zh_mad` 的 `parse_zh` / `build_tables` / `axis_values` /
|
|
13
|
-
`normalize_terms` / `intent_of` / `build_arm`——保证中文层 schema 与四轴实现
|
|
14
|
-
与 20 条基准**逐位同源**,差异只在语料规模与题库。
|
|
15
|
-
|
|
16
|
-
数据集与规模(使用者已裁决):
|
|
17
|
-
* 数据集:LoCoMo(mteb/LoCoMo BEIR 转制版)500 题。
|
|
18
|
-
* 写入侧口径:**全证据覆盖**——所有 `evidence_turns` 去重后的 turn 全部入池。
|
|
19
|
-
实测:500 题引用的证据 turn 去重 568 个,其中 **567 个在语料中存在**,
|
|
20
|
-
`scene_3_session_10_turn_19` 为数据集悬空引用(语料无此 turn,见「数据完整性」
|
|
21
|
-
条款)。
|
|
22
|
-
* 干扰项:**第一轮不加**(池仅证据 turn)。池内每一条都是某题的 gold →
|
|
23
|
-
「零干扰」理想条件,指标为**上界**,报告必须声明(干扰池为独立对照轮次)。
|
|
24
|
-
* 查询侧:500 题的中文查询词。
|
|
25
|
-
|
|
26
|
-
数据完整性条款(诚实条款,报告必须原样声明):
|
|
27
|
-
* LoCoMo 的 BEIR 转制版中,题 `scene_3_q_58`(multi_hop)引用的
|
|
28
|
-
`scene_3_session_10_turn_19` 在 `locomo_corpus.jsonl` 中**不存在**——
|
|
29
|
-
数据集自身的悬空引用。本题另有 6 条有效证据,故不影响其可命中性;
|
|
30
|
-
但「证据 turn 总数」准确值是 **567** 而非 568。
|
|
31
|
-
|
|
32
|
-
诚实条款(报告必须原样声明):
|
|
33
|
-
* **中文层与中文查询均为本次新增产出**(会话模型逐条产出),**非既有真源**——
|
|
34
|
-
与 20 条基准(`manual_zh.json` / `manual_q.json`,既有产物)性质不同。
|
|
35
|
-
* 写入侧加工(四轴)仍是**确定性纯规则**(零 LLM、零第三方依赖、同输入必同输出),
|
|
36
|
-
词表随源码公开,第三方可重放。
|
|
37
|
-
* 加工**盲于查询集**:canonical 判定只用全库 df 与中文层自身字段,
|
|
38
|
-
不读 `zh_queries/` 的任何内容——否则即数据泄漏,评测作废。
|
|
39
|
-
* 池 567 条、题 500 道;随机基线 hit@1 = 1/567 ≈ 0.18%。但池内**全是 gold**,
|
|
40
|
-
故 0.18% 不是有意义的基线——「零干扰」才是本轮真正的条件限制。
|
|
41
|
-
|
|
42
|
-
跑法:
|
|
43
|
-
python -m md_cg.bench_locomo_zh dump # 派生 raw_turns/raw_questions(零标注)
|
|
44
|
-
python -m md_cg.bench_locomo_zh status # 产出进度与缺口
|
|
45
|
-
python -m md_cg.bench_locomo_zh prepare # 合并中文层 → corpus567/questions500
|
|
46
|
-
python -m md_cg.bench_locomo_zh build # 建 5 臂
|
|
47
|
-
python -m md_cg.bench_locomo_zh run # Rust 评测(--dataset lc,5 组)
|
|
48
|
-
python -m md_cg.bench_locomo_zh seed # graph 种子口径对照
|
|
49
|
-
"""
|
|
50
|
-
import json
|
|
51
|
-
import os
|
|
52
|
-
import re
|
|
53
|
-
import sys
|
|
54
|
-
|
|
55
|
-
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
56
|
-
for _p in (HERE, os.path.join(HERE, "md_cg")):
|
|
57
|
-
if _p not in sys.path:
|
|
58
|
-
sys.path.insert(0, _p)
|
|
59
|
-
|
|
60
|
-
from md_cg import bench_zh_mad as bz # noqa: E402 复用四轴实现(同源保证)
|
|
61
|
-
|
|
62
|
-
EXT = os.path.join(HERE, "data", "external")
|
|
63
|
-
LC = os.path.join(EXT, "locomo")
|
|
64
|
-
WORK = os.path.join(EXT, "locomo_zh")
|
|
65
|
-
ZH_TURNS = os.path.join(WORK, "zh_turns") # 分批落盘:{turn_id: 中文层串}
|
|
66
|
-
ZH_QUERIES = os.path.join(WORK, "zh_queries") # 分批落盘:{qid: 中文查询词}
|
|
67
|
-
|
|
68
|
-
RAW_TURNS = os.path.join(WORK, "raw_turns.jsonl")
|
|
69
|
-
RAW_QUESTIONS = os.path.join(WORK, "raw_questions.jsonl")
|
|
70
|
-
CORPUS567 = os.path.join(WORK, "corpus567.jsonl")
|
|
71
|
-
QUESTIONS500 = os.path.join(WORK, "questions500.jsonl")
|
|
72
|
-
|
|
73
|
-
# LoCoMo 组映射(与 rust/src/main.rs `groups_of("lc")` 逐项一致)
|
|
74
|
-
GROUPS = ("precise", "temporal", "interference", "negative", "reference")
|
|
75
|
-
POS_GROUPS = ("precise", "temporal", "interference")
|
|
76
|
-
QTYPE_OF_GROUP = {
|
|
77
|
-
"precise": "single_hop",
|
|
78
|
-
"temporal": "temporal_reasoning",
|
|
79
|
-
"interference": "multi_hop",
|
|
80
|
-
"negative": "adversarial",
|
|
81
|
-
"reference": "open_domain",
|
|
82
|
-
}
|
|
83
|
-
|
|
84
|
-
# 已知的悬空引用(数据集自身缺陷,见模块头数据完整性条款)
|
|
85
|
-
KNOWN_DANGLING = {"scene_3_session_10_turn_19"}
|
|
86
|
-
|
|
87
|
-
# graph 边的 df 上限:自由参数,随池规模标定(见 rescale 说明)。
|
|
88
|
-
# 20 条基准用 5(= 25% 池);本脚本缺省按 5% 池标定,另跑固定 5 作敏感性对照。
|
|
89
|
-
MAX_DF_FIXED = 5
|
|
90
|
-
|
|
91
|
-
|
|
92
|
-
# 生效条件:任意 n_pool(含 0 等假值,源码不做校验)下返回 max(MAX_DF_FIXED, int(round(0.05 * n_pool))),结果不小于模块级常量 MAX_DF_FIXED。
|
|
93
|
-
def max_df_auto(n_pool):
|
|
94
|
-
"""池规模 → edges 的 df 上限。
|
|
95
|
-
|
|
96
|
-
原值 5 是 20 条池上的**绝对**阈值(= 池的 25%)。thousand-scale 直接照搬会
|
|
97
|
-
把常见词(人名/常用名词)全部过滤 → 图路无边可走;反之过宽则生成巨型团
|
|
98
|
-
(「全连通,零信息量只注入噪声」)。此处按池规模取 5% 作为标定值,
|
|
99
|
-
并在报告中附 max_df=5 的敏感性列——**结论不建立在单一参数上**。
|
|
100
|
-
"""
|
|
101
|
-
return max(MAX_DF_FIXED, int(round(0.05 * n_pool)))
|
|
102
|
-
|
|
103
|
-
|
|
104
|
-
# 生效条件:path 指向逐行 JSON 的 UTF-8 文本时,跳过空行并逐行 yield json.loads 结果。
|
|
105
|
-
def iter_jsonl(path):
|
|
106
|
-
with open(path, encoding="utf-8") as f:
|
|
107
|
-
for line in f:
|
|
108
|
-
line = line.strip()
|
|
109
|
-
if line:
|
|
110
|
-
yield json.loads(line)
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
# 生效条件:path 能被 open(path, encoding="utf-8") 打开时返回 json.load(f) 的解析结果。
|
|
114
|
-
def load_json(path):
|
|
115
|
-
with open(path, encoding="utf-8") as f:
|
|
116
|
-
return json.load(f)
|
|
117
|
-
|
|
118
|
-
|
|
119
|
-
# 生效条件:os.path.isdir(d) 为假时直接返回空 dict;为真时按 sorted(os.listdir(d)) 顺序读取后缀为 .json 的分片并 out.update(part)(后者覆盖前者),任一分片不是 dict 时抛 SystemExit,否则返回合并后的 out。
|
|
120
|
-
def load_chunks(d):
|
|
121
|
-
"""合并目录下全部分片 JSON(后者覆盖前者,便于修订单条)。"""
|
|
122
|
-
out = {}
|
|
123
|
-
if not os.path.isdir(d):
|
|
124
|
-
return out
|
|
125
|
-
for name in sorted(os.listdir(d)):
|
|
126
|
-
if not name.endswith(".json"):
|
|
127
|
-
continue
|
|
128
|
-
part = load_json(os.path.join(d, name))
|
|
129
|
-
if not isinstance(part, dict):
|
|
130
|
-
raise SystemExit(f"[失败] 分片非对象:{os.path.join(d, name)}")
|
|
131
|
-
out.update(part)
|
|
132
|
-
return out
|
|
133
|
-
|
|
134
|
-
|
|
135
|
-
# ------------------------------------------------------------------ dump
|
|
136
|
-
_SPEAKER_RE = re.compile(r"^([A-Za-z][A-Za-z .'\-]{0,24}): ")
|
|
137
|
-
_DATE_RE = re.compile(r"Data time: (\d{1,2}:\d{2} [AP]M on \w+ \d{1,2} \w+, \d{4})")
|
|
138
|
-
|
|
139
|
-
|
|
140
|
-
# 生效条件:调用即从 locomo_corpus.jsonl 建 id→记录映射、读 locomo_questions.jsonl,逐题取 q.get("evidence_turns") or [](缺键或假值按空列表)去重排序为证据 turn,text/title 用 c.get(...) or "" 兜假值、speaker/date 由 _SPEAKER_RE.match / _DATE_RE.search 命中才非空;verbose=True 时额外打印含 KNOWN_DANGLING 未登记告警与直接取键 q["qtype"] 的题型统计,verbose=False 时只跳过打印,写 RAW_TURNS/RAW_QUESTIONS 与返回 (rows, questions) 不变。
|
|
141
|
-
def cmd_dump(verbose=True):
|
|
142
|
-
"""确定性派生:证据 turn 全集 + 题库 → raw_turns/raw_questions(零标注)。"""
|
|
143
|
-
os.makedirs(ZH_TURNS, exist_ok=True)
|
|
144
|
-
os.makedirs(ZH_QUERIES, exist_ok=True)
|
|
145
|
-
corpus = {r["id"]: r for r in iter_jsonl(os.path.join(LC, "locomo_corpus.jsonl"))}
|
|
146
|
-
questions = list(iter_jsonl(os.path.join(LC, "locomo_questions.jsonl")))
|
|
147
|
-
|
|
148
|
-
ev = sorted({t for q in questions for t in (q.get("evidence_turns") or [])})
|
|
149
|
-
missing = sorted(t for t in ev if t not in corpus)
|
|
150
|
-
rows = []
|
|
151
|
-
for tid in ev:
|
|
152
|
-
if tid not in corpus:
|
|
153
|
-
continue
|
|
154
|
-
c = corpus[tid]
|
|
155
|
-
text = str(c.get("text") or "")
|
|
156
|
-
title = str(c.get("title") or "")
|
|
157
|
-
m = _SPEAKER_RE.match(text)
|
|
158
|
-
d = _DATE_RE.search(title)
|
|
159
|
-
rows.append({"id": tid, "text": text, "speaker": m.group(1) if m else "",
|
|
160
|
-
"date": d.group(1) if d else "", "title": title})
|
|
161
|
-
|
|
162
|
-
with open(RAW_TURNS, "w", encoding="utf-8") as f:
|
|
163
|
-
for r in rows:
|
|
164
|
-
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
165
|
-
with open(RAW_QUESTIONS, "w", encoding="utf-8") as f:
|
|
166
|
-
for q in questions:
|
|
167
|
-
f.write(json.dumps(q, ensure_ascii=False) + "\n")
|
|
168
|
-
|
|
169
|
-
if verbose:
|
|
170
|
-
print(f"证据 turn:去重 {len(ev)},语料中存在 {len(rows)},悬空 {missing}")
|
|
171
|
-
unknown = [t for t in missing if t not in KNOWN_DANGLING]
|
|
172
|
-
if unknown:
|
|
173
|
-
print(f"[警告] 出现未登记的悬空引用:{unknown}")
|
|
174
|
-
print(f"→ {RAW_TURNS}({len(rows)} 条)")
|
|
175
|
-
print(f"→ {RAW_QUESTIONS}({len(questions)} 道)")
|
|
176
|
-
ntype = {}
|
|
177
|
-
for q in questions:
|
|
178
|
-
ntype[q["qtype"]] = ntype.get(q["qtype"], 0) + 1
|
|
179
|
-
print("题型分布:" + ", ".join(f"{k}={v}" for k, v in sorted(ntype.items())))
|
|
180
|
-
return rows, questions
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
# ------------------------------------------------------------------ status
|
|
184
|
-
# 生效条件:verbose=True 时打印写入/查询侧完成度、多余 turn/qid 键与待补项(各截取前 head 项),并对已产出中文层用 bz.parse_zh 检查 identity/time/summary/terms 四项真值(并非五槽),非全真者记入 bad;verbose=False 时不打印、不做该槽位检查,返回值中 bad 恒为 []。
|
|
185
|
-
def cmd_status(verbose=True, head=12):
|
|
186
|
-
"""产出进度:中文层 / 中文查询的覆盖与缺口(分批产出时用)。"""
|
|
187
|
-
turns, questions = cmd_dump(verbose=False)
|
|
188
|
-
zt = load_chunks(ZH_TURNS)
|
|
189
|
-
zq = load_chunks(ZH_QUERIES)
|
|
190
|
-
t_ids = [r["id"] for r in turns]
|
|
191
|
-
q_ids = [q["qid"] for q in questions]
|
|
192
|
-
miss_t = [t for t in t_ids if t not in zt]
|
|
193
|
-
miss_q = [q for q in q_ids if q not in zq]
|
|
194
|
-
extra_t = [k for k in zt if k not in set(t_ids)]
|
|
195
|
-
extra_q = [k for k in zq if k not in set(q_ids)]
|
|
196
|
-
if verbose:
|
|
197
|
-
print(f"写入侧中文层:{len(t_ids) - len(miss_t)}/{len(t_ids)}"
|
|
198
|
-
f"(缺 {len(miss_t)})")
|
|
199
|
-
print(f"查询侧中文词:{len(q_ids) - len(miss_q)}/{len(q_ids)}"
|
|
200
|
-
f"(缺 {len(miss_q)})")
|
|
201
|
-
if extra_t:
|
|
202
|
-
print(f"[警告] 多余 turn 键(不在证据集内):{extra_t[:head]}")
|
|
203
|
-
if extra_q:
|
|
204
|
-
print(f"[警告] 多余 qid 键(不在题库内):{extra_q[:head]}")
|
|
205
|
-
if miss_t:
|
|
206
|
-
print(f" 待补 turn(前 {head}):{miss_t[:head]}")
|
|
207
|
-
if miss_q:
|
|
208
|
-
print(f" 待补 qid(前 {head}):{miss_q[:head]}")
|
|
209
|
-
# 结构校验:已产出的中文层必须能被 parse_zh 解出五槽
|
|
210
|
-
bad = []
|
|
211
|
-
for tid in t_ids:
|
|
212
|
-
if tid not in zt:
|
|
213
|
-
continue
|
|
214
|
-
f = bz.parse_zh(zt[tid])
|
|
215
|
-
if not (f["identity"] and f["time"] and f["summary"] and f["terms"]):
|
|
216
|
-
bad.append(tid)
|
|
217
|
-
if bad:
|
|
218
|
-
print(f"[警告] 中文层槽位不全({len(bad)} 条):{bad[:head]}")
|
|
219
|
-
return {"n_turn": len(t_ids), "n_turn_done": len(t_ids) - len(miss_t),
|
|
220
|
-
"n_q": len(q_ids), "n_q_done": len(q_ids) - len(miss_q),
|
|
221
|
-
"miss_turns": miss_t, "miss_qids": miss_q,
|
|
222
|
-
"bad": bad if verbose else []}
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
# ------------------------------------------------------------------ prepare
|
|
226
|
-
# 生效条件:miss_t 或 miss_q 任一非空即 raise SystemExit,否则按 sort_key(仅接受 scene_数字_session_数字_turn_数字 形式的 id,否则 raise SystemExit)对 turns 数值序排序,写出 corpus567.jsonl/questions500.jsonl 并返回 (corpus, out_q),其中 answer 与 evidence_turns 为假值时分别落为 "" 与 [],verbose 只控制末尾打印。
|
|
227
|
-
def cmd_prepare(verbose=True):
|
|
228
|
-
"""合并中文层 → corpus567.jsonl + questions500.jsonl(幂等覆盖)。"""
|
|
229
|
-
turns, questions = cmd_dump(verbose=False)
|
|
230
|
-
zt = load_chunks(ZH_TURNS)
|
|
231
|
-
zq = load_chunks(ZH_QUERIES)
|
|
232
|
-
|
|
233
|
-
miss_t = [r["id"] for r in turns if r["id"] not in zt]
|
|
234
|
-
miss_q = [q["qid"] for q in questions if q["qid"] not in zq]
|
|
235
|
-
if miss_t or miss_q:
|
|
236
|
-
raise SystemExit(
|
|
237
|
-
f"[失败] 中文层未产出完:缺 {len(miss_t)} 条 turn、{len(miss_q)} 道题。"
|
|
238
|
-
f"先跑 `status` 看缺口。")
|
|
239
|
-
|
|
240
|
-
# 语料:按 (scene, session, turn) 数值序 —— 「承接前一条」需要确定的时间序
|
|
241
|
-
# 生效条件:tid 匹配 ^scene_(\d+)_session_(\d+)_turn_(\d+)$ 时返回三个整数的 tuple(scene, session, turn),否则 raise SystemExit(不做其他容错或回退)。
|
|
242
|
-
def sort_key(tid):
|
|
243
|
-
m = re.match(r"scene_(\d+)_session_(\d+)_turn_(\d+)$", tid)
|
|
244
|
-
if not m:
|
|
245
|
-
raise SystemExit(f"[失败] turn id 无法解析:{tid}")
|
|
246
|
-
return tuple(int(x) for x in m.groups())
|
|
247
|
-
|
|
248
|
-
corpus = []
|
|
249
|
-
for r in sorted(turns, key=lambda r: sort_key(r["id"])):
|
|
250
|
-
raw = zt[r["id"]]
|
|
251
|
-
corpus.append({
|
|
252
|
-
"id": r["id"], "text": r["text"], "speaker": r["speaker"],
|
|
253
|
-
"date": r["date"], "zh": raw, "zh_fields": bz.parse_zh(raw),
|
|
254
|
-
})
|
|
255
|
-
|
|
256
|
-
out_q = []
|
|
257
|
-
for q in questions:
|
|
258
|
-
out_q.append({
|
|
259
|
-
"qid": q["qid"], "qtype": q["qtype"],
|
|
260
|
-
"question": zq[q["qid"]], "answer": str(q.get("answer") or ""),
|
|
261
|
-
"evidence_turns": list(q.get("evidence_turns") or []),
|
|
262
|
-
})
|
|
263
|
-
|
|
264
|
-
with open(CORPUS567, "w", encoding="utf-8") as f:
|
|
265
|
-
for r in corpus:
|
|
266
|
-
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
267
|
-
with open(QUESTIONS500, "w", encoding="utf-8") as f:
|
|
268
|
-
for q in out_q:
|
|
269
|
-
f.write(json.dumps(q, ensure_ascii=False) + "\n")
|
|
270
|
-
|
|
271
|
-
if verbose:
|
|
272
|
-
print(f"语料 {len(corpus)} 条 → {CORPUS567}")
|
|
273
|
-
print(f"题库 {len(out_q)} 条 → {QUESTIONS500}")
|
|
274
|
-
ntype = {}
|
|
275
|
-
for q in out_q:
|
|
276
|
-
ntype[q["qtype"]] = ntype.get(q["qtype"], 0) + 1
|
|
277
|
-
print("题型分布:" + ", ".join(f"{k}={v}" for k, v in sorted(ntype.items())))
|
|
278
|
-
lt = sum(len(r["zh_fields"]["terms"]) for r in corpus)
|
|
279
|
-
lc_ = sum(len(r["zh"]) for r in corpus)
|
|
280
|
-
print(f"中文层:词项合计 {lt}(均 {lt / len(corpus):.1f}/条),"
|
|
281
|
-
f"字符合计 {lc_}(均 {lc_ / len(corpus):.0f}/条)")
|
|
282
|
-
return corpus, out_q
|
|
283
|
-
|
|
284
|
-
|
|
285
|
-
# ------------------------------------------------------------------ build
|
|
286
|
-
# 生效条件:max_df 为 None(默认)时用 max_df_auto(len(corpus)) 自动取值,max_df 传出值(含 0 等假值)则直接使用;随后对 bz.ARMS 每臂调 bz.build_arm(corpus, arm, max_df=md, root_base=...) 并返回 {arm["name"]: 建库结果};verbose 形参在该符号源码段内未参与如何分支。
|
|
287
|
-
def cmd_build(max_df=None, verbose=True):
|
|
288
|
-
corpus, _ = cmd_prepare(verbose=False)
|
|
289
|
-
md = max_df if max_df is not None else max_df_auto(len(corpus))
|
|
290
|
-
print(f"== 建库(消融 {len(bz.ARMS)} 臂,池 {len(corpus)} 条,"
|
|
291
|
-
f"max_df={md}(自动,5% 池;固定基线 {MAX_DF_FIXED}))==")
|
|
292
|
-
roots = {}
|
|
293
|
-
for arm in bz.ARMS:
|
|
294
|
-
roots[arm["name"]] = bz.build_arm(
|
|
295
|
-
corpus, arm, max_df=md, root_base=f"_md_cg_eval_lczh_{arm['name']}")
|
|
296
|
-
return roots
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
# ------------------------------------------------------------------ run
|
|
300
|
-
RUST_BIN = os.path.join(HERE, "rust", "target", "release", "mdcg-eval.exe")
|
|
301
|
-
ROW_RE = re.compile(
|
|
302
|
-
r"^\s*(precise|temporal|interference|negative|reference)\s+(\d+)\s+"
|
|
303
|
-
r"([\d.]+)%\s+([\d.]+)%\s+([\d.]+)\s*$", re.M)
|
|
304
|
-
GATE_RE = re.compile(r"拒答率:([\d.]+)%")
|
|
305
|
-
LINE_RE = re.compile(r"拒答线(正例 hit@1 题 Top-1 分 p10):([\d.]+)")
|
|
306
|
-
|
|
307
|
-
|
|
308
|
-
# 生效条件:RUST_BIN 经 os.path.exists 为假时抛 SystemExit;否则以 argv 列表 [RUST_BIN, "--dataset", "lc", "--tag", name, "--lib", lib, "--qfile", qfile](extra 为真值时追加其元素)执行 subprocess.run,返回码非 0 或 stdout 未匹配 ROW_RE 时抛 SystemExit,成功时返回 (got, neg, line),其中 GATE_RE/LINE_RE 未命中时对应值为 None。
|
|
309
|
-
def run_one(name, lib, qfile, extra=None):
|
|
310
|
-
"""调用 Rust 评测器跑一臂(--dataset lc 的组映射 + 显式 lib/qfile)。
|
|
311
|
-
|
|
312
|
-
命令执行走 subprocess argv 列表 + 显式 UTF-8 + PYTHONUTF8=1(不经 Windows shell)。
|
|
313
|
-
"""
|
|
314
|
-
import subprocess
|
|
315
|
-
|
|
316
|
-
if not os.path.exists(RUST_BIN):
|
|
317
|
-
raise SystemExit(f"[失败] 未找到 Rust 评测器:{RUST_BIN}(先 cargo build --release)")
|
|
318
|
-
argv = [RUST_BIN, "--dataset", "lc", "--tag", name,
|
|
319
|
-
"--lib", lib, "--qfile", qfile]
|
|
320
|
-
if extra:
|
|
321
|
-
argv += list(extra)
|
|
322
|
-
env = dict(os.environ, PYTHONUTF8="1")
|
|
323
|
-
p = subprocess.run(argv, capture_output=True, text=True,
|
|
324
|
-
encoding="utf-8", errors="replace", env=env, cwd=HERE)
|
|
325
|
-
if p.returncode != 0:
|
|
326
|
-
raise SystemExit(f"[失败] {name}:{(p.stderr or '')[-900:]}")
|
|
327
|
-
got = {}
|
|
328
|
-
for m in ROW_RE.finditer(p.stdout):
|
|
329
|
-
g, n, h1, h5, mrr = m.groups()
|
|
330
|
-
got[g] = (int(n), float(h1), float(h5), float(mrr))
|
|
331
|
-
if not got:
|
|
332
|
-
raise SystemExit(f"[失败] {name}:未解析到组指标\n{p.stdout[-1500:]}")
|
|
333
|
-
neg = GATE_RE.search(p.stdout)
|
|
334
|
-
line = LINE_RE.search(p.stdout)
|
|
335
|
-
return got, (float(neg.group(1)) if neg else None), \
|
|
336
|
-
(float(line.group(1)) if line else None)
|
|
337
|
-
|
|
338
|
-
|
|
339
|
-
# 生效条件:对 rows 中每个 (label, got, note) 按 GROUPS 顺序打印 got[g] 的 hit@1/MRR(got 缺任一 GROUPS 键会 KeyError),并对 POS_GROUPS 组按样本数 n 加权算正例 hit@1/MRR;baseline 非 None 且某行 label 等于 baseline 时该行值记为参照,其后 label 不等于 baseline 的行附加 Δpp,无返回值。
|
|
340
|
-
def print_table(title, rows, baseline=None):
|
|
341
|
-
"""rows: [(label, got, note)];baseline = 参照行 label(算 Δ)。"""
|
|
342
|
-
print(f"\n== {title} ==")
|
|
343
|
-
hdr = (f"{'臂':<16}" + "".join(f"{g[:11]:>14}" for g in GROUPS)
|
|
344
|
-
+ f"{'正例hit@1':>12}{'正例MRR':>10}")
|
|
345
|
-
print(hdr)
|
|
346
|
-
print("-" * len(hdr))
|
|
347
|
-
ref = None
|
|
348
|
-
for label, got, note in rows:
|
|
349
|
-
cells, sh1, sn, smrr = [], 0.0, 0, 0.0
|
|
350
|
-
for g in GROUPS:
|
|
351
|
-
n, h1, _h5, mrr = got[g]
|
|
352
|
-
cells.append(f"{h1:.1f}%/{mrr:.3f}")
|
|
353
|
-
if g in POS_GROUPS:
|
|
354
|
-
sh1 += h1 / 100.0 * n
|
|
355
|
-
sn += n
|
|
356
|
-
smrr += mrr * n
|
|
357
|
-
ov_h1 = sh1 / max(1, sn)
|
|
358
|
-
ov_mrr = smrr / max(1, sn)
|
|
359
|
-
if baseline is not None and label == baseline:
|
|
360
|
-
ref = (ov_h1, ov_mrr)
|
|
361
|
-
delta = ""
|
|
362
|
-
if ref is not None and label != baseline:
|
|
363
|
-
delta = f" (Δ{(ov_h1 - ref[0]) * 100:+.1f}pp)"
|
|
364
|
-
suffix = f" {note}" if note else ""
|
|
365
|
-
print(f"{label:<16}" + "".join(f"{c:>14}" for c in cells)
|
|
366
|
-
+ f"{ov_h1 * 100:>10.1f}%{ov_mrr:>10.3f}{delta}{suffix}")
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
# 生效条件:seeds 等于 "sorted" 时给每个 run_one 追加 ["--graph-seeds", "sorted"],seeds 为其它值(含 None)时不追加;先 cmd_build(max_df=max_df) 取各臂库,再对 bz.ARMS 逐臂 run_one,neg 非 None 时把拒答率写入 note,返回 rows 并以 bz.ARMS[0]["name"] 为 baseline 打印主表。
|
|
370
|
-
def cmd_run(max_df=None, seeds=None):
|
|
371
|
-
"""消融主表(5 组;正例 hit@1/MRR = precise+temporal+interference)。"""
|
|
372
|
-
roots = cmd_build(max_df=max_df)
|
|
373
|
-
extra = ["--graph-seeds", "sorted"] if seeds == "sorted" else None
|
|
374
|
-
rows = []
|
|
375
|
-
for arm in bz.ARMS:
|
|
376
|
-
name = arm["name"]
|
|
377
|
-
got, neg, line = run_one(name, roots[name], QUESTIONS500, extra)
|
|
378
|
-
note = f"拒答率={neg:.1f}%" if neg is not None else ""
|
|
379
|
-
rows.append((name, got, note))
|
|
380
|
-
total = sum(got[g][0] for g in GROUPS for _, got, _ in rows[:1])
|
|
381
|
-
print_table(f"LoCoMo 千题中文层消融(Rust,--dataset lc,k=5,池 {total} 条证据 turn,"
|
|
382
|
-
f"种子口径={seeds or '索引序'})", rows, baseline=bz.ARMS[0]["name"])
|
|
383
|
-
print(f"\n池 {total} 条**全为某题 gold**(零干扰)→ 指标为**上界**;"
|
|
384
|
-
"干扰池为独立对照轮次,报告须声明。")
|
|
385
|
-
return rows
|
|
386
|
-
|
|
387
|
-
|
|
388
|
-
# 生效条件:cmd_build(max_df=max_df) 后取 bz.ARMS[-1]["name"] 对应的库,对同一库分别以索引序(lczh_a4_index,无额外参数)与 --graph-seeds sorted(lczh_a4_sorted)各跑一次 run_one,返回这两行结果并以 baseline="a4/index" 打印对照表。
|
|
389
|
-
def cmd_seed(max_df=None):
|
|
390
|
-
"""对照:同一个 a4 库,只切换 graph 路种子口径(隔离种子缺陷与边质量)。"""
|
|
391
|
-
roots = cmd_build(max_df=max_df)
|
|
392
|
-
lib = roots[bz.ARMS[-1]["name"]]
|
|
393
|
-
rows = [
|
|
394
|
-
("a4/index", run_one("lczh_a4_index", lib, QUESTIONS500)[0],
|
|
395
|
-
"现状:seeds=索引枚举序前 5"),
|
|
396
|
-
("a4/sorted", run_one("lczh_a4_sorted", lib, QUESTIONS500,
|
|
397
|
-
("--graph-seeds", "sorted"))[0],
|
|
398
|
-
"文档语义:seeds=top-5 词法"),
|
|
399
|
-
]
|
|
400
|
-
print_table("graph 种子口径对照(库内容完全相同,只换种子,池 567)", rows,
|
|
401
|
-
baseline="a4/index")
|
|
402
|
-
return rows
|
|
403
|
-
|
|
404
|
-
|
|
405
|
-
# 生效条件:kind 等于 "turn" 时从 RAW_TURNS 读取并按 id/speaker/date/text 打印 rows[start:end](序号自 start+1 起,text 截到 width);kind 为其它任何值时改从 RAW_QUESTIONS 按 qid/qtype/question 打印同一区间,无返回值。
|
|
406
|
-
def cmd_show(kind, start, end, width=240):
|
|
407
|
-
"""打印 [start, end) 区间的待标注项(分批产出时读原文用)。
|
|
408
|
-
|
|
409
|
-
经 python 输出而非 shell 拼接:turn 正文含 `|`、引号等,直接进 shell 会被
|
|
410
|
-
cmd.exe 解释(纪律 15 的动机)。
|
|
411
|
-
"""
|
|
412
|
-
rows = list(iter_jsonl(RAW_TURNS if kind == "turn" else RAW_QUESTIONS))
|
|
413
|
-
for i, r in enumerate(rows[start:end], start + 1):
|
|
414
|
-
if kind == "turn":
|
|
415
|
-
print(f"{i}\t{r['id']}\t{r['speaker']}\t{r['date']}\t{r['text'][:width]}")
|
|
416
|
-
else:
|
|
417
|
-
print(f"{i}\t{r['qid']}\t{r['qtype']}\t{r['question'][:width]}")
|
|
418
|
-
|
|
419
|
-
|
|
420
|
-
# 生效条件:len(argv) > 1 时 cmd 取 argv[1]、否则 cmd 为 "status";遍历 argv 时遇 "--max-df" 取 argv[i+1] 转 int 为 max_df、遇 "--seeds" 取 argv[i+1] 为 seeds;cmd 为 "show" 时调 cmd_show(argv[2], int(argv[3]), int(argv[4])),为 "dump"/"status"/"prepare" 时分别无参转调同名函数,为 "build"/"seed" 时传 max_df=max_df,为 "run" 时传 max_df=max_df 与 seeds=seeds,其余 cmd 值抛 SystemExit("未知子命令")。
|
|
421
|
-
def main(argv):
|
|
422
|
-
cmd = argv[1] if len(argv) > 1 else "status"
|
|
423
|
-
max_df = None
|
|
424
|
-
seeds = None
|
|
425
|
-
for i, a in enumerate(argv):
|
|
426
|
-
if a == "--max-df":
|
|
427
|
-
max_df = int(argv[i + 1])
|
|
428
|
-
if a == "--seeds":
|
|
429
|
-
seeds = argv[i + 1]
|
|
430
|
-
if cmd == "show":
|
|
431
|
-
cmd_show(argv[2], int(argv[3]), int(argv[4]))
|
|
432
|
-
elif cmd == "dump":
|
|
433
|
-
cmd_dump()
|
|
434
|
-
elif cmd == "status":
|
|
435
|
-
cmd_status()
|
|
436
|
-
elif cmd == "prepare":
|
|
437
|
-
cmd_prepare()
|
|
438
|
-
elif cmd == "build":
|
|
439
|
-
cmd_build(max_df=max_df)
|
|
440
|
-
elif cmd == "run":
|
|
441
|
-
cmd_run(max_df=max_df, seeds=seeds)
|
|
442
|
-
elif cmd == "seed":
|
|
443
|
-
cmd_seed(max_df=max_df)
|
|
444
|
-
else:
|
|
445
|
-
raise SystemExit(
|
|
446
|
-
f"未知子命令:{cmd}"
|
|
447
|
-
"(可用:dump / status / prepare / build / run / seed)")
|
|
448
|
-
|
|
449
|
-
|
|
450
|
-
if __name__ == "__main__":
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""LoCoMo 千题规模中文层四项能力探针(zh_mad 的 LoCoMo 扩展)。
|
|
3
|
+
|
|
4
|
+
与 `bench_zh_mad.py` 的关系:后者是 20 条 gold turn 的**方向性**探针(池 20 条,
|
|
5
|
+
±1 题 = ±5pp)。本脚本把同一套写入侧四轴消融搬到 **LoCoMo 全证据覆盖**规模:
|
|
6
|
+
|
|
7
|
+
轴 1 实体规范化 —— 全库 df 聚合出的 canonical 词 + 人物/地点/时间 → tags(entity 路)
|
|
8
|
+
轴 2 指代消解 —— **仅当本条自身无实体**时承接**前一条自身**的 canonical
|
|
9
|
+
轴 3 意图抽象 —— 摘要 → 规范意图类目;类目 + 动词原形 → tags,类目 → 正文子功能行
|
|
10
|
+
轴 4 关系图遍历 —— 人物/事件/时间/身份/地点/条件 多维关系 → edges(graph 路)
|
|
11
|
+
|
|
12
|
+
本脚本**复用** `bench_zh_mad` 的 `parse_zh` / `build_tables` / `axis_values` /
|
|
13
|
+
`normalize_terms` / `intent_of` / `build_arm`——保证中文层 schema 与四轴实现
|
|
14
|
+
与 20 条基准**逐位同源**,差异只在语料规模与题库。
|
|
15
|
+
|
|
16
|
+
数据集与规模(使用者已裁决):
|
|
17
|
+
* 数据集:LoCoMo(mteb/LoCoMo BEIR 转制版)500 题。
|
|
18
|
+
* 写入侧口径:**全证据覆盖**——所有 `evidence_turns` 去重后的 turn 全部入池。
|
|
19
|
+
实测:500 题引用的证据 turn 去重 568 个,其中 **567 个在语料中存在**,
|
|
20
|
+
`scene_3_session_10_turn_19` 为数据集悬空引用(语料无此 turn,见「数据完整性」
|
|
21
|
+
条款)。
|
|
22
|
+
* 干扰项:**第一轮不加**(池仅证据 turn)。池内每一条都是某题的 gold →
|
|
23
|
+
「零干扰」理想条件,指标为**上界**,报告必须声明(干扰池为独立对照轮次)。
|
|
24
|
+
* 查询侧:500 题的中文查询词。
|
|
25
|
+
|
|
26
|
+
数据完整性条款(诚实条款,报告必须原样声明):
|
|
27
|
+
* LoCoMo 的 BEIR 转制版中,题 `scene_3_q_58`(multi_hop)引用的
|
|
28
|
+
`scene_3_session_10_turn_19` 在 `locomo_corpus.jsonl` 中**不存在**——
|
|
29
|
+
数据集自身的悬空引用。本题另有 6 条有效证据,故不影响其可命中性;
|
|
30
|
+
但「证据 turn 总数」准确值是 **567** 而非 568。
|
|
31
|
+
|
|
32
|
+
诚实条款(报告必须原样声明):
|
|
33
|
+
* **中文层与中文查询均为本次新增产出**(会话模型逐条产出),**非既有真源**——
|
|
34
|
+
与 20 条基准(`manual_zh.json` / `manual_q.json`,既有产物)性质不同。
|
|
35
|
+
* 写入侧加工(四轴)仍是**确定性纯规则**(零 LLM、零第三方依赖、同输入必同输出),
|
|
36
|
+
词表随源码公开,第三方可重放。
|
|
37
|
+
* 加工**盲于查询集**:canonical 判定只用全库 df 与中文层自身字段,
|
|
38
|
+
不读 `zh_queries/` 的任何内容——否则即数据泄漏,评测作废。
|
|
39
|
+
* 池 567 条、题 500 道;随机基线 hit@1 = 1/567 ≈ 0.18%。但池内**全是 gold**,
|
|
40
|
+
故 0.18% 不是有意义的基线——「零干扰」才是本轮真正的条件限制。
|
|
41
|
+
|
|
42
|
+
跑法:
|
|
43
|
+
python -m md_cg.bench_locomo_zh dump # 派生 raw_turns/raw_questions(零标注)
|
|
44
|
+
python -m md_cg.bench_locomo_zh status # 产出进度与缺口
|
|
45
|
+
python -m md_cg.bench_locomo_zh prepare # 合并中文层 → corpus567/questions500
|
|
46
|
+
python -m md_cg.bench_locomo_zh build # 建 5 臂
|
|
47
|
+
python -m md_cg.bench_locomo_zh run # Rust 评测(--dataset lc,5 组)
|
|
48
|
+
python -m md_cg.bench_locomo_zh seed # graph 种子口径对照
|
|
49
|
+
"""
|
|
50
|
+
import json
|
|
51
|
+
import os
|
|
52
|
+
import re
|
|
53
|
+
import sys
|
|
54
|
+
|
|
55
|
+
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
56
|
+
for _p in (HERE, os.path.join(HERE, "md_cg")):
|
|
57
|
+
if _p not in sys.path:
|
|
58
|
+
sys.path.insert(0, _p)
|
|
59
|
+
|
|
60
|
+
from md_cg import bench_zh_mad as bz # noqa: E402 复用四轴实现(同源保证)
|
|
61
|
+
|
|
62
|
+
EXT = os.path.join(HERE, "data", "external")
|
|
63
|
+
LC = os.path.join(EXT, "locomo")
|
|
64
|
+
WORK = os.path.join(EXT, "locomo_zh")
|
|
65
|
+
ZH_TURNS = os.path.join(WORK, "zh_turns") # 分批落盘:{turn_id: 中文层串}
|
|
66
|
+
ZH_QUERIES = os.path.join(WORK, "zh_queries") # 分批落盘:{qid: 中文查询词}
|
|
67
|
+
|
|
68
|
+
RAW_TURNS = os.path.join(WORK, "raw_turns.jsonl")
|
|
69
|
+
RAW_QUESTIONS = os.path.join(WORK, "raw_questions.jsonl")
|
|
70
|
+
CORPUS567 = os.path.join(WORK, "corpus567.jsonl")
|
|
71
|
+
QUESTIONS500 = os.path.join(WORK, "questions500.jsonl")
|
|
72
|
+
|
|
73
|
+
# LoCoMo 组映射(与 rust/src/main.rs `groups_of("lc")` 逐项一致)
|
|
74
|
+
GROUPS = ("precise", "temporal", "interference", "negative", "reference")
|
|
75
|
+
POS_GROUPS = ("precise", "temporal", "interference")
|
|
76
|
+
QTYPE_OF_GROUP = {
|
|
77
|
+
"precise": "single_hop",
|
|
78
|
+
"temporal": "temporal_reasoning",
|
|
79
|
+
"interference": "multi_hop",
|
|
80
|
+
"negative": "adversarial",
|
|
81
|
+
"reference": "open_domain",
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
# 已知的悬空引用(数据集自身缺陷,见模块头数据完整性条款)
|
|
85
|
+
KNOWN_DANGLING = {"scene_3_session_10_turn_19"}
|
|
86
|
+
|
|
87
|
+
# graph 边的 df 上限:自由参数,随池规模标定(见 rescale 说明)。
|
|
88
|
+
# 20 条基准用 5(= 25% 池);本脚本缺省按 5% 池标定,另跑固定 5 作敏感性对照。
|
|
89
|
+
MAX_DF_FIXED = 5
|
|
90
|
+
|
|
91
|
+
|
|
92
|
+
# 生效条件:任意 n_pool(含 0 等假值,源码不做校验)下返回 max(MAX_DF_FIXED, int(round(0.05 * n_pool))),结果不小于模块级常量 MAX_DF_FIXED。
|
|
93
|
+
def max_df_auto(n_pool):
|
|
94
|
+
"""池规模 → edges 的 df 上限。
|
|
95
|
+
|
|
96
|
+
原值 5 是 20 条池上的**绝对**阈值(= 池的 25%)。thousand-scale 直接照搬会
|
|
97
|
+
把常见词(人名/常用名词)全部过滤 → 图路无边可走;反之过宽则生成巨型团
|
|
98
|
+
(「全连通,零信息量只注入噪声」)。此处按池规模取 5% 作为标定值,
|
|
99
|
+
并在报告中附 max_df=5 的敏感性列——**结论不建立在单一参数上**。
|
|
100
|
+
"""
|
|
101
|
+
return max(MAX_DF_FIXED, int(round(0.05 * n_pool)))
|
|
102
|
+
|
|
103
|
+
|
|
104
|
+
# 生效条件:path 指向逐行 JSON 的 UTF-8 文本时,跳过空行并逐行 yield json.loads 结果。
|
|
105
|
+
def iter_jsonl(path):
|
|
106
|
+
with open(path, encoding="utf-8") as f:
|
|
107
|
+
for line in f:
|
|
108
|
+
line = line.strip()
|
|
109
|
+
if line:
|
|
110
|
+
yield json.loads(line)
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# 生效条件:path 能被 open(path, encoding="utf-8") 打开时返回 json.load(f) 的解析结果。
|
|
114
|
+
def load_json(path):
|
|
115
|
+
with open(path, encoding="utf-8") as f:
|
|
116
|
+
return json.load(f)
|
|
117
|
+
|
|
118
|
+
|
|
119
|
+
# 生效条件:os.path.isdir(d) 为假时直接返回空 dict;为真时按 sorted(os.listdir(d)) 顺序读取后缀为 .json 的分片并 out.update(part)(后者覆盖前者),任一分片不是 dict 时抛 SystemExit,否则返回合并后的 out。
|
|
120
|
+
def load_chunks(d):
|
|
121
|
+
"""合并目录下全部分片 JSON(后者覆盖前者,便于修订单条)。"""
|
|
122
|
+
out = {}
|
|
123
|
+
if not os.path.isdir(d):
|
|
124
|
+
return out
|
|
125
|
+
for name in sorted(os.listdir(d)):
|
|
126
|
+
if not name.endswith(".json"):
|
|
127
|
+
continue
|
|
128
|
+
part = load_json(os.path.join(d, name))
|
|
129
|
+
if not isinstance(part, dict):
|
|
130
|
+
raise SystemExit(f"[失败] 分片非对象:{os.path.join(d, name)}")
|
|
131
|
+
out.update(part)
|
|
132
|
+
return out
|
|
133
|
+
|
|
134
|
+
|
|
135
|
+
# ------------------------------------------------------------------ dump
|
|
136
|
+
_SPEAKER_RE = re.compile(r"^([A-Za-z][A-Za-z .'\-]{0,24}): ")
|
|
137
|
+
_DATE_RE = re.compile(r"Data time: (\d{1,2}:\d{2} [AP]M on \w+ \d{1,2} \w+, \d{4})")
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
# 生效条件:调用即从 locomo_corpus.jsonl 建 id→记录映射、读 locomo_questions.jsonl,逐题取 q.get("evidence_turns") or [](缺键或假值按空列表)去重排序为证据 turn,text/title 用 c.get(...) or "" 兜假值、speaker/date 由 _SPEAKER_RE.match / _DATE_RE.search 命中才非空;verbose=True 时额外打印含 KNOWN_DANGLING 未登记告警与直接取键 q["qtype"] 的题型统计,verbose=False 时只跳过打印,写 RAW_TURNS/RAW_QUESTIONS 与返回 (rows, questions) 不变。
|
|
141
|
+
def cmd_dump(verbose=True):
|
|
142
|
+
"""确定性派生:证据 turn 全集 + 题库 → raw_turns/raw_questions(零标注)。"""
|
|
143
|
+
os.makedirs(ZH_TURNS, exist_ok=True)
|
|
144
|
+
os.makedirs(ZH_QUERIES, exist_ok=True)
|
|
145
|
+
corpus = {r["id"]: r for r in iter_jsonl(os.path.join(LC, "locomo_corpus.jsonl"))}
|
|
146
|
+
questions = list(iter_jsonl(os.path.join(LC, "locomo_questions.jsonl")))
|
|
147
|
+
|
|
148
|
+
ev = sorted({t for q in questions for t in (q.get("evidence_turns") or [])})
|
|
149
|
+
missing = sorted(t for t in ev if t not in corpus)
|
|
150
|
+
rows = []
|
|
151
|
+
for tid in ev:
|
|
152
|
+
if tid not in corpus:
|
|
153
|
+
continue
|
|
154
|
+
c = corpus[tid]
|
|
155
|
+
text = str(c.get("text") or "")
|
|
156
|
+
title = str(c.get("title") or "")
|
|
157
|
+
m = _SPEAKER_RE.match(text)
|
|
158
|
+
d = _DATE_RE.search(title)
|
|
159
|
+
rows.append({"id": tid, "text": text, "speaker": m.group(1) if m else "",
|
|
160
|
+
"date": d.group(1) if d else "", "title": title})
|
|
161
|
+
|
|
162
|
+
with open(RAW_TURNS, "w", encoding="utf-8") as f:
|
|
163
|
+
for r in rows:
|
|
164
|
+
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
165
|
+
with open(RAW_QUESTIONS, "w", encoding="utf-8") as f:
|
|
166
|
+
for q in questions:
|
|
167
|
+
f.write(json.dumps(q, ensure_ascii=False) + "\n")
|
|
168
|
+
|
|
169
|
+
if verbose:
|
|
170
|
+
print(f"证据 turn:去重 {len(ev)},语料中存在 {len(rows)},悬空 {missing}")
|
|
171
|
+
unknown = [t for t in missing if t not in KNOWN_DANGLING]
|
|
172
|
+
if unknown:
|
|
173
|
+
print(f"[警告] 出现未登记的悬空引用:{unknown}")
|
|
174
|
+
print(f"→ {RAW_TURNS}({len(rows)} 条)")
|
|
175
|
+
print(f"→ {RAW_QUESTIONS}({len(questions)} 道)")
|
|
176
|
+
ntype = {}
|
|
177
|
+
for q in questions:
|
|
178
|
+
ntype[q["qtype"]] = ntype.get(q["qtype"], 0) + 1
|
|
179
|
+
print("题型分布:" + ", ".join(f"{k}={v}" for k, v in sorted(ntype.items())))
|
|
180
|
+
return rows, questions
|
|
181
|
+
|
|
182
|
+
|
|
183
|
+
# ------------------------------------------------------------------ status
|
|
184
|
+
# 生效条件:verbose=True 时打印写入/查询侧完成度、多余 turn/qid 键与待补项(各截取前 head 项),并对已产出中文层用 bz.parse_zh 检查 identity/time/summary/terms 四项真值(并非五槽),非全真者记入 bad;verbose=False 时不打印、不做该槽位检查,返回值中 bad 恒为 []。
|
|
185
|
+
def cmd_status(verbose=True, head=12):
|
|
186
|
+
"""产出进度:中文层 / 中文查询的覆盖与缺口(分批产出时用)。"""
|
|
187
|
+
turns, questions = cmd_dump(verbose=False)
|
|
188
|
+
zt = load_chunks(ZH_TURNS)
|
|
189
|
+
zq = load_chunks(ZH_QUERIES)
|
|
190
|
+
t_ids = [r["id"] for r in turns]
|
|
191
|
+
q_ids = [q["qid"] for q in questions]
|
|
192
|
+
miss_t = [t for t in t_ids if t not in zt]
|
|
193
|
+
miss_q = [q for q in q_ids if q not in zq]
|
|
194
|
+
extra_t = [k for k in zt if k not in set(t_ids)]
|
|
195
|
+
extra_q = [k for k in zq if k not in set(q_ids)]
|
|
196
|
+
if verbose:
|
|
197
|
+
print(f"写入侧中文层:{len(t_ids) - len(miss_t)}/{len(t_ids)}"
|
|
198
|
+
f"(缺 {len(miss_t)})")
|
|
199
|
+
print(f"查询侧中文词:{len(q_ids) - len(miss_q)}/{len(q_ids)}"
|
|
200
|
+
f"(缺 {len(miss_q)})")
|
|
201
|
+
if extra_t:
|
|
202
|
+
print(f"[警告] 多余 turn 键(不在证据集内):{extra_t[:head]}")
|
|
203
|
+
if extra_q:
|
|
204
|
+
print(f"[警告] 多余 qid 键(不在题库内):{extra_q[:head]}")
|
|
205
|
+
if miss_t:
|
|
206
|
+
print(f" 待补 turn(前 {head}):{miss_t[:head]}")
|
|
207
|
+
if miss_q:
|
|
208
|
+
print(f" 待补 qid(前 {head}):{miss_q[:head]}")
|
|
209
|
+
# 结构校验:已产出的中文层必须能被 parse_zh 解出五槽
|
|
210
|
+
bad = []
|
|
211
|
+
for tid in t_ids:
|
|
212
|
+
if tid not in zt:
|
|
213
|
+
continue
|
|
214
|
+
f = bz.parse_zh(zt[tid])
|
|
215
|
+
if not (f["identity"] and f["time"] and f["summary"] and f["terms"]):
|
|
216
|
+
bad.append(tid)
|
|
217
|
+
if bad:
|
|
218
|
+
print(f"[警告] 中文层槽位不全({len(bad)} 条):{bad[:head]}")
|
|
219
|
+
return {"n_turn": len(t_ids), "n_turn_done": len(t_ids) - len(miss_t),
|
|
220
|
+
"n_q": len(q_ids), "n_q_done": len(q_ids) - len(miss_q),
|
|
221
|
+
"miss_turns": miss_t, "miss_qids": miss_q,
|
|
222
|
+
"bad": bad if verbose else []}
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
# ------------------------------------------------------------------ prepare
|
|
226
|
+
# 生效条件:miss_t 或 miss_q 任一非空即 raise SystemExit,否则按 sort_key(仅接受 scene_数字_session_数字_turn_数字 形式的 id,否则 raise SystemExit)对 turns 数值序排序,写出 corpus567.jsonl/questions500.jsonl 并返回 (corpus, out_q),其中 answer 与 evidence_turns 为假值时分别落为 "" 与 [],verbose 只控制末尾打印。
|
|
227
|
+
def cmd_prepare(verbose=True):
|
|
228
|
+
"""合并中文层 → corpus567.jsonl + questions500.jsonl(幂等覆盖)。"""
|
|
229
|
+
turns, questions = cmd_dump(verbose=False)
|
|
230
|
+
zt = load_chunks(ZH_TURNS)
|
|
231
|
+
zq = load_chunks(ZH_QUERIES)
|
|
232
|
+
|
|
233
|
+
miss_t = [r["id"] for r in turns if r["id"] not in zt]
|
|
234
|
+
miss_q = [q["qid"] for q in questions if q["qid"] not in zq]
|
|
235
|
+
if miss_t or miss_q:
|
|
236
|
+
raise SystemExit(
|
|
237
|
+
f"[失败] 中文层未产出完:缺 {len(miss_t)} 条 turn、{len(miss_q)} 道题。"
|
|
238
|
+
f"先跑 `status` 看缺口。")
|
|
239
|
+
|
|
240
|
+
# 语料:按 (scene, session, turn) 数值序 —— 「承接前一条」需要确定的时间序
|
|
241
|
+
# 生效条件:tid 匹配 ^scene_(\d+)_session_(\d+)_turn_(\d+)$ 时返回三个整数的 tuple(scene, session, turn),否则 raise SystemExit(不做其他容错或回退)。
|
|
242
|
+
def sort_key(tid):
|
|
243
|
+
m = re.match(r"scene_(\d+)_session_(\d+)_turn_(\d+)$", tid)
|
|
244
|
+
if not m:
|
|
245
|
+
raise SystemExit(f"[失败] turn id 无法解析:{tid}")
|
|
246
|
+
return tuple(int(x) for x in m.groups())
|
|
247
|
+
|
|
248
|
+
corpus = []
|
|
249
|
+
for r in sorted(turns, key=lambda r: sort_key(r["id"])):
|
|
250
|
+
raw = zt[r["id"]]
|
|
251
|
+
corpus.append({
|
|
252
|
+
"id": r["id"], "text": r["text"], "speaker": r["speaker"],
|
|
253
|
+
"date": r["date"], "zh": raw, "zh_fields": bz.parse_zh(raw),
|
|
254
|
+
})
|
|
255
|
+
|
|
256
|
+
out_q = []
|
|
257
|
+
for q in questions:
|
|
258
|
+
out_q.append({
|
|
259
|
+
"qid": q["qid"], "qtype": q["qtype"],
|
|
260
|
+
"question": zq[q["qid"]], "answer": str(q.get("answer") or ""),
|
|
261
|
+
"evidence_turns": list(q.get("evidence_turns") or []),
|
|
262
|
+
})
|
|
263
|
+
|
|
264
|
+
with open(CORPUS567, "w", encoding="utf-8") as f:
|
|
265
|
+
for r in corpus:
|
|
266
|
+
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
267
|
+
with open(QUESTIONS500, "w", encoding="utf-8") as f:
|
|
268
|
+
for q in out_q:
|
|
269
|
+
f.write(json.dumps(q, ensure_ascii=False) + "\n")
|
|
270
|
+
|
|
271
|
+
if verbose:
|
|
272
|
+
print(f"语料 {len(corpus)} 条 → {CORPUS567}")
|
|
273
|
+
print(f"题库 {len(out_q)} 条 → {QUESTIONS500}")
|
|
274
|
+
ntype = {}
|
|
275
|
+
for q in out_q:
|
|
276
|
+
ntype[q["qtype"]] = ntype.get(q["qtype"], 0) + 1
|
|
277
|
+
print("题型分布:" + ", ".join(f"{k}={v}" for k, v in sorted(ntype.items())))
|
|
278
|
+
lt = sum(len(r["zh_fields"]["terms"]) for r in corpus)
|
|
279
|
+
lc_ = sum(len(r["zh"]) for r in corpus)
|
|
280
|
+
print(f"中文层:词项合计 {lt}(均 {lt / len(corpus):.1f}/条),"
|
|
281
|
+
f"字符合计 {lc_}(均 {lc_ / len(corpus):.0f}/条)")
|
|
282
|
+
return corpus, out_q
|
|
283
|
+
|
|
284
|
+
|
|
285
|
+
# ------------------------------------------------------------------ build
|
|
286
|
+
# 生效条件:max_df 为 None(默认)时用 max_df_auto(len(corpus)) 自动取值,max_df 传出值(含 0 等假值)则直接使用;随后对 bz.ARMS 每臂调 bz.build_arm(corpus, arm, max_df=md, root_base=...) 并返回 {arm["name"]: 建库结果};verbose 形参在该符号源码段内未参与如何分支。
|
|
287
|
+
def cmd_build(max_df=None, verbose=True):
|
|
288
|
+
corpus, _ = cmd_prepare(verbose=False)
|
|
289
|
+
md = max_df if max_df is not None else max_df_auto(len(corpus))
|
|
290
|
+
print(f"== 建库(消融 {len(bz.ARMS)} 臂,池 {len(corpus)} 条,"
|
|
291
|
+
f"max_df={md}(自动,5% 池;固定基线 {MAX_DF_FIXED}))==")
|
|
292
|
+
roots = {}
|
|
293
|
+
for arm in bz.ARMS:
|
|
294
|
+
roots[arm["name"]] = bz.build_arm(
|
|
295
|
+
corpus, arm, max_df=md, root_base=f"_md_cg_eval_lczh_{arm['name']}")
|
|
296
|
+
return roots
|
|
297
|
+
|
|
298
|
+
|
|
299
|
+
# ------------------------------------------------------------------ run
|
|
300
|
+
RUST_BIN = os.path.join(HERE, "rust", "target", "release", "mdcg-eval.exe")
|
|
301
|
+
ROW_RE = re.compile(
|
|
302
|
+
r"^\s*(precise|temporal|interference|negative|reference)\s+(\d+)\s+"
|
|
303
|
+
r"([\d.]+)%\s+([\d.]+)%\s+([\d.]+)\s*$", re.M)
|
|
304
|
+
GATE_RE = re.compile(r"拒答率:([\d.]+)%")
|
|
305
|
+
LINE_RE = re.compile(r"拒答线(正例 hit@1 题 Top-1 分 p10):([\d.]+)")
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
# 生效条件:RUST_BIN 经 os.path.exists 为假时抛 SystemExit;否则以 argv 列表 [RUST_BIN, "--dataset", "lc", "--tag", name, "--lib", lib, "--qfile", qfile](extra 为真值时追加其元素)执行 subprocess.run,返回码非 0 或 stdout 未匹配 ROW_RE 时抛 SystemExit,成功时返回 (got, neg, line),其中 GATE_RE/LINE_RE 未命中时对应值为 None。
|
|
309
|
+
def run_one(name, lib, qfile, extra=None):
|
|
310
|
+
"""调用 Rust 评测器跑一臂(--dataset lc 的组映射 + 显式 lib/qfile)。
|
|
311
|
+
|
|
312
|
+
命令执行走 subprocess argv 列表 + 显式 UTF-8 + PYTHONUTF8=1(不经 Windows shell)。
|
|
313
|
+
"""
|
|
314
|
+
import subprocess
|
|
315
|
+
|
|
316
|
+
if not os.path.exists(RUST_BIN):
|
|
317
|
+
raise SystemExit(f"[失败] 未找到 Rust 评测器:{RUST_BIN}(先 cargo build --release)")
|
|
318
|
+
argv = [RUST_BIN, "--dataset", "lc", "--tag", name,
|
|
319
|
+
"--lib", lib, "--qfile", qfile]
|
|
320
|
+
if extra:
|
|
321
|
+
argv += list(extra)
|
|
322
|
+
env = dict(os.environ, PYTHONUTF8="1")
|
|
323
|
+
p = subprocess.run(argv, capture_output=True, text=True,
|
|
324
|
+
encoding="utf-8", errors="replace", env=env, cwd=HERE)
|
|
325
|
+
if p.returncode != 0:
|
|
326
|
+
raise SystemExit(f"[失败] {name}:{(p.stderr or '')[-900:]}")
|
|
327
|
+
got = {}
|
|
328
|
+
for m in ROW_RE.finditer(p.stdout):
|
|
329
|
+
g, n, h1, h5, mrr = m.groups()
|
|
330
|
+
got[g] = (int(n), float(h1), float(h5), float(mrr))
|
|
331
|
+
if not got:
|
|
332
|
+
raise SystemExit(f"[失败] {name}:未解析到组指标\n{p.stdout[-1500:]}")
|
|
333
|
+
neg = GATE_RE.search(p.stdout)
|
|
334
|
+
line = LINE_RE.search(p.stdout)
|
|
335
|
+
return got, (float(neg.group(1)) if neg else None), \
|
|
336
|
+
(float(line.group(1)) if line else None)
|
|
337
|
+
|
|
338
|
+
|
|
339
|
+
# 生效条件:对 rows 中每个 (label, got, note) 按 GROUPS 顺序打印 got[g] 的 hit@1/MRR(got 缺任一 GROUPS 键会 KeyError),并对 POS_GROUPS 组按样本数 n 加权算正例 hit@1/MRR;baseline 非 None 且某行 label 等于 baseline 时该行值记为参照,其后 label 不等于 baseline 的行附加 Δpp,无返回值。
|
|
340
|
+
def print_table(title, rows, baseline=None):
|
|
341
|
+
"""rows: [(label, got, note)];baseline = 参照行 label(算 Δ)。"""
|
|
342
|
+
print(f"\n== {title} ==")
|
|
343
|
+
hdr = (f"{'臂':<16}" + "".join(f"{g[:11]:>14}" for g in GROUPS)
|
|
344
|
+
+ f"{'正例hit@1':>12}{'正例MRR':>10}")
|
|
345
|
+
print(hdr)
|
|
346
|
+
print("-" * len(hdr))
|
|
347
|
+
ref = None
|
|
348
|
+
for label, got, note in rows:
|
|
349
|
+
cells, sh1, sn, smrr = [], 0.0, 0, 0.0
|
|
350
|
+
for g in GROUPS:
|
|
351
|
+
n, h1, _h5, mrr = got[g]
|
|
352
|
+
cells.append(f"{h1:.1f}%/{mrr:.3f}")
|
|
353
|
+
if g in POS_GROUPS:
|
|
354
|
+
sh1 += h1 / 100.0 * n
|
|
355
|
+
sn += n
|
|
356
|
+
smrr += mrr * n
|
|
357
|
+
ov_h1 = sh1 / max(1, sn)
|
|
358
|
+
ov_mrr = smrr / max(1, sn)
|
|
359
|
+
if baseline is not None and label == baseline:
|
|
360
|
+
ref = (ov_h1, ov_mrr)
|
|
361
|
+
delta = ""
|
|
362
|
+
if ref is not None and label != baseline:
|
|
363
|
+
delta = f" (Δ{(ov_h1 - ref[0]) * 100:+.1f}pp)"
|
|
364
|
+
suffix = f" {note}" if note else ""
|
|
365
|
+
print(f"{label:<16}" + "".join(f"{c:>14}" for c in cells)
|
|
366
|
+
+ f"{ov_h1 * 100:>10.1f}%{ov_mrr:>10.3f}{delta}{suffix}")
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
# 生效条件:seeds 等于 "sorted" 时给每个 run_one 追加 ["--graph-seeds", "sorted"],seeds 为其它值(含 None)时不追加;先 cmd_build(max_df=max_df) 取各臂库,再对 bz.ARMS 逐臂 run_one,neg 非 None 时把拒答率写入 note,返回 rows 并以 bz.ARMS[0]["name"] 为 baseline 打印主表。
|
|
370
|
+
def cmd_run(max_df=None, seeds=None):
|
|
371
|
+
"""消融主表(5 组;正例 hit@1/MRR = precise+temporal+interference)。"""
|
|
372
|
+
roots = cmd_build(max_df=max_df)
|
|
373
|
+
extra = ["--graph-seeds", "sorted"] if seeds == "sorted" else None
|
|
374
|
+
rows = []
|
|
375
|
+
for arm in bz.ARMS:
|
|
376
|
+
name = arm["name"]
|
|
377
|
+
got, neg, line = run_one(name, roots[name], QUESTIONS500, extra)
|
|
378
|
+
note = f"拒答率={neg:.1f}%" if neg is not None else ""
|
|
379
|
+
rows.append((name, got, note))
|
|
380
|
+
total = sum(got[g][0] for g in GROUPS for _, got, _ in rows[:1])
|
|
381
|
+
print_table(f"LoCoMo 千题中文层消融(Rust,--dataset lc,k=5,池 {total} 条证据 turn,"
|
|
382
|
+
f"种子口径={seeds or '索引序'})", rows, baseline=bz.ARMS[0]["name"])
|
|
383
|
+
print(f"\n池 {total} 条**全为某题 gold**(零干扰)→ 指标为**上界**;"
|
|
384
|
+
"干扰池为独立对照轮次,报告须声明。")
|
|
385
|
+
return rows
|
|
386
|
+
|
|
387
|
+
|
|
388
|
+
# 生效条件:cmd_build(max_df=max_df) 后取 bz.ARMS[-1]["name"] 对应的库,对同一库分别以索引序(lczh_a4_index,无额外参数)与 --graph-seeds sorted(lczh_a4_sorted)各跑一次 run_one,返回这两行结果并以 baseline="a4/index" 打印对照表。
|
|
389
|
+
def cmd_seed(max_df=None):
|
|
390
|
+
"""对照:同一个 a4 库,只切换 graph 路种子口径(隔离种子缺陷与边质量)。"""
|
|
391
|
+
roots = cmd_build(max_df=max_df)
|
|
392
|
+
lib = roots[bz.ARMS[-1]["name"]]
|
|
393
|
+
rows = [
|
|
394
|
+
("a4/index", run_one("lczh_a4_index", lib, QUESTIONS500)[0],
|
|
395
|
+
"现状:seeds=索引枚举序前 5"),
|
|
396
|
+
("a4/sorted", run_one("lczh_a4_sorted", lib, QUESTIONS500,
|
|
397
|
+
("--graph-seeds", "sorted"))[0],
|
|
398
|
+
"文档语义:seeds=top-5 词法"),
|
|
399
|
+
]
|
|
400
|
+
print_table("graph 种子口径对照(库内容完全相同,只换种子,池 567)", rows,
|
|
401
|
+
baseline="a4/index")
|
|
402
|
+
return rows
|
|
403
|
+
|
|
404
|
+
|
|
405
|
+
# 生效条件:kind 等于 "turn" 时从 RAW_TURNS 读取并按 id/speaker/date/text 打印 rows[start:end](序号自 start+1 起,text 截到 width);kind 为其它任何值时改从 RAW_QUESTIONS 按 qid/qtype/question 打印同一区间,无返回值。
|
|
406
|
+
def cmd_show(kind, start, end, width=240):
|
|
407
|
+
"""打印 [start, end) 区间的待标注项(分批产出时读原文用)。
|
|
408
|
+
|
|
409
|
+
经 python 输出而非 shell 拼接:turn 正文含 `|`、引号等,直接进 shell 会被
|
|
410
|
+
cmd.exe 解释(纪律 15 的动机)。
|
|
411
|
+
"""
|
|
412
|
+
rows = list(iter_jsonl(RAW_TURNS if kind == "turn" else RAW_QUESTIONS))
|
|
413
|
+
for i, r in enumerate(rows[start:end], start + 1):
|
|
414
|
+
if kind == "turn":
|
|
415
|
+
print(f"{i}\t{r['id']}\t{r['speaker']}\t{r['date']}\t{r['text'][:width]}")
|
|
416
|
+
else:
|
|
417
|
+
print(f"{i}\t{r['qid']}\t{r['qtype']}\t{r['question'][:width]}")
|
|
418
|
+
|
|
419
|
+
|
|
420
|
+
# 生效条件:len(argv) > 1 时 cmd 取 argv[1]、否则 cmd 为 "status";遍历 argv 时遇 "--max-df" 取 argv[i+1] 转 int 为 max_df、遇 "--seeds" 取 argv[i+1] 为 seeds;cmd 为 "show" 时调 cmd_show(argv[2], int(argv[3]), int(argv[4])),为 "dump"/"status"/"prepare" 时分别无参转调同名函数,为 "build"/"seed" 时传 max_df=max_df,为 "run" 时传 max_df=max_df 与 seeds=seeds,其余 cmd 值抛 SystemExit("未知子命令")。
|
|
421
|
+
def main(argv):
|
|
422
|
+
cmd = argv[1] if len(argv) > 1 else "status"
|
|
423
|
+
max_df = None
|
|
424
|
+
seeds = None
|
|
425
|
+
for i, a in enumerate(argv):
|
|
426
|
+
if a == "--max-df":
|
|
427
|
+
max_df = int(argv[i + 1])
|
|
428
|
+
if a == "--seeds":
|
|
429
|
+
seeds = argv[i + 1]
|
|
430
|
+
if cmd == "show":
|
|
431
|
+
cmd_show(argv[2], int(argv[3]), int(argv[4]))
|
|
432
|
+
elif cmd == "dump":
|
|
433
|
+
cmd_dump()
|
|
434
|
+
elif cmd == "status":
|
|
435
|
+
cmd_status()
|
|
436
|
+
elif cmd == "prepare":
|
|
437
|
+
cmd_prepare()
|
|
438
|
+
elif cmd == "build":
|
|
439
|
+
cmd_build(max_df=max_df)
|
|
440
|
+
elif cmd == "run":
|
|
441
|
+
cmd_run(max_df=max_df, seeds=seeds)
|
|
442
|
+
elif cmd == "seed":
|
|
443
|
+
cmd_seed(max_df=max_df)
|
|
444
|
+
else:
|
|
445
|
+
raise SystemExit(
|
|
446
|
+
f"未知子命令:{cmd}"
|
|
447
|
+
"(可用:dump / status / prepare / build / run / seed)")
|
|
448
|
+
|
|
449
|
+
|
|
450
|
+
if __name__ == "__main__":
|
|
451
451
|
main(sys.argv)
|