@furongjun1999/dsh-memory 0.4.11 → 0.5.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +16 -16
- package/codebuddy/CODEBUDDY.md +11 -3
- package/codebuddy/README.md +92 -90
- package/codebuddy/mcp.json +27 -27
- package/docs/GBrain/345/217/257/345/200/237/351/211/264/347/202/271_/347/201/265/346/236/242/350/220/275/347/202/271/344/272/244/346/216/245_20260919.md +169 -169
- package/docs/Pi/345/217/257/345/255/246/344/271/240/344/274/230/347/202/271_/347/201/265/346/236/242/345/244/247/350/204/221/346/224/271/350/277/233/344/272/244/346/216/245_20260915.md +144 -144
- package/docs/README.md +142 -111
- package/docs/discipline/harnesses.yaml +244 -226
- package/docs/discipline/templates/full.md.tmpl +61 -61
- package/docs/discipline/templates/rules.mdc.tmpl +68 -0
- package/docs/discipline/templates/skill.md.tmpl +23 -23
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v1.0.md +299 -299
- package/docs/eval/AGI/344/270/203/347/273/264/350/257/204/345/210/206/346/212/245/345/221/212_md_cg_v2.0.md +239 -239
- package/docs/eval//345/256/236/351/252/214/346/226/271/346/241/210_/345/255/246/344/271/240/351/227/255/347/216/257AB/344/270/216/346/250/252/350/257/204_v1.0.md +174 -174
- package/docs/eval//346/250/252/350/257/204_/345/205/255/345/256/266100/351/242/230/344/270/255/350/213/261/345/217/214/346/237/245_v1.0.md +223 -223
- package/docs/{mdcg → hive}//344/273/244/347/211/214/344/270/216/350/247/222/350/211/262/346/235/203/350/201/214/345/210/206/347/246/273_v0.1.md +165 -160
- package/docs/hive//344/273/244/347/211/214/350/257/255/344/271/211/344/277/256/346/255/243_/344/273/262/350/243/201/344/275/215_v0.1.md +76 -0
- package/docs/{mdcg → hive}//345/255/220/344/273/243/347/220/206/351/205/215/347/275/256/346/240/207/345/207/206_v0.5.md +236 -236
- package/docs/hive//346/243/200/347/264/242/346/224/266/346/225/233/345/256/236/346/265/213/344/270/216S1b/350/256/276/350/256/241_v0.1.md +43 -43
- package/docs/hive//346/243/200/347/264/242/347/256/227/346/263/225/345/217/243/345/276/204/345/257/271/347/205/247_v0.1.md +85 -0
- package/docs/hive//346/243/200/347/264/242/350/267/257/345/276/204/344/270/216/350/256/244/347/237/245/347/273/223/346/236/204/345/245/221/347/272/246_v0.1.md +102 -102
- package/docs/hive//350/234/202/345/267/242M6_ingest/345/256/236/346/226/275/350/256/241/345/210/222_v0.1.md +47 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +503 -503
- package/docs/hive//350/234/202/345/267/242/345/267/245/344/275/234/350/256/260/345/277/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +85 -85
- package/docs/hive//350/234/202/345/267/242/350/256/276/350/256/241_/347/220/206/350/256/272/345/257/271/351/275/220_v0.1.md +191 -0
- package/docs/hive//350/234/202/345/267/242/350/277/255/344/273/243_/345/256/217/350/247/202/344/270/216/347/276/244/344/275/223/350/260/203/345/272/246_v0.1.md +90 -0
- package/docs/mdcg/D_meta_/345/267/245/347/250/213/345/214/226/346/226/271/346/241/210_v0.2.md +216 -216
- package/docs/mdcg/README/350/257/246/347/273/206/347/211/210_v0.4.10.md +646 -646
- package/docs/mdcg/lingshu_tutorial.html +14449 -14449
- package/docs/mdcg/release_v0.4.11.md +49 -0
- package/docs/mdcg/release_v0.4.5.md +55 -55
- package/docs/mdcg/tool_table_v0.3.0.md +117 -117
- package/docs/mdcg//345/205/250/345/272/223/344/273/243/347/240/201/350/257/204/345/256/241/344/270/216/346/235/241/344/273/266/345/214/226/346/263/250/351/207/212_/350/256/241/345/210/222_v0.1.md +600 -600
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/mdcg//345/215/225/345/205/203/350/207/252/346/210/221/351/224/232/347/202/271_/347/263/273/347/273/237/346/217/220/347/244/272/350/257/215/346/240/207/345/207/206_v0.3.md +275 -275
- package/docs/mdcg//345/217/221/345/270/203/351/227/250/347/246/201/351/223/276_v0.1.md +54 -0
- package/docs/mdcg//346/272/220/347/240/201/347/272/247/346/236/266/346/236/204/345/256/241/350/256/241_GPT/346/211/271/350/257/204/345/257/271/347/205/247_v1.0.md +130 -130
- package/docs/mdcg//347/201/265/346/236/24282/345/267/245/345/205/267_/345/212/237/350/203/275/346/225/264/347/220/206/344/270/216/350/277/201/347/247/273/346/230/240/345/260/204_v0.1.md +278 -278
- package/docs/mdcg//347/201/265/346/236/242/350/256/260/345/277/206/345/212/250/350/257/215/345/215/217/350/256/256_v1.0-draft.md +172 -172
- package/docs/mdcg//347/274/272/345/217/243/345/215/225_P0/346/224/266/345/217/243_v0.1.md +270 -270
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_G4-G8/347/274/272/345/217/243/350/243/201/345/256/232/345/215/225_v0.1.md +491 -491
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/346/235/241/344/273/266/347/251/272/351/227/264/345/220/210/346/210/220/344/270/216/347/224/237/346/225/210/346/235/241/344/273/266/345/217/243/345/276/204_v0.1.md +340 -340
- package/docs/mdcg//350/256/244/347/237/245/345/233/276_/347/264/242/345/274/225/344/270/216/345/267/245/347/250/213/350/247/204/350/214/203/345/214/226_/350/256/241/345/210/222_v0.1.md +588 -588
- package/docs/plans/GridWorld/346/234/200/345/260/217/351/227/255/347/216/257/350/247/204/346/240/274_v0.1.md +176 -176
- package/docs/plans//345/221/275/345/220/215/346/262/273/347/220/206_/351/241/271/347/233/256/350/256/241/345/210/222.md +81 -81
- package/docs/swarm//350/234/202/347/276/244/344/272/222/350/201/224_v0.1.md +704 -704
- package/docs/swarm//350/234/202/347/276/244/345/220/214/351/224/231/346/243/200/346/265/213/345/256/236/351/252/214/345/215/217/350/256/256_v0.1.md +242 -242
- package/docs/theory//345/215/225/347/272/277/347/250/213/344/270/216/346/263/250/346/204/217/345/212/233/351/233/206/344/270/255_/346/227/240/344/272/211/350/256/256/347/220/206/350/256/272/346/226/207/346/241/243_v1.0.md +129 -129
- package/docs/theory//345/271/266/345/217/221/345/277/205/347/204/266/346/200/247/347/220/206/350/256/272_v0.2.md +134 -134
- package/docs/theory//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/docs/theory//346/246/202/345/277/265/345/210/206/345/261/202/345/257/271/351/275/220/350/241/250_v0.1.md +145 -145
- package/docs/theory//347/220/206/350/256/272_/346/234/272/345/210/266_/344/273/243/347/240/201_/345/256/236/351/252/214_/347/274/272/345/217/243/347/237/251/351/230/265_v0.1.md +144 -144
- package/docs/theory//347/220/206/350/256/272/344/273/223/346/213/206/345/210/206/344/270/216/347/231/275/347/256/261/347/237/245/350/257/206/345/272/223/345/206/205/350/277/201_v0.1.md +198 -198
- package/docs/theory//350/256/244/347/237/245/344/273/243/347/220/206/344/270/216/346/224/266/346/225/233/347/273/223/346/236/204_/346/235/241/344/273/266/350/256/272/351/207/215/346/236/204_v0.1.md +221 -221
- package/docs/world_model//344/270/226/347/225/214/346/250/241/345/236/213_/347/245/236/347/273/217/347/275/221/347/273/234/345/272/225/345/261/202/346/236/266/346/236/204_v1.0.md +237 -237
- package/docs//345/217/221/345/270/203/344/273/266/345/233/236/346/272/257/350/257/264/346/230/216_v0.1.md +82 -82
- package/docs//345/267/245/344/275/234/347/272/252/345/276/213_/350/256/244/347/237/245/345/233/276/346/235/241/347/233/256_v1.1.json +28 -2
- package/dsh/README.md +82 -82
- package/dsh/cordis.yml.example +139 -139
- package/dsh/update-lingshu.bat +11 -11
- package/lib/hooks.js +36 -2
- package/lib/lib/roleplay_web.js +427 -427
- package/md_cg/__init__.py +7 -7
- package/md_cg/audit.py +368 -368
- package/md_cg/autonomy.py +287 -287
- package/md_cg/backfill.py +1327 -1327
- package/md_cg/backfill_bigdomain.py +34 -34
- package/md_cg/bench6_arms.py +410 -410
- package/md_cg/bench6_common.py +230 -230
- package/md_cg/bench6_competitors.py +212 -212
- package/md_cg/bench_axis_domain.py +257 -257
- package/md_cg/bench_blind_comp.py +308 -308
- package/md_cg/bench_en_atoms_public.py +230 -230
- package/md_cg/bench_governance.py +348 -348
- package/md_cg/bench_lme_zh.py +410 -410
- package/md_cg/bench_locomo.py +121 -121
- package/md_cg/bench_locomo_zh.py +450 -450
- package/md_cg/bench_locomo_zh_public.py +147 -147
- package/md_cg/bench_longmem.py +112 -112
- package/md_cg/bench_membench.py +632 -632
- package/md_cg/bench_p0.py +149 -149
- package/md_cg/bench_progressive.py +287 -287
- package/md_cg/bench_role_views.py +238 -238
- package/md_cg/bench_task_ab.py +243 -243
- package/md_cg/bench_task_ab_llm.py +408 -408
- package/md_cg/bench_unified_en.py +204 -204
- package/md_cg/bench_zh_mad.py +601 -601
- package/md_cg/blindspot_tickets.py +123 -123
- package/md_cg/branches.py +285 -285
- package/md_cg/build_postings.py +73 -73
- package/md_cg/ccgc.py +1005 -948
- package/md_cg/census.py +132 -132
- package/md_cg/chain.py +300 -300
- package/md_cg/codeindex.py +531 -531
- package/md_cg/coldverify.py +292 -292
- package/md_cg/comment_gate.py +337 -337
- package/md_cg/cond_compose.py +190 -190
- package/md_cg/cond_facts.py +154 -154
- package/md_cg/cond_template.json +106 -106
- package/md_cg/condition_anchor.py +142 -142
- package/md_cg/conformance.py +726 -726
- package/md_cg/consistency.py +717 -717
- package/md_cg/consolidate.py +1536 -1439
- package/md_cg/corpus.py +110 -110
- package/md_cg/crosscheck.py +1097 -1097
- package/md_cg/crypto.py +437 -437
- package/md_cg/d_meta.py +310 -310
- package/md_cg/datapath.py +334 -334
- package/md_cg/docindex.py +473 -473
- package/md_cg/eval_common.py +575 -575
- package/md_cg/evidence.py +580 -580
- package/md_cg/evolution.py +477 -477
- package/md_cg/export.py +220 -220
- package/md_cg/forgetting.py +581 -581
- package/md_cg/fsutil.py +329 -329
- package/md_cg/hotcache.py +238 -214
- package/md_cg/hyperedge.py +251 -251
- package/md_cg/identity.py +390 -390
- package/md_cg/insight.py +500 -500
- package/md_cg/interop.py +199 -0
- package/md_cg/lexicon/build_cedict_en_zh.py +329 -329
- package/md_cg/lexicon/build_standard_en.py +171 -171
- package/md_cg/lexicon/expand_en_zh.py +211 -211
- package/md_cg/lifecycle.py +272 -272
- package/md_cg/linkref.py +280 -280
- package/md_cg/links.py +622 -622
- package/md_cg/mcp_server.py +129 -32
- package/md_cg/md_whitebox.py +345 -345
- package/md_cg/mdcg.py +176 -117
- package/md_cg/mdcos.py +79 -13
- package/md_cg/metacognition.py +591 -591
- package/md_cg/migrate.py +119 -119
- package/md_cg/migrate_aeis.py +221 -221
- package/md_cg/migrate_roleplay.py +293 -293
- package/md_cg/migrate_wisdom_graph.py +360 -360
- package/md_cg/mreview/__init__.py +25 -25
- package/md_cg/mreview/__main__.py +110 -110
- package/md_cg/mreview/bundle.py +178 -178
- package/md_cg/mreview/candidates.py +262 -262
- package/md_cg/mreview/govern.py +693 -693
- package/md_cg/mreview/locate.py +939 -939
- package/md_cg/mreview/pipeline.py +728 -728
- package/md_cg/mreview/rules/duplication.json +21 -21
- package/md_cg/mreview/rules/field_coverage.json +54 -54
- package/md_cg/mreview/rules/source_license.json +21 -21
- package/md_cg/mreview/rules/template_flow.json +21 -21
- package/md_cg/mreview/ruleset.py +252 -252
- package/md_cg/nodefile.py +575 -575
- package/md_cg/pooling.py +484 -472
- package/md_cg/postings.py +298 -298
- package/md_cg/predict.py +1100 -1100
- package/md_cg/progressive.py +123 -123
- package/md_cg/protect.py +272 -272
- package/md_cg/protocol/md_cg_gate.proto +33 -33
- package/md_cg/protocol.py +372 -372
- package/md_cg/provenance.py +582 -582
- package/md_cg/reach.py +453 -453
- package/md_cg/readcache.py +85 -0
- package/md_cg/refindex.py +833 -833
- package/md_cg/refine.py +604 -604
- package/md_cg/roleviews.py +89 -89
- package/md_cg/routing.py +365 -365
- package/md_cg/scrub.py +852 -852
- package/md_cg/security.py +274 -274
- package/md_cg/self_state.py +1029 -1029
- package/md_cg/selfreport.py +151 -151
- package/md_cg/semantic/__init__.py +10 -10
- package/md_cg/semantic/canonical.py +122 -122
- package/md_cg/semantic/en_normalizer.py +364 -364
- package/md_cg/semantic/en_zh_map.json +28694 -0
- package/md_cg/semantic/export_en_zh_map.py +64 -0
- package/md_cg/semantic/unify.py +45 -0
- package/md_cg/semantic/zh_en_atoms.py +139 -139
- package/md_cg/signer.py +562 -562
- package/md_cg/sources.py +815 -582
- package/md_cg/statushdr.py +179 -179
- package/md_cg/stg.py +48 -37
- package/md_cg/subgraph.py +729 -729
- package/md_cg/sustain.py +1138 -1138
- package/md_cg/tasks.py +470 -470
- package/md_cg/test_action_derive.py +203 -203
- package/md_cg/test_audit_rotate.py +270 -270
- package/md_cg/test_autonomy.py +143 -143
- package/md_cg/test_bench_governance.py +102 -102
- package/md_cg/test_blindspot_tickets.py +166 -166
- package/md_cg/test_branches.py +249 -249
- package/md_cg/test_ccg_perturb.py +184 -184
- package/md_cg/test_ccgc.py +433 -433
- package/md_cg/test_census_prune.py +81 -81
- package/md_cg/test_cond_compose_anchors.py +76 -76
- package/md_cg/test_cond_match.py +165 -165
- package/md_cg/test_condition_anchor.py +81 -81
- package/md_cg/test_d_meta.py +412 -412
- package/md_cg/test_datapath_root.py +199 -199
- package/md_cg/test_en_pipeline.py +166 -166
- package/md_cg/test_gain_gate.py +212 -212
- package/md_cg/test_health_scale.py +173 -173
- package/md_cg/test_hive_ingest.py +285 -0
- package/md_cg/test_hot_cold.py +215 -215
- package/md_cg/test_hyperedge.py +245 -245
- package/md_cg/test_i26_empty_first_write.py +116 -0
- package/md_cg/test_i27_e041_identity.py +128 -0
- package/md_cg/test_i28_hotcache_prodpath.py +122 -0
- package/md_cg/test_identity_attribution.py +147 -147
- package/md_cg/test_index_durability.py +224 -224
- package/md_cg/test_interop.py +93 -0
- package/md_cg/test_lifecycle.py +309 -309
- package/md_cg/test_linkref.py +306 -306
- package/md_cg/test_lock.py +43 -43
- package/md_cg/test_md_access_parity.py +255 -255
- package/md_cg/test_md_writepath.py +345 -345
- package/md_cg/test_mdstore_search_parity.py +160 -0
- package/md_cg/test_mr_m2.py +587 -587
- package/md_cg/test_mr_m3.py +710 -710
- package/md_cg/test_mr_m4.py +485 -485
- package/md_cg/test_p0.py +250 -250
- package/md_cg/test_p1.py +316 -316
- package/md_cg/test_p10_identity.py +173 -173
- package/md_cg/test_p11_consistency.py +233 -233
- package/md_cg/test_p12_metacognition.py +212 -212
- package/md_cg/test_p13_encryption.py +241 -241
- package/md_cg/test_p14_sustain.py +249 -249
- package/md_cg/test_p15_scrub.py +280 -280
- package/md_cg/test_p16_self_state.py +301 -301
- package/md_cg/test_p17_predict.py +354 -354
- package/md_cg/test_p18_whitebox.py +171 -171
- package/md_cg/test_p19_migrate_roleplay.py +149 -149
- package/md_cg/test_p20_evolution.py +315 -315
- package/md_cg/test_p21_tokens.py +293 -270
- package/md_cg/test_p22_theory.py +175 -175
- package/md_cg/test_p23_links.py +311 -311
- package/md_cg/test_p24_evidence.py +227 -227
- package/md_cg/test_p25_weights.py +156 -156
- package/md_cg/test_p26_refindex.py +416 -416
- package/md_cg/test_p27_docindex.py +765 -765
- package/md_cg/test_p28_refcheck.py +305 -305
- package/md_cg/test_p29_session_ingest_export.py +354 -333
- package/md_cg/test_p3.py +11 -2
- package/md_cg/test_p30_maintain.py +330 -330
- package/md_cg/test_p31_insight.py +534 -534
- package/md_cg/test_p32_backfill.py +298 -298
- package/md_cg/test_p33_ccg_wiring.py +293 -293
- package/md_cg/test_p34_crosscheck.py +331 -331
- package/md_cg/test_p35_conditioned_claim.py +252 -252
- package/md_cg/test_p36_kp_align.py +230 -230
- package/md_cg/test_p37_condition_space.py +248 -248
- package/md_cg/test_p38_concurrent_flush.py +102 -0
- package/md_cg/test_p38_contextualize.py +273 -273
- package/md_cg/test_p39_verify_flow.py +113 -0
- package/md_cg/test_p39_vision_evidence.py +369 -369
- package/md_cg/test_p40_refine_worklist.py +241 -241
- package/md_cg/test_p41_evolve_patrol.py +224 -224
- package/md_cg/test_p42_provenance.py +269 -269
- package/md_cg/test_p43_pooling.py +412 -398
- package/md_cg/test_p44_md_whitebox.py +231 -231
- package/md_cg/test_p45_session_identity.py +219 -219
- package/md_cg/test_p46_unit_scope.py +272 -272
- package/md_cg/test_p47_session_view.py +281 -0
- package/md_cg/test_p4_fuzzy.py +223 -223
- package/md_cg/test_p5_semantic.py +226 -226
- package/md_cg/test_p6_consolidate.py +440 -387
- package/md_cg/test_p7_goals_recent.py +202 -202
- package/md_cg/test_p8_subgraph_chain.py +200 -200
- package/md_cg/test_p9_forget_protect.py +231 -231
- package/md_cg/test_predict_beta.py +135 -135
- package/md_cg/test_preflight_failclosed.py +100 -100
- package/md_cg/test_progressive.py +146 -146
- package/md_cg/test_protocol.py +243 -243
- package/md_cg/test_reach.py +378 -378
- package/md_cg/test_reach_keys.py +201 -201
- package/md_cg/test_read_clip.py +141 -141
- package/md_cg/test_readcache_prodpath.py +155 -0
- package/md_cg/test_retr_gates_prodpath.py +140 -0
- package/md_cg/test_retr_s1.py +340 -340
- package/md_cg/test_retr_s1b.py +209 -209
- package/md_cg/test_retr_s3.py +194 -194
- package/md_cg/test_retr_s4.py +163 -163
- package/md_cg/test_retr_s5.py +200 -200
- package/md_cg/test_retr_s6.py +157 -157
- package/md_cg/test_retr_s7.py +384 -384
- package/md_cg/test_retr_s8_time.py +369 -316
- package/md_cg/test_retr_s9_edges.py +286 -286
- package/md_cg/test_retr_s9_entity_ctx.py +175 -175
- package/md_cg/test_review_conformance.py +367 -367
- package/md_cg/test_role_views.py +354 -354
- package/md_cg/test_sem_noise.py +242 -242
- package/md_cg/test_semantic_canonical.py +241 -241
- package/md_cg/test_subproc_encoding.py +192 -192
- package/md_cg/test_sustain_mutual.py +153 -153
- package/md_cg/test_tasks.py +409 -409
- package/md_cg/test_tool_face.py +189 -189
- package/md_cg/test_transfer.py +180 -180
- package/md_cg/test_trust.py +361 -361
- package/md_cg/test_twophase.py +286 -286
- package/md_cg/test_v14_fixes.py +397 -397
- package/md_cg/test_validity_filter.py +280 -280
- package/md_cg/test_verify_answer.py +138 -138
- package/md_cg/test_wisdom_md_store.py +292 -292
- package/md_cg/test_writelimit.py +197 -197
- package/md_cg/test_writepipe.py +214 -214
- package/md_cg/theory.py +273 -273
- package/md_cg/tokens.py +677 -663
- package/md_cg/tool_face.py +260 -260
- package/md_cg/trust.py +986 -950
- package/md_cg/twophase.py +231 -231
- package/md_cg/units.py +667 -667
- package/md_cg/vision_evidence.py +666 -666
- package/md_cg/weights.py +624 -624
- package/md_cg/whitebox.py +527 -527
- package/md_cg/whitebox_kb/__init__.py +37 -37
- package/md_cg/whitebox_kb/aeis_core/__init__.py +42 -42
- package/md_cg/whitebox_kb/aeis_core/semantic.py +280 -280
- package/md_cg/whitebox_kb/aeis_core/textutil.py +13 -13
- package/md_cg/whitebox_kb/engine.py +310 -310
- package/md_cg/whitebox_kb/seed_knowledge//346/231/272/350/203/275/350/256/2723.4.md +5260 -5260
- package/md_cg/whitebox_kb/wisdom/browser_units.py +2631 -2631
- package/md_cg/whitebox_kb/wisdom/causal_discover.py +432 -432
- package/md_cg/whitebox_kb/wisdom/chat_engine.py +1506 -1506
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
- package/md_cg/whitebox_kb/wisdom/compiler_code_units.py +3033 -3033
- package/md_cg/whitebox_kb/wisdom/condition_algebra.py +112 -112
- package/md_cg/whitebox_kb/wisdom/condition_frame.py +315 -315
- package/md_cg/whitebox_kb/wisdom/condition_kb.py +108 -108
- package/md_cg/whitebox_kb/wisdom/conflict_map.json +4445 -4445
- package/md_cg/whitebox_kb/wisdom/core/lexer.py +512 -512
- package/md_cg/whitebox_kb/wisdom/core/name_checker.py +1023 -1023
- package/md_cg/whitebox_kb/wisdom/cspmn.py +258 -258
- package/md_cg/whitebox_kb/wisdom/csre.py +264 -264
- package/md_cg/whitebox_kb/wisdom/danmaku_audit.py +252 -252
- package/md_cg/whitebox_kb/wisdom/distilled_condition_units.json +6417 -6417
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902b.json +1950 -1950
- package/md_cg/whitebox_kb/wisdom/docs/WB-EVAL-20260902c.json +1296 -1296
- package/md_cg/whitebox_kb/wisdom/docs/whitebox_capability_graph_demo.json +59 -59
- package/md_cg/whitebox_kb/wisdom/graph_db_units.py +3089 -3089
- package/md_cg/whitebox_kb/wisdom/knowledge_points.py +318 -318
- package/md_cg/whitebox_kb/wisdom/md_access.py +470 -470
- package/md_cg/whitebox_kb/wisdom/md_store.py +276 -251
- package/md_cg/whitebox_kb/wisdom/migrate_wisdom.py +476 -476
- package/md_cg/whitebox_kb/wisdom/navigate.py +248 -248
- package/md_cg/whitebox_kb/wisdom/neural_retrieve.py +224 -224
- package/md_cg/whitebox_kb/wisdom/os_units.py +2735 -2735
- package/md_cg/whitebox_kb/wisdom/pattern_separation.py +376 -376
- package/md_cg/whitebox_kb/wisdom/prereq_map.json +364 -364
- package/md_cg/whitebox_kb/wisdom/python_code_units.py +2766 -2766
- package/md_cg/whitebox_kb/wisdom/role_solidified.json +7 -7
- package/md_cg/whitebox_kb/wisdom/route_memory.py +244 -244
- package/md_cg/whitebox_kb/wisdom/scene_reconstruction.py +148 -148
- package/md_cg/whitebox_kb/wisdom/snr_report.json +40 -40
- package/md_cg/whitebox_kb/wisdom/test_code_compose_domains.py +7541 -7541
- package/md_cg/whitebox_kb/wisdom/test_compiler_self_bootstrap.py +354 -354
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_assembly.py +158 -158
- package/md_cg/whitebox_kb/wisdom/test_ecosystem_demos.py +193 -193
- package/md_cg/whitebox_kb/wisdom/test_graph_db.py +292 -292
- package/md_cg/whitebox_kb/wisdom/test_python_self_bootstrap.py +160 -160
- package/md_cg/whitebox_kb/wisdom/trigger_words_index.json +4366 -4366
- package/md_cg/whitebox_kb/wisdom/verifier.py +1297 -1297
- package/md_cg/writelimit.py +356 -356
- package/md_cg/writepipe.py +550 -542
- package/package.json +97 -96
- package/skills/plugin.json +54 -54
- package/skills/skills/designer-perspective/SKILL.md +158 -158
- package/skills/skills/designer-perspective/references/01-observation-position.md +66 -66
- package/skills/skills/designer-perspective/references/02-structure-recognition.md +62 -62
- package/skills/skills/designer-perspective/references/03-direction-judgment.md +55 -55
- package/skills/skills/designer-perspective/references/04-qualification-verdict.md +72 -72
- package/skills/skills/designer-perspective/references/05-condition-attribution.md +74 -74
- package/skills/skills/designer-perspective/scripts/designer.py +545 -545
- package/skills/skills/designer-perspective/tests/cases.jsonl +17 -17
- package/skills/skills/designer-perspective/tests/selftest.py +61 -61
- package/skills/skills/lingshu-browser/SKILL.md +60 -60
- package/skills/skills/lingshu-compiler/SKILL.md +56 -56
- package/skills/skills/lingshu-compiler/units/analyze-type-infer/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/check-name-real/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-assign/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-expr-tree/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-full-pipeline/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-func-def/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-if-then/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-logic-expr/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-recursive/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-scope/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-type-check/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compile-while/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0010c4bf/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0355bffb/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0361708a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-054a0414/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05a1691a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-05eeed1e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0622a1f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-08b54217/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0a62b70c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ab24d00/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e093688/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0e82b966/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0ee9b9b9/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-0f3787e6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1028685f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-11897630/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-16661b9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-1f722303/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-25be1262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2a76ba07/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2dbea54a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2df52f16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2ec2c9d2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-2f8c8f39/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-38378ac7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39457d2e/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-39fb5926/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-47f4fbfa/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4a8cd1f1/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4b230b6b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4cbbda95/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-4d5a68ff/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-56dc9bdd/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-57e76ebe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-61ff016b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-63eae588/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-64c4224b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-66377955/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-674ab4f6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6af76fd6/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6d979db2/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6de326c0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-6f1c0eea/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-724d6c9b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7690177f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7b229a7d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-7f471b3a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-815cad08/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-83c61634/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8c798c81/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-8dd747ac/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-90324d0a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94b8d72d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-94f12231/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-95937c16/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98a5625b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98b4c42f/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-98e3894a/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9bdfe4b8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9d9b4e83/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-9f7be5ad/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a6daf076/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a7995a0c/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-a8399248/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-aabbd099/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-afe169d8/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b0678bda/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b09bd196/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b1396e23/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-b9b31ce0/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-c10264a7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cb1e8e4b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ccafd438/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cda9c262/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ce648068/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-cf5776a4/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-d974e5d3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-e3979fd3/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-eb1cf2b5/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-ecb30d5b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f8c8b24b/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-f99fedbe/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fa8ff5f7/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/compiler-fe1b058d/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-chinese-program/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-dao-de-jing/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/lex-nine-chapters/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-arithmetic/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-array-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-closure-create/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-compare/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-jump/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-cond-space/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-exception/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-func-call/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-loop-run/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-profiling/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-refcount/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-run-loop/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-short-circuit/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-guard/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-stack-ops/SKILL.md +45 -45
- package/skills/skills/lingshu-compiler/units/vm-trust-accum/SKILL.md +45 -45
- package/skills/skills/lingshu-graph/SKILL.md +63 -63
- package/skills/skills/lingshu-net/SKILL.md +48 -48
- package/skills/skills/lingshu-os/SKILL.md +64 -64
- package/skills/skills/lingshu-pylang/SKILL.md +71 -71
- package/src/bridge.ts +401 -401
- package/src/hooks.ts +38 -2
- package/src/lib/datapath.ts +326 -326
- package/src/lib/mdcg_client.ts +413 -413
- package/src/lib/mutual.ts +428 -428
- package/src/lib/prompt_safety.ts +62 -62
- package/src/lib/python_path.ts +71 -71
- package/src/lib/roleplay_web.ts +932 -932
- package/src/lib/token_store.ts +192 -192
- package/src/tools.ts +212 -212
- package/zcode/AGENTS.md +11 -3
- package/zcode/README.md +41 -41
- /package/docs/{mdcg → hive}//344/270/273/344/273/243/347/220/206/345/255/220/344/273/243/347/220/206/350/256/260/345/277/206/346/236/266/346/236/204/350/256/276/350/256/241.md" +0 -0
package/md_cg/bench_zh_mad.py
CHANGED
|
@@ -1,602 +1,602 @@
|
|
|
1
|
-
# -*- coding: utf-8 -*-
|
|
2
|
-
"""中文多维探针(MAD, multi-axis)——20 条「中文层 + 英文原文」语料,Rust 侧四路检索。
|
|
3
|
-
|
|
4
|
-
与既有 zh_probe 探针的关系:前一版探针只测「中文层做索引」的单点效果;本探针
|
|
5
|
-
把写入侧加工拆成可消融的四个维度,逐一叠加,看每一步对检索指标的边际贡献:
|
|
6
|
-
|
|
7
|
-
轴 1 实体规范化 —— 全库 df 聚合出的 canonical 词 + 人物/地点/时间 → tags(entity 路)
|
|
8
|
-
轴 2 指代消解 —— **仅当本条自身无实体**时承接**前一条自身**的 canonical(保守附加)
|
|
9
|
-
轴 3 意图抽象 —— 摘要 → 规范意图类目;类目 + 动词原形 → tags,类目 → 正文子功能行
|
|
10
|
-
轴 4 关系图遍历 —— 人物/事件/时间/身份/地点/条件 多维关系 → edges(graph 路)
|
|
11
|
-
|
|
12
|
-
首轮五臂实测出现三项反直觉负增益,经取证判定为**实现/口径缺陷**而非能力缺陷
|
|
13
|
-
(结论已归档灵枢记忆;本段为修正说明):
|
|
14
|
-
* 轴2 原实现无条件承接,且 `last_canon = cur`(含继承)单调累积 → tags 退化为
|
|
15
|
-
「历史全集」,后段条目对任何查询都命中 entity 路(-15~-20pp)。本文件已改为
|
|
16
|
-
规格语义;修正后本探针集**无「自身无实体」条目 → 该轴未激活**(`承接条数=0`)。
|
|
17
|
-
* 轴3 原产出是摘要切片(非抽象),且只进正文、只喂词法路;RRF 按**名次**融合,
|
|
18
|
-
该行只改变 jaccard 分母不改名次 → 边际恒 0。本文件已改为规范类目 + tags 通道。
|
|
19
|
-
* 轴4 `_path_graph` 契约要求「种子已排序」,但 `_lexical` 在 LIKE 命中
|
|
20
|
-
≤ `GLOBAL_CAP`(500) 时返回**未排序**列表 → seeds[:5] 退化为索引枚举序前 5 条,
|
|
21
|
-
图路对多题注入同一恒定集合。`run_seed_control` 用于隔离该口径缺陷。
|
|
22
|
-
|
|
23
|
-
数据源(**全部既有真源,零新增标注**):
|
|
24
|
-
* data/external/zh_probe/manual_zh.json 20 条 gold turn 的中文层(既有产物)
|
|
25
|
-
* data/external/zh_probe/manual_q.json 20 条中文查询
|
|
26
|
-
* data/external/longmemeval/lme_s_haystack.jsonl 英文原文(按 id 取)
|
|
27
|
-
* data/external/longmemeval/lme_s_questions.jsonl qid → qtype / evidence_turns
|
|
28
|
-
|
|
29
|
-
诚实条款(报告必须原样声明):
|
|
30
|
-
* 中文层与查询均为**既有产物**(非本次生成),本脚本不改一字,只做结构解析;
|
|
31
|
-
* 写入侧加工是**确定性纯规则**(零 LLM、零第三方依赖、同输入必同输出),
|
|
32
|
-
词表随源码公开,第三方可重放;
|
|
33
|
-
* 加工**盲于查询集**:canonical 判定只用全库 df 与中文层自身字段,
|
|
34
|
-
不读 manual_q.json 的任何内容——否则即数据泄漏,评测作废;
|
|
35
|
-
* 池仅 20 条 → 随机基线 hit@1 = 5%,单项能力 ±1 题即 ±5 个百分点。
|
|
36
|
-
结论只能作**方向性证据**,不可作定量结论。
|
|
37
|
-
"""
|
|
38
|
-
import json
|
|
39
|
-
import os
|
|
40
|
-
import re
|
|
41
|
-
import sys
|
|
42
|
-
|
|
43
|
-
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
44
|
-
# 本脚本既可能被当脚本跑(sys.path[0] 是 md_cg/),也可能被当包内模块导入
|
|
45
|
-
for _p in (HERE, os.path.join(HERE, "md_cg")):
|
|
46
|
-
if _p not in sys.path:
|
|
47
|
-
sys.path.insert(0, _p)
|
|
48
|
-
|
|
49
|
-
EXT = os.path.join(HERE, "data", "external")
|
|
50
|
-
ZP = os.path.join(EXT, "zh_probe")
|
|
51
|
-
LM = os.path.join(EXT, "longmemeval")
|
|
52
|
-
|
|
53
|
-
MANUAL_ZH = os.path.join(ZP, "manual_zh.json")
|
|
54
|
-
MANUAL_Q = os.path.join(ZP, "manual_q.json")
|
|
55
|
-
LM_H = os.path.join(LM, "lme_s_haystack.jsonl")
|
|
56
|
-
LM_Q = os.path.join(LM, "lme_s_questions.jsonl")
|
|
57
|
-
|
|
58
|
-
CORPUS20 = os.path.join(ZP, "corpus20.jsonl")
|
|
59
|
-
QUESTIONS20 = os.path.join(ZP, "questions20.jsonl")
|
|
60
|
-
|
|
61
|
-
# 关系轴:用户裁决「条件链 + 人物/事件/时间/身份/地点都可以作为关系」
|
|
62
|
-
AXES = ("person", "event", "time", "identity", "place", "condition")
|
|
63
|
-
|
|
64
|
-
|
|
65
|
-
# 生效条件:path 以 UTF-8 打开后逐行读取,仅 strip 后非空的行经 json.loads 产出,空白行被跳过。
|
|
66
|
-
def iter_jsonl(path):
|
|
67
|
-
with open(path, encoding="utf-8") as f:
|
|
68
|
-
for line in f:
|
|
69
|
-
line = line.strip()
|
|
70
|
-
if line:
|
|
71
|
-
yield json.loads(line)
|
|
72
|
-
|
|
73
|
-
|
|
74
|
-
# 生效条件:path 以 UTF-8 打开成功时返回 json.load(f) 的解析结果,片段内无其它分支。
|
|
75
|
-
def load_json(path):
|
|
76
|
-
with open(path, encoding="utf-8") as f:
|
|
77
|
-
return json.load(f)
|
|
78
|
-
|
|
79
|
-
|
|
80
|
-
# 生效条件:raw 经 str() 后按“;”切分,仅含“=”的段被解析,其中键为身份/时间/摘要时写对应槽位、键为条件时按“|”与“:”写入 condition、键为词时按“,”拆出非空项写入 terms,其它键或未出现的槽位保持 out 的默认空值。
|
|
81
|
-
def parse_zh(raw):
|
|
82
|
-
"""中文层串 → 槽位 dict。
|
|
83
|
-
|
|
84
|
-
格式(manual_zh.json 既有形态,本函数只解析不改写):
|
|
85
|
-
身份=用户;时间=2023/05/24 04:49;摘要=…;
|
|
86
|
-
条件=观测位置:会话陈述|观测工具:会话记录|时间窗口:2023-05-24|存在约束:公开;
|
|
87
|
-
词=礼物,礼物清单,场合
|
|
88
|
-
"""
|
|
89
|
-
out = {"identity": "", "time": "", "summary": "", "condition": {},
|
|
90
|
-
"terms": []}
|
|
91
|
-
for seg in str(raw).split(";"):
|
|
92
|
-
seg = seg.strip()
|
|
93
|
-
if not seg or "=" not in seg:
|
|
94
|
-
continue
|
|
95
|
-
key, val = seg.split("=", 1)
|
|
96
|
-
key, val = key.strip(), val.strip()
|
|
97
|
-
if key == "身份":
|
|
98
|
-
out["identity"] = val
|
|
99
|
-
elif key == "时间":
|
|
100
|
-
out["time"] = val
|
|
101
|
-
elif key == "摘要":
|
|
102
|
-
out["summary"] = val
|
|
103
|
-
elif key == "条件":
|
|
104
|
-
for kv in val.split("|"):
|
|
105
|
-
if ":" in kv:
|
|
106
|
-
k, v = kv.split(":", 1)
|
|
107
|
-
out["condition"][k.strip()] = v.strip()
|
|
108
|
-
elif key == "词":
|
|
109
|
-
out["terms"] = [t.strip() for t in val.split(",") if t.strip()]
|
|
110
|
-
return out
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
# 生效条件:当 MANUAL_ZH/MANUAL_Q/LM_Q/LM_H 可读取、hay 覆盖 want_ids 且 meta 含 man_q 全部 qid 时写出 CORPUS20/QUESTIONS20 并返回 (corpus, questions),缺 turn 或 qid 分别 raise SystemExit,verbose(默认 True)为真值时额外打印统计、假值时静默。
|
|
114
|
-
def prepare(verbose=True):
|
|
115
|
-
"""生成 corpus20.jsonl 与 questions20.jsonl(幂等覆盖)。"""
|
|
116
|
-
man_zh = load_json(MANUAL_ZH)
|
|
117
|
-
man_q = load_json(MANUAL_Q)
|
|
118
|
-
meta = {r["qid"]: r for r in iter_jsonl(LM_Q)}
|
|
119
|
-
want_ids = set(man_zh)
|
|
120
|
-
hay = {}
|
|
121
|
-
for r in iter_jsonl(LM_H):
|
|
122
|
-
if r["id"] in want_ids:
|
|
123
|
-
hay[r["id"]] = r
|
|
124
|
-
if len(hay) == len(want_ids):
|
|
125
|
-
break
|
|
126
|
-
|
|
127
|
-
missing = want_ids - set(hay)
|
|
128
|
-
if missing:
|
|
129
|
-
raise SystemExit(f"[失败] haystack 缺以下 turn:{sorted(missing)}")
|
|
130
|
-
|
|
131
|
-
# corpus:按 turn id 的数值序(= 时间序),保证「承接前一条」有确定语义
|
|
132
|
-
# 生效条件:tid 经 rsplit("t",1) 得到至少两段且最后一段可被 int() 解析时返回该整数,否则源码未做校验会抛错。
|
|
133
|
-
def turn_no(tid):
|
|
134
|
-
return int(tid.rsplit("t", 1)[1])
|
|
135
|
-
|
|
136
|
-
corpus = []
|
|
137
|
-
for tid in sorted(man_zh, key=turn_no):
|
|
138
|
-
t = hay[tid]
|
|
139
|
-
corpus.append({
|
|
140
|
-
"id": tid,
|
|
141
|
-
"text": str(t.get("text") or ""),
|
|
142
|
-
"speaker": str(t.get("speaker") or ""),
|
|
143
|
-
"date": str(t.get("date") or ""),
|
|
144
|
-
"zh": man_zh[tid],
|
|
145
|
-
"zh_fields": parse_zh(man_zh[tid]),
|
|
146
|
-
})
|
|
147
|
-
|
|
148
|
-
# questions:qid → 中文查询 + 证据(完整 evidence_turns,池里仅 ev0 存在)
|
|
149
|
-
questions = []
|
|
150
|
-
for qid, q in man_q.items():
|
|
151
|
-
m = meta.get(qid)
|
|
152
|
-
if m is None:
|
|
153
|
-
raise SystemExit(f"[失败] questions 缺 qid={qid}")
|
|
154
|
-
questions.append({
|
|
155
|
-
"qid": qid,
|
|
156
|
-
"qtype": m["qtype"],
|
|
157
|
-
"question": q,
|
|
158
|
-
"answer": str(m.get("answer") or ""),
|
|
159
|
-
"evidence_turns": list(m.get("evidence_turns") or []),
|
|
160
|
-
})
|
|
161
|
-
|
|
162
|
-
with open(CORPUS20, "w", encoding="utf-8") as f:
|
|
163
|
-
for r in corpus:
|
|
164
|
-
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
165
|
-
with open(QUESTIONS20, "w", encoding="utf-8") as f:
|
|
166
|
-
for r in questions:
|
|
167
|
-
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
168
|
-
|
|
169
|
-
if verbose:
|
|
170
|
-
print(f"语料 {len(corpus)} 条 → {CORPUS20}")
|
|
171
|
-
print(f"题库 {len(questions)} 条 → {QUESTIONS20}")
|
|
172
|
-
n_etype = {}
|
|
173
|
-
for q in questions:
|
|
174
|
-
n_etype[q["qtype"]] = n_etype.get(q["qtype"], 0) + 1
|
|
175
|
-
print("题型分布:" + ", ".join(f"{k}={v}" for k, v in sorted(n_etype.items())))
|
|
176
|
-
return corpus, questions
|
|
177
|
-
|
|
178
|
-
|
|
179
|
-
# 生效条件:prepare(verbose=False) 返回 corpus/questions 后,对每题以 evidence_turns[0](空则 "")取 ev0,命中词限于 ev0 非空且词出现在 zh_of[ev0] 中,df1 收集池内 df==1 的词、all_df 收集 df==len(corpus) 的词,verbose(默认 True)为真值时打印后返回 per_q、为假值时直接返回 per_q。
|
|
180
|
-
def analyze(verbose=True):
|
|
181
|
-
"""诊断:查询词能否落到 gold 中文层 / 池内其他条目(决定各轴是否有可桥接的实体)。
|
|
182
|
-
|
|
183
|
-
只读既有产物,不修改任何语料。输出两类事实:
|
|
184
|
-
* 查询词在 gold 中文层的字面命中率 → 词法路的天花板;
|
|
185
|
-
* 查询词在**池内**的 df → 该词的判别力(df=1 是精确定位,df=20 则无信息量)。
|
|
186
|
-
"""
|
|
187
|
-
corpus, questions = prepare(verbose=False)
|
|
188
|
-
zh_of = {c["id"]: c["zh"] for c in corpus}
|
|
189
|
-
|
|
190
|
-
per_q = []
|
|
191
|
-
for q in questions:
|
|
192
|
-
ev0 = q["evidence_turns"][0] if q["evidence_turns"] else ""
|
|
193
|
-
terms = [t for t in str(q["question"]).split() if t]
|
|
194
|
-
hit = [t for t in terms if ev0 and t in zh_of.get(ev0, "")]
|
|
195
|
-
globalish = [t for t in terms
|
|
196
|
-
if sum(1 for c in corpus if t in zh_of[c["id"]]) == len(corpus)]
|
|
197
|
-
per_q.append({
|
|
198
|
-
"qid": q["qid"], "ev0": ev0, "qtype": q["qtype"],
|
|
199
|
-
"n_term": len(terms), "n_hit": len(hit),
|
|
200
|
-
"hit_terms": hit,
|
|
201
|
-
"df1": sorted({t for t in terms
|
|
202
|
-
if sum(1 for c in corpus if t in zh_of[c["id"]]) == 1}),
|
|
203
|
-
"all_df": globalish,
|
|
204
|
-
})
|
|
205
|
-
|
|
206
|
-
if verbose:
|
|
207
|
-
print(f"\n== 查询词覆盖诊断(池 {len(corpus)} 条,题 {len(questions)} 道)==")
|
|
208
|
-
print(f"{'qid':<16}{'ev0':<12}{'词数':>5}{'命中':>5} 命中词")
|
|
209
|
-
print("-" * 72)
|
|
210
|
-
for r in per_q:
|
|
211
|
-
print(f"{r['qid']:<16}{r['ev0']:<12}{r['n_term']:>5}{r['n_hit']:>5} "
|
|
212
|
-
f"{','.join(r['hit_terms'])[:40]}")
|
|
213
|
-
tot_t = sum(r["n_term"] for r in per_q)
|
|
214
|
-
tot_h = sum(r["n_hit"] for r in per_q)
|
|
215
|
-
zero = [r["qid"] for r in per_q if r["n_hit"] == 0]
|
|
216
|
-
print("-" * 72)
|
|
217
|
-
print(f"合计:{tot_h}/{tot_t} 个查询词落在 gold 中文层"
|
|
218
|
-
f"({tot_h / max(1, tot_t):.1%})")
|
|
219
|
-
print(f"零重叠题({len(zero)}):{', '.join(zero) if zero else '无'}")
|
|
220
|
-
return per_q
|
|
221
|
-
|
|
222
|
-
return per_q
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
# ---------------------------------------------------------------- 写入侧四轴
|
|
226
|
-
# 全部为确定性纯规则:零 LLM、零外部依赖,同输入必同输出;词表随源码公开。
|
|
227
|
-
# **盲于查询集**:以下规则只读 corpus20(语料)与中文层自身字段,从不打开 manual_q.json。
|
|
228
|
-
|
|
229
|
-
STOP_ZH = frozenset(
|
|
230
|
-
"的 了 在 是 我 你 他 她 它 和 与 或 这 那 有 没 不 就 都 也 还 要 会 能 可以 "
|
|
231
|
-
"多少 什么 哪里 哪 怎么 为什么 几 一些 一个 之 其 与 及 等 被 把 给 对 从 到".split()
|
|
232
|
-
)
|
|
233
|
-
|
|
234
|
-
PERSON_HINTS = (
|
|
235
|
-
"妹妹", "姐姐", "哥哥", "弟弟", "妈妈", "爸爸", "母亲", "父亲", "父母", "奶奶",
|
|
236
|
-
"爷爷", "外婆", "外公", "同事", "朋友", "同学", "邻居", "老板", "上司", "教练",
|
|
237
|
-
"医生", "老师", "室友", "伴侣", "丈夫", "妻子", "儿子", "女儿", "叔叔", "阿姨",
|
|
238
|
-
"表亲", "未婚夫", "未婚妻",
|
|
239
|
-
)
|
|
240
|
-
|
|
241
|
-
PLACE_HINTS = (
|
|
242
|
-
"市", "州", "城", "镇", "岛", "公园", "中心", "大学", "海滩", "广场", "机场",
|
|
243
|
-
"河", "湖", "山", "街", "路", "区", "国家", "澳洲", "欧洲", "亚洲", "美洲",
|
|
244
|
-
)
|
|
245
|
-
|
|
246
|
-
EN_STOP = frozenset("""
|
|
247
|
-
The A An And But Or For Nor So Yet This That These Those There Their They Them Then Than
|
|
248
|
-
When What Where Which While Who Whom Whose Why How However Also Always Never Often Sometimes
|
|
249
|
-
I You He She It We They Me Him Her Us My Your His Its Our Very Really Just Only Even Still
|
|
250
|
-
Have Has Had Having Do Does Did Doing Be Been Being Am Is Are Was Were Will Would Shall
|
|
251
|
-
Should Can Could May Might Must Not No Yes If In On At By To Of With From Into Onto Over
|
|
252
|
-
Under Above Below Between During After Before About Around Because Since Until Upon While
|
|
253
|
-
Monday Tuesday Wednesday Thursday Friday Saturday Sunday January February March April May
|
|
254
|
-
June July August September October November December Today Yesterday Tomorrow Last Next
|
|
255
|
-
New Old Good Bad Best Worst First Second Third One Two Three Four Five Six Seven Eight Nine
|
|
256
|
-
Ten Okay Ok Well Thanks Thank Please Hi Hello Hey
|
|
257
|
-
""".split())
|
|
258
|
-
|
|
259
|
-
EN_NAME_RE = re.compile(r"\b[A-Z][a-z]{2,}(?:[ ][A-Z][a-z]{2,})*\b")
|
|
260
|
-
DATE_RE = re.compile(r"(\d{4})[/-](\d{1,2})[/-](\d{1,2})")
|
|
261
|
-
|
|
262
|
-
|
|
263
|
-
# 生效条件:DATE_RE.search(str(raw or "")) 有匹配时返回零填充的 YYYY-MM-DD,无匹配(含 raw 为假值回落成 "")时返回 ""。
|
|
264
|
-
def norm_time(raw):
|
|
265
|
-
"""'2023/05/24 04:49' → '2023-05-24'(跨条目统一格式,使日期查询可命中)。"""
|
|
266
|
-
m = DATE_RE.search(str(raw or ""))
|
|
267
|
-
if not m:
|
|
268
|
-
return ""
|
|
269
|
-
y, mo, d = m.groups()
|
|
270
|
-
return f"{int(y):04d}-{int(mo):02d}-{int(d):02d}"
|
|
271
|
-
|
|
272
|
-
|
|
273
|
-
# 生效条件:对 str(text or "") 用 EN_NAME_RE 扫描,匹配串按空白切分后任一部分命中 EN_STOP 即跳过,其余项按首次出现顺序去重加入 out 并返回;text 为假值时扫描空串返回 []。
|
|
274
|
-
def en_names(text):
|
|
275
|
-
"""英文原文中的专名候选(跨语言桥:中文查询里的英文专名可经 entity 路命中)。"""
|
|
276
|
-
out, seen = [], set()
|
|
277
|
-
for m in EN_NAME_RE.finditer(str(text or "")):
|
|
278
|
-
w = m.group(0)
|
|
279
|
-
if any(p in EN_STOP for p in w.split()):
|
|
280
|
-
continue
|
|
281
|
-
if w not in seen:
|
|
282
|
-
seen.add(w)
|
|
283
|
-
out.append(w)
|
|
284
|
-
return out
|
|
285
|
-
|
|
286
|
-
|
|
287
|
-
# 生效条件:遍历 corpus,对每条记录的 zh_fields["terms"] 去重后逐词判断,词非空且不在 STOP_ZH 时使 df[词] 计数加一,返回 df。
|
|
288
|
-
def build_tables(corpus):
|
|
289
|
-
"""全库统计:# 词项 df(**只读语料**,与查询集无关)。"""
|
|
290
|
-
df = {}
|
|
291
|
-
for c in corpus:
|
|
292
|
-
for t in set(c["zh_fields"]["terms"]):
|
|
293
|
-
if t and t not in STOP_ZH:
|
|
294
|
-
df[t] = df.get(t, 0) + 1
|
|
295
|
-
return df
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
# 生效条件:c 含 zh_fields(terms/condition/time/identity)且 df 为词到频次的映射时,返回 person/event/time/identity/place/condition 六轴取值字典。
|
|
299
|
-
def axis_values(c, df):
|
|
300
|
-
"""一条 turn 的六个关系轴取值(用户裁决:条件链 + 人物/事件/时间/身份/地点)。"""
|
|
301
|
-
f = c["zh_fields"]
|
|
302
|
-
terms = [t for t in f["terms"] if t and t not in STOP_ZH]
|
|
303
|
-
tw = norm_time(f["condition"].get("时间窗口") or f["time"])
|
|
304
|
-
person = [t for t in terms if any(h in t for h in PERSON_HINTS)]
|
|
305
|
-
place = [t for t in terms if any(h in t for h in PLACE_HINTS)]
|
|
306
|
-
event = [t for t in terms if df.get(t, 0) >= 2]
|
|
307
|
-
return {
|
|
308
|
-
"person": person,
|
|
309
|
-
"event": event,
|
|
310
|
-
"time": [tw] if tw else [],
|
|
311
|
-
"identity": [f["identity"]] if f["identity"] else [],
|
|
312
|
-
"place": place,
|
|
313
|
-
"condition": [f"{k}:{v}" for k, v in f["condition"].items()],
|
|
314
|
-
}
|
|
315
|
-
|
|
316
|
-
|
|
317
|
-
# 意图类目:意图动词 → 规范类目(确定性词表,只读摘要,与查询集无关)
|
|
318
|
-
INTENT_CLASSES = (
|
|
319
|
-
("推荐", ("推荐", "建议", "咨询", "请教")),
|
|
320
|
-
("购置", ("购买", "买", "挑选", "选购", "下单", "购买")),
|
|
321
|
-
("整理", ("整理", "清理", "分类", "收纳", "归档")),
|
|
322
|
-
("学习", ("学习", "了解", "阅读", "查阅", "研究")),
|
|
323
|
-
("规划", ("计划", "安排", "准备", "制定")),
|
|
324
|
-
("比较", ("比较", "对比", "估算", "计算")),
|
|
325
|
-
("维修", ("维修", "修理", "更换", "保养")),
|
|
326
|
-
("出行", ("旅行", "搬家", "搬迁", "游玩", "漂流")),
|
|
327
|
-
("参加", ("参加", "报名", "出席")),
|
|
328
|
-
("健身", ("锻炼", "跑步", "训练")),
|
|
329
|
-
("取退", ("退还", "退回", "归还")),
|
|
330
|
-
("完成", ("完成", "做完", "结课")),
|
|
331
|
-
)
|
|
332
|
-
|
|
333
|
-
|
|
334
|
-
# 生效条件:按 INTENT_CLASSES 顺序检查 c["zh_fields"]["summary"],首个命中类目的首个包含于摘要的动词返回 (cls, v),全不命中返回 ('', '')。
|
|
335
|
-
def intent_of(c):
|
|
336
|
-
"""意图抽象:摘要 → (规范类目, 命中的动词原形)。
|
|
337
|
-
|
|
338
|
-
规格要求「抽象」。原实现返回摘要切片 `f"{v}:" + summary[i-6:i+10]`——
|
|
339
|
-
那只是把原文再抄一遍:不产生任何新词、不构成类目,且只写进正文,
|
|
340
|
-
而正文只喂词法路、RRF 又只按**名次**融合(见 bench 报告),故边际恒为 0。
|
|
341
|
-
此处归一为固定类目,并保留命中的动词原形,二者一并进 tags(entity 路——
|
|
342
|
-
存在性匹配,不吃分数尺度),使该轴真正获得独立检索通道。
|
|
343
|
-
"""
|
|
344
|
-
s = c["zh_fields"]["summary"]
|
|
345
|
-
for cls, verbs in INTENT_CLASSES:
|
|
346
|
-
for v in verbs:
|
|
347
|
-
if v in s:
|
|
348
|
-
return cls, v
|
|
349
|
-
return "", ""
|
|
350
|
-
|
|
351
|
-
|
|
352
|
-
# 生效条件:把 c["zh_fields"]["terms"] 与 en_names(c["text"]) 逐项 strip 后跳过空串、STOP_ZH(原形或小写)及已见项去重入 out,再用 norm_time(条件.get("时间窗口") or c 的 time) 得到非空且未出现的时窗串追加;df 形参在该片段内未参与条件判断。
|
|
353
|
-
def normalize_terms(c, df):
|
|
354
|
-
"""实体规范化:去停用 + 去重 + 附时间规范式 + 附英文专名(跨语言对齐)。"""
|
|
355
|
-
f = c["zh_fields"]
|
|
356
|
-
out, seen = [], set()
|
|
357
|
-
for t in list(f["terms"]) + en_names(c["text"]):
|
|
358
|
-
t = t.strip()
|
|
359
|
-
if not t or t in STOP_ZH or t.lower() in STOP_ZH or t in seen:
|
|
360
|
-
continue
|
|
361
|
-
seen.add(t)
|
|
362
|
-
out.append(t)
|
|
363
|
-
tw = norm_time(f["condition"].get("时间窗口") or f["time"])
|
|
364
|
-
if tw and tw not in seen:
|
|
365
|
-
out.append(tw)
|
|
366
|
-
return out
|
|
367
|
-
|
|
368
|
-
|
|
369
|
-
# 生效条件:遍历 corpus 并以 root=os.path.join(HERE, root_base or f"_md_cg_eval_zhprobe_{arm['name']}")(root_base 为 None/空串时回落)建库,graph 分支仅当 arm.get("graph") 为真且某轴取值的同值 id 数落在 [2, max_df](默认 5)时为这些 id 两两建有向边,coref 仅当 arm.get("coref") 为真、本条 own 为空(own 只在 arm.get("norm") 为真时由 normalize_terms 生成,否则恒为 [])且 prev_own 非空时承接前一条自身 canonical,intent 分支仅当 arm.get("intent") 为真且 intent_of 返回 cls 非空时写意图行并把意图词与长度>=2 的动词加入 tags,verbose(默认 True)为真值时打印臂统计、root 已是目录时先整树删除再建 MdCGOS。
|
|
370
|
-
def build_arm(corpus, arm, max_df=5, root_base=None, verbose=True):
|
|
371
|
-
"""按消融臂建库。每臂一个独立 root,互不污染。
|
|
372
|
-
|
|
373
|
-
臂的差异**只体现在库内容**(正文/tags/edges),检索路固定
|
|
374
|
-
lexical+entity+graph——这样"能力未开"就等于"库内没有对应信息",
|
|
375
|
-
归因干净:指标变化可直接归给该轴,而不是归给路开关。
|
|
376
|
-
"""
|
|
377
|
-
from md_cg.mdcos import MdCGOS
|
|
378
|
-
root = os.path.join(HERE, root_base or f"_md_cg_eval_zhprobe_{arm['name']}")
|
|
379
|
-
|
|
380
|
-
df = build_tables(corpus)
|
|
381
|
-
axes = {c["id"]: axis_values(c, df) for c in corpus}
|
|
382
|
-
|
|
383
|
-
edges = {}
|
|
384
|
-
if arm.get("graph"):
|
|
385
|
-
for ax in AXES:
|
|
386
|
-
val2ids = {}
|
|
387
|
-
for c in corpus:
|
|
388
|
-
for v in axes[c["id"]][ax]:
|
|
389
|
-
val2ids.setdefault(v, set()).add(c["id"])
|
|
390
|
-
for v, ids in val2ids.items():
|
|
391
|
-
if not (2 <= len(ids) <= max_df):
|
|
392
|
-
continue # df=1 无边;df>max_df 全连通,零信息量只注入噪声
|
|
393
|
-
for a in ids:
|
|
394
|
-
for b in ids:
|
|
395
|
-
if a != b:
|
|
396
|
-
edges.setdefault(a, {})[b] = ax
|
|
397
|
-
|
|
398
|
-
if os.path.isdir(root):
|
|
399
|
-
import shutil
|
|
400
|
-
shutil.rmtree(root, ignore_errors=True)
|
|
401
|
-
cg = MdCGOS(root, autoflush=500)
|
|
402
|
-
|
|
403
|
-
prev_own = None # 前一条**自身**的 canonical(不累积)
|
|
404
|
-
n_coref = 0
|
|
405
|
-
for c in corpus:
|
|
406
|
-
f = c["zh_fields"]
|
|
407
|
-
own = normalize_terms(c, df) if arm.get("norm") else []
|
|
408
|
-
|
|
409
|
-
# 轴2 指代消解(规格:保守附加)——**仅当本条自身无实体**时,
|
|
410
|
-
# 才承接**前一条自身**的 canonical。
|
|
411
|
-
# 原实现:`if coref and last_canon` 无条件承接,且 `last_canon = cur`
|
|
412
|
-
# 把「继承结果」也计入 → 单调累积成「历史全集」,tags 雪球污染。
|
|
413
|
-
inherit = []
|
|
414
|
-
if arm.get("coref") and not own and prev_own:
|
|
415
|
-
inherit = list(prev_own)
|
|
416
|
-
n_coref += 1
|
|
417
|
-
|
|
418
|
-
# 轴3 意图抽象:类目写正文(规格的「子功能行」),类目词 + 动词原形写 tags。
|
|
419
|
-
intent_terms = []
|
|
420
|
-
intent_line = ""
|
|
421
|
-
if arm.get("intent"):
|
|
422
|
-
cls, verb = intent_of(c)
|
|
423
|
-
if cls:
|
|
424
|
-
intent_line = f"意图:{cls}"
|
|
425
|
-
intent_terms = [f"意图:{cls}"] + ([verb] if len(verb) >= 2 else [])
|
|
426
|
-
|
|
427
|
-
body = [f"身份:{f['identity']}", f"时间:{f['time']}",
|
|
428
|
-
f"摘要:{f['summary']}", f"词:{','.join(f['terms'])}",
|
|
429
|
-
"条件:" + "|".join(f"{k}:{v}" for k, v in f["condition"].items())]
|
|
430
|
-
if intent_line:
|
|
431
|
-
body.append(intent_line)
|
|
432
|
-
if inherit:
|
|
433
|
-
body.append("承接:" + ",".join(inherit))
|
|
434
|
-
body.append("")
|
|
435
|
-
body.append(c["text"])
|
|
436
|
-
|
|
437
|
-
tags = own[:]
|
|
438
|
-
for t in intent_terms + inherit:
|
|
439
|
-
if t not in tags:
|
|
440
|
-
tags.append(t)
|
|
441
|
-
|
|
442
|
-
e = [{"target": t, "axis": a} for t, a in edges.get(c["id"], {}).items()]
|
|
443
|
-
cg.add(c["id"], "\n".join(body), layer="contextual", tags=tags,
|
|
444
|
-
edges=e, eval_src=f"zh_mad:{arm['name']}", verification_basis="data")
|
|
445
|
-
|
|
446
|
-
prev_own = own[:]
|
|
447
|
-
|
|
448
|
-
cg.flush()
|
|
449
|
-
if verbose:
|
|
450
|
-
n_e = sum(len(v) for v in edges.values())
|
|
451
|
-
print(f" [{arm['name']}] root={os.path.basename(root)} "
|
|
452
|
-
f"tags_axis={'on' if arm.get('norm') else 'off'} "
|
|
453
|
-
f"coref={'on' if arm.get('coref') else 'off'} "
|
|
454
|
-
f"intent={'on' if arm.get('intent') else 'off'} "
|
|
455
|
-
f"graph={'on' if arm.get('graph') else 'off'} 有向边={n_e} "
|
|
456
|
-
f"承接条数={n_coref}")
|
|
457
|
-
return root
|
|
458
|
-
|
|
459
|
-
|
|
460
|
-
ARMS = [
|
|
461
|
-
{"name": "a0_base", "norm": False, "coref": False, "intent": False, "graph": False},
|
|
462
|
-
{"name": "a1_norm", "norm": True, "coref": False, "intent": False, "graph": False},
|
|
463
|
-
{"name": "a2_coref", "norm": True, "coref": True, "intent": False, "graph": False},
|
|
464
|
-
{"name": "a3_intent", "norm": True, "coref": True, "intent": True, "graph": False},
|
|
465
|
-
{"name": "a4_graph", "norm": True, "coref": True, "intent": True, "graph": True},
|
|
466
|
-
]
|
|
467
|
-
|
|
468
|
-
|
|
469
|
-
# 生效条件:先执行 prepare(verbose=False) 取得 corpus,再对 ARMS 每臂以 max_df=max_df(默认 5,原样下传不做假值回落)调用 build_arm 并汇总为 roots 返回。
|
|
470
|
-
def build_all(max_df=5):
|
|
471
|
-
corpus, _ = prepare(verbose=False)
|
|
472
|
-
print(f"== 建库(消融 {len(ARMS)} 臂,max_df={max_df})==")
|
|
473
|
-
roots = {}
|
|
474
|
-
for arm in ARMS:
|
|
475
|
-
roots[arm["name"]] = build_arm(corpus, arm, max_df=max_df)
|
|
476
|
-
return roots
|
|
477
|
-
|
|
478
|
-
|
|
479
|
-
RUST_BIN = os.path.join(HERE, "rust", "target", "release", "mdcg-eval.exe")
|
|
480
|
-
ROW_RE = re.compile(
|
|
481
|
-
r"^\s*(precise|temporal|interference|reference)\s+(\d+)\s+"
|
|
482
|
-
r"([\d.]+)%\s+([\d.]+)%\s+([\d.]+)\s*$", re.M)
|
|
483
|
-
GROUPS = ["precise", "temporal", "interference", "reference"]
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
# 生效条件:os.path.exists(RUST_BIN) 为真时以 argv=[RUST_BIN,"--dataset","mad","--tag",name,"--lib",lib](extra 为真值时追加 list(extra))执行 subprocess,返回码非 0 或 ROW_RE 在 stdout 未匹配到任何组时 raise SystemExit,否则返回 {组:(n,hit@1,hit@5,MRR)}。
|
|
487
|
-
def run_one(name, lib, extra=None):
|
|
488
|
-
"""调用 **Rust 检索器** 跑一臂,返回 {组: (n, hit@1, hit@5, MRR)}。
|
|
489
|
-
|
|
490
|
-
命令执行走 subprocess argv 列表 + 显式 UTF-8 + PYTHONUTF8=1(不经 Windows shell),
|
|
491
|
-
规避 GBK 解码异常。
|
|
492
|
-
"""
|
|
493
|
-
import subprocess
|
|
494
|
-
|
|
495
|
-
if not os.path.exists(RUST_BIN):
|
|
496
|
-
raise SystemExit(f"[失败] 未找到 Rust 评测器:{RUST_BIN}(先 cargo build --release)")
|
|
497
|
-
argv = [RUST_BIN, "--dataset", "mad", "--tag", name, "--lib", lib]
|
|
498
|
-
if extra:
|
|
499
|
-
argv += list(extra)
|
|
500
|
-
env = dict(os.environ, PYTHONUTF8="1")
|
|
501
|
-
p = subprocess.run(argv, capture_output=True, text=True,
|
|
502
|
-
encoding="utf-8", errors="replace", env=env, cwd=HERE)
|
|
503
|
-
if p.returncode != 0:
|
|
504
|
-
raise SystemExit(f"[失败] {name}:{(p.stderr or '')[-900:]}")
|
|
505
|
-
got = {}
|
|
506
|
-
for m in ROW_RE.finditer(p.stdout):
|
|
507
|
-
g, n, h1, h5, mrr = m.groups()
|
|
508
|
-
got[g] = (int(n), float(h1), float(h5), float(mrr))
|
|
509
|
-
if not got:
|
|
510
|
-
raise SystemExit(f"[失败] {name}:未解析到组指标\n{p.stdout[-1200:]}")
|
|
511
|
-
return got
|
|
512
|
-
|
|
513
|
-
|
|
514
|
-
# 生效条件:先打印 title 与表头,再遍历 rows 的 (label, got, note),对 GROUPS 每组用 got[g] 取 (n,h1,h5,mrr) 并累计总体 hit@1/MRR,仅当 baseline 不为 None 且 label==baseline 时记 ref,仅当 ref 已记录且 label!=baseline 时输出 Δ。
|
|
515
|
-
def print_table(title, rows, baseline=None):
|
|
516
|
-
"""rows: [(label, got, note)];baseline = 参照行 label(算 Δ)。"""
|
|
517
|
-
print(f"\n== {title} ==")
|
|
518
|
-
hdr = (f"{'臂':<14}" + "".join(f"{g[:12]:>15}" for g in GROUPS)
|
|
519
|
-
+ f"{'总体hit@1':>12}{'总体MRR':>10}")
|
|
520
|
-
print(hdr)
|
|
521
|
-
print("-" * len(hdr))
|
|
522
|
-
ref = None
|
|
523
|
-
for label, got, note in rows:
|
|
524
|
-
cells, tot_h1, tot_n, tot_mrr = [], 0.0, 0, 0.0
|
|
525
|
-
for g in GROUPS:
|
|
526
|
-
n, h1, _h5, mrr = got[g]
|
|
527
|
-
cells.append(f"{h1:.0f}%/{mrr:.3f}")
|
|
528
|
-
tot_h1 += h1 / 100.0 * n
|
|
529
|
-
tot_n += n
|
|
530
|
-
tot_mrr += mrr * n
|
|
531
|
-
ov_h1 = tot_h1 / max(1, tot_n)
|
|
532
|
-
ov_mrr = tot_mrr / max(1, tot_n)
|
|
533
|
-
if baseline is not None and label == baseline:
|
|
534
|
-
ref = (ov_h1, ov_mrr)
|
|
535
|
-
delta = ""
|
|
536
|
-
if ref is not None and label != baseline:
|
|
537
|
-
delta = f" (Δ{(ov_h1 - ref[0]) * 100:+.1f}pp)"
|
|
538
|
-
suffix = f" {note}" if note else ""
|
|
539
|
-
print(f"{label:<14}" + "".join(f"{c:>15}" for c in cells)
|
|
540
|
-
+ f"{ov_h1 * 100:>10.1f}%{ov_mrr:>10.3f}{delta}{suffix}")
|
|
541
|
-
|
|
542
|
-
|
|
543
|
-
# 生效条件:对 ARMS 每臂以 lib=f"_md_cg_eval_zhprobe_{name}"、无 extra 调用 run_one 组成 rows,并以 ARMS[0]["name"] 为 baseline 调 print_table 后返回 rows。
|
|
544
|
-
def run_ablation():
|
|
545
|
-
"""消融主表:逐臂调用 Rust 检索器并汇总(种子口径 = 缺省,即生产现状)。"""
|
|
546
|
-
rows = []
|
|
547
|
-
for arm in ARMS:
|
|
548
|
-
name = arm["name"]
|
|
549
|
-
rows.append((name, run_one(name, f"_md_cg_eval_zhprobe_{name}"), ""))
|
|
550
|
-
print_table("消融主表(Rust 检索器,k=5,jaccard,证据命中;graph 种子=索引序)",
|
|
551
|
-
rows, baseline=ARMS[0]["name"])
|
|
552
|
-
print("\n随机基线 hit@1 = 5.0%(1/20);池仅 20 条,±1 题 = ±5pp,"
|
|
553
|
-
"结论只作方向性证据。")
|
|
554
|
-
return rows
|
|
555
|
-
|
|
556
|
-
|
|
557
|
-
# 生效条件:以最后一个臂 ARMS[-1]['name'] 对应的库路径,先无 extra 调 run_one("a4_index", lib)、再以 extra=("--graph-seeds","sorted") 调 run_one("a4_sorted", lib) 组成 rows,并以 baseline="a4/index" 调 print_table 后返回。
|
|
558
|
-
def run_seed_control():
|
|
559
|
-
"""对照:**同一个 a4 库**(写入侧与边结构完全相同),只切换 graph 路种子口径。
|
|
560
|
-
|
|
561
|
-
用于把 a4 相对 a3 的变化拆成两个因子:
|
|
562
|
-
* 边结构质量(本轴真正要测的能力);
|
|
563
|
-
* `_path_graph`「种子须已排序」契约被违反 → seeds[:5] 退化为索引枚举序前 5。
|
|
564
|
-
"""
|
|
565
|
-
lib = f"_md_cg_eval_zhprobe_{ARMS[-1]['name']}"
|
|
566
|
-
rows = [
|
|
567
|
-
("a4/index", run_one("a4_index", lib), "现状:seeds=索引枚举序前 5"),
|
|
568
|
-
("a4/sorted", run_one("a4_sorted", lib, ("--graph-seeds", "sorted")),
|
|
569
|
-
"文档语义:seeds=top-5 词法"),
|
|
570
|
-
]
|
|
571
|
-
print_table("graph 种子口径对照(库内容完全相同,只换种子)", rows,
|
|
572
|
-
baseline="a4/index")
|
|
573
|
-
return rows
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
# 生效条件:cmd 取 argv[1](len(argv)<=1 时为 "prepare"),cmd=="prepare"/"analyze"/"build"/"run"/"seed-control" 分别调用 prepare/analyze/build_all/run_ablation/run_seed_control,cmd=="all" 依次调用 prepare、analyze、build_all、run_ablation、run_seed_control,其余值 raise SystemExit。
|
|
577
|
-
def main(argv):
|
|
578
|
-
cmd = argv[1] if len(argv) > 1 else "prepare"
|
|
579
|
-
if cmd == "prepare":
|
|
580
|
-
prepare()
|
|
581
|
-
elif cmd == "analyze":
|
|
582
|
-
analyze()
|
|
583
|
-
elif cmd == "build":
|
|
584
|
-
build_all()
|
|
585
|
-
elif cmd == "run":
|
|
586
|
-
run_ablation()
|
|
587
|
-
elif cmd == "seed-control":
|
|
588
|
-
run_seed_control()
|
|
589
|
-
elif cmd == "all":
|
|
590
|
-
prepare()
|
|
591
|
-
analyze()
|
|
592
|
-
build_all()
|
|
593
|
-
run_ablation()
|
|
594
|
-
run_seed_control()
|
|
595
|
-
else:
|
|
596
|
-
raise SystemExit(
|
|
597
|
-
f"未知子命令:{cmd}"
|
|
598
|
-
"(可用:prepare / analyze / build / run / seed-control / all)")
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
if __name__ == "__main__":
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""中文多维探针(MAD, multi-axis)——20 条「中文层 + 英文原文」语料,Rust 侧四路检索。
|
|
3
|
+
|
|
4
|
+
与既有 zh_probe 探针的关系:前一版探针只测「中文层做索引」的单点效果;本探针
|
|
5
|
+
把写入侧加工拆成可消融的四个维度,逐一叠加,看每一步对检索指标的边际贡献:
|
|
6
|
+
|
|
7
|
+
轴 1 实体规范化 —— 全库 df 聚合出的 canonical 词 + 人物/地点/时间 → tags(entity 路)
|
|
8
|
+
轴 2 指代消解 —— **仅当本条自身无实体**时承接**前一条自身**的 canonical(保守附加)
|
|
9
|
+
轴 3 意图抽象 —— 摘要 → 规范意图类目;类目 + 动词原形 → tags,类目 → 正文子功能行
|
|
10
|
+
轴 4 关系图遍历 —— 人物/事件/时间/身份/地点/条件 多维关系 → edges(graph 路)
|
|
11
|
+
|
|
12
|
+
首轮五臂实测出现三项反直觉负增益,经取证判定为**实现/口径缺陷**而非能力缺陷
|
|
13
|
+
(结论已归档灵枢记忆;本段为修正说明):
|
|
14
|
+
* 轴2 原实现无条件承接,且 `last_canon = cur`(含继承)单调累积 → tags 退化为
|
|
15
|
+
「历史全集」,后段条目对任何查询都命中 entity 路(-15~-20pp)。本文件已改为
|
|
16
|
+
规格语义;修正后本探针集**无「自身无实体」条目 → 该轴未激活**(`承接条数=0`)。
|
|
17
|
+
* 轴3 原产出是摘要切片(非抽象),且只进正文、只喂词法路;RRF 按**名次**融合,
|
|
18
|
+
该行只改变 jaccard 分母不改名次 → 边际恒 0。本文件已改为规范类目 + tags 通道。
|
|
19
|
+
* 轴4 `_path_graph` 契约要求「种子已排序」,但 `_lexical` 在 LIKE 命中
|
|
20
|
+
≤ `GLOBAL_CAP`(500) 时返回**未排序**列表 → seeds[:5] 退化为索引枚举序前 5 条,
|
|
21
|
+
图路对多题注入同一恒定集合。`run_seed_control` 用于隔离该口径缺陷。
|
|
22
|
+
|
|
23
|
+
数据源(**全部既有真源,零新增标注**):
|
|
24
|
+
* data/external/zh_probe/manual_zh.json 20 条 gold turn 的中文层(既有产物)
|
|
25
|
+
* data/external/zh_probe/manual_q.json 20 条中文查询
|
|
26
|
+
* data/external/longmemeval/lme_s_haystack.jsonl 英文原文(按 id 取)
|
|
27
|
+
* data/external/longmemeval/lme_s_questions.jsonl qid → qtype / evidence_turns
|
|
28
|
+
|
|
29
|
+
诚实条款(报告必须原样声明):
|
|
30
|
+
* 中文层与查询均为**既有产物**(非本次生成),本脚本不改一字,只做结构解析;
|
|
31
|
+
* 写入侧加工是**确定性纯规则**(零 LLM、零第三方依赖、同输入必同输出),
|
|
32
|
+
词表随源码公开,第三方可重放;
|
|
33
|
+
* 加工**盲于查询集**:canonical 判定只用全库 df 与中文层自身字段,
|
|
34
|
+
不读 manual_q.json 的任何内容——否则即数据泄漏,评测作废;
|
|
35
|
+
* 池仅 20 条 → 随机基线 hit@1 = 5%,单项能力 ±1 题即 ±5 个百分点。
|
|
36
|
+
结论只能作**方向性证据**,不可作定量结论。
|
|
37
|
+
"""
|
|
38
|
+
import json
|
|
39
|
+
import os
|
|
40
|
+
import re
|
|
41
|
+
import sys
|
|
42
|
+
|
|
43
|
+
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
44
|
+
# 本脚本既可能被当脚本跑(sys.path[0] 是 md_cg/),也可能被当包内模块导入
|
|
45
|
+
for _p in (HERE, os.path.join(HERE, "md_cg")):
|
|
46
|
+
if _p not in sys.path:
|
|
47
|
+
sys.path.insert(0, _p)
|
|
48
|
+
|
|
49
|
+
EXT = os.path.join(HERE, "data", "external")
|
|
50
|
+
ZP = os.path.join(EXT, "zh_probe")
|
|
51
|
+
LM = os.path.join(EXT, "longmemeval")
|
|
52
|
+
|
|
53
|
+
MANUAL_ZH = os.path.join(ZP, "manual_zh.json")
|
|
54
|
+
MANUAL_Q = os.path.join(ZP, "manual_q.json")
|
|
55
|
+
LM_H = os.path.join(LM, "lme_s_haystack.jsonl")
|
|
56
|
+
LM_Q = os.path.join(LM, "lme_s_questions.jsonl")
|
|
57
|
+
|
|
58
|
+
CORPUS20 = os.path.join(ZP, "corpus20.jsonl")
|
|
59
|
+
QUESTIONS20 = os.path.join(ZP, "questions20.jsonl")
|
|
60
|
+
|
|
61
|
+
# 关系轴:用户裁决「条件链 + 人物/事件/时间/身份/地点都可以作为关系」
|
|
62
|
+
AXES = ("person", "event", "time", "identity", "place", "condition")
|
|
63
|
+
|
|
64
|
+
|
|
65
|
+
# 生效条件:path 以 UTF-8 打开后逐行读取,仅 strip 后非空的行经 json.loads 产出,空白行被跳过。
|
|
66
|
+
def iter_jsonl(path):
|
|
67
|
+
with open(path, encoding="utf-8") as f:
|
|
68
|
+
for line in f:
|
|
69
|
+
line = line.strip()
|
|
70
|
+
if line:
|
|
71
|
+
yield json.loads(line)
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
# 生效条件:path 以 UTF-8 打开成功时返回 json.load(f) 的解析结果,片段内无其它分支。
|
|
75
|
+
def load_json(path):
|
|
76
|
+
with open(path, encoding="utf-8") as f:
|
|
77
|
+
return json.load(f)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
# 生效条件:raw 经 str() 后按“;”切分,仅含“=”的段被解析,其中键为身份/时间/摘要时写对应槽位、键为条件时按“|”与“:”写入 condition、键为词时按“,”拆出非空项写入 terms,其它键或未出现的槽位保持 out 的默认空值。
|
|
81
|
+
def parse_zh(raw):
|
|
82
|
+
"""中文层串 → 槽位 dict。
|
|
83
|
+
|
|
84
|
+
格式(manual_zh.json 既有形态,本函数只解析不改写):
|
|
85
|
+
身份=用户;时间=2023/05/24 04:49;摘要=…;
|
|
86
|
+
条件=观测位置:会话陈述|观测工具:会话记录|时间窗口:2023-05-24|存在约束:公开;
|
|
87
|
+
词=礼物,礼物清单,场合
|
|
88
|
+
"""
|
|
89
|
+
out = {"identity": "", "time": "", "summary": "", "condition": {},
|
|
90
|
+
"terms": []}
|
|
91
|
+
for seg in str(raw).split(";"):
|
|
92
|
+
seg = seg.strip()
|
|
93
|
+
if not seg or "=" not in seg:
|
|
94
|
+
continue
|
|
95
|
+
key, val = seg.split("=", 1)
|
|
96
|
+
key, val = key.strip(), val.strip()
|
|
97
|
+
if key == "身份":
|
|
98
|
+
out["identity"] = val
|
|
99
|
+
elif key == "时间":
|
|
100
|
+
out["time"] = val
|
|
101
|
+
elif key == "摘要":
|
|
102
|
+
out["summary"] = val
|
|
103
|
+
elif key == "条件":
|
|
104
|
+
for kv in val.split("|"):
|
|
105
|
+
if ":" in kv:
|
|
106
|
+
k, v = kv.split(":", 1)
|
|
107
|
+
out["condition"][k.strip()] = v.strip()
|
|
108
|
+
elif key == "词":
|
|
109
|
+
out["terms"] = [t.strip() for t in val.split(",") if t.strip()]
|
|
110
|
+
return out
|
|
111
|
+
|
|
112
|
+
|
|
113
|
+
# 生效条件:当 MANUAL_ZH/MANUAL_Q/LM_Q/LM_H 可读取、hay 覆盖 want_ids 且 meta 含 man_q 全部 qid 时写出 CORPUS20/QUESTIONS20 并返回 (corpus, questions),缺 turn 或 qid 分别 raise SystemExit,verbose(默认 True)为真值时额外打印统计、假值时静默。
|
|
114
|
+
def prepare(verbose=True):
|
|
115
|
+
"""生成 corpus20.jsonl 与 questions20.jsonl(幂等覆盖)。"""
|
|
116
|
+
man_zh = load_json(MANUAL_ZH)
|
|
117
|
+
man_q = load_json(MANUAL_Q)
|
|
118
|
+
meta = {r["qid"]: r for r in iter_jsonl(LM_Q)}
|
|
119
|
+
want_ids = set(man_zh)
|
|
120
|
+
hay = {}
|
|
121
|
+
for r in iter_jsonl(LM_H):
|
|
122
|
+
if r["id"] in want_ids:
|
|
123
|
+
hay[r["id"]] = r
|
|
124
|
+
if len(hay) == len(want_ids):
|
|
125
|
+
break
|
|
126
|
+
|
|
127
|
+
missing = want_ids - set(hay)
|
|
128
|
+
if missing:
|
|
129
|
+
raise SystemExit(f"[失败] haystack 缺以下 turn:{sorted(missing)}")
|
|
130
|
+
|
|
131
|
+
# corpus:按 turn id 的数值序(= 时间序),保证「承接前一条」有确定语义
|
|
132
|
+
# 生效条件:tid 经 rsplit("t",1) 得到至少两段且最后一段可被 int() 解析时返回该整数,否则源码未做校验会抛错。
|
|
133
|
+
def turn_no(tid):
|
|
134
|
+
return int(tid.rsplit("t", 1)[1])
|
|
135
|
+
|
|
136
|
+
corpus = []
|
|
137
|
+
for tid in sorted(man_zh, key=turn_no):
|
|
138
|
+
t = hay[tid]
|
|
139
|
+
corpus.append({
|
|
140
|
+
"id": tid,
|
|
141
|
+
"text": str(t.get("text") or ""),
|
|
142
|
+
"speaker": str(t.get("speaker") or ""),
|
|
143
|
+
"date": str(t.get("date") or ""),
|
|
144
|
+
"zh": man_zh[tid],
|
|
145
|
+
"zh_fields": parse_zh(man_zh[tid]),
|
|
146
|
+
})
|
|
147
|
+
|
|
148
|
+
# questions:qid → 中文查询 + 证据(完整 evidence_turns,池里仅 ev0 存在)
|
|
149
|
+
questions = []
|
|
150
|
+
for qid, q in man_q.items():
|
|
151
|
+
m = meta.get(qid)
|
|
152
|
+
if m is None:
|
|
153
|
+
raise SystemExit(f"[失败] questions 缺 qid={qid}")
|
|
154
|
+
questions.append({
|
|
155
|
+
"qid": qid,
|
|
156
|
+
"qtype": m["qtype"],
|
|
157
|
+
"question": q,
|
|
158
|
+
"answer": str(m.get("answer") or ""),
|
|
159
|
+
"evidence_turns": list(m.get("evidence_turns") or []),
|
|
160
|
+
})
|
|
161
|
+
|
|
162
|
+
with open(CORPUS20, "w", encoding="utf-8") as f:
|
|
163
|
+
for r in corpus:
|
|
164
|
+
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
165
|
+
with open(QUESTIONS20, "w", encoding="utf-8") as f:
|
|
166
|
+
for r in questions:
|
|
167
|
+
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
|
168
|
+
|
|
169
|
+
if verbose:
|
|
170
|
+
print(f"语料 {len(corpus)} 条 → {CORPUS20}")
|
|
171
|
+
print(f"题库 {len(questions)} 条 → {QUESTIONS20}")
|
|
172
|
+
n_etype = {}
|
|
173
|
+
for q in questions:
|
|
174
|
+
n_etype[q["qtype"]] = n_etype.get(q["qtype"], 0) + 1
|
|
175
|
+
print("题型分布:" + ", ".join(f"{k}={v}" for k, v in sorted(n_etype.items())))
|
|
176
|
+
return corpus, questions
|
|
177
|
+
|
|
178
|
+
|
|
179
|
+
# 生效条件:prepare(verbose=False) 返回 corpus/questions 后,对每题以 evidence_turns[0](空则 "")取 ev0,命中词限于 ev0 非空且词出现在 zh_of[ev0] 中,df1 收集池内 df==1 的词、all_df 收集 df==len(corpus) 的词,verbose(默认 True)为真值时打印后返回 per_q、为假值时直接返回 per_q。
|
|
180
|
+
def analyze(verbose=True):
|
|
181
|
+
"""诊断:查询词能否落到 gold 中文层 / 池内其他条目(决定各轴是否有可桥接的实体)。
|
|
182
|
+
|
|
183
|
+
只读既有产物,不修改任何语料。输出两类事实:
|
|
184
|
+
* 查询词在 gold 中文层的字面命中率 → 词法路的天花板;
|
|
185
|
+
* 查询词在**池内**的 df → 该词的判别力(df=1 是精确定位,df=20 则无信息量)。
|
|
186
|
+
"""
|
|
187
|
+
corpus, questions = prepare(verbose=False)
|
|
188
|
+
zh_of = {c["id"]: c["zh"] for c in corpus}
|
|
189
|
+
|
|
190
|
+
per_q = []
|
|
191
|
+
for q in questions:
|
|
192
|
+
ev0 = q["evidence_turns"][0] if q["evidence_turns"] else ""
|
|
193
|
+
terms = [t for t in str(q["question"]).split() if t]
|
|
194
|
+
hit = [t for t in terms if ev0 and t in zh_of.get(ev0, "")]
|
|
195
|
+
globalish = [t for t in terms
|
|
196
|
+
if sum(1 for c in corpus if t in zh_of[c["id"]]) == len(corpus)]
|
|
197
|
+
per_q.append({
|
|
198
|
+
"qid": q["qid"], "ev0": ev0, "qtype": q["qtype"],
|
|
199
|
+
"n_term": len(terms), "n_hit": len(hit),
|
|
200
|
+
"hit_terms": hit,
|
|
201
|
+
"df1": sorted({t for t in terms
|
|
202
|
+
if sum(1 for c in corpus if t in zh_of[c["id"]]) == 1}),
|
|
203
|
+
"all_df": globalish,
|
|
204
|
+
})
|
|
205
|
+
|
|
206
|
+
if verbose:
|
|
207
|
+
print(f"\n== 查询词覆盖诊断(池 {len(corpus)} 条,题 {len(questions)} 道)==")
|
|
208
|
+
print(f"{'qid':<16}{'ev0':<12}{'词数':>5}{'命中':>5} 命中词")
|
|
209
|
+
print("-" * 72)
|
|
210
|
+
for r in per_q:
|
|
211
|
+
print(f"{r['qid']:<16}{r['ev0']:<12}{r['n_term']:>5}{r['n_hit']:>5} "
|
|
212
|
+
f"{','.join(r['hit_terms'])[:40]}")
|
|
213
|
+
tot_t = sum(r["n_term"] for r in per_q)
|
|
214
|
+
tot_h = sum(r["n_hit"] for r in per_q)
|
|
215
|
+
zero = [r["qid"] for r in per_q if r["n_hit"] == 0]
|
|
216
|
+
print("-" * 72)
|
|
217
|
+
print(f"合计:{tot_h}/{tot_t} 个查询词落在 gold 中文层"
|
|
218
|
+
f"({tot_h / max(1, tot_t):.1%})")
|
|
219
|
+
print(f"零重叠题({len(zero)}):{', '.join(zero) if zero else '无'}")
|
|
220
|
+
return per_q
|
|
221
|
+
|
|
222
|
+
return per_q
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
# ---------------------------------------------------------------- 写入侧四轴
|
|
226
|
+
# 全部为确定性纯规则:零 LLM、零外部依赖,同输入必同输出;词表随源码公开。
|
|
227
|
+
# **盲于查询集**:以下规则只读 corpus20(语料)与中文层自身字段,从不打开 manual_q.json。
|
|
228
|
+
|
|
229
|
+
STOP_ZH = frozenset(
|
|
230
|
+
"的 了 在 是 我 你 他 她 它 和 与 或 这 那 有 没 不 就 都 也 还 要 会 能 可以 "
|
|
231
|
+
"多少 什么 哪里 哪 怎么 为什么 几 一些 一个 之 其 与 及 等 被 把 给 对 从 到".split()
|
|
232
|
+
)
|
|
233
|
+
|
|
234
|
+
PERSON_HINTS = (
|
|
235
|
+
"妹妹", "姐姐", "哥哥", "弟弟", "妈妈", "爸爸", "母亲", "父亲", "父母", "奶奶",
|
|
236
|
+
"爷爷", "外婆", "外公", "同事", "朋友", "同学", "邻居", "老板", "上司", "教练",
|
|
237
|
+
"医生", "老师", "室友", "伴侣", "丈夫", "妻子", "儿子", "女儿", "叔叔", "阿姨",
|
|
238
|
+
"表亲", "未婚夫", "未婚妻",
|
|
239
|
+
)
|
|
240
|
+
|
|
241
|
+
PLACE_HINTS = (
|
|
242
|
+
"市", "州", "城", "镇", "岛", "公园", "中心", "大学", "海滩", "广场", "机场",
|
|
243
|
+
"河", "湖", "山", "街", "路", "区", "国家", "澳洲", "欧洲", "亚洲", "美洲",
|
|
244
|
+
)
|
|
245
|
+
|
|
246
|
+
EN_STOP = frozenset("""
|
|
247
|
+
The A An And But Or For Nor So Yet This That These Those There Their They Them Then Than
|
|
248
|
+
When What Where Which While Who Whom Whose Why How However Also Always Never Often Sometimes
|
|
249
|
+
I You He She It We They Me Him Her Us My Your His Its Our Very Really Just Only Even Still
|
|
250
|
+
Have Has Had Having Do Does Did Doing Be Been Being Am Is Are Was Were Will Would Shall
|
|
251
|
+
Should Can Could May Might Must Not No Yes If In On At By To Of With From Into Onto Over
|
|
252
|
+
Under Above Below Between During After Before About Around Because Since Until Upon While
|
|
253
|
+
Monday Tuesday Wednesday Thursday Friday Saturday Sunday January February March April May
|
|
254
|
+
June July August September October November December Today Yesterday Tomorrow Last Next
|
|
255
|
+
New Old Good Bad Best Worst First Second Third One Two Three Four Five Six Seven Eight Nine
|
|
256
|
+
Ten Okay Ok Well Thanks Thank Please Hi Hello Hey
|
|
257
|
+
""".split())
|
|
258
|
+
|
|
259
|
+
EN_NAME_RE = re.compile(r"\b[A-Z][a-z]{2,}(?:[ ][A-Z][a-z]{2,})*\b")
|
|
260
|
+
DATE_RE = re.compile(r"(\d{4})[/-](\d{1,2})[/-](\d{1,2})")
|
|
261
|
+
|
|
262
|
+
|
|
263
|
+
# 生效条件:DATE_RE.search(str(raw or "")) 有匹配时返回零填充的 YYYY-MM-DD,无匹配(含 raw 为假值回落成 "")时返回 ""。
|
|
264
|
+
def norm_time(raw):
|
|
265
|
+
"""'2023/05/24 04:49' → '2023-05-24'(跨条目统一格式,使日期查询可命中)。"""
|
|
266
|
+
m = DATE_RE.search(str(raw or ""))
|
|
267
|
+
if not m:
|
|
268
|
+
return ""
|
|
269
|
+
y, mo, d = m.groups()
|
|
270
|
+
return f"{int(y):04d}-{int(mo):02d}-{int(d):02d}"
|
|
271
|
+
|
|
272
|
+
|
|
273
|
+
# 生效条件:对 str(text or "") 用 EN_NAME_RE 扫描,匹配串按空白切分后任一部分命中 EN_STOP 即跳过,其余项按首次出现顺序去重加入 out 并返回;text 为假值时扫描空串返回 []。
|
|
274
|
+
def en_names(text):
|
|
275
|
+
"""英文原文中的专名候选(跨语言桥:中文查询里的英文专名可经 entity 路命中)。"""
|
|
276
|
+
out, seen = [], set()
|
|
277
|
+
for m in EN_NAME_RE.finditer(str(text or "")):
|
|
278
|
+
w = m.group(0)
|
|
279
|
+
if any(p in EN_STOP for p in w.split()):
|
|
280
|
+
continue
|
|
281
|
+
if w not in seen:
|
|
282
|
+
seen.add(w)
|
|
283
|
+
out.append(w)
|
|
284
|
+
return out
|
|
285
|
+
|
|
286
|
+
|
|
287
|
+
# 生效条件:遍历 corpus,对每条记录的 zh_fields["terms"] 去重后逐词判断,词非空且不在 STOP_ZH 时使 df[词] 计数加一,返回 df。
|
|
288
|
+
def build_tables(corpus):
|
|
289
|
+
"""全库统计:# 词项 df(**只读语料**,与查询集无关)。"""
|
|
290
|
+
df = {}
|
|
291
|
+
for c in corpus:
|
|
292
|
+
for t in set(c["zh_fields"]["terms"]):
|
|
293
|
+
if t and t not in STOP_ZH:
|
|
294
|
+
df[t] = df.get(t, 0) + 1
|
|
295
|
+
return df
|
|
296
|
+
|
|
297
|
+
|
|
298
|
+
# 生效条件:c 含 zh_fields(terms/condition/time/identity)且 df 为词到频次的映射时,返回 person/event/time/identity/place/condition 六轴取值字典。
|
|
299
|
+
def axis_values(c, df):
|
|
300
|
+
"""一条 turn 的六个关系轴取值(用户裁决:条件链 + 人物/事件/时间/身份/地点)。"""
|
|
301
|
+
f = c["zh_fields"]
|
|
302
|
+
terms = [t for t in f["terms"] if t and t not in STOP_ZH]
|
|
303
|
+
tw = norm_time(f["condition"].get("时间窗口") or f["time"])
|
|
304
|
+
person = [t for t in terms if any(h in t for h in PERSON_HINTS)]
|
|
305
|
+
place = [t for t in terms if any(h in t for h in PLACE_HINTS)]
|
|
306
|
+
event = [t for t in terms if df.get(t, 0) >= 2]
|
|
307
|
+
return {
|
|
308
|
+
"person": person,
|
|
309
|
+
"event": event,
|
|
310
|
+
"time": [tw] if tw else [],
|
|
311
|
+
"identity": [f["identity"]] if f["identity"] else [],
|
|
312
|
+
"place": place,
|
|
313
|
+
"condition": [f"{k}:{v}" for k, v in f["condition"].items()],
|
|
314
|
+
}
|
|
315
|
+
|
|
316
|
+
|
|
317
|
+
# 意图类目:意图动词 → 规范类目(确定性词表,只读摘要,与查询集无关)
|
|
318
|
+
INTENT_CLASSES = (
|
|
319
|
+
("推荐", ("推荐", "建议", "咨询", "请教")),
|
|
320
|
+
("购置", ("购买", "买", "挑选", "选购", "下单", "购买")),
|
|
321
|
+
("整理", ("整理", "清理", "分类", "收纳", "归档")),
|
|
322
|
+
("学习", ("学习", "了解", "阅读", "查阅", "研究")),
|
|
323
|
+
("规划", ("计划", "安排", "准备", "制定")),
|
|
324
|
+
("比较", ("比较", "对比", "估算", "计算")),
|
|
325
|
+
("维修", ("维修", "修理", "更换", "保养")),
|
|
326
|
+
("出行", ("旅行", "搬家", "搬迁", "游玩", "漂流")),
|
|
327
|
+
("参加", ("参加", "报名", "出席")),
|
|
328
|
+
("健身", ("锻炼", "跑步", "训练")),
|
|
329
|
+
("取退", ("退还", "退回", "归还")),
|
|
330
|
+
("完成", ("完成", "做完", "结课")),
|
|
331
|
+
)
|
|
332
|
+
|
|
333
|
+
|
|
334
|
+
# 生效条件:按 INTENT_CLASSES 顺序检查 c["zh_fields"]["summary"],首个命中类目的首个包含于摘要的动词返回 (cls, v),全不命中返回 ('', '')。
|
|
335
|
+
def intent_of(c):
|
|
336
|
+
"""意图抽象:摘要 → (规范类目, 命中的动词原形)。
|
|
337
|
+
|
|
338
|
+
规格要求「抽象」。原实现返回摘要切片 `f"{v}:" + summary[i-6:i+10]`——
|
|
339
|
+
那只是把原文再抄一遍:不产生任何新词、不构成类目,且只写进正文,
|
|
340
|
+
而正文只喂词法路、RRF 又只按**名次**融合(见 bench 报告),故边际恒为 0。
|
|
341
|
+
此处归一为固定类目,并保留命中的动词原形,二者一并进 tags(entity 路——
|
|
342
|
+
存在性匹配,不吃分数尺度),使该轴真正获得独立检索通道。
|
|
343
|
+
"""
|
|
344
|
+
s = c["zh_fields"]["summary"]
|
|
345
|
+
for cls, verbs in INTENT_CLASSES:
|
|
346
|
+
for v in verbs:
|
|
347
|
+
if v in s:
|
|
348
|
+
return cls, v
|
|
349
|
+
return "", ""
|
|
350
|
+
|
|
351
|
+
|
|
352
|
+
# 生效条件:把 c["zh_fields"]["terms"] 与 en_names(c["text"]) 逐项 strip 后跳过空串、STOP_ZH(原形或小写)及已见项去重入 out,再用 norm_time(条件.get("时间窗口") or c 的 time) 得到非空且未出现的时窗串追加;df 形参在该片段内未参与条件判断。
|
|
353
|
+
def normalize_terms(c, df):
|
|
354
|
+
"""实体规范化:去停用 + 去重 + 附时间规范式 + 附英文专名(跨语言对齐)。"""
|
|
355
|
+
f = c["zh_fields"]
|
|
356
|
+
out, seen = [], set()
|
|
357
|
+
for t in list(f["terms"]) + en_names(c["text"]):
|
|
358
|
+
t = t.strip()
|
|
359
|
+
if not t or t in STOP_ZH or t.lower() in STOP_ZH or t in seen:
|
|
360
|
+
continue
|
|
361
|
+
seen.add(t)
|
|
362
|
+
out.append(t)
|
|
363
|
+
tw = norm_time(f["condition"].get("时间窗口") or f["time"])
|
|
364
|
+
if tw and tw not in seen:
|
|
365
|
+
out.append(tw)
|
|
366
|
+
return out
|
|
367
|
+
|
|
368
|
+
|
|
369
|
+
# 生效条件:遍历 corpus 并以 root=os.path.join(HERE, root_base or f"_md_cg_eval_zhprobe_{arm['name']}")(root_base 为 None/空串时回落)建库,graph 分支仅当 arm.get("graph") 为真且某轴取值的同值 id 数落在 [2, max_df](默认 5)时为这些 id 两两建有向边,coref 仅当 arm.get("coref") 为真、本条 own 为空(own 只在 arm.get("norm") 为真时由 normalize_terms 生成,否则恒为 [])且 prev_own 非空时承接前一条自身 canonical,intent 分支仅当 arm.get("intent") 为真且 intent_of 返回 cls 非空时写意图行并把意图词与长度>=2 的动词加入 tags,verbose(默认 True)为真值时打印臂统计、root 已是目录时先整树删除再建 MdCGOS。
|
|
370
|
+
def build_arm(corpus, arm, max_df=5, root_base=None, verbose=True):
|
|
371
|
+
"""按消融臂建库。每臂一个独立 root,互不污染。
|
|
372
|
+
|
|
373
|
+
臂的差异**只体现在库内容**(正文/tags/edges),检索路固定
|
|
374
|
+
lexical+entity+graph——这样"能力未开"就等于"库内没有对应信息",
|
|
375
|
+
归因干净:指标变化可直接归给该轴,而不是归给路开关。
|
|
376
|
+
"""
|
|
377
|
+
from md_cg.mdcos import MdCGOS
|
|
378
|
+
root = os.path.join(HERE, root_base or f"_md_cg_eval_zhprobe_{arm['name']}")
|
|
379
|
+
|
|
380
|
+
df = build_tables(corpus)
|
|
381
|
+
axes = {c["id"]: axis_values(c, df) for c in corpus}
|
|
382
|
+
|
|
383
|
+
edges = {}
|
|
384
|
+
if arm.get("graph"):
|
|
385
|
+
for ax in AXES:
|
|
386
|
+
val2ids = {}
|
|
387
|
+
for c in corpus:
|
|
388
|
+
for v in axes[c["id"]][ax]:
|
|
389
|
+
val2ids.setdefault(v, set()).add(c["id"])
|
|
390
|
+
for v, ids in val2ids.items():
|
|
391
|
+
if not (2 <= len(ids) <= max_df):
|
|
392
|
+
continue # df=1 无边;df>max_df 全连通,零信息量只注入噪声
|
|
393
|
+
for a in ids:
|
|
394
|
+
for b in ids:
|
|
395
|
+
if a != b:
|
|
396
|
+
edges.setdefault(a, {})[b] = ax
|
|
397
|
+
|
|
398
|
+
if os.path.isdir(root):
|
|
399
|
+
import shutil
|
|
400
|
+
shutil.rmtree(root, ignore_errors=True)
|
|
401
|
+
cg = MdCGOS(root, autoflush=500)
|
|
402
|
+
|
|
403
|
+
prev_own = None # 前一条**自身**的 canonical(不累积)
|
|
404
|
+
n_coref = 0
|
|
405
|
+
for c in corpus:
|
|
406
|
+
f = c["zh_fields"]
|
|
407
|
+
own = normalize_terms(c, df) if arm.get("norm") else []
|
|
408
|
+
|
|
409
|
+
# 轴2 指代消解(规格:保守附加)——**仅当本条自身无实体**时,
|
|
410
|
+
# 才承接**前一条自身**的 canonical。
|
|
411
|
+
# 原实现:`if coref and last_canon` 无条件承接,且 `last_canon = cur`
|
|
412
|
+
# 把「继承结果」也计入 → 单调累积成「历史全集」,tags 雪球污染。
|
|
413
|
+
inherit = []
|
|
414
|
+
if arm.get("coref") and not own and prev_own:
|
|
415
|
+
inherit = list(prev_own)
|
|
416
|
+
n_coref += 1
|
|
417
|
+
|
|
418
|
+
# 轴3 意图抽象:类目写正文(规格的「子功能行」),类目词 + 动词原形写 tags。
|
|
419
|
+
intent_terms = []
|
|
420
|
+
intent_line = ""
|
|
421
|
+
if arm.get("intent"):
|
|
422
|
+
cls, verb = intent_of(c)
|
|
423
|
+
if cls:
|
|
424
|
+
intent_line = f"意图:{cls}"
|
|
425
|
+
intent_terms = [f"意图:{cls}"] + ([verb] if len(verb) >= 2 else [])
|
|
426
|
+
|
|
427
|
+
body = [f"身份:{f['identity']}", f"时间:{f['time']}",
|
|
428
|
+
f"摘要:{f['summary']}", f"词:{','.join(f['terms'])}",
|
|
429
|
+
"条件:" + "|".join(f"{k}:{v}" for k, v in f["condition"].items())]
|
|
430
|
+
if intent_line:
|
|
431
|
+
body.append(intent_line)
|
|
432
|
+
if inherit:
|
|
433
|
+
body.append("承接:" + ",".join(inherit))
|
|
434
|
+
body.append("")
|
|
435
|
+
body.append(c["text"])
|
|
436
|
+
|
|
437
|
+
tags = own[:]
|
|
438
|
+
for t in intent_terms + inherit:
|
|
439
|
+
if t not in tags:
|
|
440
|
+
tags.append(t)
|
|
441
|
+
|
|
442
|
+
e = [{"target": t, "axis": a} for t, a in edges.get(c["id"], {}).items()]
|
|
443
|
+
cg.add(c["id"], "\n".join(body), layer="contextual", tags=tags,
|
|
444
|
+
edges=e, eval_src=f"zh_mad:{arm['name']}", verification_basis="data")
|
|
445
|
+
|
|
446
|
+
prev_own = own[:]
|
|
447
|
+
|
|
448
|
+
cg.flush()
|
|
449
|
+
if verbose:
|
|
450
|
+
n_e = sum(len(v) for v in edges.values())
|
|
451
|
+
print(f" [{arm['name']}] root={os.path.basename(root)} "
|
|
452
|
+
f"tags_axis={'on' if arm.get('norm') else 'off'} "
|
|
453
|
+
f"coref={'on' if arm.get('coref') else 'off'} "
|
|
454
|
+
f"intent={'on' if arm.get('intent') else 'off'} "
|
|
455
|
+
f"graph={'on' if arm.get('graph') else 'off'} 有向边={n_e} "
|
|
456
|
+
f"承接条数={n_coref}")
|
|
457
|
+
return root
|
|
458
|
+
|
|
459
|
+
|
|
460
|
+
ARMS = [
|
|
461
|
+
{"name": "a0_base", "norm": False, "coref": False, "intent": False, "graph": False},
|
|
462
|
+
{"name": "a1_norm", "norm": True, "coref": False, "intent": False, "graph": False},
|
|
463
|
+
{"name": "a2_coref", "norm": True, "coref": True, "intent": False, "graph": False},
|
|
464
|
+
{"name": "a3_intent", "norm": True, "coref": True, "intent": True, "graph": False},
|
|
465
|
+
{"name": "a4_graph", "norm": True, "coref": True, "intent": True, "graph": True},
|
|
466
|
+
]
|
|
467
|
+
|
|
468
|
+
|
|
469
|
+
# 生效条件:先执行 prepare(verbose=False) 取得 corpus,再对 ARMS 每臂以 max_df=max_df(默认 5,原样下传不做假值回落)调用 build_arm 并汇总为 roots 返回。
|
|
470
|
+
def build_all(max_df=5):
|
|
471
|
+
corpus, _ = prepare(verbose=False)
|
|
472
|
+
print(f"== 建库(消融 {len(ARMS)} 臂,max_df={max_df})==")
|
|
473
|
+
roots = {}
|
|
474
|
+
for arm in ARMS:
|
|
475
|
+
roots[arm["name"]] = build_arm(corpus, arm, max_df=max_df)
|
|
476
|
+
return roots
|
|
477
|
+
|
|
478
|
+
|
|
479
|
+
RUST_BIN = os.path.join(HERE, "rust", "target", "release", "mdcg-eval.exe")
|
|
480
|
+
ROW_RE = re.compile(
|
|
481
|
+
r"^\s*(precise|temporal|interference|reference)\s+(\d+)\s+"
|
|
482
|
+
r"([\d.]+)%\s+([\d.]+)%\s+([\d.]+)\s*$", re.M)
|
|
483
|
+
GROUPS = ["precise", "temporal", "interference", "reference"]
|
|
484
|
+
|
|
485
|
+
|
|
486
|
+
# 生效条件:os.path.exists(RUST_BIN) 为真时以 argv=[RUST_BIN,"--dataset","mad","--tag",name,"--lib",lib](extra 为真值时追加 list(extra))执行 subprocess,返回码非 0 或 ROW_RE 在 stdout 未匹配到任何组时 raise SystemExit,否则返回 {组:(n,hit@1,hit@5,MRR)}。
|
|
487
|
+
def run_one(name, lib, extra=None):
|
|
488
|
+
"""调用 **Rust 检索器** 跑一臂,返回 {组: (n, hit@1, hit@5, MRR)}。
|
|
489
|
+
|
|
490
|
+
命令执行走 subprocess argv 列表 + 显式 UTF-8 + PYTHONUTF8=1(不经 Windows shell),
|
|
491
|
+
规避 GBK 解码异常。
|
|
492
|
+
"""
|
|
493
|
+
import subprocess
|
|
494
|
+
|
|
495
|
+
if not os.path.exists(RUST_BIN):
|
|
496
|
+
raise SystemExit(f"[失败] 未找到 Rust 评测器:{RUST_BIN}(先 cargo build --release)")
|
|
497
|
+
argv = [RUST_BIN, "--dataset", "mad", "--tag", name, "--lib", lib]
|
|
498
|
+
if extra:
|
|
499
|
+
argv += list(extra)
|
|
500
|
+
env = dict(os.environ, PYTHONUTF8="1")
|
|
501
|
+
p = subprocess.run(argv, capture_output=True, text=True,
|
|
502
|
+
encoding="utf-8", errors="replace", env=env, cwd=HERE)
|
|
503
|
+
if p.returncode != 0:
|
|
504
|
+
raise SystemExit(f"[失败] {name}:{(p.stderr or '')[-900:]}")
|
|
505
|
+
got = {}
|
|
506
|
+
for m in ROW_RE.finditer(p.stdout):
|
|
507
|
+
g, n, h1, h5, mrr = m.groups()
|
|
508
|
+
got[g] = (int(n), float(h1), float(h5), float(mrr))
|
|
509
|
+
if not got:
|
|
510
|
+
raise SystemExit(f"[失败] {name}:未解析到组指标\n{p.stdout[-1200:]}")
|
|
511
|
+
return got
|
|
512
|
+
|
|
513
|
+
|
|
514
|
+
# 生效条件:先打印 title 与表头,再遍历 rows 的 (label, got, note),对 GROUPS 每组用 got[g] 取 (n,h1,h5,mrr) 并累计总体 hit@1/MRR,仅当 baseline 不为 None 且 label==baseline 时记 ref,仅当 ref 已记录且 label!=baseline 时输出 Δ。
|
|
515
|
+
def print_table(title, rows, baseline=None):
|
|
516
|
+
"""rows: [(label, got, note)];baseline = 参照行 label(算 Δ)。"""
|
|
517
|
+
print(f"\n== {title} ==")
|
|
518
|
+
hdr = (f"{'臂':<14}" + "".join(f"{g[:12]:>15}" for g in GROUPS)
|
|
519
|
+
+ f"{'总体hit@1':>12}{'总体MRR':>10}")
|
|
520
|
+
print(hdr)
|
|
521
|
+
print("-" * len(hdr))
|
|
522
|
+
ref = None
|
|
523
|
+
for label, got, note in rows:
|
|
524
|
+
cells, tot_h1, tot_n, tot_mrr = [], 0.0, 0, 0.0
|
|
525
|
+
for g in GROUPS:
|
|
526
|
+
n, h1, _h5, mrr = got[g]
|
|
527
|
+
cells.append(f"{h1:.0f}%/{mrr:.3f}")
|
|
528
|
+
tot_h1 += h1 / 100.0 * n
|
|
529
|
+
tot_n += n
|
|
530
|
+
tot_mrr += mrr * n
|
|
531
|
+
ov_h1 = tot_h1 / max(1, tot_n)
|
|
532
|
+
ov_mrr = tot_mrr / max(1, tot_n)
|
|
533
|
+
if baseline is not None and label == baseline:
|
|
534
|
+
ref = (ov_h1, ov_mrr)
|
|
535
|
+
delta = ""
|
|
536
|
+
if ref is not None and label != baseline:
|
|
537
|
+
delta = f" (Δ{(ov_h1 - ref[0]) * 100:+.1f}pp)"
|
|
538
|
+
suffix = f" {note}" if note else ""
|
|
539
|
+
print(f"{label:<14}" + "".join(f"{c:>15}" for c in cells)
|
|
540
|
+
+ f"{ov_h1 * 100:>10.1f}%{ov_mrr:>10.3f}{delta}{suffix}")
|
|
541
|
+
|
|
542
|
+
|
|
543
|
+
# 生效条件:对 ARMS 每臂以 lib=f"_md_cg_eval_zhprobe_{name}"、无 extra 调用 run_one 组成 rows,并以 ARMS[0]["name"] 为 baseline 调 print_table 后返回 rows。
|
|
544
|
+
def run_ablation():
|
|
545
|
+
"""消融主表:逐臂调用 Rust 检索器并汇总(种子口径 = 缺省,即生产现状)。"""
|
|
546
|
+
rows = []
|
|
547
|
+
for arm in ARMS:
|
|
548
|
+
name = arm["name"]
|
|
549
|
+
rows.append((name, run_one(name, f"_md_cg_eval_zhprobe_{name}"), ""))
|
|
550
|
+
print_table("消融主表(Rust 检索器,k=5,jaccard,证据命中;graph 种子=索引序)",
|
|
551
|
+
rows, baseline=ARMS[0]["name"])
|
|
552
|
+
print("\n随机基线 hit@1 = 5.0%(1/20);池仅 20 条,±1 题 = ±5pp,"
|
|
553
|
+
"结论只作方向性证据。")
|
|
554
|
+
return rows
|
|
555
|
+
|
|
556
|
+
|
|
557
|
+
# 生效条件:以最后一个臂 ARMS[-1]['name'] 对应的库路径,先无 extra 调 run_one("a4_index", lib)、再以 extra=("--graph-seeds","sorted") 调 run_one("a4_sorted", lib) 组成 rows,并以 baseline="a4/index" 调 print_table 后返回。
|
|
558
|
+
def run_seed_control():
|
|
559
|
+
"""对照:**同一个 a4 库**(写入侧与边结构完全相同),只切换 graph 路种子口径。
|
|
560
|
+
|
|
561
|
+
用于把 a4 相对 a3 的变化拆成两个因子:
|
|
562
|
+
* 边结构质量(本轴真正要测的能力);
|
|
563
|
+
* `_path_graph`「种子须已排序」契约被违反 → seeds[:5] 退化为索引枚举序前 5。
|
|
564
|
+
"""
|
|
565
|
+
lib = f"_md_cg_eval_zhprobe_{ARMS[-1]['name']}"
|
|
566
|
+
rows = [
|
|
567
|
+
("a4/index", run_one("a4_index", lib), "现状:seeds=索引枚举序前 5"),
|
|
568
|
+
("a4/sorted", run_one("a4_sorted", lib, ("--graph-seeds", "sorted")),
|
|
569
|
+
"文档语义:seeds=top-5 词法"),
|
|
570
|
+
]
|
|
571
|
+
print_table("graph 种子口径对照(库内容完全相同,只换种子)", rows,
|
|
572
|
+
baseline="a4/index")
|
|
573
|
+
return rows
|
|
574
|
+
|
|
575
|
+
|
|
576
|
+
# 生效条件:cmd 取 argv[1](len(argv)<=1 时为 "prepare"),cmd=="prepare"/"analyze"/"build"/"run"/"seed-control" 分别调用 prepare/analyze/build_all/run_ablation/run_seed_control,cmd=="all" 依次调用 prepare、analyze、build_all、run_ablation、run_seed_control,其余值 raise SystemExit。
|
|
577
|
+
def main(argv):
|
|
578
|
+
cmd = argv[1] if len(argv) > 1 else "prepare"
|
|
579
|
+
if cmd == "prepare":
|
|
580
|
+
prepare()
|
|
581
|
+
elif cmd == "analyze":
|
|
582
|
+
analyze()
|
|
583
|
+
elif cmd == "build":
|
|
584
|
+
build_all()
|
|
585
|
+
elif cmd == "run":
|
|
586
|
+
run_ablation()
|
|
587
|
+
elif cmd == "seed-control":
|
|
588
|
+
run_seed_control()
|
|
589
|
+
elif cmd == "all":
|
|
590
|
+
prepare()
|
|
591
|
+
analyze()
|
|
592
|
+
build_all()
|
|
593
|
+
run_ablation()
|
|
594
|
+
run_seed_control()
|
|
595
|
+
else:
|
|
596
|
+
raise SystemExit(
|
|
597
|
+
f"未知子命令:{cmd}"
|
|
598
|
+
"(可用:prepare / analyze / build / run / seed-control / all)")
|
|
599
|
+
|
|
600
|
+
|
|
601
|
+
if __name__ == "__main__":
|
|
602
602
|
main(sys.argv)
|