@furongjun1999/dsh-memory 0.5.0 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/docs/README.md +1 -0
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +19 -3
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +530 -443
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/audit.py +12 -1
- package/md_cg/backfill.py +16 -15
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/branches.py +18 -2
- package/md_cg/ccgc.py +3 -2
- package/md_cg/chain.py +19 -4
- package/md_cg/consolidate.py +7 -6
- package/md_cg/crosscheck.py +4 -3
- package/md_cg/crypto.py +439 -437
- package/md_cg/datapath.py +395 -335
- package/md_cg/evidence.py +582 -580
- package/md_cg/export.py +3 -1
- package/md_cg/forgetting.py +2 -2
- package/md_cg/fsutil.py +48 -0
- package/md_cg/hotcache.py +255 -238
- package/md_cg/interop.py +161 -22
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/links.py +655 -622
- package/md_cg/mcp_server.py +280 -29
- package/md_cg/mdcg.py +616 -125
- package/md_cg/mdcos.py +430 -66
- package/md_cg/mreview/govern.py +5 -4
- package/md_cg/postings.py +4 -2
- package/md_cg/readcache.py +76 -18
- package/md_cg/reconcile.py +228 -0
- package/md_cg/review_cli.py +170 -0
- package/md_cg/routing.py +28 -0
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +862 -852
- package/md_cg/security.py +385 -275
- package/md_cg/selfreport.py +3 -2
- package/md_cg/signer.py +565 -562
- package/md_cg/sources.py +3 -2
- package/md_cg/stg.py +6 -0
- package/md_cg/sustain.py +1168 -1138
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +259 -249
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +186 -166
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +228 -147
- package/md_cg/test_index_durability.py +238 -224
- package/md_cg/test_interop.py +4 -2
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p27_docindex.py +774 -765
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p32_backfill.py +304 -298
- package/md_cg/test_p39_verify_flow.py +90 -50
- package/md_cg/test_p47_session_view.py +60 -25
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +55 -7
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_s1.py +6 -2
- package/md_cg/test_retr_s1b.py +67 -0
- package/md_cg/test_retr_s7.py +8 -0
- package/md_cg/test_retr_s9_entity_ctx.py +181 -175
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_semantic_canonical.py +255 -241
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_v14_fixes.py +415 -397
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/theory.py +276 -273
- package/md_cg/tokens.py +734 -677
- package/md_cg/units.py +3 -2
- package/md_cg/vision_evidence.py +4 -3
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/writepipe.py +554 -550
- package/package.json +6 -2
- package/src/bridge.ts +434 -401
- package/src/index.ts +526 -518
- package/src/lib/roleplay_web.ts +1019 -932
- package/src/lib/token_store.ts +202 -192
|
@@ -0,0 +1,532 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""端到端干扰池评测:确定性裁决层 vs LLM-as-judge(2026-09-23 · 外部建议里程碑)
|
|
3
|
+
|
|
4
|
+
评测问题:公开数据集 locomo-zh-500 上,当候选池里「错误答案长得像正确答案」时——
|
|
5
|
+
arm_base 裸五路融合排序(无裁决对照)
|
|
6
|
+
arm_firewall 灵枢确定性裁决层(judge_qualification 四态 + judge_ranking 防火墙)
|
|
7
|
+
arm_llm LLM-as-judge(同一候选面逐卡资格裁决,语义阅读)
|
|
8
|
+
三臂谁能把 gold 守在 top1?退化曲线随干扰浓度如何变化?
|
|
9
|
+
|
|
10
|
+
公平性三原则:
|
|
11
|
+
1) 同管线:gold 与全部合成干扰走同一 CCG 六要素写入函数,生效条件由节点自身
|
|
12
|
+
字段机械合成(主体/时间窗口/主题),不看题面——两个裁决器都无题面捷径可抄。
|
|
13
|
+
2) 同候选:三臂共享同一融合候选(五路 sum RRF);firewall 与 LLM 都只做资格
|
|
14
|
+
裁决,裁决机制不同(词面条件四态 vs 语义阅读)。
|
|
15
|
+
3) 可证伪:合成干扰不设任何「方便防火墙剔除」的标记(不适用条件一律诚实写
|
|
16
|
+
「无」)——防火墙若无效、LLM 若翻车,都如实报,不粉饰。
|
|
17
|
+
|
|
18
|
+
干扰家族(每题合成、按家族顺序截断到 level 条;id 前缀 x_ 保证排在 gold 之后,
|
|
19
|
+
平局兜底偏向 gold——对我方保守):
|
|
20
|
+
D2 实体互换 摘要中双方名字互换 + 主体换人:「谁说的/谁做的」答案错,词面几乎同款
|
|
21
|
+
D3p/D3m 时间错位 同一事实 ±1 年:查询不含时间词时原则上不可区分(诚实边界)
|
|
22
|
+
D4 否定澄清 「澄清:此前所说「…」并不属实」:否认该事实,词面高度重叠
|
|
23
|
+
D1 同话题硬负例不合成——语料其余 566 条 turn 即天然干扰池,全 level 在场
|
|
24
|
+
|
|
25
|
+
浓度梯度:level∈{0,1,2,4} = 每题注入前 level 条合成干扰(500 题全部注入、池共享;
|
|
26
|
+
level=0 即零干扰锚点池)。家族级归因在 level=4 池上做。
|
|
27
|
+
|
|
28
|
+
指标:hit@1/5/10、MRR(各臂排序);top1 置换归因(gold/x_d2/x_d3/x_d4/天然);
|
|
29
|
+
gold 被合成干扰压过率;LLM 臂加报 gold-qualified 率 / 弃权率 / 两轮自一致性。
|
|
30
|
+
|
|
31
|
+
诚实边界:
|
|
32
|
+
· 本评测为 CCG 化写入(已发布 locomo 主链路成绩是五槽裸串口径),零干扰锚点
|
|
33
|
+
与 94.6% 只做量级对照,不做逐位复现声明。
|
|
34
|
+
· 题面为关键词串(与检索同源);LLM 拿到的查询与引擎完全同面。位置偏差以
|
|
35
|
+
逐位 qualified 率诊断呈现,v1 不做候选旋转。
|
|
36
|
+
· LLM temperature=0 仍非严格确定,以两轮独立调用的一致性量化;结果含缓存
|
|
37
|
+
(--data-root/llm_cache,键含轮次),重跑命中缓存不重复计费。
|
|
38
|
+
|
|
39
|
+
跑法:
|
|
40
|
+
python -X utf8 -m md_cg.bench_e2e_judge --quick # 冒烟 20题×level 0,1×无LLM
|
|
41
|
+
python -X utf8 -m md_cg.bench_e2e_judge # 全量 120题×4级×3臂+LLM两轮
|
|
42
|
+
python -X utf8 -m md_cg.bench_e2e_judge --skip-llm # 只跑确定性两臂
|
|
43
|
+
环境:DEEPSEEK_API_KEY(LLM 臂);数据落 --data-root(默认 D:/program/test/e2e_judge)
|
|
44
|
+
"""
|
|
45
|
+
from __future__ import annotations
|
|
46
|
+
|
|
47
|
+
import argparse
|
|
48
|
+
import hashlib
|
|
49
|
+
import json
|
|
50
|
+
import os
|
|
51
|
+
import random
|
|
52
|
+
import re
|
|
53
|
+
import shutil
|
|
54
|
+
import sys
|
|
55
|
+
import time
|
|
56
|
+
import urllib.request
|
|
57
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
58
|
+
|
|
59
|
+
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
60
|
+
if HERE not in sys.path:
|
|
61
|
+
sys.path.insert(0, HERE)
|
|
62
|
+
|
|
63
|
+
from md_cg.mdcos import MdCGOS # noqa: E402
|
|
64
|
+
from md_cg import nodefile # noqa: E402
|
|
65
|
+
|
|
66
|
+
DATA_SRC = os.path.join(HERE, "data", "benchmarks", "locomo-zh-500")
|
|
67
|
+
DEFAULT_ROOT = r"D:\program\test\e2e_judge"
|
|
68
|
+
FUSION_KW = dict(paths=("lexical", "bucket", "entity", "graph", "fuzzy"),
|
|
69
|
+
fusion="sum")
|
|
70
|
+
LEVELS_DEFAULT = (0, 1, 2, 4)
|
|
71
|
+
FAMILIES = ("D2", "D3p", "D3m", "D4") # 每题合成顺序;level 截断
|
|
72
|
+
FAM_PREFIX = {"D2": "x_d2", "D3p": "x_d3p", "D3m": "x_d3m", "D4": "x_d4"}
|
|
73
|
+
|
|
74
|
+
|
|
75
|
+
# ---------------------------------------------------------------- 数据装载
|
|
76
|
+
def load_src():
|
|
77
|
+
corpus = [json.loads(ln) for ln
|
|
78
|
+
in open(os.path.join(DATA_SRC, "corpus567.jsonl"),
|
|
79
|
+
encoding="utf-8") if ln.strip()]
|
|
80
|
+
questions = [json.loads(ln) for ln
|
|
81
|
+
in open(os.path.join(DATA_SRC, "questions500.jsonl"),
|
|
82
|
+
encoding="utf-8") if ln.strip()]
|
|
83
|
+
return corpus, questions
|
|
84
|
+
|
|
85
|
+
|
|
86
|
+
def speakers_by_scene(corpus):
|
|
87
|
+
m = {}
|
|
88
|
+
for c in corpus:
|
|
89
|
+
scene = c["id"].split("_session")[0]
|
|
90
|
+
m.setdefault(scene, set()).add(c.get("speaker") or "")
|
|
91
|
+
return {k: sorted(v) for k, v in m.items() if len(v) >= 2}
|
|
92
|
+
|
|
93
|
+
|
|
94
|
+
# ---------------------------------------------------------------- CCG 写入面
|
|
95
|
+
def ccg_card(summary, identity, time_s, terms, node_id):
|
|
96
|
+
"""gold 与合成干扰共用同一写入函数(公平性原则 1)。
|
|
97
|
+
|
|
98
|
+
生效条件只含节点自身字段(主体/时间窗口/主题),不含任何题面信息;
|
|
99
|
+
不适用条件一律诚实写「无」——合成干扰不携带可被负条件路剔除的标记。
|
|
100
|
+
"""
|
|
101
|
+
terms_s = "、".join(terms)
|
|
102
|
+
lines = [f"{summary}({identity},{time_s})",
|
|
103
|
+
f"# 功能名:会话记忆 {node_id}",
|
|
104
|
+
f"# 生效条件:主体={identity};时间窗口={time_s};主题={terms_s}",
|
|
105
|
+
"# 子功能:episodic 会话片段(对话摘要转写)",
|
|
106
|
+
"# 执行:陈述会话事实",
|
|
107
|
+
"# 验证方式:据会话记录核对",
|
|
108
|
+
"# 不适用条件:无"]
|
|
109
|
+
return "\n".join(lines) + "\n"
|
|
110
|
+
|
|
111
|
+
|
|
112
|
+
def _shift_year(time_s, delta):
|
|
113
|
+
return re.sub(r"(\d{4})", lambda m: str(int(m.group(1)) + delta),
|
|
114
|
+
str(time_s), count=1)
|
|
115
|
+
|
|
116
|
+
|
|
117
|
+
def _swap_names(text, a, b):
|
|
118
|
+
return text.replace(a, "\x00").replace(b, a).replace("\x00", b)
|
|
119
|
+
|
|
120
|
+
|
|
121
|
+
def synth_for_question(q, corpus_map, spk):
|
|
122
|
+
"""一题的合成干扰(全家族),返回 [(fam, node_id, card), ...]。"""
|
|
123
|
+
evs = [corpus_map[t] for t in (q.get("evidence_turns") or [])
|
|
124
|
+
if t in corpus_map]
|
|
125
|
+
if not evs:
|
|
126
|
+
return []
|
|
127
|
+
ev = evs[0]
|
|
128
|
+
zf = ev.get("zh_fields") or {}
|
|
129
|
+
summary = str(zf.get("summary") or ev.get("zh") or "")
|
|
130
|
+
identity = str(zf.get("identity") or ev.get("speaker") or "")
|
|
131
|
+
time_s = str(zf.get("time") or "")
|
|
132
|
+
terms = [str(t) for t in (zf.get("terms") or [])]
|
|
133
|
+
scene = ev["id"].split("_session")[0]
|
|
134
|
+
pair = spk.get(scene) or []
|
|
135
|
+
other = next((s for s in pair if s and s != identity), None)
|
|
136
|
+
qid = q["qid"]
|
|
137
|
+
out = []
|
|
138
|
+
if other: # D2 实体互换
|
|
139
|
+
s2 = _swap_names(summary, identity, other)
|
|
140
|
+
t2 = [_swap_names(t, identity, other) for t in terms]
|
|
141
|
+
out.append(("D2", f"{FAM_PREFIX['D2']}_{qid}",
|
|
142
|
+
ccg_card(s2, other, time_s, t2, f"{FAM_PREFIX['D2']}_{qid}")))
|
|
143
|
+
if time_s: # D3p / D3m 时间错位
|
|
144
|
+
for fam, d in (("D3p", 1), ("D3m", -1)):
|
|
145
|
+
t3 = _shift_year(time_s, d)
|
|
146
|
+
nid = f"{FAM_PREFIX[fam]}_{qid}"
|
|
147
|
+
out.append((fam, nid, ccg_card(summary, identity, t3, terms, nid)))
|
|
148
|
+
s4 = f"澄清:此前所说「{summary}」并不属实" # D4 否定澄清
|
|
149
|
+
nid = f"{FAM_PREFIX['D4']}_{qid}"
|
|
150
|
+
out.append(("D4", nid, ccg_card(s4, identity, time_s, terms, nid)))
|
|
151
|
+
return out
|
|
152
|
+
|
|
153
|
+
|
|
154
|
+
# ---------------------------------------------------------------- 池构建
|
|
155
|
+
def build_pool(level, corpus, questions, corpus_map, spk, root_dir):
|
|
156
|
+
"""level=每题注入的合成干扰条数(按 FAMILIES 顺序截断)。带指纹缓存。"""
|
|
157
|
+
synth_all = [] # 全 500 题都合成(池共享、更真实)
|
|
158
|
+
for q in questions:
|
|
159
|
+
synth_all.append(synth_for_question(q, corpus_map, spk))
|
|
160
|
+
n_synth = sum(len(s[:level]) for s in synth_all)
|
|
161
|
+
fp_src = hashlib.md5(
|
|
162
|
+
json.dumps([len(corpus), len(questions)]).encode()).hexdigest()[:8]
|
|
163
|
+
root = os.path.join(root_dir, "pools", f"level{level}")
|
|
164
|
+
manifest = os.path.join(root, "_manifest.json")
|
|
165
|
+
want = {"fp": fp_src, "level": level, "corpus": len(corpus),
|
|
166
|
+
"synth": n_synth}
|
|
167
|
+
if os.path.isfile(manifest):
|
|
168
|
+
try:
|
|
169
|
+
if json.load(open(manifest, encoding="utf-8")) == want \
|
|
170
|
+
and os.path.isdir(os.path.join(root, "mem")):
|
|
171
|
+
return MdCGOS(os.path.join(root, "mem")), n_synth
|
|
172
|
+
except Exception:
|
|
173
|
+
pass # 陈化即重建
|
|
174
|
+
if os.path.isdir(root):
|
|
175
|
+
shutil.rmtree(root)
|
|
176
|
+
os.makedirs(root, exist_ok=True)
|
|
177
|
+
cg = MdCGOS(os.path.join(root, "mem"))
|
|
178
|
+
cards = {}
|
|
179
|
+
for c in corpus: # gold 天然面
|
|
180
|
+
zf = c.get("zh_fields") or {}
|
|
181
|
+
nid = c["id"]
|
|
182
|
+
cards[nid] = ccg_card(str(zf.get("summary") or c.get("zh") or ""),
|
|
183
|
+
str(zf.get("identity") or c.get("speaker") or ""),
|
|
184
|
+
str(zf.get("time") or ""),
|
|
185
|
+
[str(t) for t in (zf.get("terms") or [])], nid)
|
|
186
|
+
for s in synth_all:
|
|
187
|
+
for fam, nid, card in s[:level]:
|
|
188
|
+
assert nid not in cards, f"干扰 id 冲突 {nid}"
|
|
189
|
+
cards[nid] = card
|
|
190
|
+
bad = [nid for nid, cd in cards.items()
|
|
191
|
+
if not nodefile.ccg_completeness(cd)["complete"]]
|
|
192
|
+
assert not bad, f"CCG 六要素不全(不应发生): {bad[:3]}"
|
|
193
|
+
for nid, card in cards.items():
|
|
194
|
+
cg.add(nid, card, layer="knowledge", verification_basis="data")
|
|
195
|
+
json.dump(want, open(manifest, "w", encoding="utf-8"))
|
|
196
|
+
return cg, n_synth
|
|
197
|
+
|
|
198
|
+
|
|
199
|
+
# ---------------------------------------------------------------- 检索两臂
|
|
200
|
+
def run_det_arms(cg, q, k=10):
|
|
201
|
+
gold = set(q.get("evidence_turns") or [])
|
|
202
|
+
base, _mb = cg.search_rrf(q["question"], k=k, judge=False, **FUSION_KW)
|
|
203
|
+
fw, _mf = cg.search_rrf(q["question"], k=k, judge=True,
|
|
204
|
+
judge_ranking=True, **FUSION_KW)
|
|
205
|
+
return {"gold": gold,
|
|
206
|
+
"base": [r[0]["id"] for r in base],
|
|
207
|
+
"firewall": [r[0]["id"] for r in fw],
|
|
208
|
+
"fw_states": [((r[2] or {}).get("state") if len(r) > 2 else None)
|
|
209
|
+
for r in fw],
|
|
210
|
+
"base_nodes": [r[0] for r in base],
|
|
211
|
+
"fw_nodes": [r[0] for r in fw]}
|
|
212
|
+
|
|
213
|
+
|
|
214
|
+
# ---------------------------------------------------------------- LLM 裁决
|
|
215
|
+
LLM_SYS = (
|
|
216
|
+
"你是记忆系统的资格裁决器。给定一个查询和若干条候选记忆卡,逐条判断该卡"
|
|
217
|
+
"是否有资格作为回答该查询的证据。判定标准:卡的内容真实包含查询所问的事实"
|
|
218
|
+
"(主题一致且事实成立;澄清/否认类内容不构成该事实的证据;主体或时间与查询"
|
|
219
|
+
"冲突时同样不合格)。\n"
|
|
220
|
+
"只输出一行 JSON,不要解释、不要代码围栏,格式:\n"
|
|
221
|
+
'{"verdicts":[{"no":1,"qualified":true,"reason":"不超过20字"},...]}'
|
|
222
|
+
)
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def llm_chat(model, base, key, messages, timeout=120, max_tokens=1500,
|
|
226
|
+
temperature=0.0, extra_payload=None):
|
|
227
|
+
body = {"model": model, "messages": messages,
|
|
228
|
+
"temperature": temperature, "max_tokens": max_tokens}
|
|
229
|
+
if extra_payload:
|
|
230
|
+
body.update(extra_payload)
|
|
231
|
+
payload = json.dumps(body).encode("utf-8")
|
|
232
|
+
req = urllib.request.Request(
|
|
233
|
+
base.rstrip("/") + "/chat/completions", data=payload,
|
|
234
|
+
headers={"Content-Type": "application/json",
|
|
235
|
+
"Authorization": f"Bearer {key}"})
|
|
236
|
+
t0 = time.time()
|
|
237
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
238
|
+
data = json.loads(resp.read().decode("utf-8"))
|
|
239
|
+
content = data["choices"][0]["message"].get("content") or ""
|
|
240
|
+
return content, (data.get("usage") or {}), time.time() - t0
|
|
241
|
+
|
|
242
|
+
|
|
243
|
+
def llm_user_prompt(q, cand_nodes):
|
|
244
|
+
lines = [f"查询:{q['question']}", "", "候选记忆卡:"]
|
|
245
|
+
for i, nd in enumerate(cand_nodes, 1):
|
|
246
|
+
content = str(nd.get("content") or "")[:400]
|
|
247
|
+
lines.append(f"【{i}】{content}")
|
|
248
|
+
lines.append("")
|
|
249
|
+
lines.append("请逐卡输出资格判定 JSON。")
|
|
250
|
+
return "\n".join(lines)
|
|
251
|
+
|
|
252
|
+
|
|
253
|
+
def parse_verdicts(raw, n):
|
|
254
|
+
"""解析 LLM 裁决 JSON → {no: qualified};解析失败返回 None(≠空判定)。"""
|
|
255
|
+
s = raw or ""
|
|
256
|
+
i, j = s.find("{"), s.rfind("}")
|
|
257
|
+
verdicts = {}
|
|
258
|
+
if i >= 0 and j > i:
|
|
259
|
+
try:
|
|
260
|
+
obj = json.loads(s[i:j + 1])
|
|
261
|
+
for v in obj.get("verdicts") or []:
|
|
262
|
+
no = v.get("no")
|
|
263
|
+
if isinstance(no, int) and 1 <= no <= n:
|
|
264
|
+
verdicts[no] = bool(v.get("qualified"))
|
|
265
|
+
except ValueError:
|
|
266
|
+
pass
|
|
267
|
+
if not verdicts: # 解析失败=该题无裁决(计入 err)
|
|
268
|
+
return None
|
|
269
|
+
return verdicts
|
|
270
|
+
|
|
271
|
+
|
|
272
|
+
def llm_judge_one(cfg, q, cand_nodes, rnd, cache_dir):
|
|
273
|
+
"""一轮 LLM 资格裁决(含缓存)。返回 (qualified_ids, tokens, dt, cached)。"""
|
|
274
|
+
prompt = llm_user_prompt(q, cand_nodes)
|
|
275
|
+
key_hex = hashlib.md5(json.dumps(
|
|
276
|
+
[cfg["model"], cfg["base"], prompt, rnd], ensure_ascii=False)
|
|
277
|
+
.encode("utf-8")).hexdigest()
|
|
278
|
+
cpath = os.path.join(cache_dir, key_hex + ".json")
|
|
279
|
+
if os.path.isfile(cpath):
|
|
280
|
+
d = json.load(open(cpath, encoding="utf-8"))
|
|
281
|
+
return (d["qualified"], d["tokens"], 0.0, True)
|
|
282
|
+
raw, usage, dt = llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
283
|
+
[{"role": "system", "content": LLM_SYS},
|
|
284
|
+
{"role": "user", "content": prompt}],
|
|
285
|
+
timeout=cfg["timeout"])
|
|
286
|
+
v = parse_verdicts(raw, len(cand_nodes))
|
|
287
|
+
if v is None:
|
|
288
|
+
raise ValueError(f"LLM 输出不可解析:{raw[:120]!r}")
|
|
289
|
+
ids = [cand_nodes[no - 1]["id"] for no in sorted(v) if v[no]]
|
|
290
|
+
d = {"qualified": ids, "raw": raw[:800],
|
|
291
|
+
"tokens": int(usage.get("total_tokens") or 0), "round": rnd}
|
|
292
|
+
json.dump(d, open(cpath, "w", encoding="utf-8"), ensure_ascii=False)
|
|
293
|
+
return ids, d["tokens"], dt, False
|
|
294
|
+
|
|
295
|
+
|
|
296
|
+
def llm_reorder(base_ids, qualified):
|
|
297
|
+
"""qualified 优先(保持基序),其余殿后——与防火墙 keep-降权同构。"""
|
|
298
|
+
qs = [i for i in base_ids if i in set(qualified)]
|
|
299
|
+
rest = [i for i in base_ids if i not in set(qualified)]
|
|
300
|
+
return qs + rest
|
|
301
|
+
|
|
302
|
+
|
|
303
|
+
# ---------------------------------------------------------------- 指标
|
|
304
|
+
def rank_of(gold, ids):
|
|
305
|
+
return next((i for i, nid in enumerate(ids, 1) if nid in gold), 0)
|
|
306
|
+
|
|
307
|
+
|
|
308
|
+
def classify(nid):
|
|
309
|
+
for fam, p in FAM_PREFIX.items():
|
|
310
|
+
if nid.startswith(p + "_"):
|
|
311
|
+
return fam
|
|
312
|
+
return "natural" if not nid.startswith("x_") else "x_other"
|
|
313
|
+
|
|
314
|
+
|
|
315
|
+
def arm_metrics(rows, key):
|
|
316
|
+
n = len(rows)
|
|
317
|
+
st = {"h1": 0, "h5": 0, "h10": 0, "mrr": 0.0,
|
|
318
|
+
"top1": {"gold": 0, "D2": 0, "D3p": 0, "D3m": 0, "D4": 0,
|
|
319
|
+
"natural": 0, "x_other": 0},
|
|
320
|
+
"gold_outranked_by_synth": 0}
|
|
321
|
+
for r in rows:
|
|
322
|
+
ids = r[key]
|
|
323
|
+
gold = r["gold"]
|
|
324
|
+
rk = rank_of(gold, ids)
|
|
325
|
+
if rk == 1:
|
|
326
|
+
st["h1"] += 1
|
|
327
|
+
if 0 < rk <= 5:
|
|
328
|
+
st["h5"] += 1
|
|
329
|
+
if 0 < rk <= 10:
|
|
330
|
+
st["h10"] += 1
|
|
331
|
+
if rk:
|
|
332
|
+
st["mrr"] += 1.0 / rk
|
|
333
|
+
st["top1"][classify(ids[0])] += 1 if ids else 0
|
|
334
|
+
gr = rk
|
|
335
|
+
synth_before_gold = any(
|
|
336
|
+
nid.startswith("x_")
|
|
337
|
+
for nid in ids[:(gr - 1 if gr else len(ids))])
|
|
338
|
+
if synth_before_gold:
|
|
339
|
+
st["gold_outranked_by_synth"] += 1
|
|
340
|
+
return st
|
|
341
|
+
|
|
342
|
+
|
|
343
|
+
def fmt_row(name, st, n):
|
|
344
|
+
t1 = st["top1"]
|
|
345
|
+
return (f" {name:<10} hit@1={100*st['h1']/n:5.1f}% hit@5={100*st['h5']/n:5.1f}%"
|
|
346
|
+
f" hit@10={100*st['h10']/n:5.1f}% MRR={st['mrr']/n:.4f}"
|
|
347
|
+
f" top1置换[D2={t1['D2']} D3={t1['D3p']+t1['D3m']} D4={t1['D4']}"
|
|
348
|
+
f" 天然={t1['natural']}]")
|
|
349
|
+
|
|
350
|
+
|
|
351
|
+
# ---------------------------------------------------------------- 主流程
|
|
352
|
+
def main(argv=None):
|
|
353
|
+
ap = argparse.ArgumentParser(
|
|
354
|
+
description="端到端干扰池评测:确定性裁决层 vs LLM-as-judge")
|
|
355
|
+
ap.add_argument("--data-root", default=DEFAULT_ROOT)
|
|
356
|
+
ap.add_argument("--n", type=int, default=120, help="评测题数(种子采样)")
|
|
357
|
+
ap.add_argument("--seed", type=int, default=7)
|
|
358
|
+
ap.add_argument("--levels", default="0,1,2,4")
|
|
359
|
+
ap.add_argument("--arms", default="base,firewall,llm")
|
|
360
|
+
ap.add_argument("--rounds", type=int, default=2, help="LLM 独立轮数")
|
|
361
|
+
ap.add_argument("--model", default="deepseek-chat")
|
|
362
|
+
ap.add_argument("--base-url", default="https://api.deepseek.com/v1")
|
|
363
|
+
ap.add_argument("--key", default="")
|
|
364
|
+
ap.add_argument("--workers", type=int, default=6)
|
|
365
|
+
ap.add_argument("--timeout", type=int, default=120)
|
|
366
|
+
ap.add_argument("--k", type=int, default=10)
|
|
367
|
+
ap.add_argument("--quick", action="store_true", help="冒烟:20题×0,1级×无LLM")
|
|
368
|
+
ap.add_argument("--skip-llm", action="store_true")
|
|
369
|
+
a = ap.parse_args(argv)
|
|
370
|
+
|
|
371
|
+
if a.quick:
|
|
372
|
+
a.n, a.levels, a.skip_llm = 20, "0,1", True
|
|
373
|
+
levels = [int(x) for x in a.levels.split(",") if x.strip() != ""]
|
|
374
|
+
arms = [x.strip() for x in a.arms.split(",") if x.strip()]
|
|
375
|
+
use_llm = ("llm" in arms) and not a.skip_llm
|
|
376
|
+
key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
|
|
377
|
+
if use_llm and not key:
|
|
378
|
+
print("[error] LLM 臂需要 DEEPSEEK_API_KEY 或 --key(或 --skip-llm)")
|
|
379
|
+
return 2
|
|
380
|
+
|
|
381
|
+
os.makedirs(os.path.join(a.data_root, "results"), exist_ok=True)
|
|
382
|
+
cache_dir = os.path.join(a.data_root, "llm_cache")
|
|
383
|
+
os.makedirs(cache_dir, exist_ok=True)
|
|
384
|
+
|
|
385
|
+
corpus, questions = load_src()
|
|
386
|
+
corpus_map = {c["id"]: c for c in corpus}
|
|
387
|
+
spk = speakers_by_scene(corpus)
|
|
388
|
+
rnd = random.Random(a.seed)
|
|
389
|
+
sample = sorted(rnd.sample(questions, min(a.n, len(questions))),
|
|
390
|
+
key=lambda q: q["qid"])
|
|
391
|
+
print(f"[e2e_judge] 语料 {len(corpus)} · 题目 {len(questions)} · "
|
|
392
|
+
f"采样 {len(sample)}(seed={a.seed}) · levels={levels} · arms={arms}"
|
|
393
|
+
+ (f" · LLM={a.model}×{a.rounds}轮" if use_llm else " · 无LLM"))
|
|
394
|
+
print(f"[e2e_judge] 数据根:{a.data_root}")
|
|
395
|
+
|
|
396
|
+
all_out = {"meta": {"seed": a.seed, "n": len(sample), "levels": levels,
|
|
397
|
+
"model": a.model if use_llm else None,
|
|
398
|
+
"k": a.k}, "levels": {}}
|
|
399
|
+
for lv in levels:
|
|
400
|
+
t0 = time.time()
|
|
401
|
+
cg, n_synth = build_pool(lv, corpus, questions, corpus_map, spk,
|
|
402
|
+
a.data_root)
|
|
403
|
+
print(f"\n===== level {lv}(池={len(corpus)+n_synth} 节点,"
|
|
404
|
+
f"合成干扰 {n_synth} 条)· 建池 {time.time()-t0:.1f}s =====")
|
|
405
|
+
rows = []
|
|
406
|
+
for q in sample:
|
|
407
|
+
det = run_det_arms(cg, q, k=a.k)
|
|
408
|
+
rows.append({"qid": q["qid"], "qtype": q.get("qtype"),
|
|
409
|
+
"question": q["question"], **det})
|
|
410
|
+
n = len(rows)
|
|
411
|
+
lv_out = {"pool": len(corpus) + n_synth, "synth": n_synth,
|
|
412
|
+
"rows_count": n}
|
|
413
|
+
for arm in ("base", "firewall"):
|
|
414
|
+
if arm in arms:
|
|
415
|
+
st = arm_metrics(rows, arm)
|
|
416
|
+
print(fmt_row(arm, st, n))
|
|
417
|
+
lv_out[arm] = st
|
|
418
|
+
# 四态分布(firewall 诊断面)
|
|
419
|
+
if "firewall" in arms:
|
|
420
|
+
sd = {"ACCEPT": 0, "DEFER": 0, "REJECT": 0, "BLINDSPOT": 0}
|
|
421
|
+
for r in rows:
|
|
422
|
+
for s in r["fw_states"]:
|
|
423
|
+
sd[s] = sd.get(s, 0) + 1
|
|
424
|
+
tot = sum(sd.values()) or 1
|
|
425
|
+
print(" firewall四态(top10): " + " ".join(
|
|
426
|
+
f"{k2}={v}({100*v/tot:.0f}%)" for k2, v in sd.items() if v))
|
|
427
|
+
lv_out["fw_states"] = sd
|
|
428
|
+
# LLM 臂
|
|
429
|
+
if use_llm:
|
|
430
|
+
cfg = {"model": a.model, "base": a.base_url, "key": key,
|
|
431
|
+
"timeout": a.timeout}
|
|
432
|
+
llm_rows = []
|
|
433
|
+
t1 = time.time()
|
|
434
|
+
|
|
435
|
+
def work(args):
|
|
436
|
+
r, rd = args
|
|
437
|
+
qobj = {"qid": r["qid"], "question": r["question"]}
|
|
438
|
+
try:
|
|
439
|
+
qids, toks, dt, cached = llm_judge_one(
|
|
440
|
+
cfg, qobj, r["base_nodes"], rd, cache_dir)
|
|
441
|
+
return {"qid": r["qid"], "round": rd, "qualified": qids,
|
|
442
|
+
"tokens": toks, "latency": dt, "cached": cached,
|
|
443
|
+
"order": llm_reorder(r["base"], qids),
|
|
444
|
+
"gold": r["gold"], "base": r["base"]}
|
|
445
|
+
except Exception as exc: # noqa: BLE001
|
|
446
|
+
return {"qid": r["qid"], "round": rd, "error":
|
|
447
|
+
f"{type(exc).__name__}: {exc}", "gold": r["gold"],
|
|
448
|
+
"base": r["base"], "order": r["base"],
|
|
449
|
+
"qualified": None, "tokens": 0, "latency": 0.0,
|
|
450
|
+
"cached": False}
|
|
451
|
+
tasks = [(r, rd) for rd in range(1, a.rounds + 1) for r in rows]
|
|
452
|
+
with ThreadPoolExecutor(max_workers=a.workers) as ex:
|
|
453
|
+
llm_rows = list(ex.map(work, tasks))
|
|
454
|
+
for rd in range(1, a.rounds + 1):
|
|
455
|
+
rr = [x for x in llm_rows if x["round"] == rd]
|
|
456
|
+
for x in rr:
|
|
457
|
+
x_ref = next(y for y in rows if y["qid"] == x["qid"])
|
|
458
|
+
x["order"] = (llm_reorder(x_ref["base"], x["qualified"])
|
|
459
|
+
if x["qualified"] is not None
|
|
460
|
+
else x_ref["base"])
|
|
461
|
+
rd_rows = [{"gold": x["gold"], "llm": x["order"]} for x in rr]
|
|
462
|
+
st = arm_metrics(
|
|
463
|
+
[{"gold": d["gold"], "llm": d["llm"]} for d in rd_rows],
|
|
464
|
+
"llm")
|
|
465
|
+
errs = sum(1 for x in rr if x.get("error"))
|
|
466
|
+
cached = sum(1 for x in rr if x.get("cached"))
|
|
467
|
+
toks = sum(x["tokens"] for x in rr)
|
|
468
|
+
print(fmt_row(f"llm(r{rd})", st, n)
|
|
469
|
+
+ f" err={errs} cache={cached} tok={toks}")
|
|
470
|
+
lv_out[f"llm_r{rd}"] = st
|
|
471
|
+
lv_out[f"llm_r{rd}_diag"] = {"errors": errs,
|
|
472
|
+
"cached": cached, "tokens": toks}
|
|
473
|
+
# gold-qualified 率 + 弃权率(round1)
|
|
474
|
+
r1 = {x["qid"]: x for x in llm_rows if x["round"] == 1}
|
|
475
|
+
gq = ab = 0
|
|
476
|
+
for x in r1.values():
|
|
477
|
+
if x["qualified"] is None:
|
|
478
|
+
continue
|
|
479
|
+
if not x["qualified"]:
|
|
480
|
+
ab += 1
|
|
481
|
+
elif set(x["qualified"]) & x["gold"]:
|
|
482
|
+
gq += 1
|
|
483
|
+
print(f" llm诊断: gold获qualified={100*gq/n:.1f}% "
|
|
484
|
+
f"全弃权={100*ab/n:.1f}%")
|
|
485
|
+
lv_out["llm_diag"] = {"gold_qualified": gq, "abstain": ab}
|
|
486
|
+
# 两轮自一致性
|
|
487
|
+
if a.rounds >= 2:
|
|
488
|
+
r2 = {x["qid"]: x for x in llm_rows if x["round"] == 2}
|
|
489
|
+
same_top = agree_v = pair = 0
|
|
490
|
+
for qid, x1 in r1.items():
|
|
491
|
+
x2 = r2.get(qid)
|
|
492
|
+
if not x2 or x1["qualified"] is None \
|
|
493
|
+
or x2["qualified"] is None:
|
|
494
|
+
continue
|
|
495
|
+
pair += 1
|
|
496
|
+
if x1["order"][:1] == x2["order"][:1]:
|
|
497
|
+
same_top += 1
|
|
498
|
+
s1, s2 = set(x1["qualified"]), set(x2["qualified"])
|
|
499
|
+
if s1 == s2:
|
|
500
|
+
agree_v += 1
|
|
501
|
+
if pair:
|
|
502
|
+
print(f" llm自一致性: top1一致={100*same_top/pair:.1f}% "
|
|
503
|
+
f"qualified集一致={100*agree_v/pair:.1f}% "
|
|
504
|
+
f"(n={pair})")
|
|
505
|
+
lv_out["llm_selfconsistency"] = {
|
|
506
|
+
"top1": same_top, "set": agree_v, "pairs": pair}
|
|
507
|
+
print(f" LLM 臂耗时 {time.time()-t1:.1f}s")
|
|
508
|
+
# 逐题留档(gold 序列化-safe)
|
|
509
|
+
for x in llm_rows:
|
|
510
|
+
if isinstance(x.get("gold"), set):
|
|
511
|
+
x["gold"] = sorted(x["gold"])
|
|
512
|
+
lv_out["llm_rows"] = llm_rows
|
|
513
|
+
all_out["levels"][lv] = lv_out
|
|
514
|
+
# 行级明细落盘(报告取证用;文件名含采样指纹,防不同规模运行互相覆盖)
|
|
515
|
+
detail = [{"qid": r["qid"], "qtype": r["qtype"], "gold":
|
|
516
|
+
sorted(r["gold"]), "base": r["base"],
|
|
517
|
+
"firewall": r["firewall"]} for r in rows]
|
|
518
|
+
json.dump(detail, open(os.path.join(
|
|
519
|
+
a.data_root, "results",
|
|
520
|
+
f"detail_L{lv}_n{len(sample)}_s{a.seed}.json"), "w",
|
|
521
|
+
encoding="utf-8"), ensure_ascii=False, indent=1)
|
|
522
|
+
|
|
523
|
+
out_path = os.path.join(a.data_root, "results",
|
|
524
|
+
f"summary_{time.strftime('%Y%m%d_%H%M%S')}.json")
|
|
525
|
+
json.dump(all_out, open(out_path, "w", encoding="utf-8"),
|
|
526
|
+
ensure_ascii=False, indent=1)
|
|
527
|
+
print(f"\n[e2e_judge] 完成。汇总 → {out_path}")
|
|
528
|
+
return 0
|
|
529
|
+
|
|
530
|
+
|
|
531
|
+
if __name__ == "__main__":
|
|
532
|
+
sys.exit(main())
|