@furongjun1999/dsh-memory 0.5.0 → 0.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +552 -465
- package/docs/README.md +1 -0
- package/docs/eval/bench_lingshu_self/bench_self.py +140 -0
- package/docs/eval/bench_lingshu_self/self_bench_result.json +404 -0
- package/docs/eval/bench_lingshu_self//347/201/265/346/236/242/350/207/252/345/272/223/347/253/257/345/210/260/347/253/257/346/243/200/347/264/242/345/256/236/346/265/213_v1.0.md +38 -0
- package/docs/eval//344/270/215/345/217/257/351/235/240/346/200/247/350/220/275/345/234/260_P0_v1.0.md +185 -0
- package/docs/eval//346/225/205/351/232/234/346/263/250/345/205/245/345/256/236/346/265/213_v1.0.md +422 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257LoCoMoQA/345/220/214/345/217/243/345/276/204/345/257/271/347/205/247_v1.0.md +100 -0
- package/docs/eval//347/253/257/345/210/260/347/253/257/345/271/262/346/211/260/346/261/240/350/257/204/346/265/213_/347/241/256/345/256/232/346/200/247/350/243/201/345/206/263vsLLM_judge_v1.1.md +197 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v1.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v10.md +210 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v11.md +227 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v12.md +203 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v13.md +233 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v14.md +191 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v15.md +213 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v16.md +214 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v17.md +199 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v2.md +156 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v3.md +152 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v4.md +128 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v5.md +114 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v6.md +192 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v7.md +187 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v8.md +207 -0
- package/docs/eval//347/274/272/351/231/267/346/214/226/346/216/230_/350/207/252/344/270/273/350/277/255/344/273/243_v9.md +203 -0
- package/docs/hive//345/256/211/345/205/250/345/256/241/350/256/241/345/256/236/351/224/232_v0.1.md +202 -0
- package/docs/hive//350/234/202/345/267/242/345/217/214/345/256/236/344/276/213/344/272/222/351/252/214_/350/256/276/350/256/241/345/256/232/347/250/277.md +19 -3
- package/docs/images/lingshu-moonlight-covenant-poster-preview.jpg +0 -0
- package/docs/images/lingshu-moonlight-covenant-poster.png +0 -0
- package/docs/mdcg//345/212/237/350/203/275/350/260/203/347/224/250/346/230/240/345/260/204/350/241/250_v0.1.md +40 -40
- package/docs/theory//344/270/215/345/217/257/351/235/240/345/256/232/347/220/206/344/270/216/345/244/261/346/225/210/344/274/230/345/205/210/346/241/206/346/236/266_v0.3.md +365 -0
- package/docs//344/270/215/345/217/257/351/235/240/346/200/247/347/220/206/350/256/272_v0.1.md +343 -0
- package/lib/bridge.d.ts +9 -0
- package/lib/bridge.js +35 -0
- package/lib/index.js +7 -1
- package/lib/lib/roleplay_web.js +530 -443
- package/lib/lib/token_store.d.ts +7 -1
- package/lib/lib/token_store.js +12 -3
- package/md_cg/audit.py +12 -1
- package/md_cg/backfill.py +16 -15
- package/md_cg/backfill_bucket_zh.py +35 -0
- package/md_cg/bench_e2e_judge.py +532 -0
- package/md_cg/bench_e2e_locomo_qa.py +368 -0
- package/md_cg/bench_e2e_qa.py +256 -0
- package/md_cg/branches.py +18 -2
- package/md_cg/ccgc.py +3 -2
- package/md_cg/chain.py +19 -4
- package/md_cg/consolidate.py +7 -6
- package/md_cg/crosscheck.py +4 -3
- package/md_cg/crypto.py +439 -437
- package/md_cg/datapath.py +395 -335
- package/md_cg/evidence.py +582 -580
- package/md_cg/export.py +3 -1
- package/md_cg/forgetting.py +2 -2
- package/md_cg/fsutil.py +48 -0
- package/md_cg/hotcache.py +255 -238
- package/md_cg/interop.py +161 -22
- package/md_cg/judgment_manifest.py +177 -0
- package/md_cg/links.py +655 -622
- package/md_cg/mcp_server.py +280 -29
- package/md_cg/mdcg.py +616 -125
- package/md_cg/mdcos.py +430 -66
- package/md_cg/mreview/govern.py +5 -4
- package/md_cg/postings.py +4 -2
- package/md_cg/readcache.py +76 -18
- package/md_cg/reconcile.py +228 -0
- package/md_cg/review_cli.py +170 -0
- package/md_cg/routing.py +28 -0
- package/md_cg/run_tests.py +211 -0
- package/md_cg/scrub.py +862 -852
- package/md_cg/security.py +385 -275
- package/md_cg/selfreport.py +3 -2
- package/md_cg/signer.py +565 -562
- package/md_cg/sources.py +3 -2
- package/md_cg/stg.py +6 -0
- package/md_cg/sustain.py +1168 -1138
- package/md_cg/test_access_hints.py +147 -0
- package/md_cg/test_branch_discard_tombstone.py +136 -0
- package/md_cg/test_branches.py +259 -249
- package/md_cg/test_chain_read_isolate.py +168 -0
- package/md_cg/test_datapath_device_name.py +203 -0
- package/md_cg/test_emit_negtail_cache.py +156 -0
- package/md_cg/test_en_pipeline.py +186 -166
- package/md_cg/test_govern_directread.py +421 -0
- package/md_cg/test_i32_hotcache_env_key.py +218 -0
- package/md_cg/test_identity_attribution.py +228 -147
- package/md_cg/test_index_durability.py +238 -224
- package/md_cg/test_interop.py +4 -2
- package/md_cg/test_interop_judgment.py +228 -0
- package/md_cg/test_issue39_utf8_stdio.py +273 -0
- package/md_cg/test_links_concurrent_write.py +188 -0
- package/md_cg/test_merge_upsert.py +168 -0
- package/md_cg/test_n123_derive_expiry_chain.py +205 -0
- package/md_cg/test_n130_verify_falsified_protect.py +185 -0
- package/md_cg/test_n131_merge_gate.py +205 -0
- package/md_cg/test_p1x_ref_root.py +160 -0
- package/md_cg/test_p27_docindex.py +774 -765
- package/md_cg/test_p2_mcp.py +3 -0
- package/md_cg/test_p32_backfill.py +304 -298
- package/md_cg/test_p39_verify_flow.py +90 -50
- package/md_cg/test_p47_session_view.py +60 -25
- package/md_cg/test_propose_tail_index.py +157 -0
- package/md_cg/test_read_scope_b27.py +277 -0
- package/md_cg/test_readcache_default_on.py +168 -0
- package/md_cg/test_readcache_precise_inval.py +270 -0
- package/md_cg/test_readcache_prodpath.py +55 -7
- package/md_cg/test_reconcile_v0.py +294 -0
- package/md_cg/test_retr_s1.py +6 -2
- package/md_cg/test_retr_s1b.py +67 -0
- package/md_cg/test_retr_s7.py +8 -0
- package/md_cg/test_retr_s9_entity_ctx.py +181 -175
- package/md_cg/test_retr_score_once.py +208 -0
- package/md_cg/test_review_onepass.py +170 -0
- package/md_cg/test_rrf_graph_seed_cache.py +154 -0
- package/md_cg/test_security_audit.py +155 -0
- package/md_cg/test_security_audit_b26.py +161 -0
- package/md_cg/test_security_audit_v21.py +250 -0
- package/md_cg/test_semantic_canonical.py +255 -241
- package/md_cg/test_session_isolation.py +168 -0
- package/md_cg/test_snapshot_autoclose.py +187 -0
- package/md_cg/test_tail_watermark_race.py +208 -0
- package/md_cg/test_tenant_env_override_warn.py +139 -0
- package/md_cg/test_tenant_registry_corrupt_warn.py +151 -0
- package/md_cg/test_v14_fixes.py +415 -397
- package/md_cg/test_verify_dirty_reconcile.py +157 -0
- package/md_cg/theory.py +276 -273
- package/md_cg/tokens.py +734 -677
- package/md_cg/units.py +3 -2
- package/md_cg/vision_evidence.py +4 -3
- package/md_cg/whitebox_kb/wisdom/code_solidified.json +6195 -6195
- package/md_cg/whitebox_kb/wisdom/multilang_ir.py +128 -124
- package/md_cg/writepipe.py +554 -550
- package/package.json +6 -2
- package/src/bridge.ts +434 -401
- package/src/index.ts +526 -518
- package/src/lib/roleplay_web.ts +1019 -932
- package/src/lib/token_store.ts +202 -192
|
@@ -0,0 +1,368 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""LoCoMo QA 同口径对照评测:完整对话做记忆 · 自然问句做查询(2026-09-23)
|
|
3
|
+
|
|
4
|
+
目的(使用者口径):与 Mem0/Letta/Zep 等同设定直接比数字——**对话做记忆、自然问句做
|
|
5
|
+
查询、LLM reader + LLM judge**,不再以「摘要卡/关键词串」口径差异回避对照。
|
|
6
|
+
|
|
7
|
+
与同行的对齐面:
|
|
8
|
+
· 记忆面:LoCoMo 全量 5882 turn 完整对话(mteb/LoCoMo BEIR corpus,id 与仓内
|
|
9
|
+
evidence_turns 逐字对齐已验证)→ LLM 译为中文(人名保留)→ 逐 turn 写入灵枢
|
|
10
|
+
(对话原文,不做摘要加工——写入侧加工差异正是各记忆系统的被测对象本身)
|
|
11
|
+
· 查询面:上游英文自然问句 → LLM 译为中文自然问句
|
|
12
|
+
· 答题面:reader 只据注入记忆作答(不足答「无法确定」)→ judge 对上游 gold
|
|
13
|
+
语义判等(correct/incorrect/refused;adversarial gold 空=正确行为拒答)
|
|
14
|
+
· 两臂:
|
|
15
|
+
retrieval 灵枢检索注入(cg.search 阶梯路 top-10,与生产主形态同路)
|
|
16
|
+
full_context 整段对话全量注入(对标 Mem0 论文 Table 2 的 Full-context 72.90)
|
|
17
|
+
|
|
18
|
+
标尺(Mem0 论文 arXiv:2504.19413 Table 2,agent=GPT-4o-mini):
|
|
19
|
+
Full-context 72.90 · Mem0ᵍ 68.44 · Mem0 66.88 · Zep 65.99 · LangMem 58.10 ·
|
|
20
|
+
OpenAI memory 52.90 · A-Mem 48.38;Letta Filesystem 复测 74.0(letta.com 2025-08)。
|
|
21
|
+
本评测 reader/judge=deepseek-chat(与同行 GPT-4o-mini 不同档,绝对分含模型差,
|
|
22
|
+
对照时声明;同标尺内两臂之差是检索质量的净效应,与模型无关)。
|
|
23
|
+
|
|
24
|
+
跑法:
|
|
25
|
+
python -X utf8 -m md_cg.bench_e2e_locomo_qa --translate # 翻译 5882 turn + 500 问句
|
|
26
|
+
python -X utf8 -m md_cg.bench_e2e_locomo_qa # 全量 500 题 × 2 臂
|
|
27
|
+
python -X utf8 -m md_cg.bench_e2e_locomo_qa --quick # 冒烟 3 题(须先完成翻译)
|
|
28
|
+
环境:DEEPSEEK_API_KEY;建议 MDCG_UNIFY_QUERY=0(中文问句含英文名时批次15归一有
|
|
29
|
+
形态缺陷,见 docs/eval/端到端干扰池评测_v1.1 §4)。
|
|
30
|
+
"""
|
|
31
|
+
from __future__ import annotations
|
|
32
|
+
|
|
33
|
+
import argparse
|
|
34
|
+
import hashlib
|
|
35
|
+
import json
|
|
36
|
+
import os
|
|
37
|
+
import re
|
|
38
|
+
import sys
|
|
39
|
+
import time
|
|
40
|
+
import urllib.request
|
|
41
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
42
|
+
|
|
43
|
+
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
44
|
+
if HERE not in sys.path:
|
|
45
|
+
sys.path.insert(0, HERE)
|
|
46
|
+
|
|
47
|
+
from md_cg.bench_e2e_judge import DEFAULT_ROOT, llm_chat # noqa: E402
|
|
48
|
+
from md_cg.bench_e2e_qa import ( # noqa: E402
|
|
49
|
+
JUDGE_SYS, READER_SYS, _cache_call, parse_json_field)
|
|
50
|
+
|
|
51
|
+
UP = os.path.join(DEFAULT_ROOT, "upstream")
|
|
52
|
+
CORPUS_PQ = os.path.join(UP, "corpus.parquet")
|
|
53
|
+
ANSWERS = os.path.join(UP, "answers_map.json")
|
|
54
|
+
ZH_TURNS = os.path.join(UP, "zh_turns.json")
|
|
55
|
+
ZH_QUERIES = os.path.join(UP, "zh_queries.json")
|
|
56
|
+
T_CACHE = os.path.join(UP, "translate_cache")
|
|
57
|
+
|
|
58
|
+
TRANS_TURN_SYS = (
|
|
59
|
+
"你是对话翻译器。把英文对话 turn 逐条译成中文:忠实原义、不压缩不增删;"
|
|
60
|
+
"人名与专有名词保留英文原文(如 Caroline / Ed Sheeran);说话人前缀保留"
|
|
61
|
+
"(格式「名字: 译文」)。\n只输出一行 JSON:{\"x1\": \"名字: 译文\", ...}"
|
|
62
|
+
)
|
|
63
|
+
TRANS_Q_SYS = (
|
|
64
|
+
"你是问题翻译器。把英文问句译成自然的中文问句:语气自然、不逐词直译;"
|
|
65
|
+
"人名与专有名词保留英文原文。\n只输出一行 JSON:{\"x1\": \"中文问句\", ...}"
|
|
66
|
+
)
|
|
67
|
+
|
|
68
|
+
|
|
69
|
+
def load_corpus():
|
|
70
|
+
import pyarrow.parquet as pq
|
|
71
|
+
t = pq.read_table(CORPUS_PQ)
|
|
72
|
+
ids = t.column("id").to_pylist()
|
|
73
|
+
txts = t.column("text").to_pylist()
|
|
74
|
+
return ids, txts
|
|
75
|
+
|
|
76
|
+
|
|
77
|
+
def _batch(lst, n):
|
|
78
|
+
for i in range(0, len(lst), n):
|
|
79
|
+
yield lst[i:i + n]
|
|
80
|
+
|
|
81
|
+
|
|
82
|
+
def translate_all(key, model_cfg, items, batch_n, sys_prompt, out_path,
|
|
83
|
+
max_tokens=4000):
|
|
84
|
+
"""items=[(k, en_text)] → {k: zh},分片缓存、断点续跑。"""
|
|
85
|
+
os.makedirs(T_CACHE, exist_ok=True)
|
|
86
|
+
out = {}
|
|
87
|
+
if os.path.isfile(out_path):
|
|
88
|
+
out = json.load(open(out_path, encoding="utf-8"))
|
|
89
|
+
todo = [(k, t) for k, t in items if k not in out]
|
|
90
|
+
print(f"[translate:{key}] 已有 {len(out)} / 待译 {len(todo)}"
|
|
91
|
+
f" · engine={model_cfg['model']}@{model_cfg['base']}")
|
|
92
|
+
jobs = list(enumerate(_batch(todo, batch_n)))
|
|
93
|
+
|
|
94
|
+
def call_one(items, mt):
|
|
95
|
+
# 行前缀从 x1 起(与系统提示示例 {"x1": …} 一致):模型对 x0 起头
|
|
96
|
+
# 的批量会整体偏移成 x1 起,导致键错位解析失败(214 条顽固缺口的根因)
|
|
97
|
+
mapping = {f"x{i + 1}": k for i, (k, _t) in enumerate(items)}
|
|
98
|
+
user = "\n".join(f"x{i + 1}. {t}" for i, (_k, t) in enumerate(items))
|
|
99
|
+
try:
|
|
100
|
+
raw, _u, _d = llm_chat(
|
|
101
|
+
model_cfg["model"], model_cfg["base"], model_cfg["key"],
|
|
102
|
+
[{"role": "system", "content": sys_prompt},
|
|
103
|
+
{"role": "user", "content": user}],
|
|
104
|
+
timeout=model_cfg["timeout"], max_tokens=mt,
|
|
105
|
+
extra_payload=model_cfg.get("extra"))
|
|
106
|
+
except Exception: # noqa: BLE001
|
|
107
|
+
return {}
|
|
108
|
+
s = re.sub(r"<think>.*?</think>", "", str(raw), flags=re.S)
|
|
109
|
+
obj = {}
|
|
110
|
+
_i, _j = s.find("{"), s.rfind("}")
|
|
111
|
+
if _i >= 0 and _j > _i:
|
|
112
|
+
try:
|
|
113
|
+
obj = json.loads(s[_i:_j + 1])
|
|
114
|
+
except ValueError:
|
|
115
|
+
obj = {}
|
|
116
|
+
res = {}
|
|
117
|
+
for tag, zh in obj.items():
|
|
118
|
+
if tag in mapping and isinstance(zh, str) and zh.strip():
|
|
119
|
+
res[mapping[tag]] = zh.strip()
|
|
120
|
+
return res
|
|
121
|
+
|
|
122
|
+
def work(item):
|
|
123
|
+
idx, job = item
|
|
124
|
+
# 缓存按**内容寻址**(分片键集 hash):早前按 todo 重排 idx 命名,
|
|
125
|
+
# 续传轮 idx 语义漂移会命中**旧内容的分片**(返回已在 out 的旧键,
|
|
126
|
+
# n_ok 虚涨而总数停滞——180 条顽固缺口的第二重根因)。内容寻址
|
|
127
|
+
# 跨轮稳定,同内容分片天然命中、不同内容互不误撞。
|
|
128
|
+
chash = hashlib.md5(",".join(k for k, _ in job)
|
|
129
|
+
.encode("utf-8")).hexdigest()[:12]
|
|
130
|
+
cpath = os.path.join(T_CACHE, f"{key}_{chash}.json")
|
|
131
|
+
if os.path.isfile(cpath):
|
|
132
|
+
return json.load(open(cpath, encoding="utf-8"))
|
|
133
|
+
res = call_one(job, max_tokens)
|
|
134
|
+
if len(res) < len(job):
|
|
135
|
+
# 分片级失败兜底:逐条重译(长 turn 挤爆批量输出导致 JSON 截断
|
|
136
|
+
# 的形态,单条请求给足生成空间即可收敛)
|
|
137
|
+
for one in job:
|
|
138
|
+
if one[0] not in res:
|
|
139
|
+
res.update(call_one([one], 1000))
|
|
140
|
+
if len(res) == len(job):
|
|
141
|
+
json.dump(res, open(cpath, "w", encoding="utf-8"),
|
|
142
|
+
ensure_ascii=False)
|
|
143
|
+
return res
|
|
144
|
+
|
|
145
|
+
n_ok = 0
|
|
146
|
+
with ThreadPoolExecutor(max_workers=model_cfg["workers"]) as ex:
|
|
147
|
+
for res in ex.map(work, jobs):
|
|
148
|
+
out.update(res)
|
|
149
|
+
n_ok += len(res)
|
|
150
|
+
json.dump(out, open(out_path, "w", encoding="utf-8"), ensure_ascii=False)
|
|
151
|
+
print(f"[translate:{key}] 本轮完成 {n_ok},总 {len(out)}/{len(items)}")
|
|
152
|
+
return out
|
|
153
|
+
|
|
154
|
+
|
|
155
|
+
def build_dialog_pool(ids, zh_turns, root_dir):
|
|
156
|
+
"""5882 中文 turn 逐条写入灵枢(对话原文做记忆,无摘要加工)。"""
|
|
157
|
+
root = os.path.join(root_dir, "pools", "dialog_zh")
|
|
158
|
+
if os.path.isfile(os.path.join(root, "_manifest.json")):
|
|
159
|
+
try:
|
|
160
|
+
m = json.load(open(os.path.join(root, "_manifest.json"),
|
|
161
|
+
encoding="utf-8"))
|
|
162
|
+
if m == {"turns": len(ids)}:
|
|
163
|
+
from md_cg.mdcos import MdCGOS
|
|
164
|
+
return MdCGOS(os.path.join(root, "mem"))
|
|
165
|
+
except Exception:
|
|
166
|
+
pass
|
|
167
|
+
import shutil
|
|
168
|
+
from md_cg.mdcos import MdCGOS
|
|
169
|
+
if os.path.isdir(root):
|
|
170
|
+
shutil.rmtree(root)
|
|
171
|
+
os.makedirs(root, exist_ok=True)
|
|
172
|
+
cg = MdCGOS(os.path.join(root, "mem"))
|
|
173
|
+
for nid in ids:
|
|
174
|
+
cg.add(nid, zh_turns.get(nid) or "", layer="knowledge",
|
|
175
|
+
verification_basis="data")
|
|
176
|
+
json.dump({"turns": len(ids)},
|
|
177
|
+
open(os.path.join(root, "_manifest.json"), "w", encoding="utf-8"))
|
|
178
|
+
return cg
|
|
179
|
+
|
|
180
|
+
|
|
181
|
+
READER_USER = "问题:{q}\n\n候选记忆:\n{cards}"
|
|
182
|
+
|
|
183
|
+
|
|
184
|
+
def run_one(cfg, cache_dir, qid, qtype, zh_q, en_q, gold, cards_text, arm):
|
|
185
|
+
row = {"qid": qid, "qtype": qtype, "arm": arm, "gold": gold}
|
|
186
|
+
|
|
187
|
+
def call_reader():
|
|
188
|
+
return llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
189
|
+
[{"role": "system", "content": READER_SYS},
|
|
190
|
+
{"role": "user", "content": READER_USER.format(
|
|
191
|
+
q=zh_q, cards=cards_text)}],
|
|
192
|
+
timeout=cfg["timeout"], max_tokens=300)
|
|
193
|
+
|
|
194
|
+
def call_judge():
|
|
195
|
+
return llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
196
|
+
[{"role": "system", "content": JUDGE_SYS},
|
|
197
|
+
{"role": "user", "content":
|
|
198
|
+
f"问题:{en_q}\ngold:{gold or ''}\n"
|
|
199
|
+
f"模型回答:{row['answer']}"}],
|
|
200
|
+
timeout=cfg["timeout"], max_tokens=200)
|
|
201
|
+
|
|
202
|
+
try:
|
|
203
|
+
pkey = f"{cfg['model']}|{arm}|{qid}|{cards_text[:200]}"
|
|
204
|
+
raw, _t1, _d1, _c1 = _cache_call(cache_dir, "reader",
|
|
205
|
+
hashlib.md5(
|
|
206
|
+
pkey.encode()).hexdigest(),
|
|
207
|
+
call_reader)
|
|
208
|
+
a = parse_json_field(raw, "answer")
|
|
209
|
+
row["answer"] = a if a is not None else (raw or "").strip()[:80]
|
|
210
|
+
jkey = f"{cfg['model']}|{en_q}|{gold}|{row['answer']}"
|
|
211
|
+
raw2, _t2, _d2, _c2 = _cache_call(
|
|
212
|
+
cache_dir, "judge", hashlib.md5(jkey.encode()).hexdigest(),
|
|
213
|
+
call_judge)
|
|
214
|
+
v = parse_json_field(raw2, "verdict")
|
|
215
|
+
row["verdict"] = v if v in ("correct", "incorrect", "refused") else None
|
|
216
|
+
if row["verdict"] is None:
|
|
217
|
+
row["error"] = f"judge 不可解析 {(raw2 or '')[:60]}"
|
|
218
|
+
except Exception as exc: # noqa: BLE001
|
|
219
|
+
row["error"] = f"{type(exc).__name__}: {exc}"
|
|
220
|
+
row.setdefault("answer", "")
|
|
221
|
+
row["verdict"] = None
|
|
222
|
+
return row
|
|
223
|
+
|
|
224
|
+
|
|
225
|
+
def report(rows, title):
|
|
226
|
+
ok = [r for r in rows if r.get("verdict")]
|
|
227
|
+
n = max(len(ok), 1)
|
|
228
|
+
acc = sum(1 for r in ok if r["verdict"] == "correct")
|
|
229
|
+
out = {"n": len(rows), "judged": len(ok), "acc": acc,
|
|
230
|
+
"acc_pct": 100.0 * acc / n, "refused": sum(
|
|
231
|
+
1 for r in ok if r["verdict"] == "refused")}
|
|
232
|
+
by = {}
|
|
233
|
+
for qt in ("single_hop", "multi_hop", "temporal_reasoning",
|
|
234
|
+
"adversarial", "open_domain"):
|
|
235
|
+
rs = [r for r in ok if r["qtype"] == qt]
|
|
236
|
+
if rs:
|
|
237
|
+
c = sum(1 for r in rs if r["verdict"] == "correct")
|
|
238
|
+
by[qt] = f"{100.0*c/len(rs):.1f}%({c}/{len(rs)})"
|
|
239
|
+
print(f" {title:<14} QA准确率={out['acc_pct']:5.1f}%"
|
|
240
|
+
f"({acc}/{len(ok)}) 拒答={out['refused']} 分题型: {by}")
|
|
241
|
+
return out
|
|
242
|
+
|
|
243
|
+
|
|
244
|
+
def main(argv=None):
|
|
245
|
+
ap = argparse.ArgumentParser(description="LoCoMo QA 同口径对照(中文全对话)")
|
|
246
|
+
ap.add_argument("--data-root", default=DEFAULT_ROOT)
|
|
247
|
+
ap.add_argument("--n", type=int, default=0, help="题数(0=全量500)")
|
|
248
|
+
ap.add_argument("--model", default="deepseek-chat")
|
|
249
|
+
ap.add_argument("--base-url", default="https://api.deepseek.com/v1")
|
|
250
|
+
ap.add_argument("--key", default="")
|
|
251
|
+
ap.add_argument("--workers", type=int, default=6)
|
|
252
|
+
ap.add_argument("--timeout", type=int, default=180)
|
|
253
|
+
ap.add_argument("--k", type=int, default=10)
|
|
254
|
+
ap.add_argument("--translate", action="store_true",
|
|
255
|
+
help="只做翻译阶段(5882 turn + 500 问句)")
|
|
256
|
+
ap.add_argument("--lm-studio", action="store_true",
|
|
257
|
+
help="翻译走 LM Studio 本地模型(localhost:1234/v1,"
|
|
258
|
+
"自动探测模型名;reader/judge 仍用 --model)")
|
|
259
|
+
ap.add_argument("--batch-turns", type=int, default=8,
|
|
260
|
+
help="翻译每请求 turn 数(LM Studio 建议 5)")
|
|
261
|
+
ap.add_argument("--batch-q", type=int, default=10,
|
|
262
|
+
help="问句翻译每请求数(LM Studio 建议 6)")
|
|
263
|
+
ap.add_argument("--quick", action="store_true")
|
|
264
|
+
a = ap.parse_args(argv)
|
|
265
|
+
if a.quick:
|
|
266
|
+
a.n = 3
|
|
267
|
+
key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
|
|
268
|
+
if not key:
|
|
269
|
+
print("[error] 需要 DEEPSEEK_API_KEY 或 --key")
|
|
270
|
+
return 2
|
|
271
|
+
cfg = {"model": a.model, "base": a.base_url, "key": key,
|
|
272
|
+
"timeout": a.timeout, "workers": a.workers}
|
|
273
|
+
print(f"[locomo_qa] 归一层开关 MDCG_UNIFY_QUERY="
|
|
274
|
+
f"{os.environ.get('MDCG_UNIFY_QUERY', '1')}(建议 0,见模块头注)")
|
|
275
|
+
|
|
276
|
+
ids, txts = load_corpus()
|
|
277
|
+
answers = json.load(open(ANSWERS, encoding="utf-8"))
|
|
278
|
+
ours = [json.loads(l) for l in open(os.path.join(
|
|
279
|
+
HERE, "data", "benchmarks", "locomo-zh-500", "questions500.jsonl"),
|
|
280
|
+
encoding="utf-8") if l.strip()]
|
|
281
|
+
qs = ours[:a.n] if a.n else ours
|
|
282
|
+
|
|
283
|
+
if a.translate or not (os.path.isfile(ZH_TURNS)
|
|
284
|
+
and os.path.isfile(ZH_QUERIES)):
|
|
285
|
+
tcfg = cfg
|
|
286
|
+
if a.lm_studio:
|
|
287
|
+
with urllib.request.urlopen(
|
|
288
|
+
"http://localhost:1234/v1/models", timeout=8) as r:
|
|
289
|
+
mids = [m["id"] for m in
|
|
290
|
+
json.loads(r.read().decode("utf-8"))["data"]
|
|
291
|
+
if "embed" not in m["id"].lower()]
|
|
292
|
+
if not mids:
|
|
293
|
+
print("[error] LM Studio 无已加载 LLM(仅有 embedding 模型)")
|
|
294
|
+
return 2
|
|
295
|
+
tcfg = {"model": mids[0], "base": "http://localhost:1234/v1",
|
|
296
|
+
"key": "lm-studio", "timeout": 600,
|
|
297
|
+
"workers": min(a.workers, 3),
|
|
298
|
+
# 顶层 reasoning_effort=none 是实测唯一能压住思考的传法
|
|
299
|
+
# (chat_template_kwargs/enable_thinking 均无效,思考仍吃满
|
|
300
|
+
# max_tokens;矩阵实验 2026-09-23:none → 0 reasoning 4s/批)
|
|
301
|
+
"extra": {"reasoning_effort": "none"}}
|
|
302
|
+
print(f"[lm-studio] 使用本地模型 {mids[0]}(并发 {tcfg['workers']})")
|
|
303
|
+
items_t = list(zip(ids, txts))
|
|
304
|
+
zh_turns = translate_all("turn", tcfg, items_t, a.batch_turns,
|
|
305
|
+
TRANS_TURN_SYS, ZH_TURNS,
|
|
306
|
+
max_tokens=2000 if a.lm_studio else 4000)
|
|
307
|
+
items_q = [(q["qid"], answers[q["qid"]]["en_q"]) for q in ours
|
|
308
|
+
if q["qid"] in answers]
|
|
309
|
+
zh_qs = translate_all("q", tcfg, items_q, a.batch_q,
|
|
310
|
+
TRANS_Q_SYS, ZH_QUERIES,
|
|
311
|
+
max_tokens=2000 if a.lm_studio else 4000)
|
|
312
|
+
miss_t = len(ids) - len(zh_turns)
|
|
313
|
+
miss_q = len(items_q) - len(zh_qs)
|
|
314
|
+
print(f"[translate] 缺口 turn={miss_t} q={miss_q}"
|
|
315
|
+
f"(缺口>0 可重跑 --translate 续传)")
|
|
316
|
+
if a.translate:
|
|
317
|
+
return 0 if (miss_t == 0 and miss_q == 0) else 1
|
|
318
|
+
|
|
319
|
+
zh_turns = json.load(open(ZH_TURNS, encoding="utf-8"))
|
|
320
|
+
zh_qs = json.load(open(ZH_QUERIES, encoding="utf-8"))
|
|
321
|
+
cache_dir = os.path.join(a.data_root, "lq_cache")
|
|
322
|
+
os.makedirs(cache_dir, exist_ok=True)
|
|
323
|
+
from md_cg.mdcos import MdCGOS
|
|
324
|
+
cg = build_dialog_pool(ids, zh_turns, a.data_root)
|
|
325
|
+
# scene → 全对话中文(full-context 臂用)
|
|
326
|
+
scene_turns = {}
|
|
327
|
+
for nid in ids:
|
|
328
|
+
scene_turns.setdefault(nid.split("_session")[0], []).append(nid)
|
|
329
|
+
|
|
330
|
+
tasks = []
|
|
331
|
+
for q in qs:
|
|
332
|
+
qid = q["qid"]
|
|
333
|
+
if qid not in answers or qid not in zh_qs:
|
|
334
|
+
continue
|
|
335
|
+
gold = answers[qid]["answer"] or ""
|
|
336
|
+
zh_q = zh_qs[qid]
|
|
337
|
+
en_q = answers[qid]["en_q"]
|
|
338
|
+
# 臂1:灵枢检索注入(生产主形态阶梯路;judge 关——对话原文无 CCG 面)
|
|
339
|
+
res, _m = cg.search(zh_q, k=a.k, judge=False, record=False)
|
|
340
|
+
cards = "\n\n".join(f"【{i}】{zh_turns.get(r[0]['id'], '')}"
|
|
341
|
+
for i, r in enumerate(res, 1))
|
|
342
|
+
tasks.append((qid, q.get("qtype"), zh_q, en_q, gold, cards,
|
|
343
|
+
"retrieval"))
|
|
344
|
+
# 臂2:full-context(该 scene 全对话)
|
|
345
|
+
scene = qid.split("_q_")[0]
|
|
346
|
+
full = "\n".join(zh_turns.get(t, "") for t in scene_turns[scene])
|
|
347
|
+
tasks.append((qid, q.get("qtype"), zh_q, en_q, gold, full[:110000],
|
|
348
|
+
"full_context"))
|
|
349
|
+
t0 = time.time()
|
|
350
|
+
with ThreadPoolExecutor(max_workers=a.workers) as ex:
|
|
351
|
+
rows = list(ex.map(lambda t: run_one(cfg, cache_dir, *t), tasks))
|
|
352
|
+
print(f"[locomo_qa] {len(tasks)} 次 reader+judge 耗时 {time.time()-t0:.0f}s")
|
|
353
|
+
errs = sum(1 for r in rows if r.get("error"))
|
|
354
|
+
print(f"判分失败 {errs} 行\n")
|
|
355
|
+
summary = {"meta": {"model": a.model, "n_q": len(qs), "k": a.k,
|
|
356
|
+
"unify": os.environ.get("MDCG_UNIFY_QUERY", "1")}}
|
|
357
|
+
for arm in ("retrieval", "full_context"):
|
|
358
|
+
summary[arm] = report([r for r in rows if r["arm"] == arm], arm)
|
|
359
|
+
out = os.path.join(a.data_root, "results",
|
|
360
|
+
f"locomo_qa_{time.strftime('%Y%m%d_%H%M%S')}.json")
|
|
361
|
+
json.dump({"summary": summary, "rows": rows},
|
|
362
|
+
open(out, "w", encoding="utf-8"), ensure_ascii=False, indent=1)
|
|
363
|
+
print(f"\n[locomo_qa] 明细 → {out}")
|
|
364
|
+
return 0
|
|
365
|
+
|
|
366
|
+
|
|
367
|
+
if __name__ == "__main__":
|
|
368
|
+
sys.exit(main())
|
|
@@ -0,0 +1,256 @@
|
|
|
1
|
+
# -*- coding: utf-8 -*-
|
|
2
|
+
"""端到端 QA 评测:pinpoint / answerability / decision-making(2026-09-23 · 外部建议)
|
|
3
|
+
|
|
4
|
+
评测问题(外部建议口径:hit@k 是上界,端到端要答对题):
|
|
5
|
+
pinpoint 干扰浓度升高(level 0 → 4),reader 从候选记忆中抽取正确事实的
|
|
6
|
+
QA 准确率退化多少?
|
|
7
|
+
answerability adversarial 题(上游设计 gold 为空、正确行为=拒答):reader 会不会
|
|
8
|
+
被干扰节点骗去编答案?
|
|
9
|
+
decision-making 裁决层是否改变下游行为——同一 reader,候选分别来自
|
|
10
|
+
arm_base(裸融合 top10)与 arm_firewall(证据防火墙重排 top10),
|
|
11
|
+
QA 准确率有无差异?
|
|
12
|
+
|
|
13
|
+
管线(全部真实 LLM,deepseek-chat,temperature=0,带缓存):
|
|
14
|
+
检索(与 bench_e2e_judge 同池同采样同臂)→ reader 只据候选卡作答
|
|
15
|
+
(不足则答「无法确定」)→ judge 对 gold 语义判等(correct/incorrect/refused)。
|
|
16
|
+
gold 答案来自上游 LoCoMo 原始标注(mteb/LoCoMo 1976 题英文题面 ↔
|
|
17
|
+
snap-research/locomo10.json 逐题文本精确匹配,500/500 覆盖;
|
|
18
|
+
113 题 adversarial 的 gold 为空 = 拒答语义)。
|
|
19
|
+
|
|
20
|
+
诚实边界:
|
|
21
|
+
· reader 同时拿中文关键词串与上游英文原题(真实用户问题面),记忆卡为中文,
|
|
22
|
+
judge 对英文 gold 做跨语语义判等——这是双语口径,非纯单语评测。
|
|
23
|
+
· judge 亦为 LLM(LLM-as-a-judge),其判分自身有方差;refused 由 judge 判
|
|
24
|
+
reader 文本,边界例(半猜测半拒答)按 judge 裁决计入,不人工复核(v1)。
|
|
25
|
+
· level 2 不入 QA 层(0 vs 4 已给退化区间,节省调用)。
|
|
26
|
+
|
|
27
|
+
跑法:
|
|
28
|
+
python -X utf8 -m md_cg.bench_e2e_qa --quick # 冒烟 3 题
|
|
29
|
+
python -X utf8 -m md_cg.bench_e2e_qa # 全量 120 题 × 4 组
|
|
30
|
+
依赖:DEEPSEEK_API_KEY;answers_map 由上游对齐生成(见 --answers 默认路径)。
|
|
31
|
+
"""
|
|
32
|
+
from __future__ import annotations
|
|
33
|
+
|
|
34
|
+
import argparse
|
|
35
|
+
import hashlib
|
|
36
|
+
import json
|
|
37
|
+
import os
|
|
38
|
+
import sys
|
|
39
|
+
import time
|
|
40
|
+
from concurrent.futures import ThreadPoolExecutor
|
|
41
|
+
|
|
42
|
+
HERE = os.path.dirname(os.path.dirname(os.path.abspath(__file__)))
|
|
43
|
+
if HERE not in sys.path:
|
|
44
|
+
sys.path.insert(0, HERE)
|
|
45
|
+
|
|
46
|
+
from md_cg.bench_e2e_judge import ( # noqa: E402
|
|
47
|
+
DEFAULT_ROOT, load_src, speakers_by_scene, build_pool, run_det_arms,
|
|
48
|
+
llm_chat)
|
|
49
|
+
|
|
50
|
+
READER_SYS = (
|
|
51
|
+
"你是记忆问答器。只根据给定的记忆卡回答问题,不要用记忆卡以外的知识。"
|
|
52
|
+
"记忆卡不足以确定答案时,回答「无法确定」。答案尽量简短(不超过20字),"
|
|
53
|
+
"不要解释。\n只输出一行 JSON:{\"answer\": \"…\"}"
|
|
54
|
+
)
|
|
55
|
+
|
|
56
|
+
JUDGE_SYS = (
|
|
57
|
+
"你是 QA 评分器。给定问题、参考答案(gold)与模型回答,判一个结论:\n"
|
|
58
|
+
"- correct:回答与 gold 语义等价(中英跨语判等;日期/人名/实体须一致,"
|
|
59
|
+
"表述差异可容忍)\n"
|
|
60
|
+
"- incorrect:回答了但语义与 gold 不等价(含被错误记忆误导的答案)\n"
|
|
61
|
+
"- refused:回答表示无法确定/不知道/记忆未提及\n"
|
|
62
|
+
"gold 为空字符串时(对抗题):回答「无法确定」= correct,给出任何具体"
|
|
63
|
+
"答案 = incorrect。\n只输出一行 JSON:{\"verdict\": \"correct\"}"
|
|
64
|
+
)
|
|
65
|
+
|
|
66
|
+
|
|
67
|
+
def _cache_call(cache_dir, kind, payload_key, fn):
|
|
68
|
+
"""LLM 调用缓存:payload_key 相同直接回放(qa 层可复现性与省钱)。"""
|
|
69
|
+
hexk = hashlib.md5(payload_key.encode("utf-8")).hexdigest()
|
|
70
|
+
path = os.path.join(cache_dir, f"{kind}_{hexk}.json")
|
|
71
|
+
if os.path.isfile(path):
|
|
72
|
+
d = json.load(open(path, encoding="utf-8"))
|
|
73
|
+
return d["out"], d.get("tokens", 0), 0.0, True
|
|
74
|
+
out, usage, dt = fn()
|
|
75
|
+
json.dump({"out": out, "tokens": int(usage.get("total_tokens") or 0)},
|
|
76
|
+
open(path, "w", encoding="utf-8"), ensure_ascii=False)
|
|
77
|
+
return out, int(usage.get("total_tokens") or 0), dt, False
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def parse_json_field(raw, field):
|
|
81
|
+
s = raw or ""
|
|
82
|
+
i, j = s.find("{"), s.rfind("}")
|
|
83
|
+
if i >= 0 and j > i:
|
|
84
|
+
try:
|
|
85
|
+
return json.loads(s[i:j + 1]).get(field)
|
|
86
|
+
except ValueError:
|
|
87
|
+
pass
|
|
88
|
+
return None
|
|
89
|
+
|
|
90
|
+
|
|
91
|
+
def run_qa_one(cfg, cache_dir, q, ans, cand_nodes, arm, level):
|
|
92
|
+
"""一题一组:reader 作答 → judge 判分。返回行(异常行 error 字段)。"""
|
|
93
|
+
cards = "\n\n".join(
|
|
94
|
+
f"【{i}】{str(nd.get('content') or '')[:400]}"
|
|
95
|
+
for i, nd in enumerate(cand_nodes, 1))
|
|
96
|
+
r_user = (f"问题(中文关键词):{q['question']}\n"
|
|
97
|
+
f"问题(英文原题):{ans.get('en_q') or ''}\n\n候选记忆卡:\n{cards}")
|
|
98
|
+
tag = f"{q['qid']}|{arm}|L{level}"
|
|
99
|
+
|
|
100
|
+
def call_reader():
|
|
101
|
+
return llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
102
|
+
[{"role": "system", "content": READER_SYS},
|
|
103
|
+
{"role": "user", "content": r_user}],
|
|
104
|
+
timeout=cfg["timeout"], max_tokens=300)
|
|
105
|
+
|
|
106
|
+
def call_judge():
|
|
107
|
+
return llm_chat(cfg["model"], cfg["base"], cfg["key"],
|
|
108
|
+
[{"role": "system", "content": JUDGE_SYS},
|
|
109
|
+
{"role": "user", "content":
|
|
110
|
+
f"问题:{ans.get('en_q') or q['question']}\n"
|
|
111
|
+
f"gold:{ans.get('answer') or ''}\n"
|
|
112
|
+
f"模型回答:{row['answer']}"}],
|
|
113
|
+
timeout=cfg["timeout"], max_tokens=200)
|
|
114
|
+
|
|
115
|
+
row = {"qid": q["qid"], "qtype": q.get("qtype"), "arm": arm,
|
|
116
|
+
"level": level, "gold": ans.get("answer") or ""}
|
|
117
|
+
try:
|
|
118
|
+
raw, tok_r, dt_r, _c = _cache_call(
|
|
119
|
+
cache_dir, "reader", cfg["model"] + "|" + r_user, call_reader)
|
|
120
|
+
ans_txt = parse_json_field(raw, "answer")
|
|
121
|
+
if ans_txt is None:
|
|
122
|
+
ans_txt = (raw or "").strip()[:80]
|
|
123
|
+
row["answer"] = ans_txt
|
|
124
|
+
raw2, tok_j, dt_j, _c2 = _cache_call(
|
|
125
|
+
cache_dir, "judge",
|
|
126
|
+
cfg["model"] + "|" + row["gold"] + "|" + ans_txt + "|"
|
|
127
|
+
+ (ans.get("en_q") or ""), call_judge)
|
|
128
|
+
verdict = parse_json_field(raw2, "verdict")
|
|
129
|
+
row["verdict"] = verdict if verdict in ("correct", "incorrect",
|
|
130
|
+
"refused") else None
|
|
131
|
+
row["tokens"] = tok_r + tok_j
|
|
132
|
+
row["latency"] = dt_r + dt_j
|
|
133
|
+
if row["verdict"] is None:
|
|
134
|
+
row["error"] = f"judge 不可解析:{(raw2 or '')[:80]}"
|
|
135
|
+
except Exception as exc: # noqa: BLE001
|
|
136
|
+
row["error"] = f"{type(exc).__name__}: {exc}"
|
|
137
|
+
row.setdefault("answer", "")
|
|
138
|
+
row["verdict"] = None
|
|
139
|
+
row["tokens"] = row.get("tokens", 0)
|
|
140
|
+
row["latency"] = row.get("latency", 0.0)
|
|
141
|
+
row["_tag"] = tag
|
|
142
|
+
return row
|
|
143
|
+
|
|
144
|
+
|
|
145
|
+
def qa_report(rows):
|
|
146
|
+
n = len(rows)
|
|
147
|
+
ok = [r for r in rows if r.get("verdict")]
|
|
148
|
+
adv = [r for r in ok if r.get("qtype") == "adversarial"]
|
|
149
|
+
norm = [r for r in ok if r.get("qtype") != "adversarial"]
|
|
150
|
+
c = lambda rs, v: sum(1 for r in rs if r["verdict"] == v) # noqa: E731
|
|
151
|
+
st = {
|
|
152
|
+
"n": n, "judged": len(ok), "errors": n - len(ok),
|
|
153
|
+
"acc": c(ok, "correct") / max(len(ok), 1),
|
|
154
|
+
"incorrect": c(ok, "incorrect"), "refused": c(ok, "refused"),
|
|
155
|
+
"norm_n": len(norm),
|
|
156
|
+
"norm_acc": c(norm, "correct") / max(len(norm), 1),
|
|
157
|
+
"norm_refused": c(norm, "refused"),
|
|
158
|
+
"adv_n": len(adv),
|
|
159
|
+
"adv_correct_refusal": c(adv, "correct"),
|
|
160
|
+
"adv_fooled": c(adv, "incorrect"),
|
|
161
|
+
"tokens": sum(r.get("tokens", 0) for r in rows),
|
|
162
|
+
}
|
|
163
|
+
return st
|
|
164
|
+
|
|
165
|
+
|
|
166
|
+
def main(argv=None):
|
|
167
|
+
ap = argparse.ArgumentParser(description="端到端 QA:pinpoint/answerability")
|
|
168
|
+
ap.add_argument("--data-root", default=DEFAULT_ROOT)
|
|
169
|
+
ap.add_argument("--answers",
|
|
170
|
+
default=os.path.join(DEFAULT_ROOT, "upstream",
|
|
171
|
+
"answers_map.json"))
|
|
172
|
+
ap.add_argument("--n", type=int, default=120)
|
|
173
|
+
ap.add_argument("--seed", type=int, default=7)
|
|
174
|
+
ap.add_argument("--levels", default="0,4")
|
|
175
|
+
ap.add_argument("--model", default="deepseek-chat")
|
|
176
|
+
ap.add_argument("--base-url", default="https://api.deepseek.com/v1")
|
|
177
|
+
ap.add_argument("--key", default="")
|
|
178
|
+
ap.add_argument("--workers", type=int, default=6)
|
|
179
|
+
ap.add_argument("--timeout", type=int, default=120)
|
|
180
|
+
ap.add_argument("--k", type=int, default=10)
|
|
181
|
+
ap.add_argument("--quick", action="store_true")
|
|
182
|
+
a = ap.parse_args(argv)
|
|
183
|
+
if a.quick:
|
|
184
|
+
a.n = 3
|
|
185
|
+
key = a.key or os.environ.get("DEEPSEEK_API_KEY") or ""
|
|
186
|
+
if not key:
|
|
187
|
+
print("[error] 需要 DEEPSEEK_API_KEY 或 --key")
|
|
188
|
+
return 2
|
|
189
|
+
if not os.path.isfile(a.answers):
|
|
190
|
+
print(f"[error] 找不到答案映射 {a.answers}(先跑上游对齐生成)")
|
|
191
|
+
return 2
|
|
192
|
+
|
|
193
|
+
levels = [int(x) for x in a.levels.split(",") if x.strip() != ""]
|
|
194
|
+
cache_dir = os.path.join(a.data_root, "qa_cache")
|
|
195
|
+
os.makedirs(cache_dir, exist_ok=True)
|
|
196
|
+
res_dir = os.path.join(a.data_root, "results")
|
|
197
|
+
os.makedirs(res_dir, exist_ok=True)
|
|
198
|
+
|
|
199
|
+
import random
|
|
200
|
+
corpus, questions = load_src()
|
|
201
|
+
corpus_map = {c["id"]: c for c in corpus}
|
|
202
|
+
spk = speakers_by_scene(corpus)
|
|
203
|
+
answers = json.load(open(a.answers, encoding="utf-8"))
|
|
204
|
+
sample = sorted(random.Random(a.seed).sample(
|
|
205
|
+
questions, min(a.n, len(questions))), key=lambda q: q["qid"])
|
|
206
|
+
print(f"[e2e_qa] 采样 {len(sample)} 题(seed={a.seed})× levels={levels}"
|
|
207
|
+
f" × 候选臂 base/firewall × {a.model}")
|
|
208
|
+
|
|
209
|
+
cfg = {"model": a.model, "base": a.base_url, "key": key,
|
|
210
|
+
"timeout": a.timeout}
|
|
211
|
+
all_rows, summary = [], {"meta": {"seed": a.seed, "n": len(sample),
|
|
212
|
+
"model": a.model, "levels": levels}}
|
|
213
|
+
for lv in levels:
|
|
214
|
+
cg, n_synth = build_pool(lv, corpus, questions, corpus_map, spk,
|
|
215
|
+
a.data_root)
|
|
216
|
+
det = []
|
|
217
|
+
for q in sample:
|
|
218
|
+
r = run_det_arms(cg, q, k=a.k)
|
|
219
|
+
r["q"] = q
|
|
220
|
+
det.append(r)
|
|
221
|
+
tasks = []
|
|
222
|
+
for r in det:
|
|
223
|
+
for arm in ("base", "firewall"):
|
|
224
|
+
nodes = r["base_nodes"][:a.k] if arm == "base" \
|
|
225
|
+
else r["fw_nodes"][:a.k]
|
|
226
|
+
tasks.append((r["q"], answers.get(r["q"]["qid"], {}),
|
|
227
|
+
nodes, arm, lv))
|
|
228
|
+
t0 = time.time()
|
|
229
|
+
with ThreadPoolExecutor(max_workers=a.workers) as ex:
|
|
230
|
+
rows = list(ex.map(lambda t: run_qa_one(cfg, cache_dir, *t),
|
|
231
|
+
tasks))
|
|
232
|
+
all_rows.extend(rows)
|
|
233
|
+
print(f"\n===== level {lv}(池={len(corpus)+n_synth})"
|
|
234
|
+
f" · QA 耗时 {time.time()-t0:.1f}s =====")
|
|
235
|
+
print(f"{'臂':<10}{'QA准确率':>9}{'常规准确率':>10}{'常规拒答':>8}"
|
|
236
|
+
f"{'对抗正确拒答':>12}{'对抗被骗':>8}{'判分失败':>8}")
|
|
237
|
+
for arm in ("base", "firewall"):
|
|
238
|
+
st = qa_report([r for r in rows if r["arm"] == arm])
|
|
239
|
+
print(f"{arm:<10}{st['acc']*100:>8.1f}%{st['norm_acc']*100:>9.1f}%"
|
|
240
|
+
f"{st['norm_refused']:>7d}/{st['norm_n']}"
|
|
241
|
+
f"{st['adv_correct_refusal']:>10d}/{st['adv_n']}"
|
|
242
|
+
f"{st['adv_fooled']:>7d}"
|
|
243
|
+
f"{st['errors']:>7d}")
|
|
244
|
+
summary[f"L{lv}_{arm}"] = st
|
|
245
|
+
out = os.path.join(res_dir,
|
|
246
|
+
f"qa_{time.strftime('%Y%m%d_%H%M%S')}.json")
|
|
247
|
+
json.dump({"summary": summary,
|
|
248
|
+
"rows": [{k: v for k, v in r.items() if k != "q"}
|
|
249
|
+
for r in all_rows]},
|
|
250
|
+
open(out, "w", encoding="utf-8"), ensure_ascii=False, indent=1)
|
|
251
|
+
print(f"\n[e2e_qa] 完成。明细 → {out}")
|
|
252
|
+
return 0
|
|
253
|
+
|
|
254
|
+
|
|
255
|
+
if __name__ == "__main__":
|
|
256
|
+
sys.exit(main())
|
package/md_cg/branches.py
CHANGED
|
@@ -23,6 +23,8 @@ import os
|
|
|
23
23
|
import re
|
|
24
24
|
import uuid
|
|
25
25
|
|
|
26
|
+
from .fsutil import publish
|
|
27
|
+
|
|
26
28
|
#: discard 的 branch_summary 必填标记(复用 NEG_MEMORY_MARKS 的必填防呆模式:
|
|
27
29
|
#: 「假设 + 结果 + 教训」缺一不收——放弃分支必须留下可复用的教训)。
|
|
28
30
|
BRANCH_MARKS = ("分支假设", "实验结果", "教训")
|
|
@@ -254,9 +256,23 @@ def discard(cg, branch_id, summary: str) -> dict:
|
|
|
254
256
|
if path:
|
|
255
257
|
src = os.path.join(cg.root, str(path).replace("/", os.sep))
|
|
256
258
|
if os.path.exists(src):
|
|
257
|
-
|
|
259
|
+
publish(src, os.path.join(cold, os.path.basename(src)))
|
|
258
260
|
moved += 1
|
|
259
|
-
|
|
261
|
+
# 摘索引必须**落盘**(写删除记录,forget 的口径——mdcos.py forget
|
|
262
|
+
# 注释明载同构坑):只 pop 内存时,fork 阶段仍在 _dirty 的分支条目
|
|
263
|
+
# 会被下面的 flush 作为 upsert 持久化进 _index_log;close 的 compact
|
|
264
|
+
# 计数对账触发重扫后,_apply_log 把该 upsert 叠回干净扫描之上——
|
|
265
|
+
# 幽灵条目连同目录指纹固化进 _index.json 快照,重开永久复活。
|
|
266
|
+
# _unstage:pop 内存 + _dirty[nid]=None(tombstone)+ 立即 flush,
|
|
267
|
+
# 重放按序 pop 掉 fork upsert(含 fork 后曾显式 flush 的时序)。
|
|
268
|
+
unst = getattr(cg, "_unstage", None)
|
|
269
|
+
if unst is not None:
|
|
270
|
+
try:
|
|
271
|
+
unst(nid)
|
|
272
|
+
except Exception: # noqa: BLE001
|
|
273
|
+
(cg.index.get("nodes") or {}).pop(nid, None)
|
|
274
|
+
else:
|
|
275
|
+
(cg.index.get("nodes") or {}).pop(nid, None)
|
|
260
276
|
fl = getattr(cg, "flush", None)
|
|
261
277
|
if fl is not None:
|
|
262
278
|
try:
|