sophhub 0.4.65 → 0.4.67
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/agents/ai-cs-admin/.config.json +6 -1
- package/agents/ai-cs-admin/AGENTS.md +54 -8
- package/agents/ai-cs-qa/.config.json +7 -2
- package/agents/ai-cs-qa/AGENTS.md +38 -81
- package/agents/ai-cs-qa/BOOTSTRAP.md +1 -2
- package/agents/ai-cs-qa/SOUL.md +3 -2
- package/agents/ai-cs-qa/TOOLS.md +8 -8
- package/agents/ai-cs-qa/scripts/setup_links.sh +14 -0
- package/package.json +1 -1
- package/skills/knowledge-search/skill.json +26 -0
- package/skills/knowledge-search/src/SKILL.md +63 -0
- package/skills/knowledge-search/src/pyproject.toml +8 -0
- package/skills/knowledge-search/src/scripts/__init__.py +1 -0
- package/skills/knowledge-search/src/scripts/bge_client.py +139 -0
- package/skills/knowledge-search/src/scripts/bm25.py +60 -0
- package/skills/knowledge-search/src/scripts/index_loader.py +35 -0
- package/skills/knowledge-search/src/scripts/ksearch.py +141 -0
- package/skills/knowledge-search/src/scripts/ranker.py +75 -0
- package/skills/knowledge-search-admin/skill.json +31 -0
- package/skills/knowledge-search-admin/src/SKILL.md +79 -0
- package/skills/knowledge-search-admin/src/pyproject.toml +8 -0
- package/skills/knowledge-search-admin/src/scripts/__init__.py +1 -0
- package/skills/knowledge-search-admin/src/scripts/bge_client.py +139 -0
- package/skills/knowledge-search-admin/src/scripts/bm25.py +37 -0
- package/skills/knowledge-search-admin/src/scripts/chunker.py +210 -0
- package/skills/knowledge-search-admin/src/scripts/index_store.py +147 -0
- package/skills/knowledge-search-admin/src/scripts/ksearch.py +218 -0
- package/skills/knowledge-search-admin/src/scripts/ranker.py +75 -0
|
@@ -0,0 +1,139 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""BGE-M3 embedding 与 bge-reranker-v2-m3 客户端。
|
|
3
|
+
|
|
4
|
+
通过 stdlib urllib 调用 Sophnet 平台接口,返回 L2 归一化后的向量与 rerank 分数。
|
|
5
|
+
平台 ApiKey 在运行时由 sophnet_tools.get_api_key() 获取,不再硬编码。
|
|
6
|
+
"""
|
|
7
|
+
|
|
8
|
+
import json
|
|
9
|
+
import os
|
|
10
|
+
import urllib.error
|
|
11
|
+
import urllib.request
|
|
12
|
+
from typing import List, Tuple
|
|
13
|
+
|
|
14
|
+
import numpy as np
|
|
15
|
+
|
|
16
|
+
# Sophnet 平台接口地址
|
|
17
|
+
BASE_URL = "https://www.sophnet.com/api/open-apis"
|
|
18
|
+
EMBED_URL = BASE_URL + "/projects/easyllms/embeddings"
|
|
19
|
+
RERANK_URL = BASE_URL + "/projects/rerank"
|
|
20
|
+
|
|
21
|
+
EMBED_MODEL = "bge-m3"
|
|
22
|
+
RERANK_MODEL = "bge-reranker-v2-m3"
|
|
23
|
+
EMBED_DIM = 1024
|
|
24
|
+
EMBED_BATCH = 8 # embedding 单次输入条数上限(按需保守取 8)
|
|
25
|
+
RERANK_BATCH = 256 # rerank 接口单次 documents 上限
|
|
26
|
+
DEFAULT_TIMEOUT = 60
|
|
27
|
+
|
|
28
|
+
|
|
29
|
+
class BgeError(RuntimeError):
|
|
30
|
+
pass
|
|
31
|
+
|
|
32
|
+
|
|
33
|
+
def _get_api_key() -> str:
|
|
34
|
+
"""运行时获取平台 ApiKey:优先 sophnet_tools.get_api_key(),回退环境变量 SOPHNET_API_KEY。"""
|
|
35
|
+
try:
|
|
36
|
+
import sophnet_tools # noqa: S404 - 平台运行时提供
|
|
37
|
+
key = sophnet_tools.get_api_key()
|
|
38
|
+
if key:
|
|
39
|
+
return key
|
|
40
|
+
except ImportError:
|
|
41
|
+
pass
|
|
42
|
+
key = os.environ.get("SOPHNET_API_KEY")
|
|
43
|
+
if not key:
|
|
44
|
+
raise BgeError("未找到 Sophnet 平台 ApiKey(sophnet_tools 不可用且 SOPHNET_API_KEY 未设置),请联系客户支持")
|
|
45
|
+
return key
|
|
46
|
+
|
|
47
|
+
|
|
48
|
+
def _post_json(url: str, key: str, payload: dict, timeout: int = DEFAULT_TIMEOUT) -> dict:
|
|
49
|
+
body = json.dumps(payload).encode("utf-8")
|
|
50
|
+
req = urllib.request.Request(
|
|
51
|
+
url=url,
|
|
52
|
+
data=body,
|
|
53
|
+
method="POST",
|
|
54
|
+
headers={
|
|
55
|
+
"Content-Type": "application/json",
|
|
56
|
+
"Accept": "application/json",
|
|
57
|
+
"Authorization": "Bearer %s" % key,
|
|
58
|
+
},
|
|
59
|
+
)
|
|
60
|
+
try:
|
|
61
|
+
with urllib.request.urlopen(req, timeout=timeout) as resp:
|
|
62
|
+
raw = resp.read()
|
|
63
|
+
except urllib.error.HTTPError as e:
|
|
64
|
+
snippet = e.read()[:300].decode("utf-8", "replace")
|
|
65
|
+
raise BgeError("HTTP %s 调用 %s 失败:%s" % (e.code, url, snippet))
|
|
66
|
+
except urllib.error.URLError as e:
|
|
67
|
+
raise BgeError("网络错误,无法访问 %s:%s" % (url, e.reason))
|
|
68
|
+
try:
|
|
69
|
+
return json.loads(raw.decode("utf-8"))
|
|
70
|
+
except ValueError:
|
|
71
|
+
raise BgeError("响应非 JSON:%s" % raw[:300].decode("utf-8", "replace"))
|
|
72
|
+
|
|
73
|
+
|
|
74
|
+
def _l2_normalize(matrix: np.ndarray) -> np.ndarray:
|
|
75
|
+
norms = np.linalg.norm(matrix, axis=1, keepdims=True)
|
|
76
|
+
norms[norms == 0] = 1.0
|
|
77
|
+
return (matrix / norms).astype(np.float32)
|
|
78
|
+
|
|
79
|
+
|
|
80
|
+
def embed(texts: List[str], timeout: int = DEFAULT_TIMEOUT) -> np.ndarray:
|
|
81
|
+
"""批量编码文本,返回 L2 归一化后的 float32 矩阵 (N, 1024)。"""
|
|
82
|
+
texts = [t for t in texts]
|
|
83
|
+
if not texts:
|
|
84
|
+
return np.zeros((0, EMBED_DIM), dtype=np.float32)
|
|
85
|
+
|
|
86
|
+
key = _get_api_key()
|
|
87
|
+
vectors: List[List[float]] = []
|
|
88
|
+
for start in range(0, len(texts), EMBED_BATCH):
|
|
89
|
+
batch = texts[start:start + EMBED_BATCH]
|
|
90
|
+
payload = {
|
|
91
|
+
"model": EMBED_MODEL,
|
|
92
|
+
"input_texts": batch,
|
|
93
|
+
"dimensions": EMBED_DIM,
|
|
94
|
+
}
|
|
95
|
+
data = _post_json(EMBED_URL, key, payload, timeout=timeout)
|
|
96
|
+
# OpenAI 风格:{"data":[{"index":i,"embedding":[...]}, ...]}
|
|
97
|
+
items = data.get("data") if isinstance(data, dict) else None
|
|
98
|
+
if not isinstance(items, list) or len(items) != len(batch):
|
|
99
|
+
raise BgeError("embed 响应形状异常:期望 %d 条,实际 %s" % (len(batch), type(data).__name__))
|
|
100
|
+
# 按 index 排序后取 embedding,确保与输入顺序对齐
|
|
101
|
+
items_sorted = sorted(items, key=lambda x: int(x.get("index", 0)))
|
|
102
|
+
vectors.extend([item["embedding"] for item in items_sorted])
|
|
103
|
+
matrix = np.asarray(vectors, dtype=np.float32)
|
|
104
|
+
if matrix.ndim != 2 or matrix.shape[1] != EMBED_DIM:
|
|
105
|
+
raise BgeError("embed 向量维度异常:%s" % (matrix.shape,))
|
|
106
|
+
return _l2_normalize(matrix)
|
|
107
|
+
|
|
108
|
+
|
|
109
|
+
def rerank(query: str, texts: List[str], timeout: int = DEFAULT_TIMEOUT) -> List[Tuple[int, float]]:
|
|
110
|
+
"""对候选文本按与 query 的相关性打分,返回 [(原下标, score)] 降序。"""
|
|
111
|
+
if not texts:
|
|
112
|
+
return []
|
|
113
|
+
key = _get_api_key()
|
|
114
|
+
|
|
115
|
+
def _rerank_batch(batch: List[str], offset: int) -> List[Tuple[int, float]]:
|
|
116
|
+
payload = {
|
|
117
|
+
"model": RERANK_MODEL,
|
|
118
|
+
"query": query,
|
|
119
|
+
"documents": batch,
|
|
120
|
+
"return_documents": False,
|
|
121
|
+
}
|
|
122
|
+
data = _post_json(RERANK_URL, key, payload, timeout=timeout)
|
|
123
|
+
# Result 外层:{"status":0,"result":[{"index":i,"score":s}, ...]}
|
|
124
|
+
if isinstance(data, dict) and data.get("status", 0) != 0:
|
|
125
|
+
raise BgeError("rerank 接口返回错误:%s" % data.get("message", data))
|
|
126
|
+
result = data.get("result") if isinstance(data, dict) else None
|
|
127
|
+
if not isinstance(result, list):
|
|
128
|
+
raise BgeError("rerank 响应形状异常:%s" % type(data).__name__)
|
|
129
|
+
return [(int(item["index"]) + offset, float(item["score"])) for item in result]
|
|
130
|
+
|
|
131
|
+
# 接口单次最多 256 条;超过则分批打分后合并再统一排序。
|
|
132
|
+
if len(texts) <= RERANK_BATCH:
|
|
133
|
+
results = _rerank_batch(texts, 0)
|
|
134
|
+
else:
|
|
135
|
+
results: List[Tuple[int, float]] = []
|
|
136
|
+
for start in range(0, len(texts), RERANK_BATCH):
|
|
137
|
+
results.extend(_rerank_batch(texts[start:start + RERANK_BATCH], start))
|
|
138
|
+
results.sort(key=lambda x: x[1], reverse=True)
|
|
139
|
+
return results
|
|
@@ -0,0 +1,60 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""BM25 稀疏检索:混合 char-bigram 分词 + 内存倒排打分(search 侧)。
|
|
3
|
+
|
|
4
|
+
分词策略与 knowledge-search-admin skill 一致。
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import math
|
|
8
|
+
import re
|
|
9
|
+
from typing import List
|
|
10
|
+
|
|
11
|
+
_ASCII_RUN = re.compile(r"[A-Za-z0-9_-]+")
|
|
12
|
+
_CJK_RUN = re.compile(r"[㐀-鿿]+")
|
|
13
|
+
|
|
14
|
+
|
|
15
|
+
def tokenize(text: str) -> List[str]:
|
|
16
|
+
if not text:
|
|
17
|
+
return []
|
|
18
|
+
tokens: List[str] = []
|
|
19
|
+
for m in _ASCII_RUN.finditer(text):
|
|
20
|
+
tokens.append(m.group(0).lower())
|
|
21
|
+
for m in _CJK_RUN.finditer(text):
|
|
22
|
+
seg = m.group(0)
|
|
23
|
+
for i in range(len(seg) - 1):
|
|
24
|
+
tokens.append(seg[i:i + 2])
|
|
25
|
+
return tokens
|
|
26
|
+
|
|
27
|
+
|
|
28
|
+
def bm25_scores(query_tokens: List[str], doc_tokens: List[List[str]],
|
|
29
|
+
k1: float = 1.5, b: float = 0.75) -> List[float]:
|
|
30
|
+
"""对每篇 doc 按 BM25 打分,返回与 doc_tokens 同序的分值列表。"""
|
|
31
|
+
N = len(doc_tokens)
|
|
32
|
+
if N == 0:
|
|
33
|
+
return []
|
|
34
|
+
df: dict = {}
|
|
35
|
+
doc_len: List[int] = []
|
|
36
|
+
for toks in doc_tokens:
|
|
37
|
+
seen = set(toks)
|
|
38
|
+
for t in seen:
|
|
39
|
+
df[t] = df.get(t, 0) + 1
|
|
40
|
+
doc_len.append(len(toks))
|
|
41
|
+
avgdl = (sum(doc_len) / N) or 1.0
|
|
42
|
+
if not query_tokens:
|
|
43
|
+
return [0.0] * N
|
|
44
|
+
scores = [0.0] * N
|
|
45
|
+
for i, toks in enumerate(doc_tokens):
|
|
46
|
+
tf: dict = {}
|
|
47
|
+
for t in toks:
|
|
48
|
+
tf[t] = tf.get(t, 0) + 1
|
|
49
|
+
dl = doc_len[i] or 1
|
|
50
|
+
s = 0.0
|
|
51
|
+
for t in query_tokens:
|
|
52
|
+
d = df.get(t, 0)
|
|
53
|
+
if d == 0:
|
|
54
|
+
continue
|
|
55
|
+
idf = math.log(1 + (N - d + 0.5) / (d + 0.5))
|
|
56
|
+
f = tf.get(t, 0)
|
|
57
|
+
denom = f + k1 * (1 - b + b * dl / avgdl)
|
|
58
|
+
s += idf * (f * (k1 + 1)) / denom
|
|
59
|
+
scores[i] = s
|
|
60
|
+
return scores
|
|
@@ -0,0 +1,35 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""只读加载索引(chunks.json + vectors.npy + bm25.json)。
|
|
3
|
+
|
|
4
|
+
QA skill 仅检索不构建。bm25.json 缺失(老索引)时返回空 token 列表,向后兼容。
|
|
5
|
+
"""
|
|
6
|
+
|
|
7
|
+
import json
|
|
8
|
+
from pathlib import Path
|
|
9
|
+
from typing import List, Tuple
|
|
10
|
+
|
|
11
|
+
import numpy as np
|
|
12
|
+
|
|
13
|
+
from bge_client import EMBED_DIM
|
|
14
|
+
|
|
15
|
+
CHUNKS_FILE = "chunks.json"
|
|
16
|
+
VECTORS_FILE = "vectors.npy"
|
|
17
|
+
BM25_FILE = "bm25.json"
|
|
18
|
+
|
|
19
|
+
|
|
20
|
+
def load_index(index_dir: Path) -> Tuple[List[dict], np.ndarray, List[List[str]]]:
|
|
21
|
+
"""返回 (chunks, vectors, token_lists)。索引不存在则返回空。"""
|
|
22
|
+
chunks_path = index_dir / CHUNKS_FILE
|
|
23
|
+
vectors_path = index_dir / VECTORS_FILE
|
|
24
|
+
if not chunks_path.exists() or not vectors_path.exists():
|
|
25
|
+
return [], np.zeros((0, EMBED_DIM), dtype=np.float32), []
|
|
26
|
+
with chunks_path.open("r", encoding="utf-8") as f:
|
|
27
|
+
chunks = json.load(f)
|
|
28
|
+
vectors = np.load(str(vectors_path))
|
|
29
|
+
bm25_path = index_dir / BM25_FILE
|
|
30
|
+
if bm25_path.exists():
|
|
31
|
+
with bm25_path.open("r", encoding="utf-8") as f:
|
|
32
|
+
token_lists = json.load(f)
|
|
33
|
+
else:
|
|
34
|
+
token_lists = [[] for _ in chunks]
|
|
35
|
+
return chunks, vectors, token_lists
|
|
@@ -0,0 +1,141 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""知识库模糊检索(QA Agent 用)。
|
|
3
|
+
|
|
4
|
+
链路:load 索引 → dense 召回 + BM25 召回 → RRF 融合 → rerank 精排
|
|
5
|
+
→ sigmoid 归一化 → 类型加权 → 文件级分散 → 阈值闸口 → 输出。
|
|
6
|
+
只读索引,不写任何文件。
|
|
7
|
+
|
|
8
|
+
输出(stdout):
|
|
9
|
+
STATUS=succeeded\nRESULTS=[...]
|
|
10
|
+
STATUS=failed\nERROR_MESSAGE=...
|
|
11
|
+
"""
|
|
12
|
+
|
|
13
|
+
import argparse
|
|
14
|
+
import json
|
|
15
|
+
import sys
|
|
16
|
+
from pathlib import Path
|
|
17
|
+
|
|
18
|
+
import numpy as np
|
|
19
|
+
|
|
20
|
+
from bge_client import embed, rerank
|
|
21
|
+
from bm25 import bm25_scores, tokenize
|
|
22
|
+
from index_loader import load_index
|
|
23
|
+
from ranker import (
|
|
24
|
+
FAQ_BOOST_DEFAULT, OVERVIEW_BOOST_DEFAULT, THRESHOLD_DEFAULT,
|
|
25
|
+
apply_type_boost, diversify_by_file, rrf_fuse, sigmoid_norm,
|
|
26
|
+
)
|
|
27
|
+
|
|
28
|
+
DEFAULT_RECALL = 30
|
|
29
|
+
DEFAULT_RECALL_TOP = 50
|
|
30
|
+
DEFAULT_TOP = 10
|
|
31
|
+
DEFAULT_MAX_PER_FILE = 2
|
|
32
|
+
|
|
33
|
+
|
|
34
|
+
def _emit(status: str, **fields) -> None:
|
|
35
|
+
lines = ["STATUS=%s" % status]
|
|
36
|
+
for k, v in fields.items():
|
|
37
|
+
if isinstance(v, (dict, list)):
|
|
38
|
+
lines.append("%s=%s" % (k, json.dumps(v, ensure_ascii=False)))
|
|
39
|
+
else:
|
|
40
|
+
lines.append("%s=%s" % (k, v))
|
|
41
|
+
sys.stdout.write("\n".join(lines) + "\n")
|
|
42
|
+
sys.stdout.flush()
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def main() -> int:
|
|
46
|
+
parser = argparse.ArgumentParser(description="知识库模糊检索:混合召回 + rerank 精排")
|
|
47
|
+
parser.add_argument("--query", required=True, help="查询文本")
|
|
48
|
+
parser.add_argument("--index-dir", required=True, help="索引目录(只读)")
|
|
49
|
+
parser.add_argument("--recall", type=int, default=DEFAULT_RECALL, help="每路召回数(默认 30)")
|
|
50
|
+
parser.add_argument("--top", type=int, default=DEFAULT_TOP, help="最终返回数(默认 10)")
|
|
51
|
+
parser.add_argument("--threshold", type=float, default=THRESHOLD_DEFAULT,
|
|
52
|
+
help="相关度闸口,作用于归一化 norm(默认 0.5;0=不过滤)")
|
|
53
|
+
parser.add_argument("--max-per-file", type=int, default=DEFAULT_MAX_PER_FILE,
|
|
54
|
+
help="同一文件最多进榜条数(默认 2)")
|
|
55
|
+
parser.add_argument("--faq-boost", type=float, default=FAQ_BOOST_DEFAULT,
|
|
56
|
+
help="FAQ 加权幅度(默认 0.15)")
|
|
57
|
+
parser.add_argument("--no-bm25", action="store_true", help="关闭 BM25,纯 dense")
|
|
58
|
+
parser.add_argument("--no-faq-boost", action="store_true", help="关闭类型加权")
|
|
59
|
+
parser.add_argument("--no-snippet", action="store_true", help="不输出片段文本")
|
|
60
|
+
parser.add_argument("--timeout", type=int, default=60, help="单次 API 超时秒数")
|
|
61
|
+
args = parser.parse_args()
|
|
62
|
+
|
|
63
|
+
index_dir = Path(args.index_dir).resolve()
|
|
64
|
+
try:
|
|
65
|
+
chunks, vectors, token_lists = load_index(index_dir)
|
|
66
|
+
except Exception as e: # noqa: BLE001
|
|
67
|
+
_emit("failed", ERROR_MESSAGE="加载索引失败:%s" % e)
|
|
68
|
+
return 1
|
|
69
|
+
if not chunks:
|
|
70
|
+
_emit("succeeded", RESULTS=[])
|
|
71
|
+
return 0
|
|
72
|
+
|
|
73
|
+
try:
|
|
74
|
+
q_vec = embed([args.query], timeout=args.timeout)[0]
|
|
75
|
+
except Exception as e: # noqa: BLE001
|
|
76
|
+
_emit("failed", ERROR_MESSAGE="编码 query 失败:%s" % e)
|
|
77
|
+
return 1
|
|
78
|
+
|
|
79
|
+
n = len(chunks)
|
|
80
|
+
recall_n = min(args.recall, n)
|
|
81
|
+
|
|
82
|
+
sims = vectors @ q_vec
|
|
83
|
+
dense_rank = np.argsort(-sims)[:recall_n].tolist()
|
|
84
|
+
|
|
85
|
+
if args.no_bm25 or not token_lists or not any(token_lists):
|
|
86
|
+
bm25_rank = []
|
|
87
|
+
else:
|
|
88
|
+
q_tokens = tokenize(args.query)
|
|
89
|
+
if q_tokens:
|
|
90
|
+
scores = bm25_scores(q_tokens, token_lists)
|
|
91
|
+
bm25_rank = [i for i in np.argsort(-np.asarray(scores))[:recall_n].tolist()]
|
|
92
|
+
else:
|
|
93
|
+
bm25_rank = []
|
|
94
|
+
|
|
95
|
+
if bm25_rank:
|
|
96
|
+
cand_idx = rrf_fuse(dense_rank, bm25_rank, top=DEFAULT_RECALL_TOP)
|
|
97
|
+
else:
|
|
98
|
+
cand_idx = dense_rank[:DEFAULT_RECALL_TOP]
|
|
99
|
+
|
|
100
|
+
cand_texts = [chunks[i]["text"] for i in cand_idx]
|
|
101
|
+
try:
|
|
102
|
+
ranked = rerank(args.query, cand_texts, timeout=args.timeout)
|
|
103
|
+
raw_scores = {local: s for local, s in ranked}
|
|
104
|
+
except Exception:
|
|
105
|
+
raw_scores = {j: float(sims[cand_idx[j]]) for j in range(len(cand_idx))}
|
|
106
|
+
|
|
107
|
+
items = []
|
|
108
|
+
for local, idx in enumerate(cand_idx):
|
|
109
|
+
norm = sigmoid_norm(float(raw_scores.get(local, 0.0)))
|
|
110
|
+
items.append({
|
|
111
|
+
"idx": idx,
|
|
112
|
+
"norm": norm,
|
|
113
|
+
"type": chunks[idx].get("type", "normal"),
|
|
114
|
+
"file": chunks[idx]["file"],
|
|
115
|
+
})
|
|
116
|
+
|
|
117
|
+
faq_boost = 0.0 if args.no_faq_boost else args.faq_boost
|
|
118
|
+
over_boost = 0.0 if args.no_faq_boost else OVERVIEW_BOOST_DEFAULT
|
|
119
|
+
boosted = apply_type_boost(items, faq_boost=faq_boost, overview_boost=over_boost)
|
|
120
|
+
final = diversify_by_file(boosted, max_per_file=args.max_per_file,
|
|
121
|
+
top=args.top, threshold=args.threshold)
|
|
122
|
+
|
|
123
|
+
results = []
|
|
124
|
+
for it in final:
|
|
125
|
+
c = chunks[it["idx"]]
|
|
126
|
+
item = {
|
|
127
|
+
"file": c["file"],
|
|
128
|
+
"heading_path": c["heading_path"],
|
|
129
|
+
"loc": {"start_line": c["start_line"], "end_line": c["end_line"]},
|
|
130
|
+
"score": round(float(it["norm"]), 5),
|
|
131
|
+
}
|
|
132
|
+
if not args.no_snippet:
|
|
133
|
+
item["snippet"] = c["text"][:200]
|
|
134
|
+
results.append(item)
|
|
135
|
+
|
|
136
|
+
_emit("succeeded", RESULTS=results)
|
|
137
|
+
return 0
|
|
138
|
+
|
|
139
|
+
|
|
140
|
+
if __name__ == "__main__":
|
|
141
|
+
sys.exit(main())
|
|
@@ -0,0 +1,75 @@
|
|
|
1
|
+
#!/usr/bin/env python3
|
|
2
|
+
"""检索排序管线:RRF 融合 + sigmoid 归一化 + 类型加权 + 文件级分散 + 阈值。
|
|
3
|
+
|
|
4
|
+
与 knowledge-search skill 中的 ranker.py 保持一致。
|
|
5
|
+
|
|
6
|
+
排序流程:
|
|
7
|
+
dense 召回 Top-N + bm25 召回 Top-N
|
|
8
|
+
→ rrf_fuse 融合取 Top-F(候选下标,按 final 排序前的池子)
|
|
9
|
+
→ 对候选调 rerank 得原始分 → sigmoid_norm 归一化为 norm (0~1)
|
|
10
|
+
→ apply_type_boost:final = norm + boost(type),按 final 降序
|
|
11
|
+
→ diversify_by_file:按 final 降序、同 file 上限、norm 阈值过滤、取 Top-K
|
|
12
|
+
输出 score 字段 = norm(不含 boost),FAQ 优先体现在排序而非 score 值。
|
|
13
|
+
"""
|
|
14
|
+
|
|
15
|
+
import math
|
|
16
|
+
from typing import List, Sequence
|
|
17
|
+
|
|
18
|
+
FAQ_BOOST_DEFAULT = 0.15
|
|
19
|
+
OVERVIEW_BOOST_DEFAULT = 0.05
|
|
20
|
+
THRESHOLD_DEFAULT = 0.6
|
|
21
|
+
RRF_K = 60
|
|
22
|
+
|
|
23
|
+
|
|
24
|
+
def rrf_fuse(dense_rank: Sequence[int], bm25_rank: Sequence[int],
|
|
25
|
+
k: int = RRF_K, top: int = 50) -> List[int]:
|
|
26
|
+
"""两路排名 RRF 融合,返回融合后按分降序的下标列表(最多 top 个,去重)。"""
|
|
27
|
+
scores: dict = {}
|
|
28
|
+
for rank, idx in enumerate(dense_rank):
|
|
29
|
+
scores[idx] = scores.get(idx, 0.0) + 1.0 / (k + rank + 1)
|
|
30
|
+
for rank, idx in enumerate(bm25_rank):
|
|
31
|
+
scores[idx] = scores.get(idx, 0.0) + 1.0 / (k + rank + 1)
|
|
32
|
+
ordered = sorted(scores.items(), key=lambda kv: kv[1], reverse=True)
|
|
33
|
+
return [idx for idx, _ in ordered[:top]]
|
|
34
|
+
|
|
35
|
+
|
|
36
|
+
def sigmoid_norm(raw: float) -> float:
|
|
37
|
+
"""rerank 原始分(无界)→ (0,1) 归一化。"""
|
|
38
|
+
if raw >= 0:
|
|
39
|
+
z = math.exp(-raw)
|
|
40
|
+
return 1.0 / (1.0 + z)
|
|
41
|
+
z = math.exp(raw)
|
|
42
|
+
return z / (1.0 + z)
|
|
43
|
+
|
|
44
|
+
|
|
45
|
+
def apply_type_boost(items: List[dict], faq_boost: float, overview_boost: float) -> List[dict]:
|
|
46
|
+
"""给每个 item 加 final = norm + boost(type),按 final 降序返回。"""
|
|
47
|
+
out: List[dict] = []
|
|
48
|
+
for it in items:
|
|
49
|
+
norm = it["norm"]
|
|
50
|
+
t = it.get("type", "normal")
|
|
51
|
+
boost = faq_boost if t == "faq" else (overview_boost if t == "overview" else 0.0)
|
|
52
|
+
enriched = dict(it)
|
|
53
|
+
enriched["final"] = norm + boost
|
|
54
|
+
out.append(enriched)
|
|
55
|
+
# round 排序键以抵消浮点误差(如 0.55+0.05 略大于 0.6+0.0),同分时稳定保序
|
|
56
|
+
out.sort(key=lambda x: round(x["final"], 9), reverse=True)
|
|
57
|
+
return out
|
|
58
|
+
|
|
59
|
+
|
|
60
|
+
def diversify_by_file(items: List[dict], max_per_file: int, top: int,
|
|
61
|
+
threshold: float = THRESHOLD_DEFAULT) -> List[dict]:
|
|
62
|
+
"""按 final 降序遍历,同 file 计数上限 max_per_file;norm < threshold 丢弃;取 top。"""
|
|
63
|
+
per_file: dict = {}
|
|
64
|
+
out: List[dict] = []
|
|
65
|
+
for it in items: # items 已按 final 降序
|
|
66
|
+
f = it.get("file", "")
|
|
67
|
+
if per_file.get(f, 0) >= max_per_file:
|
|
68
|
+
continue
|
|
69
|
+
if it.get("norm", 1.0) < threshold:
|
|
70
|
+
continue
|
|
71
|
+
per_file[f] = per_file.get(f, 0) + 1
|
|
72
|
+
out.append(it)
|
|
73
|
+
if len(out) >= top:
|
|
74
|
+
break
|
|
75
|
+
return out
|
|
@@ -0,0 +1,31 @@
|
|
|
1
|
+
{
|
|
2
|
+
"name": "knowledge-search-admin",
|
|
3
|
+
"version": "1.3.0",
|
|
4
|
+
"types": ["store"],
|
|
5
|
+
"displayName": "知识库向量检索管理",
|
|
6
|
+
"description": "构建与检索知识库向量索引(BGE-M3 embedding + bge-reranker)。当管理员要求对 knowledge 目录建库、重建向量索引、增量更新、或做模糊检索自测时使用。",
|
|
7
|
+
"changelog": [
|
|
8
|
+
{
|
|
9
|
+
"version": "1.3.0",
|
|
10
|
+
"date": "2026-07-29",
|
|
11
|
+
"changes": ["新增 BM25 稀疏召回与 RRF 融合(补 dense 词项盲区);FAQ 文档按 ## Q: pair 切块并打 type=faq;chunk 类型加权(FAQ +0.15、INDEX 概览 +0.05),rerank 分 sigmoid 归一化、阈值作用于 norm;文件级分散解决广度查询覆盖不全;索引新增 bm25.json"]
|
|
12
|
+
},
|
|
13
|
+
{
|
|
14
|
+
"version": "1.2.0",
|
|
15
|
+
"date": "2026-07-29",
|
|
16
|
+
"changes": ["BGE 接口切换到 Sophnet 平台(/projects/easyllms/embeddings、/projects/rerank),请求/响应字段按新文档适配;ApiKey 改由 sophnet_tools.get_api_key() 运行时获取,不再硬编码;embedding 单次 batch 降至 8"]
|
|
17
|
+
},
|
|
18
|
+
{
|
|
19
|
+
"version": "1.1.0",
|
|
20
|
+
"date": "2026-07-27",
|
|
21
|
+
"changes": ["切块策略改为固定窗口滑窗(窗口 800 / 重叠 150),替换原 Markdown 标题切块;保留 heading_path 与起止行号定位能力"]
|
|
22
|
+
},
|
|
23
|
+
{
|
|
24
|
+
"version": "1.0.0",
|
|
25
|
+
"date": "2026-07-27",
|
|
26
|
+
"changes": ["初次提交"]
|
|
27
|
+
}
|
|
28
|
+
],
|
|
29
|
+
"createdAt": "2026-07-27",
|
|
30
|
+
"updatedAt": "2026-07-29"
|
|
31
|
+
}
|
|
@@ -0,0 +1,79 @@
|
|
|
1
|
+
---
|
|
2
|
+
name: knowledge-search-admin
|
|
3
|
+
description: 构建与检索知识库向量索引(BGE-M3 embedding + bge-reranker)。当管理员要求对 knowledge 目录建库、重建向量索引、增量更新、或做模糊检索自测时使用。
|
|
4
|
+
---
|
|
5
|
+
|
|
6
|
+
# 知识库向量检索管理
|
|
7
|
+
|
|
8
|
+
对知识库文档目录构建本地向量索引 + BM25 token 索引(按文档类型分流切块:FAQ 按 ## Q: pair 切、其余固定窗口滑窗 → BGE-M3 embedding + char-bigram 分词 → 落本地),并提供混合召回(dense+BM25 RRF)+ reranker 精排 + 类型加权 + 文件分散的检索自测能力。供知识库管理主 Agent 使用;问答 Agent 的在线检索请用配套的 `knowledge-search` skill。
|
|
9
|
+
|
|
10
|
+
## 用法
|
|
11
|
+
|
|
12
|
+
```bash
|
|
13
|
+
# 1. 构建/增量更新索引(扫描 knowledge/ 下所有 .md/.txt)
|
|
14
|
+
uv run {baseDir}/scripts/ksearch.py build \
|
|
15
|
+
--doc-dir knowledge/ \
|
|
16
|
+
--index-dir knowledge-index/
|
|
17
|
+
|
|
18
|
+
# 2. 全量重建(忽略已有索引,全部重新 embedding)
|
|
19
|
+
uv run {baseDir}/scripts/ksearch.py build \
|
|
20
|
+
--doc-dir knowledge/ --index-dir knowledge-index/ --rebuild
|
|
21
|
+
|
|
22
|
+
# 3. 检索自测:dense+BM25 召回 → RRF 融合 → rerank 精排 → 类型加权 → Top-10
|
|
23
|
+
uv run {baseDir}/scripts/ksearch.py search \
|
|
24
|
+
--query "怎么办理退款" \
|
|
25
|
+
--index-dir knowledge-index/ \
|
|
26
|
+
--recall 30 --top 10
|
|
27
|
+
```
|
|
28
|
+
|
|
29
|
+
## 参数
|
|
30
|
+
|
|
31
|
+
build:
|
|
32
|
+
- `--doc-dir`:知识库文档目录(必填)
|
|
33
|
+
- `--index-dir`:索引产物目录(必填)
|
|
34
|
+
- `--rebuild`:全量重建
|
|
35
|
+
- `--window`:切块窗口字符数(默认 800)
|
|
36
|
+
- `--overlap`:相邻 chunk 重叠字符数(默认 150,步长 = 窗口 - 重叠 = 650)
|
|
37
|
+
- `--timeout`:单次 API 超时秒数(默认 60)
|
|
38
|
+
|
|
39
|
+
search:
|
|
40
|
+
- `--query`:查询文本(必填)
|
|
41
|
+
- `--index-dir`:索引目录(必填)
|
|
42
|
+
- `--recall`:向量召回数(默认 30)
|
|
43
|
+
- `--top`:最终返回条数(默认 10)
|
|
44
|
+
- `--threshold`:相关度闸口,作用于归一化 norm(默认 0.6;0=不过滤)
|
|
45
|
+
- `--max-per-file`:同一文件最多进榜条数(默认 2)
|
|
46
|
+
- `--faq-boost`:FAQ 加权幅度(默认 0.15)
|
|
47
|
+
- `--no-bm25`:关闭 BM25,纯 dense
|
|
48
|
+
- `--no-faq-boost`:关闭类型加权
|
|
49
|
+
- `--no-snippet`:不输出片段文本
|
|
50
|
+
- `--timeout`:单次 API 超时秒数(默认 60)
|
|
51
|
+
|
|
52
|
+
## 输出格式
|
|
53
|
+
|
|
54
|
+
stdout 输出结构化键值对,便于 LLM 解析:
|
|
55
|
+
|
|
56
|
+
```
|
|
57
|
+
# build 成功
|
|
58
|
+
STATUS=succeeded
|
|
59
|
+
STATS={"doc_total":12,"reused_files":10,"reembedded_files":2,"chunk_total":86}
|
|
60
|
+
INDEX_DIR=/abs/path/to/knowledge-index
|
|
61
|
+
|
|
62
|
+
# search 成功
|
|
63
|
+
STATUS=succeeded
|
|
64
|
+
RESULTS=[{"file":"knowledge/售前流程.md","heading_path":"售前流程.md > 退货 > 退款时效","loc":{"start_line":42,"end_line":58},"score":0.86,"snippet":"..."}]
|
|
65
|
+
|
|
66
|
+
# 失败
|
|
67
|
+
STATUS=failed
|
|
68
|
+
ERROR_MESSAGE=...
|
|
69
|
+
```
|
|
70
|
+
|
|
71
|
+
## 注意事项
|
|
72
|
+
|
|
73
|
+
- 索引产物为 `chunks.json`(chunk 元信息,含 type)+ `vectors.npy`(N×1024 float32,L2 归一化)+ `bm25.json`(每行 chunk 的 token 列表)。三个文件行序对齐,不可单独改动。
|
|
74
|
+
- 增量更新按文件 md5 hash 判断变更;未变更文件复用已有向量,仅对变更/新增文件重 embedding。
|
|
75
|
+
- 切块策略:固定窗口滑窗,默认窗口 800 字符、重叠 150 字符(步长 650);每个 chunk 记录起始位置所属的标题路径与起止行号,便于结果定位。末尾不足 100 字符的碎块并入上一个 chunk。FAQ 文档(文件名 FAQ.md 或路径含 `/faq/`,且含 `## Q:` pair)按 pair 切块、type=faq;INDEX.md → type=overview;其余固定窗口滑窗、type=normal。
|
|
76
|
+
- FAQ 导入格式约定 `## Q: 问题 / A: 答案 / 出处: ...`,管理员导入时统一规范化。
|
|
77
|
+
- 平台 ApiKey 运行时由 `sophnet_tools.get_api_key()` 获取,不硬编码;接口走 Sophnet 平台 `https://www.sophnet.com/api/open-apis`。
|
|
78
|
+
- embedding 单次输入 ≤ 8 条、rerank 单次 ≤ 256 条,脚本已自动分批。
|
|
79
|
+
- 仅处理 `.md` / `.txt`;图片、附件等不纳入向量索引。
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
# knowledge-search-admin
|