@a9i5k4/dsh-auto-memory 2.2.6 → 2.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +20 -10
- package/README.zh-CN.md +22 -10
- package/docs/CONTINUITY-FLOW.md +222 -0
- package/docs/HANDBOOK.md +354 -0
- package/docs/INTEGRATION-ANALYSIS.md +348 -0
- package/docs/M-CM7-HANDOFF-LAYERED-RETRIEVAL.md +311 -0
- package/docs/M8-MEMORY-HUB.md +1 -1
- package/docs/PROMPT-PACK-LAYERED-RECALL.md +474 -0
- package/docs/PROMPT-SET-STRICT.md +389 -0
- package/docs/RELEASE-GO-NOGO.md +82 -0
- package/docs/ROADMAP.md +162 -0
- package/docs/STATUS-BOARD.md +147 -0
- package/docs/USER-GUIDE.en.md +382 -0
- package/docs/USER-GUIDE.zh-CN.md +382 -289
- package/docs/prompts/EXEC-ORDER.md +77 -0
- package/docs/prompts/FEEDING-SCRIPT.md +174 -0
- package/docs/prompts/FEEDING-SEQUENCE.md +61 -0
- package/docs/prompts/FIX-AGENT-M8-2b.md +119 -0
- package/docs/prompts/FIX-AGENT-P11.md +97 -0
- package/docs/prompts/FIX-AGENT-P12-FULL-REGRESSION.md +135 -0
- package/docs/prompts/FIX-AGENT-P12.md +113 -0
- package/docs/prompts/FIX-AGENT-P13-PYTHON-RANK.md +100 -0
- package/docs/prompts/FIX-AGENT-P8.md +120 -0
- package/docs/prompts/FIX-AGENT-P9.md +110 -0
- package/docs/prompts/FIX-AGENT-P9a.md +94 -0
- package/docs/prompts/FIX-AGENT-P9d.md +114 -0
- package/docs/prompts/FIX-AGENT-TEMPORAL-ARM.md +148 -0
- package/docs/prompts/LIVE-VERIFY-ZCODE.md +105 -0
- package/docs/prompts/M8-1-fact-metadata.md +45 -0
- package/docs/prompts/M8-2-ADJUDICATION.md +98 -0
- package/docs/prompts/M8-2-importance-wiring.md +42 -0
- package/docs/prompts/M8-2b-evidence-pipeline.md +52 -0
- package/docs/prompts/M8-3-enable-verify.md +49 -0
- package/docs/prompts/M8-R-REPORT.md +156 -0
- package/docs/prompts/M8-R-research.md +67 -0
- package/docs/prompts/P1-l0-index.md +30 -0
- package/docs/prompts/P10-importance-calibration.md +45 -0
- package/docs/prompts/P11-silent-catch-observability.md +43 -0
- package/docs/prompts/P2-semantic-recall.md +30 -0
- package/docs/prompts/P3-fusion.md +28 -0
- package/docs/prompts/P4-l0-response.md +28 -0
- package/docs/prompts/P5-handoff-anchor.md +28 -0
- package/docs/prompts/P6-ledger-weight.md +27 -0
- package/docs/prompts/P7-write-fix.md +26 -0
- package/docs/prompts/P8-rrf-wiring.md +47 -0
- package/docs/prompts/P9-REVIEW-DECISION.md +95 -0
- package/docs/prompts/P9-evidence-write-coverage.md +113 -0
- package/docs/prompts/README.md +105 -0
- package/docs/prompts/ZCODE-DROPIN.md +229 -0
- package/docs/prompts/_COMMON.md +88 -0
- package/lib/client.js +36 -2
- package/lib/context-host.js +77 -2
- package/lib/evidence-agg.js +81 -0
- package/lib/fact-store.js +32 -0
- package/lib/handoff-anchor.js +114 -0
- package/lib/index.js +402 -43
- package/lib/l0-extract.js +149 -0
- package/lib/l0-index.js +239 -0
- package/lib/m7-wire.js +4 -3
- package/lib/memory-importance.js +70 -0
- package/lib/python-setup.js +16 -4
- package/lib/recall-fusion.js +99 -0
- package/lib/shadow-retrieval.js +2 -2
- package/lib/storage-manage.js +17 -0
- package/lib/subagent-gc.js +8 -1
- package/lib/temporal-parse.js +159 -0
- package/package.json +1 -1
- package/python/worker_semantic_v1.py +28 -1
|
@@ -0,0 +1,149 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* L0 抽取纯核心(l0_extract_v1)—— 分层语义唤回的地基。
|
|
3
|
+
*
|
|
4
|
+
* 2026-09-08 建立。目的:为每条记忆生成廉价摘要(L0),使检索可先在小空间
|
|
5
|
+
* 收敛候选,再按 id 下钻原文,从而把 token 开销与索引构建成本降约一个数量级
|
|
6
|
+
* (实测:条目平均 814 字符 → L0 约 118 字符,压缩比 6.9:1)。
|
|
7
|
+
*
|
|
8
|
+
* 组成:
|
|
9
|
+
* 1) parseMemoryItemsPre —— 按 `<!-- memory:mem_<32hex> -->` 锚点切分记忆条目
|
|
10
|
+
* 2) extractL0Pre —— 单条记忆的 L0 抽取(三级 fallback,见下)
|
|
11
|
+
* 3) buildL0IndexPre —— 文件级 L0 索引(确定性排序)
|
|
12
|
+
*
|
|
13
|
+
* L0 抽取优先级(零 LLM,纯解析):
|
|
14
|
+
* ① `## 主题块标题` —— 已由写入侧概括,质量最好(去掉 `(HH:MM)` 后缀)
|
|
15
|
+
* ② 首个 `- ` 条目首句 —— 去掉 `- HH:MM ` 时间戳前缀后取首句
|
|
16
|
+
* ③ 截断兜底 —— 前 maxChars 字符
|
|
17
|
+
* 若首句过短(< minChars)则继续并接后续句子,直到达标或触顶。
|
|
18
|
+
*
|
|
19
|
+
* 边界:纯函数、零 IO、零依赖;对同输入逐字段确定;非法输入 fail closed(返回
|
|
20
|
+
* 空串/空数组),不抛异常。所有新增文本 UTF-8 无 BOM。
|
|
21
|
+
*/
|
|
22
|
+
|
|
23
|
+
export const L0_EXTRACT_VERSION = 'l0_extract_v1'
|
|
24
|
+
|
|
25
|
+
/** 锚点:`<!-- memory:mem_<32hex> -->`(允许空白浮动)。 */
|
|
26
|
+
const MEM_ANCHOR_RE = /<!--\s*memory:(mem_[0-9a-f]{32})\s*-->/g
|
|
27
|
+
|
|
28
|
+
/** 行首时间戳:`12:01 ` / `14:5x ` 等(日志条目惯例)。 */
|
|
29
|
+
const LEAD_TIME_RE = /^\s*\d{1,2}:\d{2}[a-z]?\s+/
|
|
30
|
+
|
|
31
|
+
/** 句末/句读分隔符(含中文)。 */
|
|
32
|
+
const SENTENCE_SPLIT_RE = /[。;;!!??\n]/
|
|
33
|
+
|
|
34
|
+
/** 主题块标题后缀:`(12:02)` 或 `(12:02)`。 */
|
|
35
|
+
const HEADING_SUFFIX_RE = /\s*[((]\s*\d{1,2}:\d{2}\s*[))]\s*$/
|
|
36
|
+
|
|
37
|
+
export const L0_DEFAULTS = Object.freeze({
|
|
38
|
+
maxChars: 160,
|
|
39
|
+
minChars: 15,
|
|
40
|
+
hardChars: 480,
|
|
41
|
+
})
|
|
42
|
+
|
|
43
|
+
const clean = (s) => String(s == null ? '' : s).replace(/\u0000/g, '').trim()
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* 按锚点切分记忆条目。
|
|
47
|
+
*
|
|
48
|
+
* 约定:锚点标记**其后**的内容(实测文件结构为 `<!-- A -->内容A<!-- B -->内容B`),
|
|
49
|
+
* 因此第 i 个内容对应第 i 个锚点。锚点之前的游离内容(若有)归入 `preamble`。
|
|
50
|
+
*
|
|
51
|
+
* @param {string} text 文件内容
|
|
52
|
+
* @returns {{items: Array<{id:string, body:string}>, preamble: string, anchors: number}}
|
|
53
|
+
*/
|
|
54
|
+
export function parseMemoryItemsPre(text) {
|
|
55
|
+
const src = typeof text === 'string' ? text : ''
|
|
56
|
+
const items = []
|
|
57
|
+
if (!src) return { items, preamble: '', anchors: 0 }
|
|
58
|
+
|
|
59
|
+
MEM_ANCHOR_RE.lastIndex = 0
|
|
60
|
+
const marks = []
|
|
61
|
+
let m
|
|
62
|
+
while ((m = MEM_ANCHOR_RE.exec(src)) !== null) {
|
|
63
|
+
marks.push({ id: m[1], start: m.index, end: m.index + m[0].length })
|
|
64
|
+
if (m.index === MEM_ANCHOR_RE.lastIndex) MEM_ANCHOR_RE.lastIndex++
|
|
65
|
+
}
|
|
66
|
+
if (!marks.length) return { items, preamble: clean(src), anchors: 0 }
|
|
67
|
+
|
|
68
|
+
for (let i = 0; i < marks.length; i++) {
|
|
69
|
+
const from = marks[i].end
|
|
70
|
+
const to = i + 1 < marks.length ? marks[i + 1].start : src.length
|
|
71
|
+
const body = clean(src.slice(from, to))
|
|
72
|
+
if (body) items.push({ id: marks[i].id, body })
|
|
73
|
+
}
|
|
74
|
+
return { items, preamble: clean(src.slice(0, marks[0].start)), anchors: marks.length }
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
/**
|
|
78
|
+
* 抽取单条记忆的 L0。纯函数,永不抛异常。
|
|
79
|
+
*
|
|
80
|
+
* @param {string} body 条目正文
|
|
81
|
+
* @param {{maxChars?:number, minChars?:number}} opts
|
|
82
|
+
* @returns {{l0: string, source: 'heading'|'firstSentence'|'truncate'|'empty'}}
|
|
83
|
+
*/
|
|
84
|
+
export function extractL0Pre(body, opts = {}) {
|
|
85
|
+
const maxChars = Math.max(16, Number(opts.maxChars) || L0_DEFAULTS.maxChars)
|
|
86
|
+
const minChars = Math.max(0, Number(opts.minChars) || L0_DEFAULTS.minChars)
|
|
87
|
+
const text = clean(body)
|
|
88
|
+
if (!text) return { l0: '', source: 'empty' }
|
|
89
|
+
|
|
90
|
+
const lines = text.split(/\r?\n/)
|
|
91
|
+
|
|
92
|
+
// ① 主题块标题
|
|
93
|
+
for (const line of lines) {
|
|
94
|
+
const h = /^\s{0,3}#{1,6}\s+(.+?)\s*$/.exec(line)
|
|
95
|
+
if (h) {
|
|
96
|
+
const title = clean(h[1]).replace(HEADING_SUFFIX_RE, '')
|
|
97
|
+
if (title) return { l0: cut(title, maxChars), source: 'heading' }
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
|
|
101
|
+
// ② 首个 `- ` 条目:先取首句,过短再并接(并接时剥列表标记与时间戳)
|
|
102
|
+
for (const line of lines) {
|
|
103
|
+
const b = /^\s*[-*+]\s+(.+?)\s*$/.exec(line)
|
|
104
|
+
if (!b) continue
|
|
105
|
+
let s = clean(b[1]).replace(LEAD_TIME_RE, '')
|
|
106
|
+
if (!s) continue
|
|
107
|
+
const first = clean(s.split(SENTENCE_SPLIT_RE)[0])
|
|
108
|
+
s = growToMin(first || s, text, minChars, maxChars)
|
|
109
|
+
return { l0: cut(s, maxChars), source: 'firstSentence' }
|
|
110
|
+
}
|
|
111
|
+
|
|
112
|
+
// ③ 兜底:正文截断
|
|
113
|
+
const flat = clean(text.replace(/\s+/g, ' '))
|
|
114
|
+
return { l0: cut(flat, maxChars), source: 'truncate' }
|
|
115
|
+
}
|
|
116
|
+
|
|
117
|
+
/** 构建文件级 L0 索引(按 id 升序,确定性)。 */
|
|
118
|
+
export function buildL0IndexPre(text, opts = {}) {
|
|
119
|
+
const { items } = parseMemoryItemsPre(text)
|
|
120
|
+
const out = items.map((it) => {
|
|
121
|
+
const r = extractL0Pre(it.body, opts)
|
|
122
|
+
return { id: it.id, l0: r.l0, source: r.source, chars: r.l0.length, bodyChars: it.body.length }
|
|
123
|
+
})
|
|
124
|
+
out.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0))
|
|
125
|
+
return out
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
// ---------- 内部工具 ----------
|
|
129
|
+
|
|
130
|
+
function cut(s, n) {
|
|
131
|
+
if (s.length <= n) return s
|
|
132
|
+
return s.slice(0, Math.max(1, n - 1)) + '…'
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
/** 首句过短时,并接后续句子直到 minChars 或 maxChars。每句先剥列表标记与时间戳前缀。 */
|
|
136
|
+
function growToMin(first, full, minChars, maxChars) {
|
|
137
|
+
if (first.length >= minChars) return first
|
|
138
|
+
const flat = clean(full.replace(/\s+/g, ' '))
|
|
139
|
+
if (!flat || flat.length <= first.length) return first
|
|
140
|
+
const parts = flat.split(SENTENCE_SPLIT_RE)
|
|
141
|
+
.map((x) => clean(String(x).replace(/^\s*[-*+]\s+/, '').replace(LEAD_TIME_RE, '')))
|
|
142
|
+
.filter(Boolean)
|
|
143
|
+
let acc = ''
|
|
144
|
+
for (const p of parts) {
|
|
145
|
+
acc = acc ? acc + '。' + p : p
|
|
146
|
+
if (acc.length >= minChars || acc.length >= maxChars) break
|
|
147
|
+
}
|
|
148
|
+
return acc || first
|
|
149
|
+
}
|
package/lib/l0-index.js
ADDED
|
@@ -0,0 +1,239 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* L0 向量索引(l0_index_v1)—— 为 T1 产出的 L0 建立向量索引,供语义检索使用。
|
|
3
|
+
*
|
|
4
|
+
* 2026-09-09 建立(P1)。参照 OpenViking「Vector Index 只存 URI+向量+元数据,不含文件内容」:
|
|
5
|
+
* 每条目仅 {id, vector, l0, source, l0Hash, updatedAt},**绝不存记忆原文**。
|
|
6
|
+
*
|
|
7
|
+
* 组成(工厂 createL0IndexPre,IO 与 embedding 全注入,模块层零 fs/零模型):
|
|
8
|
+
* 1) buildFull —— 全量建索引:buildL0IndexPre(text) × embedPassages(l0s),一次性写盘
|
|
9
|
+
* 2) update —— 增量:逐条比 l0Hash,新增/重算/跳过/移除,仅重算变化条
|
|
10
|
+
* 3) remove —— 显式按 ids 移除失效条目
|
|
11
|
+
* 4) load —— fail-soft 读取:文件缺失/schema 不符/条目非法 → {ok:false, entries:[]},绝不抛
|
|
12
|
+
* 5) status —— 索引概况(count/version/updatedAt)
|
|
13
|
+
*
|
|
14
|
+
* 身份与版本约定(沿用仓库惯例):
|
|
15
|
+
* - l0Hash = sha256(l0),增量重算的唯一判据(l0 不变 → 跳过该条)
|
|
16
|
+
* - l0IndexVersion = 'l0idx_' + first32hex(sha256(canonical sorted [id,l0Hash] tuples))
|
|
17
|
+
* (前缀 l0idx_ 有意区别于 corpus 的 idx_ memoryIndexVersion,避免两套版本语义混淆)
|
|
18
|
+
* - 向量维度由 embedder 决定(真实 C2 引擎为 384 维已归一化),本模块不硬编码维度
|
|
19
|
+
* - 向量以纯 number 数组存盘(JSON 安全);embedder 返回 Float32Array 时经 Array.from 转换
|
|
20
|
+
*
|
|
21
|
+
* 边界:纯函数 + IO 注入、零 npm 依赖;所有 API 在边界处 fail-soft 返回 {ok:false,error},
|
|
22
|
+
* 不向调用方抛异常(不阻塞任何调用方);非法输入返回空结果。UTF-8 无 BOM。
|
|
23
|
+
*/
|
|
24
|
+
import { createHash } from 'node:crypto'
|
|
25
|
+
import { buildL0IndexPre } from './l0-extract.js'
|
|
26
|
+
|
|
27
|
+
export const L0_INDEX_VERSION = 'l0_index_v1'
|
|
28
|
+
export const L0_INDEX_SCHEMA_VERSION = 1
|
|
29
|
+
|
|
30
|
+
const sha256Hex = (s) => createHash('sha256').update(String(s == null ? '' : s), 'utf8').digest('hex')
|
|
31
|
+
const HEX64_RE = /^[0-9a-f]{64}$/
|
|
32
|
+
|
|
33
|
+
/** 规范化向量:Float32Array/number[] → 纯 number 数组;非法输入返回 null。 */
|
|
34
|
+
function normalizeVector(vec) {
|
|
35
|
+
if (!(vec && (Array.isArray(vec) || vec instanceof Float32Array))) return null
|
|
36
|
+
const out = []
|
|
37
|
+
for (let i = 0; i < vec.length; i++) {
|
|
38
|
+
const n = Number(vec[i])
|
|
39
|
+
if (!Number.isFinite(n)) return null
|
|
40
|
+
out.push(n)
|
|
41
|
+
}
|
|
42
|
+
return out.length ? out : null
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* l0IndexVersion:canonical sorted [id,l0Hash] tuples → 'l0idx_' + first32hex(sha256)。
|
|
47
|
+
* 同内容同版本(确定性);任一条目 l0 变化或条目增删 → 版本变化。
|
|
48
|
+
*/
|
|
49
|
+
export function computeL0IndexVersionPre(entries) {
|
|
50
|
+
const canon = (Array.isArray(entries) ? entries : [])
|
|
51
|
+
.map((e) => [String(e && e.id) || '', String(e && e.l0Hash) || ''])
|
|
52
|
+
.sort((a, b) => (a[0] < b[0] ? -1 : a[0] > b[0] ? 1 : 0))
|
|
53
|
+
return 'l0idx_' + sha256Hex(JSON.stringify(canon)).slice(0, 32)
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
/**
|
|
57
|
+
* 工厂:创建 L0 索引操作器。
|
|
58
|
+
* @param {object} opts
|
|
59
|
+
* @param {{readJson(path:any):any, writeJson(path:any, obj:any):void, exists?(path:any):boolean}} opts.io
|
|
60
|
+
* 磁盘 IO 全注入;readJson 对缺失文件应返回 null 或抛错(两种都被模块层容错)。
|
|
61
|
+
* @param {{embedPassages(texts:string[]):Promise<Float32Array[]|number[][]>}} opts.embedder
|
|
62
|
+
* embedding 注入(真实 C2 引擎或测试假 embedder);前缀由引擎内部负责,调用方传裸文本。
|
|
63
|
+
* @param {()=>number} [opts.now] 时间注入(默认 Date.now;测试确定性用)。
|
|
64
|
+
*/
|
|
65
|
+
export function createL0IndexPre(opts = {}) {
|
|
66
|
+
const io = opts.io
|
|
67
|
+
const embedder = opts.embedder
|
|
68
|
+
const nowMs = () => (typeof opts.now === 'function' ? Number(opts.now()) || 0 : Date.now())
|
|
69
|
+
if (!io || typeof io.readJson !== 'function' || typeof io.writeJson !== 'function') {
|
|
70
|
+
// fail closed:工厂级配置错误直接抛(调用方组装错误,不属于运行期 fail-soft 范畴)
|
|
71
|
+
throw new Error('l0-index: io.readJson/io.writeJson required')
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
/** fail-soft 读取+整文件校验。任何异常/不符 → {ok:false, reason, entries:[]}。 */
|
|
75
|
+
function load({ path } = {}) {
|
|
76
|
+
try {
|
|
77
|
+
const obj = io.readJson(path)
|
|
78
|
+
if (!obj || typeof obj !== 'object') return { ok: false, reason: 'missing', entries: [] }
|
|
79
|
+
if (obj.schemaVersion !== L0_INDEX_SCHEMA_VERSION) return { ok: false, reason: 'schema', entries: [] }
|
|
80
|
+
if (typeof obj.l0IndexVersion !== 'string' || !obj.l0IndexVersion.startsWith('l0idx_')) {
|
|
81
|
+
return { ok: false, reason: 'version', entries: [] }
|
|
82
|
+
}
|
|
83
|
+
if (!Array.isArray(obj.entries)) return { ok: false, reason: 'entries', entries: [] }
|
|
84
|
+
const out = []
|
|
85
|
+
for (const e of obj.entries) {
|
|
86
|
+
if (!e || typeof e !== 'object') return { ok: false, reason: 'entry', entries: [] }
|
|
87
|
+
if (typeof e.id !== 'string' || !e.id) return { ok: false, reason: 'entry.id', entries: [] }
|
|
88
|
+
if (typeof e.l0 !== 'string') return { ok: false, reason: 'entry.l0', entries: [] }
|
|
89
|
+
if (typeof e.l0Hash !== 'string' || !HEX64_RE.test(e.l0Hash)) return { ok: false, reason: 'entry.l0Hash', entries: [] }
|
|
90
|
+
if (typeof e.source !== 'string') return { ok: false, reason: 'entry.source', entries: [] }
|
|
91
|
+
if (!Number.isFinite(Number(e.updatedAt))) return { ok: false, reason: 'entry.updatedAt', entries: [] }
|
|
92
|
+
const vec = normalizeVector(e.vector)
|
|
93
|
+
if (!vec) return { ok: false, reason: 'entry.vector', entries: [] }
|
|
94
|
+
out.push({ id: e.id, vector: vec, l0: e.l0, source: e.source, l0Hash: e.l0Hash, updatedAt: Number(e.updatedAt) })
|
|
95
|
+
}
|
|
96
|
+
return { ok: true, entries: out, l0IndexVersion: obj.l0IndexVersion, updatedAt: Number(obj.updatedAt) || 0 }
|
|
97
|
+
} catch (e) {
|
|
98
|
+
return { ok: false, reason: 'read-error', error: String(e && e.message ? e.message : e), entries: [] }
|
|
99
|
+
}
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
/** 由 buildL0IndexPre 的条目 + 批量 embedding 组装索引文件对象(不写盘)。 */
|
|
103
|
+
async function assemble(items, prevById) {
|
|
104
|
+
const texts = items.map((it) => it.l0)
|
|
105
|
+
const vecs = await embedder.embedPassages(texts)
|
|
106
|
+
if (!Array.isArray(vecs) || vecs.length !== items.length) {
|
|
107
|
+
throw new Error('l0-index: embedder returned ' + (Array.isArray(vecs) ? vecs.length : 'non-array') + ' vectors for ' + items.length + ' passages')
|
|
108
|
+
}
|
|
109
|
+
const ts = nowMs()
|
|
110
|
+
const entries = items.map((it, i) => {
|
|
111
|
+
const vector = normalizeVector(vecs[i])
|
|
112
|
+
if (!vector) throw new Error('l0-index: embedder produced invalid vector at index ' + i)
|
|
113
|
+
const l0Hash = sha256Hex(it.l0)
|
|
114
|
+
const prev = prevById ? prevById.get(it.id) : null
|
|
115
|
+
// 复用语义:prev 的 l0Hash 一致才整条复用(保留原 updatedAt);否则按新条处理
|
|
116
|
+
const reused = prev && prev.l0Hash === l0Hash
|
|
117
|
+
return {
|
|
118
|
+
id: it.id,
|
|
119
|
+
vector: reused ? prev.vector : vector,
|
|
120
|
+
l0: it.l0,
|
|
121
|
+
source: it.source,
|
|
122
|
+
l0Hash,
|
|
123
|
+
updatedAt: reused ? prev.updatedAt : ts,
|
|
124
|
+
}
|
|
125
|
+
})
|
|
126
|
+
entries.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0))
|
|
127
|
+
return {
|
|
128
|
+
schemaVersion: L0_INDEX_SCHEMA_VERSION,
|
|
129
|
+
l0IndexVersion: computeL0IndexVersionPre(entries),
|
|
130
|
+
updatedAt: ts,
|
|
131
|
+
entries,
|
|
132
|
+
}
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
async function writeIndex(path, fileObj) {
|
|
136
|
+
io.writeJson(path, fileObj)
|
|
137
|
+
return fileObj
|
|
138
|
+
}
|
|
139
|
+
|
|
140
|
+
return {
|
|
141
|
+
version: L0_INDEX_VERSION,
|
|
142
|
+
|
|
143
|
+
/** 全量建索引(整文件重建,所有条目重算)。返回 {ok, count, added, recomputed, removed, skipped, l0IndexVersion} 或 {ok:false, error}。 */
|
|
144
|
+
async buildFull({ path, text, maxChars, minChars } = {}) {
|
|
145
|
+
try {
|
|
146
|
+
if (!embedder || typeof embedder.embedPassages !== 'function') throw new Error('embedder.embedPassages required')
|
|
147
|
+
const items = buildL0IndexPre(typeof text === 'string' ? text : '', { maxChars, minChars })
|
|
148
|
+
const fileObj = await assemble(items, null)
|
|
149
|
+
await writeIndex(path, fileObj)
|
|
150
|
+
return {
|
|
151
|
+
ok: true, count: fileObj.entries.length, added: fileObj.entries.length,
|
|
152
|
+
recomputed: fileObj.entries.length, removed: 0, skipped: 0,
|
|
153
|
+
l0IndexVersion: fileObj.l0IndexVersion,
|
|
154
|
+
}
|
|
155
|
+
} catch (e) {
|
|
156
|
+
return { ok: false, error: String(e && e.message ? e.message : e) }
|
|
157
|
+
}
|
|
158
|
+
},
|
|
159
|
+
|
|
160
|
+
/**
|
|
161
|
+
* 增量更新:新 L0 与旧索引逐条比 l0Hash——
|
|
162
|
+
* 新增(旧无此 id)/变化(hash 不同)→ 仅这些条重算;不变 → 原样保留;消失(旧有新无)→ 移除。
|
|
163
|
+
* 旧索引损坏/非法 → fail-soft 退化为全量重建(recovered:true),不阻塞调用方。
|
|
164
|
+
*/
|
|
165
|
+
async update({ path, text, maxChars, minChars } = {}) {
|
|
166
|
+
try {
|
|
167
|
+
if (!embedder || typeof embedder.embedPassages !== 'function') throw new Error('embedder.embedPassages required')
|
|
168
|
+
const prev = load({ path })
|
|
169
|
+
const recovered = !prev.ok
|
|
170
|
+
const prevById = new Map()
|
|
171
|
+
if (prev.ok) for (const e of prev.entries) prevById.set(e.id, e)
|
|
172
|
+
const items = buildL0IndexPre(typeof text === 'string' ? text : '', { maxChars, minChars })
|
|
173
|
+
// 先做哈希分类,只为计数;实际组装仍走 assemble(其内部同样按 hash 决定复用)
|
|
174
|
+
const nextById = new Map()
|
|
175
|
+
let unchanged = 0
|
|
176
|
+
let changed = 0
|
|
177
|
+
for (const it of items) {
|
|
178
|
+
const h = sha256Hex(it.l0)
|
|
179
|
+
nextById.set(it.id, h)
|
|
180
|
+
const p = prevById.get(it.id)
|
|
181
|
+
if (p && p.l0Hash === h) unchanged++
|
|
182
|
+
else changed++
|
|
183
|
+
}
|
|
184
|
+
let removed = 0
|
|
185
|
+
for (const id of prevById.keys()) if (!nextById.has(id)) removed++
|
|
186
|
+
const fileObj = await assemble(items, prevById)
|
|
187
|
+
await writeIndex(path, fileObj)
|
|
188
|
+
return {
|
|
189
|
+
ok: true,
|
|
190
|
+
count: fileObj.entries.length,
|
|
191
|
+
added: items.filter((it) => !prevById.has(it.id)).length,
|
|
192
|
+
recomputed: changed,
|
|
193
|
+
removed,
|
|
194
|
+
skipped: unchanged,
|
|
195
|
+
recovered: recovered || undefined,
|
|
196
|
+
l0IndexVersion: fileObj.l0IndexVersion,
|
|
197
|
+
}
|
|
198
|
+
} catch (e) {
|
|
199
|
+
return { ok: false, error: String(e && e.message ? e.message : e) }
|
|
200
|
+
}
|
|
201
|
+
},
|
|
202
|
+
|
|
203
|
+
/** 显式移除失效条目。ids 中不存在的自动忽略。 */
|
|
204
|
+
async remove({ path, ids } = {}) {
|
|
205
|
+
try {
|
|
206
|
+
const prev = load({ path })
|
|
207
|
+
if (!prev.ok) return { ok: false, error: 'index not loadable: ' + (prev.reason || '?'), removed: 0 }
|
|
208
|
+
const drop = new Set((Array.isArray(ids) ? ids : []).map(String))
|
|
209
|
+
const kept = prev.entries.filter((e) => !drop.has(e.id))
|
|
210
|
+
const ts = nowMs()
|
|
211
|
+
const fileObj = {
|
|
212
|
+
schemaVersion: L0_INDEX_SCHEMA_VERSION,
|
|
213
|
+
l0IndexVersion: computeL0IndexVersionPre(kept),
|
|
214
|
+
updatedAt: ts,
|
|
215
|
+
entries: kept,
|
|
216
|
+
}
|
|
217
|
+
await writeIndex(path, fileObj)
|
|
218
|
+
return {
|
|
219
|
+
ok: true,
|
|
220
|
+
removed: prev.entries.length - kept.length,
|
|
221
|
+
count: kept.length,
|
|
222
|
+
l0IndexVersion: fileObj.l0IndexVersion,
|
|
223
|
+
}
|
|
224
|
+
} catch (e) {
|
|
225
|
+
return { ok: false, error: String(e && e.message ? e.message : e), removed: 0 }
|
|
226
|
+
}
|
|
227
|
+
},
|
|
228
|
+
|
|
229
|
+
/** fail-soft 读取(见 load)。 */
|
|
230
|
+
load,
|
|
231
|
+
|
|
232
|
+
/** 索引概况(不修改任何状态)。 */
|
|
233
|
+
status({ path } = {}) {
|
|
234
|
+
const r = load({ path })
|
|
235
|
+
if (!r.ok) return { ok: false, count: 0, path, reason: r.reason }
|
|
236
|
+
return { ok: true, count: r.entries.length, path, l0IndexVersion: r.l0IndexVersion, updatedAt: r.updatedAt }
|
|
237
|
+
},
|
|
238
|
+
}
|
|
239
|
+
}
|
package/lib/m7-wire.js
CHANGED
|
@@ -30,14 +30,14 @@ export const M7_TRANSPORT_BUDGET_V1 = Object.freeze({
|
|
|
30
30
|
breakerCooldownMs: 30000,
|
|
31
31
|
})
|
|
32
32
|
|
|
33
|
-
/** JS→Python frame 类型(§7.2;index_sync_* 属 M7-1)。 */
|
|
33
|
+
/** JS→Python frame 类型(§7.2;index_sync_* 属 M7-1;recall_rank 属 P13 recall C3 语义臂)。 */
|
|
34
34
|
export const JS_FRAME_TYPES_V1 = Object.freeze([
|
|
35
35
|
'health', 'context_push', 'index_sync_begin', 'index_sync_page', 'index_sync_commit',
|
|
36
|
-
'cancel', 'close_session',
|
|
36
|
+
'cancel', 'close_session', 'recall_rank',
|
|
37
37
|
])
|
|
38
38
|
/** Python→JS frame 类型。 */
|
|
39
39
|
export const PY_FRAME_TYPES_V1 = Object.freeze([
|
|
40
|
-
'health_result', 'context_ack', 'index_ack', 'activation_request', 'error',
|
|
40
|
+
'health_result', 'context_ack', 'index_ack', 'activation_request', 'error', 'recall_rank_result',
|
|
41
41
|
])
|
|
42
42
|
const ALL_FRAME_TYPES = new Set([...JS_FRAME_TYPES_V1, ...PY_FRAME_TYPES_V1])
|
|
43
43
|
/** 请求→响应 type 对应(cancel/close_session 刻意无响应帧)。 */
|
|
@@ -47,6 +47,7 @@ export const RESPONSE_TYPE_FOR_V1 = Object.freeze({
|
|
|
47
47
|
index_sync_begin: 'index_ack',
|
|
48
48
|
index_sync_page: 'index_ack',
|
|
49
49
|
index_sync_commit: 'index_ack',
|
|
50
|
+
recall_rank: 'recall_rank_result',
|
|
50
51
|
})
|
|
51
52
|
|
|
52
53
|
// ========== canonical JSON(与 python/worker_v1.py 逐字节一致) ==========
|
|
@@ -0,0 +1,70 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* memory-importance-pre —— evidence → importance 纯核心(M8-2, 2026-09-09; memory_importance_v1)。
|
|
3
|
+
*
|
|
4
|
+
* 背景(M8-2 任务书):六类证据 seen/read/cite/reuse/success/correction 已由 M5 写入
|
|
5
|
+
* (lib/context-bridge.js:45 枚举),消费侧聚合为 fact-store.js:341 evidenceFor(memoryId)
|
|
6
|
+
* → {memoryId, total, distinctSessions, seen, read, cite, reuse, success, correction},
|
|
7
|
+
* 但此前只用于 M-04 技能晋升(procedure-store.js promote),从未接入记忆检索排序。
|
|
8
|
+
*
|
|
9
|
+
* 口径纪律(禁止项):correctionRate **逐字复用** procedure-store.js:268-269 的口径——
|
|
10
|
+
* total = seen+read+cite+reuse+success+correction
|
|
11
|
+
* correctionRate = total > 0 ? correction / total : 0
|
|
12
|
+
* 不另立纠正口径。
|
|
13
|
+
*
|
|
14
|
+
* 公式(确定性,∈[0,1]):
|
|
15
|
+
* neutral(无任何证据) → importance = 0.5(中性,不置顶不垫底)
|
|
16
|
+
* pos = 0.5×min(1, distinctSessions/3) + 0.5×min(1, (success+reuse)/4) // 正向:跨会话多样性 + 成功/复用饱和
|
|
17
|
+
* importance = clamp01(0.5 + 0.3×pos − 0.5×correctionRate) // correction 负向,权重最大
|
|
18
|
+
* 值域:正向满格 → 0.8;correctionRate=1 → 0;cite/seen/read 仅稀释 correctionRate(与 promote 口径一致)。
|
|
19
|
+
*
|
|
20
|
+
* 边界:纯函数、零依赖、零 IO;非法输入 fail closed(按全零处理 → 中性);同输入逐字节确定。
|
|
21
|
+
* 本段(2026-09-09)只交付纯函数与测试,不接线——接线点检索结论:现役融合 fuseD6Pre 位于
|
|
22
|
+
* _jsDecide 的 fv2 决策 margin(非检索排序),P3 rankFusionRRFPre 未接线,shadow 管线 evidence
|
|
23
|
+
* 数据不可达;待 P3 RRF 实际接线时作为加权因子之一落点(importance 禁止作为唯一排序依据)。
|
|
24
|
+
*/
|
|
25
|
+
|
|
26
|
+
export const MEMORY_IMPORTANCE_VERSION = 'memory_importance_v1'
|
|
27
|
+
|
|
28
|
+
/** 中性值:无 evidence 记录时返回,保证不因此置顶或垫底。 */
|
|
29
|
+
export const IMPORTANCE_NEUTRAL_V1 = 0.5
|
|
30
|
+
|
|
31
|
+
/** 权重常数(冻结;调整即新版本)。 */
|
|
32
|
+
export const IMPORTANCE_WEIGHTS_V1 = Object.freeze({
|
|
33
|
+
diversityDivisor: 3, // distinctSessions 饱和点:≥3 个不同会话记满
|
|
34
|
+
successReuseDivisor: 4, // success+reuse 饱和点:合计 ≥4 次记满
|
|
35
|
+
posGain: 0.3, // 正向增益上限(0.5 + 0.3 = 0.8)
|
|
36
|
+
negGain: 0.5, // correctionRate 惩罚权重(负向,口径复用 promote)
|
|
37
|
+
})
|
|
38
|
+
|
|
39
|
+
const num0 = (v) => (typeof v === 'number' && Number.isFinite(v) && v > 0 ? v : 0)
|
|
40
|
+
const clamp01 = (x) => Math.min(1, Math.max(0, x))
|
|
41
|
+
|
|
42
|
+
/**
|
|
43
|
+
* evidence 聚合 → importance ∈ [0,1]。
|
|
44
|
+
* @param {{distinctSessions?:number, seen?:number, read?:number, cite?:number,
|
|
45
|
+
* reuse?:number, success?:number, correction?:number}} agg evidenceFor() 形状的聚合(缺字段按 0)。
|
|
46
|
+
* @returns {{importance:number, correctionRate:number, neutral:boolean, total:number}}
|
|
47
|
+
*/
|
|
48
|
+
export function computeImportancePre(agg) {
|
|
49
|
+
const a = agg && typeof agg === 'object' ? agg : {}
|
|
50
|
+
const seen = num0(a.seen)
|
|
51
|
+
const read = num0(a.read)
|
|
52
|
+
const cite = num0(a.cite)
|
|
53
|
+
const reuse = num0(a.reuse)
|
|
54
|
+
const success = num0(a.success)
|
|
55
|
+
const correction = num0(a.correction)
|
|
56
|
+
const distinctSessions = num0(a.distinctSessions)
|
|
57
|
+
const total = seen + read + cite + reuse + success + correction
|
|
58
|
+
// 无任何证据 → 中性(0.5):既不置顶也不垫底(验收 3)
|
|
59
|
+
if (total === 0 && distinctSessions === 0) {
|
|
60
|
+
return { importance: IMPORTANCE_NEUTRAL_V1, correctionRate: 0, neutral: true, total: 0 }
|
|
61
|
+
}
|
|
62
|
+
// correctionRate 口径与 procedure-store.js:268-269 逐字一致(禁止另立)
|
|
63
|
+
const correctionRate = total > 0 ? correction / total : 0
|
|
64
|
+
const w = IMPORTANCE_WEIGHTS_V1
|
|
65
|
+
const diversity = Math.min(1, distinctSessions / w.diversityDivisor)
|
|
66
|
+
const successReuse = Math.min(1, (success + reuse) / w.successReuseDivisor)
|
|
67
|
+
const pos = 0.5 * diversity + 0.5 * successReuse
|
|
68
|
+
const importance = clamp01(IMPORTANCE_NEUTRAL_V1 + w.posGain * pos - w.negGain * correctionRate)
|
|
69
|
+
return { importance, correctionRate, neutral: false, total }
|
|
70
|
+
}
|
package/lib/python-setup.js
CHANGED
|
@@ -23,9 +23,11 @@ import { readFile, writeFile, rm } from 'node:fs/promises'
|
|
|
23
23
|
|
|
24
24
|
const execFileP = promisify(execFile)
|
|
25
25
|
|
|
26
|
-
/** BGE-M3 int8 单文件(与 bench 夹具同源;hf-mirror 为主源,HF 官方为备源)。
|
|
26
|
+
/** BGE-M3 int8 单文件(与 bench 夹具同源;hf-mirror 为主源,HF 官方为备源)。
|
|
27
|
+
* 2026-09-09 修复(issue #27):repo 原为 'Xenova/bge-m3-int8'(仓库不存在→HF 恒 401),正确仓库名是 'Xenova/bge-m3';
|
|
28
|
+
* INT8 量化文件 = onnx/model_int8.onnx(568,456,694 字节,仓库清单已确认存在;fp16/q4/uint8 等非本档目标)。 */
|
|
27
29
|
const MODEL_SPEC = {
|
|
28
|
-
repo: 'Xenova/bge-m3
|
|
30
|
+
repo: 'Xenova/bge-m3',
|
|
29
31
|
file: 'onnx/model_int8.onnx',
|
|
30
32
|
bytes: 568456694,
|
|
31
33
|
sha256: '', // 见 MODEL_SHA256:远端 LFS 校验不可靠时以 size 下限+可执行性兜底;sha256 由首版发布冻结后填入
|
|
@@ -35,6 +37,10 @@ const MODEL_SPEC = {
|
|
|
35
37
|
],
|
|
36
38
|
}
|
|
37
39
|
|
|
40
|
+
/** issue #28:AutoTokenizer.from_pretrained(modelDir) 需要完整 tokenizer 套件,HF_HUB_OFFLINE=1 时 transformers
|
|
41
|
+
* 无法运行时补拉。5 文件均已确认存在于 Xenova/bge-m3 仓库根,与 model_int8.onnx 同目录下载。 */
|
|
42
|
+
const TOKENIZER_FILES = ['config.json', 'tokenizer.json', 'tokenizer_config.json', 'special_tokens_map.json', 'sentencepiece.bpe.model']
|
|
43
|
+
|
|
38
44
|
/** venv 内 pip 依赖:#19 实测(int8 档 encode_ids 全程 numpy+onnxruntime,tokenizer 走 Rust 快速路径)——基础集无 torch,venv 体积 ~400MB。GPU 推理开关追加 onnxruntime-gpu(CUDAExecutionProvider)。池化/精度契约冻结自 R@5 0.925 基线,fastembed 注册表无 BGE-M3 且池化契约不同,不可用。 */
|
|
39
45
|
const PIP_DEPS_CPU = ['transformers', 'onnxruntime']
|
|
40
46
|
const PIP_DEPS_GPU_EXTRA = ['onnxruntime-gpu']
|
|
@@ -101,7 +107,8 @@ export function createPythonSetupPre(opts = {}) {
|
|
|
101
107
|
// 既有成果快照
|
|
102
108
|
st.venvOk = existsSync(venvPython())
|
|
103
109
|
st.depsOk = st.venvOk ? (await probe(venvPython(), ['-c', 'import transformers, onnxruntime, torch; print("deps-ok")'])).ok : false
|
|
104
|
-
|
|
110
|
+
// modelReady 判定(2026-09-09):onnx + tokenizer 5 件全齐才算就绪,避免 UI 显示 ✓ 但 sidecar 起不来
|
|
111
|
+
st.modelReady = existsSync(modelPath()) && TOKENIZER_FILES.every((f) => existsSync(path.join(modelsDir(), f)))
|
|
105
112
|
st.pythons = out
|
|
106
113
|
st.phase = 'idle'
|
|
107
114
|
return snapshot()
|
|
@@ -188,8 +195,13 @@ export function createPythonSetupPre(opts = {}) {
|
|
|
188
195
|
st.phase = 'verifying'
|
|
189
196
|
const size = statSync(modelPath()).size
|
|
190
197
|
if (size < 100 * 1024 * 1024) throw new Error('下载不完整(' + size + ' bytes)')
|
|
198
|
+
// issue #28:补下 tokenizer 套件到同目录(缺任一文件即失败,走 catch → 下一镜像)
|
|
199
|
+
for (const tf of TOKENIZER_FILES) {
|
|
200
|
+
const tu = mirror.url(MODEL_SPEC.repo + '/resolve/main/' + tf)
|
|
201
|
+
await downloadWithResume(tu, path.join(modelsDir(), tf), () => {}, () => st.cancelled)
|
|
202
|
+
}
|
|
191
203
|
st.modelReady = true; st.phase = 'ready'
|
|
192
|
-
diagOf('python-setup: model ready at ' +
|
|
204
|
+
diagOf('python-setup: model+tokens ready at ' + modelsDir() + ' (model=' + size + ' bytes, mirror=' + mirror.id + ')')
|
|
193
205
|
return snapshot()
|
|
194
206
|
} catch (e) {
|
|
195
207
|
if (st.cancelled) { st.phase = 'idle'; return snapshot() }
|
|
@@ -0,0 +1,99 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* recall-fusion-pre —— rank-space 融合(P3, 2026-09-09; rrf_fusion_v1)。
|
|
3
|
+
*
|
|
4
|
+
* 背景:旧 fuseD6Pre(semantic-js.js)为 minmax 加权融合,存在三宗罪:
|
|
5
|
+
* ① 分数随候选集漂移(归一化域=当前候选集) ② 矮子里拔将军(零极差臂全员 0.5)
|
|
6
|
+
* ③ 候选 ≤1 时退化为常数 0.5(排序失效)。
|
|
7
|
+
* 本模块提供 **rank-space RRF** 并存实现(不替换 fuseD6Pre,由调用方按需选用):
|
|
8
|
+
* score = Σ_arms 1/(k + rank/divisor),k=60(Hindsight issue #3956 实测安全值)。
|
|
9
|
+
*
|
|
10
|
+
* 决策/排序解耦(验收 1):融合分数**只用于排序**;"是否注入"的决策必须使用
|
|
11
|
+
* 绝对分数(如 rank.scores 的稠密余弦)与校准阈值比较 —— rank-space 分数本身
|
|
12
|
+
* 仍是候选集内的相对量,不承担决策职责。
|
|
13
|
+
*
|
|
14
|
+
* 边界:纯函数、零依赖、零 IO;非法输入 fail closed 返回空数组;确定性
|
|
15
|
+
* (同输入同输出;同分按 memoryId 升序)。无 score-space 加权(禁止项);无父分数传播。
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
/** RRF 常数 k(Hindsight issue #3956 实测: k=60 时动态范围安全,加权会退化排序)。 */
|
|
19
|
+
export const FUSION_RRF_K_V1 = 60
|
|
20
|
+
|
|
21
|
+
/** rank 尺度常数:rank/divisor 把秩归到 (0,1] 量级(默认与 k 同值)。 */
|
|
22
|
+
export const FUSION_RRF_DIVISOR_V1 = 60
|
|
23
|
+
|
|
24
|
+
/** 版本标识。 */
|
|
25
|
+
export const RECALL_FUSION_VERSION = 'rrf_fusion_v1'
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* rank-space RRF 融合。
|
|
29
|
+
*
|
|
30
|
+
* @param {Array<{memoryId:string, dense?:number|null, lex?:number|null}>} pairs
|
|
31
|
+
* 与 fuseD6Pre 同形:两臂分数可缺失(null/undefined/非有限数视为该臂缺席)。
|
|
32
|
+
* @param {{k?:number, divisor?:number}} [opts] 可覆盖 k 与 divisor(默认均 60)。
|
|
33
|
+
* @returns {Array<{memoryId:string, fused:number, rrfDense:number, rrfLex:number,
|
|
34
|
+
* denseRaw:number|null, lexRaw:number|null, rankDense:number|null, rankLex:number|null}>}
|
|
35
|
+
* 按 fused 降序、平局 memoryId 升序(确定性)。原始分数逐条保留供审计。
|
|
36
|
+
*
|
|
37
|
+
* 性质:
|
|
38
|
+
* - 候选 <3(含单候选)不退化:每条 fused = 1/(k + rank/divisor),良定义非常数
|
|
39
|
+
* - 候选集增删只平移秩,不改既有条目的相对序(rank-space 关键性质,minmax 不具备)
|
|
40
|
+
* - 缺失臂贡献 0(不是 minmax 的 0.5)
|
|
41
|
+
*/
|
|
42
|
+
export function rankFusionRRFPre(pairs, opts = {}) {
|
|
43
|
+
const k = Number.isFinite(opts.k) && opts.k >= 0 ? opts.k : FUSION_RRF_K_V1
|
|
44
|
+
const divisor = Number.isFinite(opts.divisor) && opts.divisor > 0 ? opts.divisor : FUSION_RRF_DIVISOR_V1
|
|
45
|
+
const list = Array.isArray(pairs) ? pairs.filter((p) => p && typeof p.memoryId === 'string') : []
|
|
46
|
+
|
|
47
|
+
// 每臂独立排名:非空有限值按分数降序(平局 memoryId 升序)取秩(1 起);缺席者无秩。
|
|
48
|
+
const rankArm = (key) => {
|
|
49
|
+
const entries = []
|
|
50
|
+
for (const p of list) {
|
|
51
|
+
const v = p[key]
|
|
52
|
+
if (typeof v === 'number' && Number.isFinite(v)) entries.push({ memoryId: p.memoryId, v })
|
|
53
|
+
}
|
|
54
|
+
entries.sort((a, b) => (b.v !== a.v ? b.v - a.v : (a.memoryId < b.memoryId ? -1 : a.memoryId > b.memoryId ? 1 : 0)))
|
|
55
|
+
const ranks = new Map()
|
|
56
|
+
entries.forEach((e, i) => ranks.set(e.memoryId, i + 1))
|
|
57
|
+
return ranks
|
|
58
|
+
}
|
|
59
|
+
const denseRanks = rankArm('dense')
|
|
60
|
+
const lexRanks = rankArm('lex')
|
|
61
|
+
|
|
62
|
+
// 时间臂(P12 后新增,2026-09-09):可选第三臂。temp 为 1(命中时间范围)/0(未命中)/
|
|
63
|
+
// 缺失(无时间臂或非日期来源)。所有 pair 均缺 temp **或 temp 全相等**(全 0/全 1)→
|
|
64
|
+
// 不构建 temp 秩,行为与两臂版逐字节一致(fused 不加 0 项、输出对象不含新字段)——
|
|
65
|
+
// 全 0 = 查询有时间表达但无候选命中,必须零扰动。rank-space,禁止 score-space 加权(#3956)。
|
|
66
|
+
const temps = []
|
|
67
|
+
for (const p of list) { if (typeof p.temp === 'number' && Number.isFinite(p.temp)) temps.push(p.temp) }
|
|
68
|
+
const hasTemp = temps.length > 0 && temps.some((v) => v !== temps[0])
|
|
69
|
+
const tempRanks = hasTemp ? rankArm('temp') : null
|
|
70
|
+
|
|
71
|
+
return list
|
|
72
|
+
.map((p) => {
|
|
73
|
+
const rd = denseRanks.get(p.memoryId)
|
|
74
|
+
const rl = lexRanks.get(p.memoryId)
|
|
75
|
+
const rrfDense = rd === undefined ? 0 : 1 / (k + rd / divisor)
|
|
76
|
+
const rrfLex = rl === undefined ? 0 : 1 / (k + rl / divisor)
|
|
77
|
+
const base = {
|
|
78
|
+
memoryId: p.memoryId,
|
|
79
|
+
fused: rrfDense + rrfLex,
|
|
80
|
+
rrfDense,
|
|
81
|
+
rrfLex,
|
|
82
|
+
denseRaw: typeof p.dense === 'number' && Number.isFinite(p.dense) ? p.dense : null,
|
|
83
|
+
lexRaw: typeof p.lex === 'number' && Number.isFinite(p.lex) ? p.lex : null,
|
|
84
|
+
rankDense: rd === undefined ? null : rd,
|
|
85
|
+
rankLex: rl === undefined ? null : rl,
|
|
86
|
+
}
|
|
87
|
+
if (!hasTemp) return base
|
|
88
|
+
const rt = tempRanks.get(p.memoryId)
|
|
89
|
+
const rrfTemp = rt === undefined ? 0 : 1 / (k + rt / divisor)
|
|
90
|
+
return {
|
|
91
|
+
...base,
|
|
92
|
+
fused: rrfDense + rrfLex + rrfTemp,
|
|
93
|
+
rrfTemp,
|
|
94
|
+
tempRaw: typeof p.temp === 'number' && Number.isFinite(p.temp) ? p.temp : null,
|
|
95
|
+
rankTemp: rt === undefined ? null : rt,
|
|
96
|
+
}
|
|
97
|
+
})
|
|
98
|
+
.sort((x, y) => (y.fused !== x.fused ? y.fused - x.fused : (x.memoryId < y.memoryId ? -1 : x.memoryId > y.memoryId ? 1 : 0)))
|
|
99
|
+
}
|
package/lib/shadow-retrieval.js
CHANGED
|
@@ -242,8 +242,8 @@ export function buildQueryPlan(snapshot, opts = {}) {
|
|
|
242
242
|
seenTerms.set(t, { term: t, weight: wgt, origin })
|
|
243
243
|
}
|
|
244
244
|
}
|
|
245
|
-
//
|
|
246
|
-
const terms = [...seenTerms.values()].sort((a, b) => (a.term < b.term ? -1 : a.term > b.term ? 1 : 0))
|
|
245
|
+
// P12 排序:weight 降序优先(截断保高权重词),term 字典序升序作稳定 tiebreak;截断逻辑与 truncated 标记不变
|
|
246
|
+
const terms = [...seenTerms.values()].sort((a, b) => (b.weight - a.weight) || (a.term < b.term ? -1 : a.term > b.term ? 1 : 0))
|
|
247
247
|
let truncated = false
|
|
248
248
|
if (terms.length > SHADOW_LEXICAL_BUDGET_V1.queryTerms) {
|
|
249
249
|
terms.length = SHADOW_LEXICAL_BUDGET_V1.queryTerms
|