@a9i5k4/dsh-auto-memory 2.2.6 → 2.3.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (68) hide show
  1. package/README.md +20 -10
  2. package/README.zh-CN.md +22 -10
  3. package/docs/CONTINUITY-FLOW.md +222 -0
  4. package/docs/HANDBOOK.md +354 -0
  5. package/docs/INTEGRATION-ANALYSIS.md +348 -0
  6. package/docs/M-CM7-HANDOFF-LAYERED-RETRIEVAL.md +311 -0
  7. package/docs/M8-MEMORY-HUB.md +1 -1
  8. package/docs/PROMPT-PACK-LAYERED-RECALL.md +474 -0
  9. package/docs/PROMPT-SET-STRICT.md +389 -0
  10. package/docs/RELEASE-GO-NOGO.md +82 -0
  11. package/docs/ROADMAP.md +162 -0
  12. package/docs/STATUS-BOARD.md +147 -0
  13. package/docs/USER-GUIDE.en.md +382 -0
  14. package/docs/USER-GUIDE.zh-CN.md +382 -289
  15. package/docs/prompts/EXEC-ORDER.md +77 -0
  16. package/docs/prompts/FEEDING-SCRIPT.md +174 -0
  17. package/docs/prompts/FEEDING-SEQUENCE.md +61 -0
  18. package/docs/prompts/FIX-AGENT-M8-2b.md +119 -0
  19. package/docs/prompts/FIX-AGENT-P11.md +97 -0
  20. package/docs/prompts/FIX-AGENT-P12-FULL-REGRESSION.md +135 -0
  21. package/docs/prompts/FIX-AGENT-P12.md +113 -0
  22. package/docs/prompts/FIX-AGENT-P13-PYTHON-RANK.md +100 -0
  23. package/docs/prompts/FIX-AGENT-P8.md +120 -0
  24. package/docs/prompts/FIX-AGENT-P9.md +110 -0
  25. package/docs/prompts/FIX-AGENT-P9a.md +94 -0
  26. package/docs/prompts/FIX-AGENT-P9d.md +114 -0
  27. package/docs/prompts/FIX-AGENT-TEMPORAL-ARM.md +148 -0
  28. package/docs/prompts/LIVE-VERIFY-ZCODE.md +105 -0
  29. package/docs/prompts/M8-1-fact-metadata.md +45 -0
  30. package/docs/prompts/M8-2-ADJUDICATION.md +98 -0
  31. package/docs/prompts/M8-2-importance-wiring.md +42 -0
  32. package/docs/prompts/M8-2b-evidence-pipeline.md +52 -0
  33. package/docs/prompts/M8-3-enable-verify.md +49 -0
  34. package/docs/prompts/M8-R-REPORT.md +156 -0
  35. package/docs/prompts/M8-R-research.md +67 -0
  36. package/docs/prompts/P1-l0-index.md +30 -0
  37. package/docs/prompts/P10-importance-calibration.md +45 -0
  38. package/docs/prompts/P11-silent-catch-observability.md +43 -0
  39. package/docs/prompts/P2-semantic-recall.md +30 -0
  40. package/docs/prompts/P3-fusion.md +28 -0
  41. package/docs/prompts/P4-l0-response.md +28 -0
  42. package/docs/prompts/P5-handoff-anchor.md +28 -0
  43. package/docs/prompts/P6-ledger-weight.md +27 -0
  44. package/docs/prompts/P7-write-fix.md +26 -0
  45. package/docs/prompts/P8-rrf-wiring.md +47 -0
  46. package/docs/prompts/P9-REVIEW-DECISION.md +95 -0
  47. package/docs/prompts/P9-evidence-write-coverage.md +113 -0
  48. package/docs/prompts/README.md +105 -0
  49. package/docs/prompts/ZCODE-DROPIN.md +229 -0
  50. package/docs/prompts/_COMMON.md +88 -0
  51. package/lib/client.js +36 -2
  52. package/lib/context-host.js +77 -2
  53. package/lib/evidence-agg.js +81 -0
  54. package/lib/fact-store.js +32 -0
  55. package/lib/handoff-anchor.js +114 -0
  56. package/lib/index.js +402 -43
  57. package/lib/l0-extract.js +149 -0
  58. package/lib/l0-index.js +239 -0
  59. package/lib/m7-wire.js +4 -3
  60. package/lib/memory-importance.js +70 -0
  61. package/lib/python-setup.js +16 -4
  62. package/lib/recall-fusion.js +99 -0
  63. package/lib/shadow-retrieval.js +2 -2
  64. package/lib/storage-manage.js +17 -0
  65. package/lib/subagent-gc.js +8 -1
  66. package/lib/temporal-parse.js +159 -0
  67. package/package.json +1 -1
  68. package/python/worker_semantic_v1.py +28 -1
@@ -0,0 +1,229 @@
1
+ > **给使用者的说明**:**直接整段复制扔进 Zcode**(从「你是独立第三方验收 Agent」开始到文件末尾)。
2
+ > 本 prompt 已包含全部导航信息:面板在哪、设置项在哪个分区、控件长什么样、改了会怎样、后端怎么确认。
3
+ > ⚠️ 已写入 `--no-open` 硬性要求。
4
+
5
+ ---
6
+
7
+ 你是独立第三方验收 Agent。你有**电脑控制能力**(操作 UI、重启应用)和 **DSH 后台日志读取能力**。
8
+
9
+ **你的立场**:你不参与前面的开发,**不采信任何"已完成 / ✅ 通过"的结论**。前面汇报里的通过对你无效——每项都必须你亲自操作、亲自取证。
10
+
11
+ # 任务:验收 dsh-auto-memory 插件,给出发版 Go / No-Go 结论
12
+
13
+ ## 0. 你要知道的背景(先读,别跳过)
14
+
15
+ | 概念 | 含义 |
16
+ |---|---|
17
+ | **DSH** | DeepSeek Harness,宿主应用。CLI 名 `dsh`,`dsh web` 提供浏览器 UI |
18
+ | **本插件** | `dsh-auto-memory`,给 DSH 加长期记忆。工程根 `D:\dsh-auto-memory` |
19
+ | **主动联想** | 不等用户问,系统主动判断"该想起什么"并注入——本插件核心能力 |
20
+ | **水位** | 上下文占用比例 0-1。达 **0.75** 触发接续(官方压缩阈值是 0.80,必须留余量) |
21
+ | **接续 / handoff** | 上下文将满时把进度交接给新会话 |
22
+ | **L0** | 记忆摘要(约 93 字符)。检索默认只返回 L0,需原文时用 `expand="mem_xxx"` 展开 |
23
+ | **M8 / 记忆中枢** | 三层记忆:fact 事实 / episodic 经历 / procedure 技能 |
24
+ | **evidence** | 记忆被使用的记录,六类:`seen` `read` `cite` `reuse` `success` `correction` |
25
+ | **C1 / C2 / C3** | 检索三档:C1 词法(0GB 保底)/C2 内置语义(约 130MB 量化模型)/C3 Python BGE-M3(约 563MB,深度用户可选) |
26
+
27
+ **必读材料**(里面有完整术语表与端点清单):
28
+ 1. `D:\dsh-auto-memory\docs\HANDBOOK.md`
29
+ 2. `D:\dsh-auto-memory\docs\STATUS-BOARD.md`
30
+
31
+ ## 1. 启动(**硬性 `--no-open`**)
32
+
33
+ > ⚠️ **任何启动/打开 DSH 的命令都必须带 `--no-open`**,禁止自动弹浏览器打断用户桌面工作。
34
+ > 官方选项,help 原文:`--no-open do not open the Web UI in the default browser`
35
+
36
+ ```bash
37
+ dsh web --no-open # 推荐
38
+ dsh web --no-open --port 0 # 需要指定端口时(0 = 系统分配空闲端口)
39
+ ```
40
+
41
+ - 从**终端输出**读实际监听地址与端口(不要猜)
42
+ - 需要 UI 时**由你手动打开**该地址
43
+ - 确认启动日志**无** `SyntaxError` / `ReferenceError` / 模块加载失败
44
+ - 报错 → **立即停止回报**
45
+
46
+ ## 2. 界面导航图(照着找,别乱点)
47
+
48
+ ### 2.1 记忆面板(侧栏)
49
+
50
+ - **入口**:DSH 界面侧栏的记忆按钮(DOM:`data-dam-sidebar-btn`);面板本体 DOM:`data-dam-panel`
51
+ - **页签**(面板内):
52
+ - **记忆中枢**(DOM:`hub`)—— 技能 / 事实 / 经历 三栏
53
+ - **唤起回顾** —— 复核每次"是否该想起"的打分
54
+ - 日历、问候等相关页签
55
+ - **面板内按钮**:「一键接续」「一键反思」「一键更新」等
56
+
57
+ ### 2.2 设置页(8 个分区)
58
+
59
+ 设置页 DOM:`data-dam-settings`;左侧导航 `data-dam-settings-nav`;每个分区是一个 `<section id="dam-settings-<key>" data-dam-settings-group>`;配置项行 DOM:`data-dam-settings-row`。
60
+
61
+ | 分区 key | 界面标题(中文) | 主要管什么 |
62
+ |---|---|---|
63
+ | `semantic` | **自动记忆引擎** | 检索模式(自动 / 仅词法 / 内置语义 / 高级 Python)、思维链监听、记忆唤起档位 |
64
+ | `memoryHub` | **记忆中枢** | 三层记忆总开关、技能固化与晋升 |
65
+ | `injection` | **记忆窗口** | 周期记忆快照注入 |
66
+ | `automation` | **自动化** | 暂离问候、夜间/批量自动托管、每日反思、定时总结、记忆固化 |
67
+ | `context` | **上下文管理** | 水位相关、接续策略 |
68
+ | `storage` | **存储** | 存储位置与管理 |
69
+ | `appearance` | **外观** | 面板外观 |
70
+ | `maintenance` | **维护** | 维护/恢复默认(含「一键恢复默认」) |
71
+
72
+ **改设置的通用操作**:设置页 → 点左侧导航跳到分区 → 找到对应行 → 切换开关/改数值/选下拉 → 通常即时保存(改后按第 4 节方法验证是否真生效)。
73
+
74
+ ### 2.3 语义引擎检测面板
75
+
76
+ - DOM:`data-dam-detect-panel`
77
+ - 用途:显示 C1/C2/C3 检测结果、模型是否就绪、可就地下载安装
78
+
79
+ ## 3. 关键配置项(在哪个分区、改成什么、改了会怎样)
80
+
81
+ | 配置键 | 所在分区 | 控件 | 默认 | 改成某值后的预期 | 后端怎么确认 |
82
+ |---|---|---|---|---|---|
83
+ | `associativeMemoryEnabled` | 自动记忆引擎 | 开关 | false | 开 → 会话中可能出现主动联想注入 | 端点 `/config` 回写;对话中出现注入 |
84
+ | `activationEmitMode` | 自动记忆引擎 | 模式开关 | canary | 开=canary(显式回忆注入,推荐)/关=shadow(只记录不注入) | 日志出现 `emit explicit_lane` 或仅 shadow 记录 |
85
+ | `reasoningObserverEnabled` | 自动记忆引擎 | 开关 | false | 开 → 监听思维链作检索信号(更敏感,默认关) | `/config` 回写 |
86
+ | `semanticEngineMode` | 自动记忆引擎 | 下拉 | `'auto'` | 可选:自动 / 仅词法 / 内置语义(C2) / 高级 Python(C3) | `/semantic-status` 显示档位 |
87
+ | `procedurePromotionEnabled` | 记忆中枢 | 开关 | false | 开 → 重复流程固化为 checklist,跨会话验证后晋升(在记忆中枢页审批) | `procedures.json` 条目变化 |
88
+ | `memoryHubEnabled` | 记忆中枢 | 开关 | **true** | 关 → 三栏停止更新;开 → episodes/facts/procedures 落盘 | `GET /api/dsh-auto-memory-pre/memory-hub` 200 + 概览 |
89
+ | `autoConsolidate` | 自动化 | 开关 | — | 开 → 每轮结束把结论沉淀进日志与记忆店 | 日志文件新增条目 |
90
+ | `autoPopupEnabled` | 自动化 | 开关 | — | 开 → 暂离超 1 小时回归自动弹面板问候 | 观察到弹窗 |
91
+ | `unattendedAuto` | 自动化 | 开关 | false | 开 → 22:00-08:00 或检测到托管任务自动进入托管(零寒暄、上下文冻结) | `/config` 回写 |
92
+ | `reflectEnabled` | 自动化 | 开关 | — | 开 → 每天首次会话呈现前一天反思 | 出现反思内容 |
93
+ | `autoSummaryTimes` | 自动化 | 多选 | 空 | 勾 12:00 / 18:00 / 22:00 → 到点自动总结 | 到点出现总结 |
94
+ | `waterLevelThreshold` | 上下文管理 | 数值 | **0.75** | 调低→更早提示;**不得超过 0.80** | `/state` 或 `/config` |
95
+ | `autoContinueEnabled` | 上下文管理 | 开关 | true | 关 → 水位达标不自动接续 | 水位达标不触发 |
96
+ | `autoContinueThreshold` | 上下文管理 | 数值 | **0.75** | 同上,须与水位同步、低于 0.80 | `/config` |
97
+ | `handoffLedgerChars` | 上下文管理 | 数值 | 800 | 快照注入的账本字符预算(**与接续材料里的 8000 不同源**) | 注入文本长度变化 |
98
+
99
+ > 面板上还会显示「上下文水位」实时值。
100
+
101
+ ## 4. 后端状态来源与"前端改动是否真生效"的判定法
102
+
103
+ ### 4.1 端点(前缀 `/api/dsh-auto-memory-pre/`,共 39 个)
104
+
105
+ | 端点 | 用途 | 关键返回 |
106
+ |---|---|---|
107
+ | `/memory-hub` | M8 三层总览 | `policyVersion` / `stats` / `episodic` / `facts` / `procedures` |
108
+ | `/config` | 读写配置 | 各配置键当前值 |
109
+ | `/state` | 运行状态 | 水位、启用状态 |
110
+ | `/debug` | 调试视图 | 含 `shadowRetrieval` 等 |
111
+ | `/semantic-status` | 语义引擎档位 | C1/C2/C3、模型就绪状态 |
112
+ | `/auto-continue-state` `/auto-continue-decide` `/handoff-continue` | 接续状态与决策 | 水位、是否触发、材料层 |
113
+ | `/smart-recall` | 智能检索 | 命中列表 |
114
+ | `/storage-manage` `/workspaces` `/calendar` `/file` 等 | 辅助 | — |
115
+
116
+ ### 4.2 磁盘与日志
117
+
118
+ - 记忆根:`C:\Users\JH Z\.dsh\memory`
119
+ - `workspaces\<工作区>\*.md` —— 日志/笔记/白板/账本
120
+ - `hub-pre\episodes.json` `facts.json` `procedures.json` —— M8 三层(原子写)
121
+ - `evidence-pre\events\YYYY-MM-DD.jsonl` —— 证据事件(按日)
122
+ - 诊断日志:`C:\Users\JH Z\.dsh\memory\dsh-auto-memory-pre-diagnose.log`
123
+ - **证据事件单行结构**(实测):顶层 `kind` / `memoryId` / `recordedAt` / `anchorId` …;**时间戳在 `event.ts`(顶层无 `ts` / `createdAt`)**
124
+
125
+ ### 4.3 判定法(三步,至少两项一致变化才算生效)
126
+
127
+ 1. **操作前**:`GET /config`(或 `/state`)+ 记录相关文件条目数/mtime
128
+ 2. **在前端操作**(改开关、点按钮)
129
+ 3. **操作后再取**:对比配置值、端点返回、磁盘文件
130
+
131
+ **只有 UI 变了、后端/磁盘没变 = 未真正生效**(未保存或需重启)。
132
+
133
+ ## 5. 九项 live 验收(每项必须有证据)
134
+
135
+ > 证据三选一:**UI 截图 / 后台日志原文 / 数据快照**(events JSONL 行、端点响应、top-5 id 序列、条目计数)。
136
+ > **禁止**"应该是正常的""看起来没问题"——出现即判**未通过**。
137
+
138
+ ### L1 主动联想(本版改动直接影响面)
139
+ - **前置**:设置 → 自动记忆引擎 → 开 `associativeMemoryEnabled`;`activationEmitMode` = canary
140
+ - **步骤**:新开会话,聊一段与已有记忆主题相关的话(**不主动查询**)
141
+ - **通过**:记忆被**主动**注入;诊断日志有对应记录;保留词含高权重来源(user/trigger)
142
+ - **证据**:注入内容截图 + 日志片段
143
+
144
+ ### L2 语义检索
145
+ - **步骤**:调 `memory_recall_pre`,查一个**词法不重合但语义相关**的查询(记忆里写"npm 发布报 ENEEDAUTH",查"发布凭证问题")
146
+ - **通过**:能召回。**若失败 → 切 `legacy` 融合再测对比**
147
+ - **判定**:legacy 正常而 rrf 失败 = **P8 融合回归,停止**
148
+ - **证据**:两次返回的 top-5 `id` 序列
149
+
150
+ ### L3 L0 返回与展开
151
+ - **步骤**:同 L2;再用返回的 `id` 调 `expand="mem_xxx"`
152
+ - **通过**:默认返回 L0 列表(含 `id`/`score`/`match_reason`);expand 能取到原文且**不串条**
153
+
154
+ ### T1 时间臂(本版新功能)
155
+ - **步骤**:查含时间表达的查询(如"上周");再查一条**不含时间词**的同类查询做对照
156
+ - **通过**:含时间词 → 对应时间段条目排序**上升**;不含时间词 → 排序**与对照一致**(零行为变更)
157
+ - **证据**:两次 top-5 `id` 序列对比
158
+
159
+ ### L4 M8 记忆中枢
160
+ - **步骤**:打开记忆面板 → 记忆中枢页签;`GET /api/dsh-auto-memory-pre/memory-hub`
161
+ - **通过**:三栏有内容或正确空态;端点 200 且含 `policyVersion`/`stats`
162
+ - **异常场景**:关闭 `memoryHubEnabled` → 重载 → 对话若干轮 → 应无新落盘;再开启 → 应有新条目
163
+
164
+ ### L5 证据链
165
+ - **步骤**:对话若干轮后看当日 `evidence-pre\events\*.jsonl`;**再触发一次用户纠正**(对话里说"不对,你记错了"之类)
166
+ - **通过**:新增 `seen`;纠正后新增 `kind:"correction"`(且归因到最近被 cite/read 的记忆)
167
+ - **异常场景**:events 目录缺失/损坏 → 检索不报错(fail-soft)
168
+
169
+ ### L6 跨窗口接续
170
+ - **前置**:`waterLevelThreshold` / `autoContinueThreshold` = 0.75,`autoContinueEnabled` = true
171
+ - **步骤**:让上下文增长到 75%(或点面板「一键接续」)
172
+ - **通过**:新会话收到**分层**材料(白板/账本/近期线程);首条注入**不再要求"先 read 转写"**;对话能连续推进
173
+
174
+ ### L7 写入与持久化
175
+ - **步骤**:对话若干轮 → 看工作区日志 `*.md` 与 `hub-pre\*.json` 有新条目 → **再重启一次**(同样带 `--no-open`)
176
+ - **通过**:重启后数据 **restore 不丢**
177
+ - **异常场景**:磁盘不可写 → 不崩溃,日志有记录
178
+
179
+ ### P 性能
180
+ - 全程观察,无卡顿/内存异常
181
+
182
+ ### 附加:UI 完整性自查(顺手验,发现问题记录)
183
+ - 设置页左侧导航 8 个分区是否都在、标题是否正常显示
184
+ - ⚠️ **已知可疑点**:`semantic`(自动记忆引擎)分区的标题可能显示为空白/undefined(代码里取的是 `sectionLabels.secSemantic`,而该对象的键是 `semantic`)。**请实际看一眼并确认**,若标题空白即为 UI 缺陷,记录回报
185
+
186
+ ## 6. 输出结论
187
+
188
+ ```
189
+ 结论:【可以发版 / 暂缓发版】
190
+
191
+ L1 主动联想 : 通过 / 未通过 — <证据摘要>
192
+ L2 语义检索 : 通过 / 未通过 — <证据摘要>
193
+ L3 L0 返回展开 : 通过 / 未通过
194
+ T1 时间臂 : 通过 / 未通过 — <含时间词 vs 对照的 top-5 对比>
195
+ L4 M8 记忆中枢 : 通过 / 未通过
196
+ L5 证据链 : 通过 / 未通过
197
+ L6 跨窗口接续 : 通过 / 未通过
198
+ L7 写入 restore: 通过 / 未通过
199
+ P 性能 : 正常 / 异常
200
+ UI 完整性 : 正常 / <发现的问题>
201
+
202
+ 未通过项:<现象 + 复现步骤 + 你尝试过的处置>
203
+ ```
204
+
205
+ ### 暂缓判据(任一成立即暂缓)
206
+
207
+ 1. **L2 失败但 `legacy` 下正常** → P8 融合回归
208
+ 2. **L1 主动联想失效**或重启后插件加载报错 → P12 回归
209
+ 3. **T1 中"无时间词查询"的排序也变了** → 违反"零行为变更"约束
210
+ 4. **L7 重启后数据丢失**
211
+
212
+ ## 7. 边界(你只验收,不改代码)
213
+
214
+ - ❌ **禁止修改** `D:\dsh-auto-memory` 下任何文件(不改代码、不改测试、不提交)
215
+ - ❌ **禁止修改** `C:\Users\JH Z\.dsh\memory` 下记忆数据(只读取证;正常对话产生的写入除外)
216
+ - ✅ 允许:重启 dsh web(**必须 `--no-open`**)、操作 UI、调用只读工具(recall / 端点 GET)
217
+ - 发现问题 → **记录并回报**,不要自己动手修
218
+
219
+ ## 8. 回报必须包含
220
+
221
+ 1. 前置检查(git log / status)
222
+ 2. 启动命令与日志关键行(**确认带了 `--no-open`**)
223
+ 3. 九项 + UI 自查,逐项结果与证据
224
+ 4. **Go / No-Go 结论**
225
+ 5. 未通过项:现象、复现步骤、尝试过的处置
226
+
227
+ ---
228
+
229
+ > 端口、配置默认值、端点全量清单、故障排查都在 `HANDBOOK.md`,以它为准,不要凭常识推断。
@@ -0,0 +1,88 @@
1
+ ## 通用前置约束(每一段 prompt 都必须遵守,投喂时一并附上)
2
+
3
+ ### 1. 搜索优先原则(违反即打回)
4
+
5
+ 1. **动手前必须先搜索定位**。禁止凭 prompt 描述臆造文件路径、函数名、字段名。
6
+ 2. 必用手段:`Grep`(正则)、`Glob`、`Read`。每个要改动的符号都要**实际 grep 到并记下真实行号**。
7
+ 3. prompt 中给出的行号是**上一次观测值,可能已漂移**,必须先重新定位再动手;若行号对不上,以实际搜索结果为准并在回报中说明。
8
+ 4. **引用规范**:回报中每一处改动必须写成 `文件路径:行号 — 原内容 → 新内容`。
9
+
10
+ ### 2. 禁止事项
11
+
12
+ - **不得臆造 API**:任何调用的函数、字段、配置项,必须先在仓库中 grep 到定义处。搜不到就**停下回报**。
13
+ - **不得整文件重写**:一律最小 diff。禁止"顺手重构""统一风格""优化命名"。
14
+ - **不得引入新依赖**:项目 `dependencies` 为空(零运行时依赖承诺)。
15
+ - **不得改 API 签名**:既有导出函数的参数列表不得变更(可用可选参数扩展)。
16
+ - **不得删除既有测试断言**。
17
+ - 涉及 OpenViking(AGPLv3):**不得复制、翻译、逐行改写其源码**,只可参考公开文档算法思路。
18
+
19
+ ### 3. 集成位置正确性论证(回报必写)
20
+
21
+ 每处改动必须说明:
22
+ - **为什么选这个位置**(上游数据来源、下游消费者分别是谁,用 grep 到的调用链证明)
23
+ - **上下游影响**(哪些函数/模块会受影响,列出调用点行号)
24
+ - **回滚方式**(精确到命令或操作)
25
+
26
+ ### 4. 无法定位时
27
+
28
+ **立即停止并回报**,格式:
29
+ ```
30
+ 停止原因:未能定位 <符号/文件>
31
+ 已尝试:<搜索词 1>、<搜索词 2>、<路径>
32
+ 需要:<澄清问题>
33
+ ```
34
+ **禁止猜测、禁止"应该是"**。
35
+
36
+ ### 5. 自检清单模板(每段完成后逐项执行并贴结果)
37
+
38
+ ```bash
39
+ cd D:\dsh-auto-memory
40
+
41
+ # ① 编译
42
+ node --check <改动涉及的每个 lib/*.js>
43
+
44
+ # ② 测试(必跑,数字不得下降)
45
+ node tests/smoke/smoke-test-l0-extract-pre.mjs # 基线 18
46
+ node tests/smoke/smoke-test-handoff-pre.mjs # 基线 51
47
+ node tests/smoke/smoke-test-continue-chain-pre.mjs # 基线 58
48
+ node tests/smoke/smoke-test-water-step-pre.mjs # 基线 12
49
+ node tests/smoke/smoke-test-autocont-host-pre.mjs # 基线 29
50
+
51
+ # ③ 接口一致性(grep 校验,不得出现孤儿调用/断链)
52
+ grep -rn "<新增/改动的符号>" lib/ | head -20
53
+
54
+ # ④ 改动范围
55
+ git status --short
56
+ git diff --stat
57
+ ```
58
+
59
+ ### 6.5 接线类任务的额外约束(2026-09-09 新增,针对「把已交付模块接入既有管线」的段)
60
+
61
+ > 起因:M8-2 的接线开关被设在"P3 是否完成交付"上,而 P3 交付的是**并存函数、从未接线** → 执行侧按 prompt 交付纯核心后,管道悬空一整轮;随后 M8-2b 又因裸 `readdirSync` + 静默 catch 全程失效而**单测全绿**。
62
+
63
+ 1. **开关判据 = 目标管线的运行时状态**,不得用"上游段是否完成"代替。
64
+ - ❌ 错误示例:`若 P3 已完成则接线`
65
+ - ✅ 正确示例:`若 RRF 已实际被 recall() 调用则接线`
66
+ - 判据必须是**能被 grep 到的调用点**,不是某个段的完工状态。
67
+
68
+ 2. **验收必须含「接线代码在生产路径真实执行」的证据**(缺此项视为未完成):
69
+ - 调用链证据:调用方 `文件:行号` → 被调函数定义 `文件:行号`
70
+ - 运行证据:一次真实执行的输出 / 日志 / 实测数据(如真实 events 数据跑出的非空聚合结果)
71
+ - 若环境确实无法运行:**允许用静态证据(grep 到调用点)但必须显式标注「未实证」**,禁止默认视为已完成。
72
+
73
+ 3. **fail-soft 降级必须可观测**:接线处的 `catch` 不得为静默空捕获(`catch (e) {}`),至少 `diag('…降级: ' + 摘要)`。**保持 fail-soft 语义不变,只加日志**。
74
+
75
+ 4. **纯函数冒烟 ≠ 接线完成**:冒烟测的是模块本身(IO 注入夹具),**从不执行调用点**。交付接线后必须单独确认"生产路径真的会走到这里"。
76
+
77
+ ### 7. 项目事实速查(已核实,可直接引用)
78
+
79
+ | 项 | 值 |
80
+ |---|---|
81
+ | 根 | `D:\dsh-auto-memory`(pre 线) |
82
+ | 版本 / 协议 | 2.2.6 / BSD-3-Clause |
83
+ | 依赖 | peer `@deepseek-ai/cordis ^4.0.1`;optional `@huggingface/transformers ^3.7.6`;**dependencies 为空** |
84
+ | 约定 | 新增模块一律 `lib/xxx-pre.js`(纯函数、零 IO、IO 注入) |
85
+ | 编码 | UTF-8 无 BOM;仓库 CRLF;`*-pre.js` 与 `*.js` 成对存在 |
86
+ | 五条不变量 | I1 前缀缓存字节稳定 / I2 不替 host 决定压缩 / I3 凭证永不进提示词 / I4 绝不阻塞接续 / I5 水位测量在 pre-step |
87
+
88
+ ---
package/lib/client.js CHANGED
@@ -161,6 +161,7 @@ window.__ModuleLoader__.load({
161
161
  pyWizDetect: '开始检测', pyWizRedetect: '重新检测', pyWizCreate: '创建虚拟环境', pyWizInstall: '安装依赖', pyWizDownload: '开始下载', pyWizCancel: '取消下载', pyWizDone: '全部就绪 ✓ — 回上方启用即可', pyWizOk: 'ok', pyWizMissing: '未找到', pyWizTooOld: '版本不兼容(需 3.9-3.12)', pyWizVenvRec: '推荐', pyWizEta: '剩余', pyWizSec: '秒', pyWizModelHint: '模型与虚拟环境安装在 ~/.dsh/python-engine/(用户目录),升级/重装插件不受影响。',
162
162
  semModeHint: '自动=内置语义就绪即用,否则词法保底;高级 Python 需另行安装。',
163
163
  fAssocEngine: '启用自动记忆引擎', fAssocEngineHint: '总开关。开启后自动观测上下文、语义检索并适时唤起记忆注入(消费少量 token)。关闭则整个引擎不运行——不检索、不判定、不注入、不生成唤起记录。介意 token 消耗或担心动作跑偏的用户可关闭。默认关。',
164
+ fAnchorIndex: '记忆锚定索引(语料健康/存储管理)', fAnchorIndexHint: '开启后为记忆正文建立 sidecar 索引副本:存储管理页可做语料健康比对、「修复 stale」与「删除记忆」(三联动)。关闭时这些动作不可用(修复会提示 no-doc-store),但不影响记忆读写与检索。默认关。',
164
165
  secMemoryHubHint: '记忆中枢 = 三层记忆(经历/事实/技能)的编排器。开启后自动从对话沉淀经历、固化事实、把反复成功的流程固化为技能(skill),并在相似场景自动召回注入。',
165
166
  fMemoryHub: '启用记忆中枢', fMemoryHubHint: '总开关。开启后三层记忆(episodic 经历 / semantic 事实 / procedural 技能)开始运行;关闭则只保留已有记忆,不再沉淀新内容。默认关。',
166
167
  fEpisodicMin: '经历最少对话段数', fEpisodicMinHint: '一次经历(episode)至少积累多少段对话才巩固为记忆。太少=噪声多,太多=小对话被丢弃。默认 2。',
@@ -309,6 +310,7 @@ window.__ModuleLoader__.load({
309
310
  pyWizTitle: 'One-click Python engine setup', pyWizStep1: '1. Detect Python environment', pyWizStep2: '2. Create isolated venv (~/.dsh/python-engine/.venv, system untouched)', pyWizStep3: '3. Install engine deps (transformers + onnxruntime + torch)', pyWizStep4: '4. Download BGE-M3 int8 model (~539MB, resumable, auto mirror failover)',
310
311
  pyWizDetect: 'Detect', pyWizRedetect: 'Re-detect', pyWizCreate: 'Create venv', pyWizInstall: 'Install deps', pyWizDownload: 'Start download', pyWizCancel: 'Cancel download', pyWizDone: 'All ready - enable it above', pyWizOk: 'ok', pyWizMissing: 'not found', pyWizTooOld: 'incompatible (need 3.9-3.12)', pyWizVenvRec: 'recommended', pyWizEta: 'eta', pyWizSec: 's', pyWizModelHint: 'Model and venv live in ~/.dsh/python-engine/ (user dir) - plugin upgrades never touch them.',
311
312
  fAssocEngine: 'Enable automatic memory engine', fAssocEngineHint: 'Master switch. On = auto-observe context, semantic retrieval, and timely memory-activation injection (costs a little token). Off = the whole engine stops — no retrieval, no decide, no injection, no activation records. For users concerned about token cost or off-course actions. Default off.',
313
+ fAnchorIndex: 'Memory anchor index (corpus health / storage mgmt)', fAnchorIndexHint: 'When on, each memory file gets a sidecar index copy: the Storage tab can run corpus-health comparisons, "repair stale", and delete memories (cascading). When off those actions are unavailable (repair reports no-doc-store) but memory read/write/recall are unaffected. Default off.',
312
314
  secMemoryHubHint: 'Memory Hub = the orchestrator for three memory layers (episodic / semantic / procedural). When on, it distills episodes from dialogue, solidifies facts, and turns repeatedly-successful workflows into skills that are auto-recalled in similar contexts.',
313
315
  fMemoryHub: 'Enable Memory Hub', fMemoryHubHint: 'Master switch. On = the three memory layers (episodic / semantic / procedural) start running; Off = keep existing memories but stop distilling new ones. Default off.',
314
316
  fEpisodicMin: 'Min segments per episode', fEpisodicMinHint: 'How many dialogue segments an episode needs before it is consolidated. Too low = noise; too high = small talks discarded. Default 2.',
@@ -561,6 +563,29 @@ window.__ModuleLoader__.load({
561
563
 
562
564
  // ───────────────────────── 更新弹窗 / 首次指导 ─────────────────────────
563
565
  var CHANGELOG = {
566
+ '2.3.0': { zh: [
567
+ '★ 检索改为分层语义唤回:memory_recall 先返回 L0 摘要列表(每条记忆约 90 字,含 id/得分/匹配原因),需要原文再按 id 展开——此前全文直接进 embedding,超过模型 512 token 上限的条目尾部全部丢失(实测最长一条 9,822 字符);现在摘要永远在限制内,长记忆不再有损。',
568
+ '★ 检索融合改为三臂 rank-space:词法 + 语义 + 时间 三路各自排名后按 RRF(k=60)融合,取代旧的 minmax 归一化加权——旧方案分数只反映「这一批里排第几」,候选集一换分数就漂移、候选只剩 1 条时退化为常数,而「要不要注入」的决策恰恰依赖分数与阈值比较。现在排序与决策解耦,分数不再说谎。',
569
+ '★ 新增时间检索臂:查询里出现「上周」「三天前」「上个月」「最近 N 天」等中文时间表达时,命中时间范围的记忆会在排序中上升(软提升,不做硬过滤);查询不含时间词时行为与旧版逐字节一致。',
570
+ '★ 修复主动联想的词项截断:QueryPlan 构建时按 term 字典序排序后截断前 32 个,导致权重 1.0 的 trigger 词可能被丢、权重 0.2 的 assistant 词反而留下;现改为按来源权重降序保留(trigger 1.0 > user 0.8 > tool-result 0.6…),同权重按字典序稳定。',
571
+ '★ C3/Python 档也有语义召回了:此前 Python(BGE-M3)只服务主动联想,recall 在该档只有词法;现把 worker 已有的 dense_search 暴露为 recall 的语义臂,设置 → 自动记忆引擎 切到「高级 Python」后召回同样带语义分;auto 档自动择优(Python 可用则 C3,否则 C2),Python 不可用时静默回退 C2/词法。',
572
+ '★ M8 三层记忆默认开启:fact(事实)/episodic(经历)/procedure(技能)三层记忆店与「记忆中枢」页签随装即用;fact 带时间三价(事件发生/陈述/入库)与认识论状态(证据/推断/指令),冲突不再互相覆盖。',
573
+ '★ 记忆有了重要性权重:六类使用证据(曝光/读取/引用/复用/成功/纠正)聚合为重要性分,接入检索排序——常被引用、跨会话复现的记忆排得更靠前,被频繁纠正的下沉;它不随查询变化,是绝对量。',
574
+ '★ 交接默认开启并分层:水位阈值 0.75(官方压缩 0.80,留余量),四层交接材料(白板/账本/近期线程/完整转写按需 read)改为「按需取用而非通读」;账本按段赋权截断(失败原因 > 下一步 > 目标 > 状态),预算不足先截低价值段。',
575
+ '★ 证据链修复两条:①用户纠正此前要求消息里含完整 32 位记忆 id(几乎不可能触发),现归因到最近被引用/读取的记忆;②success 事件时间戳取错字段(在 event.ts 而非顶层)导致成功证据恒为 0,已修正。',
576
+ '修复:设置页「自动记忆引擎」分区标题空白(sectionLabels 键名错误);bge-m3 模型仓库名错误导致 HF 401(#27)与 tokenizer 离线缺件(#28);DSH 0.1.2 in-process 子代理回收失效(localAgent 字段更名)与 result 卡死泄漏(withTimeout 兜底);importance 管道因未导入 readdirSync 静默失效(补导入+降级日志)。',
577
+ ], en: [
578
+ '★ Layered semantic recall: memory_recall now returns an L0 digest list first (~90 characters per memory, with id/score/match reason); expand any id for the full text. Previously the full text went straight into embedding and anything past the model\'s 512-token limit was silently dropped (longest observed: 9,822 characters); digests always fit.',
579
+ '★ Three-arm rank-space fusion: lexical + semantic + time arms are ranked separately and fused with RRF (k=60), replacing minmax normalization — the old scores only reflected "rank within this batch", drifted whenever the candidate set changed, and collapsed to a constant with a single candidate, while the inject-or-not decision depends on comparing scores to thresholds. Ranking and decision are now decoupled.',
580
+ '★ Time-aware retrieval: queries containing Chinese time expressions ("last week", "three days ago", "last month", "the last N days") now boost memories from the matching window (soft boost, no hard filtering); queries without a time expression behave byte-for-byte as before.',
581
+ '★ Fixed proactive-recall term truncation: the QueryPlan sorted terms by dictionary order and kept the first 32, so a weight-1.0 trigger term could be dropped while a weight-0.2 assistant term survived; terms are now kept by source weight descending (trigger 1.0 > user 0.8 > tool-result 0.6 …), with dictionary order as a stable tiebreak.',
582
+ '★ The C3/Python tier now has semantic recall too: Python (BGE-M3) used to serve only proactive association, leaving recall lexical-only on that tier; the worker\'s existing dense_search is now exposed as recall\'s semantic arm. Settings → Semantic engine → Advanced Python gets semantic scores in recall; auto picks the best available (C3 when Python is ready, else C2), with silent fallback to C2/lexical when Python is unavailable.',
583
+ '★ M8 three-tier memory on by default: fact / episodic / procedure stores and the Memory Hub tab work out of the box; facts carry three-valued time (occurred / mentioned / ingested) and an epistemic status (evidence / inference / directive), so conflicts no longer overwrite each other.',
584
+ '★ Memories now carry importance: six usage signals (seen / read / cited / reused / succeeded / corrected) aggregate into an importance score that feeds retrieval ranking — frequently cited, cross-session memories rank higher, heavily corrected ones sink; it does not vary with the query, making it an absolute quantity.',
585
+ '★ Handoff on by default and layered: the water-level threshold sits at 0.75 (official compaction at 0.80, headroom kept), the four-layer handoff material (whiteboard / ledger / recent thread / full transcript on demand) is consumed "on demand, not read-through", and the ledger is truncated by section weight (dead ends > next steps > goals > status) so budget cuts hit low-value sections first.',
586
+ '★ Two evidence-chain fixes: (1) user corrections used to require the full 32-character memory id inside the message (practically never fired) — they now attribute to the most recently cited/read memory; (2) success events read the timestamp from the wrong field (event.ts, not top-level), so success evidence was structurally always zero — fixed.',
587
+ 'Fixes: the "Semantic engine" settings section title rendered blank (wrong sectionLabels key); the bge-m3 model repo name pointed to a non-existent repo causing HF 401 (#27) plus missing tokenizer files offline (#28); DSH 0.1.2 in-process subagent recycling never fired (field renamed to localAgent) with result hangs leaking (withTimeout fallback); the importance pipeline silently died on a missing readdirSync import (import added + downgrade logging).',
588
+ ] },
564
589
  '2.2.6': { zh: [
565
590
  '★ 修复:交接测量终于站到官方压缩的同一条边界上——官方自动压缩挂在每个 step 边界(pre-step,阈值 80%),而插件此前只在「一轮结束」测量;某一轮把水位从阈值下推到 80% 以上时,官方会在该轮就压缩完,交接白板/账本来不及写。现在 pre-step 也做同一测量(同一会话 5 秒节流),水位一到阈值就先写交接材料,不再被官方抢跑。',
566
591
  '★ 自动接续改为宿主兜底:建新会话不再依赖浏览器页面——水位达标后在宿主侧开倒计时,确认卡照常弹出(同意/拒绝都直接发给宿主);页面被后台节流、标签页关闭或人不在时,倒计时一到宿主自己完成「刷新白板/账本 → 建新会话 → 沿用模型与工作区 → 注入交接材料」,不再因浏览器休眠而断档。',
@@ -1885,7 +1910,11 @@ window.__ModuleLoader__.load({
1885
1910
  headers: { 'Content-Type': 'application/json' }, body: JSON.stringify(body) })
1886
1911
  .then(function (r) { return r.json() })
1887
1912
  .then(function (j) {
1888
- setMsg(action + ': ' + (j && (j.reason || (j.ok === false ? 'rejected' : 'ok')) || 'done'))
1913
+ var reason = (j && (j.reason || (j.ok === false ? 'rejected' : 'ok'))) || 'done'
1914
+ if (j && j.ok === false && j.reason === 'no-doc-store') {
1915
+ reason = locale === 'zh' ? '记忆锚定索引未启用:请在 设置 → 自动记忆引擎 开启「记忆锚定索引」后重试;或忽略此提示(不影响记忆读写与检索)' : 'anchor index disabled: enable "Memory anchor index" under Settings → Semantic engine, or ignore (read/write/recall unaffected)'
1916
+ }
1917
+ setMsg(action + ': ' + reason)
1889
1918
  if (onDone) onDone(j)
1890
1919
  var again = fetch('/api/dsh-auto-memory/storage-manage').then(function (r2) { return r2.json() })
1891
1920
  again.then(function (j2) { setData(j2 || null) })
@@ -1897,6 +1926,10 @@ window.__ModuleLoader__.load({
1897
1926
  if (data.error) return h('div', { 'data-dam-hint': '' }, String(data.error))
1898
1927
  var counts = data.counts || { total: 0, ok: 0, stale: 0, unrepairable: 0 }
1899
1928
  var rows = []
1929
+ if (data.indexEnabled === false) {
1930
+ rows.push(h('div', { 'data-dam-hint': '', style: { marginTop: '6px', color: 'var(--dsw-alias-warn, #e8c584)' } },
1931
+ locale === 'zh' ? '⚠ 记忆锚定索引未启用:语料健康不做 sidecar 比对(无 stale 可修)。如需「修复 stale / 删除记忆」,请到 设置 → 自动记忆引擎 开启「记忆锚定索引」。此提示不影响记忆读写与检索。' : '⚠ Memory anchor index disabled: corpus health performs no sidecar comparison (nothing stale). To use repair/delete, enable "Memory anchor index" under Settings → Semantic engine. Read/write/recall are unaffected.'))
1932
+ }
1900
1933
  rows.push(h('div', { 'data-dam-hint': '', style: { marginTop: '8px' } }, t('storageScanHint')))
1901
1934
  rows.push(h(Card, { title: (locale === 'zh' ? '语料健康' : 'Corpus health') + ' (' + counts.ok + '/' + counts.total + ' ok)' },
1902
1935
  h('div', null,
@@ -3867,8 +3900,9 @@ window.__ModuleLoader__.load({
3867
3900
  return h('button', { key: key, 'data-dam-btn': '', 'data-active': settingsSection === key ? 'true' : undefined, onClick: function () { jumpToSection(key) } }, sectionLabels[key])
3868
3901
  })),
3869
3902
  h('div', { 'data-dam-settings-content': '' },
3870
- section('semantic', sectionLabels.secSemantic, [
3903
+ section('semantic', sectionLabels.semantic, [
3871
3904
  field(t('fAssocEngine'), h('input', { type: 'checkbox', checked: !!cfg.associativeMemoryEnabled, onChange: function (e) { set('associativeMemoryEnabled', e.target.checked) } }), t('fAssocEngineHint')),
3905
+ field(t('fAnchorIndex'), h('input', { type: 'checkbox', checked: !!cfg.memoryAnchorEnabled, onChange: function (e) { set('memoryAnchorEnabled', e.target.checked) } }), t('fAnchorIndexHint')),
3872
3906
  field(t('fJsCooldown'), h('input', { 'data-dam-input': '', type: 'number', min: 0, max: 60, value: cfg.jsDecideCooldownRounds === undefined ? 1 : cfg.jsDecideCooldownRounds, onChange: function (e) { set('jsDecideCooldownRounds', (function () { var v = Number(e.target.value); return (Number.isFinite(v) && v >= 0) ? v : 1 })()) /* #19: 0=不冷却 是合法值,|| 1 会吃掉 0 */ } }), t('fJsCooldownHint')),
3873
3907
  field(t('fJsDelta'), h('input', { 'data-dam-input': '', type: 'number', min: 0, max: 1, step: 0.005, value: cfg.jsDecideDeltaExp === undefined ? 0.01 : cfg.jsDecideDeltaExp, onChange: function (e) { var v = Number(e.target.value); set('jsDecideDeltaExp', Number.isFinite(v) && v >= 0 ? v : 0.01) } }), t('fJsDeltaHint')),
3874
3908
  field(t('fJsExcerpt'), h('input', { 'data-dam-input': '', type: 'number', min: 20, max: 480, value: cfg.jsDecideExcerptChars === undefined ? 40 : cfg.jsDecideExcerptChars, onChange: function (e) { set('jsDecideExcerptChars', Math.max(20, Math.min(480, Number(e.target.value) || 40))) } }), t('fJsExcerptHint')),
@@ -19,6 +19,7 @@ import {
19
19
  buildContextPushEnvelopePre, buildAuthorizedMemoryRefFromRecord, createAccessEvidencePre,
20
20
  createCiteEvidencesFromText, createCorrectionEvidencesFromText, computeReadCoverage,
21
21
  createContextPushBridge, createNullContextSinkPre, createFakeContextSinkPre,
22
+ CORRECTION_LEXICON_V1,
22
23
  CONTEXT_BRIDGE_BUDGET_V1, CONTEXT_BRIDGE_POLICY_VERSION, EVIDENCE_POLICY_VERSION,
23
24
  } from './context-bridge.js'
24
25
  import { EvidenceEventStore, rebuildAggregates, workspaceRefOf } from './evidence-store.js'
@@ -41,6 +42,49 @@ function diagCtx(msg) {
41
42
  } catch (e) {}
42
43
  }
43
44
 
45
+ /**
46
+ * P9a correction 归因选择器(纯函数,零 IO;precision-first):
47
+ * 从 evidence store 事件(loadEvents 投影形态)里选「最近一条被 cite/read 的记忆」作为
48
+ * correction 归因对象。用户纠正时几乎不可能手打完整 memoryId,故归因到最近被引用/读取
49
+ * 的记忆而非用户消息内新找的 id(原 createCorrectionEvidencesFromText 路径保留不动)。
50
+ * - 只认 kind∈{cite,read} 且带 memoryId;时间窗口 [now-windowMs, +∞),默认 5 分钟;
51
+ * - 按 ts 降序取第 1 条;ts 平局按 memoryId 升序保证确定性;
52
+ * - 窗口内已有 correction 的 memoryId 跳过(同一记忆一轮最多一条,防连续段重复惩罚);
53
+ * - ts 取 event.ts(投影形态);兼容内存形态顶层 ts/createdAt;
54
+ * - 任何异常/非法输入返回 null(调用方静默不发)。
55
+ */
56
+ export function selectCorrectionAttributionPre(input) {
57
+ try {
58
+ const events = Array.isArray(input && input.events) ? input.events : []
59
+ const now = Number.isFinite(input && input.now) ? input.now : Date.now()
60
+ const windowMs = Number.isFinite(input && input.windowMs) && input.windowMs > 0 ? input.windowMs : 300000
61
+ const cutoff = now - windowMs
62
+ const correctedRecently = new Set()
63
+ let best = null
64
+ let bestTs = -1
65
+ for (const e of events) {
66
+ if (!e || typeof e !== 'object') continue
67
+ if (e.kind === 'correction' && e.memoryId) {
68
+ const cts = Number(e.event && e.event.ts) || Number(e.ts) || Number(e.createdAt) || 0
69
+ if (cts >= cutoff) correctedRecently.add(String(e.memoryId))
70
+ }
71
+ }
72
+ for (const e of events) {
73
+ if (!e || typeof e !== 'object') continue
74
+ if (e.kind !== 'cite' && e.kind !== 'read') continue
75
+ if (!e.memoryId) continue
76
+ if (correctedRecently.has(String(e.memoryId))) continue
77
+ const ets = Number(e.event && e.event.ts) || Number(e.ts) || Number(e.createdAt) || 0
78
+ if (ets < cutoff) continue
79
+ if (!best || ets > bestTs || (ets === bestTs && String(e.memoryId) < String(best.memoryId))) {
80
+ best = e
81
+ bestTs = ets
82
+ }
83
+ }
84
+ return best
85
+ } catch (_) { return null }
86
+ }
87
+
44
88
  export function createContextHost(opts = {}) {
45
89
  const engine = opts.engine
46
90
  if (!engine) throw new Error('context-host: engine required')
@@ -497,7 +541,36 @@ export function createContextHost(opts = {}) {
497
541
  const corrections = seg.kind === 'user'
498
542
  ? createCorrectionEvidencesFromText({ text: seg.text, knownRecords: corpusSnap.records, coords })
499
543
  : []
500
- await persistEvidence([...cites, ...corrections])
544
+ // P9a:用户段命中纠正词典但文本不含完整 memoryId 时(现实常态),把 correction 归因到
545
+ // 最近一条被 cite/read 的记忆(5 分钟窗口,同记忆一轮最多一条);无命中静默不发。
546
+ // 原 cite/correction 产出零改动;store 读取仅在词典命中时发生;fail-soft + diag(不记用户原文)。
547
+ const attributed = []
548
+ if (seg.kind === 'user') {
549
+ let lexHits = 0
550
+ let attributedPrefix = 'none'
551
+ try {
552
+ const norm = String(seg.text == null ? '' : seg.text).normalize('NFKC').replace(/[A-Z]/g, (c) => c.toLowerCase())
553
+ lexHits = CORRECTION_LEXICON_V1.filter((w) => norm.includes(w)).length
554
+ if (lexHits > 0) {
555
+ const st = storeFor()
556
+ const recent = selectCorrectionAttributionPre({ events: st.loadEvents().events, now: Date.now() })
557
+ if (recent) {
558
+ const r = createAccessEvidencePre({
559
+ ...coords, kind: 'correction', memoryId: recent.memoryId, anchorId: recent.anchorId, scope: recent.scope,
560
+ sourceRef: recent.source && recent.source.sourceRef, sourceEpoch: recent.source && recent.source.sourceEpoch,
561
+ sourceVersion: recent.source && recent.source.sourceVersion,
562
+ fileDigest: recent.source && recent.source.fileDigest, recordDigest: recent.source && recent.source.recordDigest,
563
+ })
564
+ if (r.ok) { attributed.push(r.evidence); attributedPrefix = String(recent.memoryId).slice(0, 12) }
565
+ }
566
+ }
567
+ } catch (eP9a) {
568
+ diagCtx('p9a correction attribution error: ' + String(eP9a && eP9a.message || eP9a).slice(0, 100))
569
+ }
570
+ // 隐私:diag 只记词典词计数 + 归因 memoryId 前 12 位,绝不记录用户原文。
571
+ diagCtx('p9a correction attribution: lexHits=' + lexHits + ' attributed=' + attributedPrefix)
572
+ }
573
+ await persistEvidence([...cites, ...corrections, ...attributed])
501
574
  }
502
575
 
503
576
  /**
@@ -696,7 +769,9 @@ export function createContextHost(opts = {}) {
696
769
  const cutoff = Date.now() - (Number(windowMs) || 300000)
697
770
  const out = new Map()
698
771
  for (const e of events) {
699
- const ets = e.ts || e.createdAt || 0
772
+ // P9d:store 投影事件 ts 在 event.ts(顶层无 ts/createdAt),与 selectCorrectionAttributionPre 同口径;
773
+ // 旧写法 e.ts || e.createdAt || 0 恒为 0 → 窗口全跳过 → success 证据链结构性断裂。
774
+ const ets = Number(e.event && e.event.ts) || Number(e.ts) || Number(e.createdAt) || 0
700
775
  if (ets < cutoff) continue
701
776
  if (e.kind !== 'read' && e.kind !== 'cite') continue
702
777
  if (!e.memoryId) continue
@@ -0,0 +1,81 @@
1
+ /**
2
+ * evidence-agg-pre —— evidence 事件 → 聚合 → importance 输入契约(M8-2b, 2026-09-09)。
3
+ *
4
+ * 管道:evidence/events/*.jsonl(写入侧 context-bridge-pre,只读)→ 有界扫描 →
5
+ * 按 memoryId 聚合六类计数 + distinctSessions → 交给 memory-importance-pre.computeImportancePre
6
+ * → 作为 recall() L0 融合的加权因子之一(P8 融合入口)。
7
+ *
8
+ * 契约:
9
+ * - 纯函数 + IO 注入(io = { listFiles(), readFile(name) }),模块零内置 IO、零写入。
10
+ * - 有界读取:文件名日期在窗口内(默认近 7 天)且每文件只取末 N 行(默认 400),绝不全量扫描历史。
11
+ * 文件名约定 YYYY-MM-DD.jsonl(EvidenceEventStore 落盘惯例,实测样本确认);无法解析日期的文件跳过(确定性)。
12
+ * - 事件行结构(实测 2026-09-09.jsonl 确认):kind/memoryId 在顶层,会话= event.sessionRef,时间= event.ts。
13
+ * - 聚合输出形状 = memory-importance-pre.computeImportancePre 的输入契约(以其源码为准,不另立)。
14
+ * - fail-soft:行损坏跳过;io 抛错由调用方处理(模块内不吞 IO 异常——io 是注入的,调用方决定降级)。
15
+ */
16
+
17
+ export const EVIDENCE_AGG_VERSION = 'evidence_agg_v1'
18
+
19
+ /** 有界读取默认值。 */
20
+ export const EVIDENCE_AGG_DEFAULTS_V1 = Object.freeze({
21
+ maxAgeDays: 7, // 文件名日期距 now 的最大天数
22
+ maxLinesPerFile: 400, // 每文件末 N 行(沿用 index.js 证据读取范式 slice(-400))
23
+ })
24
+
25
+ const KINDS = ['seen', 'read', 'cite', 'reuse', 'success', 'correction']
26
+
27
+ /**
28
+ * 有界扫描 evidence 事件(只读)。io 注入;文件按名字日期过滤 + 每文件末 N 行。
29
+ * @param {{listFiles:Function, readFile:Function}} io listFiles()→文件名数组;readFile(name)→全文
30
+ * @param {{maxAgeDays?:number, maxLinesPerFile?:number, now?:Function}} opts
31
+ * @returns {Array<object>} 解析后的事件对象(损坏行跳过;任何字段缺失由聚合层兜底)
32
+ */
33
+ export function scanEvidenceEventsPre(io, opts = {}) {
34
+ const d = Object.assign({}, EVIDENCE_AGG_DEFAULTS_V1, opts)
35
+ const now = d.now || Date.now
36
+ const files = (io.listFiles() || []).slice().sort()
37
+ const cutoff = now() - d.maxAgeDays * 86400000
38
+ const out = []
39
+ for (const name of files) {
40
+ const m = /^(\d{4})-(\d{2})-(\d{2})\.jsonl$/.exec(String(name))
41
+ if (!m) continue // 非日期命名 → 跳过(确定性;不做全量兜底)
42
+ const fileTs = Date.UTC(Number(m[1]), Number(m[2]) - 1, Number(m[3]), 23, 59, 59)
43
+ if (fileTs < cutoff) continue
44
+ const lines = String(io.readFile(name) || '').split('\n').filter(Boolean)
45
+ for (const ln of lines.slice(-d.maxLinesPerFile)) {
46
+ try { out.push(JSON.parse(ln)) } catch (_) {}
47
+ }
48
+ }
49
+ return out
50
+ }
51
+
52
+ /**
53
+ * 按 memoryId 聚合六类计数与去重会话数。输出形状 = computeImportancePre 的输入契约。
54
+ * @param {Array<{kind?:string, memoryId?:string, event?:{sessionRef?:string}, sessionRef?:string}>} events
55
+ * @returns {Map<string, {distinctSessions:number, seen:number, read:number, cite:number, reuse:number, success:number, correction:number}>}
56
+ */
57
+ export function aggregateEvidenceEventsPre(events) {
58
+ const list = Array.isArray(events) ? events : []
59
+ const byId = new Map()
60
+ for (const e of list) {
61
+ if (!e || typeof e.memoryId !== 'string' || !e.memoryId) continue
62
+ const kind = e.kind
63
+ if (!KINDS.includes(kind)) continue // 非六类事件不计数(与 evidenceFor 口径一致)
64
+ let agg = byId.get(e.memoryId)
65
+ if (!agg) {
66
+ agg = { distinctSessions: 0, seen: 0, read: 0, cite: 0, reuse: 0, success: 0, correction: 0 }
67
+ byId.set(e.memoryId, agg)
68
+ }
69
+ agg[kind]++
70
+ const sessionRef = (e.event && e.event.sessionRef) || e.sessionRef
71
+ if (sessionRef) {
72
+ if (!agg._sessions) agg._sessions = new Set()
73
+ agg._sessions.add(sessionRef)
74
+ }
75
+ }
76
+ for (const agg of byId.values()) {
77
+ agg.distinctSessions = agg._sessions ? agg._sessions.size : 0
78
+ delete agg._sessions
79
+ }
80
+ return byId
81
+ }