@a9i5k4/dsh-auto-memory 2.5.3 → 3.0.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (167) hide show
  1. package/README.md +189 -7
  2. package/README.zh-CN.md +189 -7
  3. package/docs/CONTRIBUTORS.html +471 -0
  4. package/docs/FRONTEND-CO-CREATION.md +191 -0
  5. package/docs/GM53-HOMEPAGE-PROMPT.md +323 -0
  6. package/docs/HANDOFF-CRITERIA.md +92 -0
  7. package/docs/HOMEPAGE-CONTENT-FOR-GM53.md +299 -0
  8. package/docs/INTEGRATION-ANALYSIS.md +350 -348
  9. package/docs/PROMO-PROMPT-3.0.md +100 -0
  10. package/docs/USER-GUIDE.en.md +58 -3
  11. package/docs/USER-GUIDE.zh-CN.md +59 -4
  12. package/docs/WHITEPAPER.md +207 -0
  13. package/docs/internal/ACCEPT-35-LIVE.md +143 -0
  14. package/docs/internal/ACCEPTANCE-20260914.md +90 -0
  15. package/docs/internal/ARCH-REVIEW-BRIEF.md +411 -0
  16. package/docs/internal/ARCH-REVIEW-REQUEST.md +201 -0
  17. package/docs/internal/ARCH-REVIEW-ROUND2.md +169 -0
  18. package/docs/internal/ARCH-REVIEW-ROUND3.md +206 -0
  19. package/docs/internal/ARCHITECTURE-FOR-ZCODE-20260920.md +397 -0
  20. package/docs/internal/ART-DIRECTION-DEEPSEEK-20260920.md +351 -0
  21. package/docs/internal/ART-DIRECTION-WIREFRAME.md +191 -181
  22. package/docs/internal/ART-DIRECTION-WIREFRAME.md.bak-superseded +181 -0
  23. package/docs/internal/AUDIT-WB-GRAPH-FULL-20260916.md +314 -0
  24. package/docs/internal/BATTLE-PLAN-20260917.md +871 -0
  25. package/docs/internal/CONCURRENCY-INVESTIGATION-20260917.md +192 -0
  26. package/docs/internal/CROSS-SESSION-SEARCH-PATH-DECISION.md +72 -0
  27. package/docs/internal/CROSS-SESSION-SEARCH-RESEARCH.md +131 -0
  28. package/docs/internal/DECISIONS-20260914-SESSION.md +269 -0
  29. package/docs/internal/DESIGN-P1-STATE-COMMIT-20260915.md +219 -0
  30. package/docs/internal/DIRECTION-CHECK-WB-GRAPH-20260916.md +132 -0
  31. package/docs/internal/FEATURE-INVENTORY.md +531 -0
  32. package/docs/internal/FEEDBACK-TO-DSHAPI-RELAY.md +13 -0
  33. package/docs/internal/G-SERIES-EXECUTION-20260917.md +248 -0
  34. package/docs/internal/G3-DESIGN-20260918.md +82 -0
  35. package/docs/internal/G3-DISK-FORMAT-GAP-20260919.md +92 -0
  36. package/docs/internal/GH-DISCUSSION-5732-COMMENT.md +74 -0
  37. package/docs/internal/GPT-ACCEPTANCE-PROMPT-20260916.md +352 -0
  38. package/docs/internal/GPT-REVIEW-PROMPT.md +216 -0
  39. package/docs/internal/GROUP-WEBHOOK-SETUP.md +33 -0
  40. package/docs/internal/HANDOFF-TO-ZCODE-20260920.md +309 -0
  41. package/docs/internal/HERMES-DATA-VERIFICATION-20260919.md +120 -0
  42. package/docs/internal/HERMES-LEGACY-STATUS-20260919.md +74 -0
  43. package/docs/internal/ISSUE-55-58-VERIFICATION-20260918.md +175 -0
  44. package/docs/internal/ISSUE10-FIX-EXECUTION-20260919.md +389 -0
  45. package/docs/internal/ISSUE10-PLAN-20260919.md +254 -0
  46. package/docs/internal/ISSUE10B-FORENSICS-20260919.md +468 -0
  47. package/docs/internal/ISSUE9-PURGE-AND-R1-PLAIN-20260919.md +150 -0
  48. package/docs/internal/ISSUE9-RESIDUAL-FORENSICS-20260919.md +114 -0
  49. package/docs/internal/KICKOFF-P0.md +254 -0
  50. package/docs/internal/LESSON-TO-CANDIDATE-STATUS-20260919.md +79 -0
  51. package/docs/internal/MASTER-PLAN-3.0.md +411 -0
  52. package/docs/internal/MEMORY-GOVERNANCE-20260917.md +309 -0
  53. package/docs/internal/MEMORY-MUTATION-AND-INDEX-DESIGN.md +85 -0
  54. package/docs/internal/MERGE-CONFLICT-SCAN-20260914.md +222 -0
  55. package/docs/internal/PENDING-FIXES-20260916.md +289 -0
  56. package/docs/internal/PRE-FRONTEND-CHECKLIST-20260919.md +705 -0
  57. package/docs/internal/PRE-FRONTEND-CHECKLIST-20260919.md.bak-s10 +649 -0
  58. package/docs/internal/PROCEDURAL-MEMORY-AND-APPROVAL-DESIGN-20260918.md +225 -0
  59. package/docs/internal/PROGRESS-20260917.md +93 -0
  60. package/docs/internal/PROMPT-GAP-AUDIT-20260920.md +128 -0
  61. package/docs/internal/R1-DEGRADE-AUDIT-20260918.md +163 -0
  62. package/docs/internal/R1-READABILITY-FORENSICS-20260919.md +127 -0
  63. package/docs/internal/R2-EVIDENCE-DEEP-AUDIT-20260918.md +140 -0
  64. package/docs/internal/R3-DEGRADE-LEDGER-DESIGN-20260918.md +138 -0
  65. package/docs/internal/R4-RECALL-QUOTA-PLAN-20260918.md +218 -0
  66. package/docs/internal/RAG-KARPATHY-PROGRAM.md +229 -0
  67. package/docs/internal/REPORT-P0-NIGHTLY.md +212 -0
  68. package/docs/internal/REPORT-P5-ACCEPTANCE.md +31 -0
  69. package/docs/internal/REPORT-WB-GRAPH-NIGHTLY.md +153 -0
  70. package/docs/internal/RESUME-20260918.md +171 -0
  71. package/docs/internal/RESUME-20260919.md +104 -0
  72. package/docs/internal/REVIEW-WB-GRAPH-SELF.md +81 -0
  73. package/docs/internal/RHINELAB-TO-DEEPSEEK-FEASIBILITY.md +198 -0
  74. package/docs/internal/ROADMAP-20260917-WEEK.md +439 -0
  75. package/docs/internal/ROADMAP.md +106 -0
  76. package/docs/internal/RUN-P0-NIGHTLY.md +227 -0
  77. package/docs/internal/S10-CONSTRUCTION-HANDOFF-20260917.md +185 -0
  78. package/docs/internal/S10-GAP-INVENTORY-20260917.md +239 -0
  79. package/docs/internal/S10-GAPS-PLAIN-20260917.md +125 -0
  80. package/docs/internal/SEMANTIC-ARCHITECTURE-SPEC.md +360 -0
  81. package/docs/internal/SESSION-FILE-REPAIR-PROTOCOL.md +90 -0
  82. package/docs/internal/T6-EXECUTION-20260920.md +130 -0
  83. package/docs/internal/TELEMETRY-EFFECT-REPORT-DESIGN-20260918.md +146 -0
  84. package/docs/internal/THESIS-GAP-ANALYSIS-20260918.md +89 -0
  85. package/docs/internal/THESIS-OUTLINE-20260918.md +147 -0
  86. package/docs/internal/THREE-LAYER-CONTRACT.md +219 -0
  87. package/docs/internal/TODO-BACKLOG.md +263 -142
  88. package/docs/internal/TODO-GRAPH.html +715 -0
  89. package/docs/internal/TODO-GRAPH.html.bak-20260914-v2 +493 -0
  90. package/docs/internal/TODO-GRAPH.html.bak-20260915-alsfix +710 -0
  91. package/docs/internal/TODO-GRAPH.html.bak-20260915-p1 +710 -0
  92. package/docs/internal/TODO-GRAPH.html.bak-20260915-p6a-rev +703 -0
  93. package/docs/internal/TODO-GRAPH.html.bak-20260915-wshint +710 -0
  94. package/docs/internal/TODO-GRAPH.html.bak-20260916-batch +715 -0
  95. package/docs/internal/UPSTREAM-ISSUE-PR-TRIAGE-20260919.md +297 -0
  96. package/docs/internal/UPSTREAM-ISSUES-3RD-AUDIT-20260920.md +104 -0
  97. package/docs/internal/WB-FORMAT-CONVENTION.md +112 -0
  98. package/docs/internal/WB-GRAPH-DECISIONS-20260914.md +71 -0
  99. package/docs/internal/reviews/CLAIM-VERIFICATION-20260914.md +56 -0
  100. package/docs/internal/reviews/PLAN-gpt6astra-round2-20260914.md +787 -0
  101. package/docs/internal/reviews/REVIEW-gpt6astra-20260914.md +112 -0
  102. package/docs/internal/reviews/ROUND3-REVIEW-INTEGRATION-20260914.md +230 -0
  103. package/docs/prompts/M8-3-enable-verify.md +49 -49
  104. package/docs/screenshots/promo/promo-0-banner-v3.png +0 -0
  105. package/lib/acceptance.js +71 -0
  106. package/lib/activation-host.js +153 -18
  107. package/lib/activation-inbox.js +25 -7
  108. package/lib/board-mode.js +30 -0
  109. package/lib/client.js +1758 -90
  110. package/lib/config-io.js +156 -0
  111. package/lib/context-bridge.js +5 -2
  112. package/lib/context-host.js +86 -15
  113. package/lib/degrade.js +385 -0
  114. package/lib/dsh-home.js +143 -0
  115. package/lib/engine-identity.js +149 -0
  116. package/lib/engine-switch.js +247 -0
  117. package/lib/episodic-store.js +63 -12
  118. package/lib/evidence-store.js +10 -3
  119. package/lib/fact-store.js +22 -3
  120. package/lib/fs-retry.js +46 -0
  121. package/lib/index-sync.js +13 -1
  122. package/lib/index.js +3446 -263
  123. package/lib/intent-clean-safe.js +258 -0
  124. package/lib/intent-clean.js +12 -16
  125. package/lib/l0-extract.js +478 -149
  126. package/lib/l0-index-sync.js +195 -0
  127. package/lib/l0-index.js +349 -239
  128. package/lib/ledger-criteria.js +142 -0
  129. package/lib/m4-corpus.js +8 -2
  130. package/lib/m7-index-sync-host.js +73 -5
  131. package/lib/m7-wire.js +3 -3
  132. package/lib/memory-anchor.js +56 -1
  133. package/lib/memory-envelope.js +257 -0
  134. package/lib/memory-hub.js +138 -13
  135. package/lib/memory-index.js +4 -2
  136. package/lib/memory-mutation.js +246 -0
  137. package/lib/memory-writer.js +204 -24
  138. package/lib/note-status-apply.js +118 -0
  139. package/lib/note-status.js +196 -0
  140. package/lib/procedure-observation.js +48 -0
  141. package/lib/procedure-store.js +118 -20
  142. package/lib/python-setup.js +1 -1
  143. package/lib/python-sidecar-client.js +29 -3
  144. package/lib/recall-fusion.js +83 -12
  145. package/lib/rerank-host.js +160 -0
  146. package/lib/rules-edit.js +159 -0
  147. package/lib/rules-layer.js +261 -0
  148. package/lib/semantic-decide.js +41 -8
  149. package/lib/semantic-js.js +66 -6
  150. package/lib/shadow-host.js +3 -5
  151. package/lib/shadow-retrieval.js +3 -3
  152. package/lib/skill-export-host.js +153 -0
  153. package/lib/skill-export.js +239 -0
  154. package/lib/state-commit.js +245 -0
  155. package/lib/storage-manage.js +6 -0
  156. package/lib/subagent-gc.js +4 -8
  157. package/lib/temporal-parse.js +191 -159
  158. package/lib/tier-layer-inject.js +650 -0
  159. package/lib/tier0-catalog.js +735 -0
  160. package/lib/water-window.js +263 -186
  161. package/lib/wb-contract.js +691 -0
  162. package/lib/wb-sidecar.js +890 -0
  163. package/lib/ws-overview-rank.js +2 -2
  164. package/package.json +1 -1
  165. package/python/m7_embedding_v1.py +5 -5
  166. package/python/worker_semantic_v1.py +17 -6
  167. package/python/worker_v1.py +38 -4
package/lib/l0-extract.js CHANGED
@@ -1,149 +1,478 @@
1
- /**
2
- * L0 抽取纯核心(l0_extract_v1)—— 分层语义唤回的地基。
3
- *
4
- * 2026-09-08 建立。目的:为每条记忆生成廉价摘要(L0),使检索可先在小空间
5
- * 收敛候选,再按 id 下钻原文,从而把 token 开销与索引构建成本降约一个数量级
6
- * (实测:条目平均 814 字符 → L0 约 118 字符,压缩比 6.9:1)。
7
- *
8
- * 组成:
9
- * 1) parseMemoryItemsPre —— 按 `<!-- memory:mem_<32hex> -->` 锚点切分记忆条目
10
- * 2) extractL0Pre —— 单条记忆的 L0 抽取(三级 fallback,见下)
11
- * 3) buildL0IndexPre —— 文件级 L0 索引(确定性排序)
12
- *
13
- * L0 抽取优先级(零 LLM,纯解析):
14
- * ① `## 主题块标题` —— 已由写入侧概括,质量最好(去掉 `(HH:MM)` 后缀)
15
- * ② 首个 `- ` 条目首句 —— 去掉 `- HH:MM ` 时间戳前缀后取首句
16
- * ③ 截断兜底 —— 前 maxChars 字符
17
- * 若首句过短(< minChars)则继续并接后续句子,直到达标或触顶。
18
- *
19
- * 边界:纯函数、零 IO、零依赖;对同输入逐字段确定;非法输入 fail closed(返回
20
- * 空串/空数组),不抛异常。所有新增文本 UTF-8 无 BOM。
21
- */
22
-
23
- export const L0_EXTRACT_VERSION = 'l0_extract_v1'
24
-
25
- /** 锚点:`<!-- memory:mem_<32hex> -->`(允许空白浮动)。 */
26
- const MEM_ANCHOR_RE = /<!--\s*memory:(mem_[0-9a-f]{32})\s*-->/g
27
-
28
- /** 行首时间戳:`12:01 ` / `14:5x ` 等(日志条目惯例)。 */
29
- const LEAD_TIME_RE = /^\s*\d{1,2}:\d{2}[a-z]?\s+/
30
-
31
- /** 句末/句读分隔符(含中文)。 */
32
- const SENTENCE_SPLIT_RE = /[。;;!!??\n]/
33
-
34
- /** 主题块标题后缀:`(12:02)` 或 `(12:02)`。 */
35
- const HEADING_SUFFIX_RE = /\s*[((]\s*\d{1,2}:\d{2}\s*[))]\s*$/
36
-
37
- export const L0_DEFAULTS = Object.freeze({
38
- maxChars: 160,
39
- minChars: 15,
40
- hardChars: 480,
41
- })
42
-
43
- const clean = (s) => String(s == null ? '' : s).replace(/\u0000/g, '').trim()
44
-
45
- /**
46
- * 按锚点切分记忆条目。
47
- *
48
- * 约定:锚点标记**其后**的内容(实测文件结构为 `<!-- A -->内容A<!-- B -->内容B`),
49
- * 因此第 i 个内容对应第 i 个锚点。锚点之前的游离内容(若有)归入 `preamble`。
50
- *
51
- * @param {string} text 文件内容
52
- * @returns {{items: Array<{id:string, body:string}>, preamble: string, anchors: number}}
53
- */
54
- export function parseMemoryItemsPre(text) {
55
- const src = typeof text === 'string' ? text : ''
56
- const items = []
57
- if (!src) return { items, preamble: '', anchors: 0 }
58
-
59
- MEM_ANCHOR_RE.lastIndex = 0
60
- const marks = []
61
- let m
62
- while ((m = MEM_ANCHOR_RE.exec(src)) !== null) {
63
- marks.push({ id: m[1], start: m.index, end: m.index + m[0].length })
64
- if (m.index === MEM_ANCHOR_RE.lastIndex) MEM_ANCHOR_RE.lastIndex++
65
- }
66
- if (!marks.length) return { items, preamble: clean(src), anchors: 0 }
67
-
68
- for (let i = 0; i < marks.length; i++) {
69
- const from = marks[i].end
70
- const to = i + 1 < marks.length ? marks[i + 1].start : src.length
71
- const body = clean(src.slice(from, to))
72
- if (body) items.push({ id: marks[i].id, body })
73
- }
74
- return { items, preamble: clean(src.slice(0, marks[0].start)), anchors: marks.length }
75
- }
76
-
77
- /**
78
- * 抽取单条记忆的 L0。纯函数,永不抛异常。
79
- *
80
- * @param {string} body 条目正文
81
- * @param {{maxChars?:number, minChars?:number}} opts
82
- * @returns {{l0: string, source: 'heading'|'firstSentence'|'truncate'|'empty'}}
83
- */
84
- export function extractL0Pre(body, opts = {}) {
85
- const maxChars = Math.max(16, Number(opts.maxChars) || L0_DEFAULTS.maxChars)
86
- const minChars = Math.max(0, Number(opts.minChars) || L0_DEFAULTS.minChars)
87
- const text = clean(body)
88
- if (!text) return { l0: '', source: 'empty' }
89
-
90
- const lines = text.split(/\r?\n/)
91
-
92
- // ① 主题块标题
93
- for (const line of lines) {
94
- const h = /^\s{0,3}#{1,6}\s+(.+?)\s*$/.exec(line)
95
- if (h) {
96
- const title = clean(h[1]).replace(HEADING_SUFFIX_RE, '')
97
- if (title) return { l0: cut(title, maxChars), source: 'heading' }
98
- }
99
- }
100
-
101
- // ② 首个 `- ` 条目:先取首句,过短再并接(并接时剥列表标记与时间戳)
102
- for (const line of lines) {
103
- const b = /^\s*[-*+]\s+(.+?)\s*$/.exec(line)
104
- if (!b) continue
105
- let s = clean(b[1]).replace(LEAD_TIME_RE, '')
106
- if (!s) continue
107
- const first = clean(s.split(SENTENCE_SPLIT_RE)[0])
108
- s = growToMin(first || s, text, minChars, maxChars)
109
- return { l0: cut(s, maxChars), source: 'firstSentence' }
110
- }
111
-
112
- // ③ 兜底:正文截断
113
- const flat = clean(text.replace(/\s+/g, ' '))
114
- return { l0: cut(flat, maxChars), source: 'truncate' }
115
- }
116
-
117
- /** 构建文件级 L0 索引(按 id 升序,确定性)。 */
118
- export function buildL0IndexPre(text, opts = {}) {
119
- const { items } = parseMemoryItemsPre(text)
120
- const out = items.map((it) => {
121
- const r = extractL0Pre(it.body, opts)
122
- return { id: it.id, l0: r.l0, source: r.source, chars: r.l0.length, bodyChars: it.body.length }
123
- })
124
- out.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0))
125
- return out
126
- }
127
-
128
- // ---------- 内部工具 ----------
129
-
130
- function cut(s, n) {
131
- if (s.length <= n) return s
132
- return s.slice(0, Math.max(1, n - 1)) + '…'
133
- }
134
-
135
- /** 首句过短时,并接后续句子直到 minChars 或 maxChars。每句先剥列表标记与时间戳前缀。 */
136
- function growToMin(first, full, minChars, maxChars) {
137
- if (first.length >= minChars) return first
138
- const flat = clean(full.replace(/\s+/g, ' '))
139
- if (!flat || flat.length <= first.length) return first
140
- const parts = flat.split(SENTENCE_SPLIT_RE)
141
- .map((x) => clean(String(x).replace(/^\s*[-*+]\s+/, '').replace(LEAD_TIME_RE, '')))
142
- .filter(Boolean)
143
- let acc = ''
144
- for (const p of parts) {
145
- acc = acc ? acc + '。' + p : p
146
- if (acc.length >= minChars || acc.length >= maxChars) break
147
- }
148
- return acc || first
149
- }
1
+ /**
2
+ * L0 抽取纯核心(l0_extract_v1)—— 分层语义唤回的地基。
3
+ *
4
+ * 2026-09-08 建立。目的:为每条记忆生成廉价摘要(L0),使检索可先在小空间
5
+ * 收敛候选,再按 id 下钻原文,从而把 token 开销与索引构建成本降约一个数量级
6
+ * (实测:条目平均 814 字符 → L0 约 118 字符,压缩比 6.9:1)。
7
+ *
8
+ * 组成:
9
+ * 1) parseMemoryItemsPre —— 按 `<!-- memory:mem_<32hex> -->` 锚点切分记忆条目
10
+ * 2) extractL0Pre —— 单条记忆的 L0 抽取(三级 fallback,见下)
11
+ * 3) buildL0IndexPre —— 文件级 L0 索引(确定性排序)
12
+ * 4) classifyLayerPre —— 来源 → 分层归属(2026-09-14,三层契约 C1,见下)
13
+ *
14
+ * 分层归属与状态(2026-09-14 · THREE-LAYER-CONTRACT §2 / C1):
15
+ * `layer` ∈ { user | project | log | reflection | whiteboard },由**来源标识**判定
16
+ * (userMemoryPath → user;workspaceMemoryPath 与旧版 {ws}/.dsh-memory/MEMORY.md → project;
17
+ * 日志(todayLogPath / 历史日志) → log;reflections/ → reflection;
18
+ * handoff/PLAN.md 与交接账本 → whiteboard)。
19
+ * `status` ∈ { current | superseded | retracted },本轮**只输出 current**(写入侧见
20
+ * buildL0IndexPre 内 TODO)。
21
+ *
22
+ * L0 抽取优先级(零 LLM,纯解析):
23
+ * ① `## 主题块标题` —— 已由写入侧概括,质量最好(去掉 `(HH:MM)` 后缀)
24
+ * ② 首个 `- ` 条目首句 —— 去掉 `- HH:MM ` 时间戳前缀后取首句
25
+ * ③ 截断兜底 —— 前 maxChars 字符
26
+ * 若首句过短(< minChars)则继续并接后续句子,直到达标或触顶。
27
+ *
28
+ * 边界:纯函数、零 IO、零依赖;对同输入逐字段确定;非法输入 fail closed(返回
29
+ * 空串/空数组),不抛异常。所有新增文本 UTF-8 无 BOM。
30
+ */
31
+
32
+ export const L0_EXTRACT_VERSION = 'l0_extract_v1'
33
+
34
+ /** 锚点:`<!-- memory:mem_<32hex> -->`(允许空白浮动)。 */
35
+ const MEM_ANCHOR_RE = /<!--\s*memory:(mem_[0-9a-f]{32})\s*-->/g
36
+
37
+ /** 行首时间戳:`12:01 ` / `14:5x ` 等(日志条目惯例)。 */
38
+ const LEAD_TIME_RE = /^\s*\d{1,2}:\d{2}[a-z]?\s+/
39
+
40
+ /** 句末/句读分隔符(含中文)。 */
41
+ const SENTENCE_SPLIT_RE = /[。;;!!??\n]/
42
+
43
+ /** 主题块标题后缀:`(12:02)` 或 `(12:02)`。 */
44
+ const HEADING_SUFFIX_RE = /\s*[((]\s*\d{1,2}:\d{2}\s*[))]\s*$/
45
+
46
+ /**
47
+ * ★M2.5a(2026-09-18):**退化标题判据**。
48
+ *
49
+ * 规则① 原先假设「有标题 ⇒ 标题即摘要」,该假设**只对日志成立**:
50
+ * - 日志 `## 主题(12:02)` → 去掉时间后缀 = 真摘要 ✅
51
+ * - **项目/用户级笔记 `## 2026-09-17`(纯日期)→ 去后缀无效 ⇒ L0 = 一个日期** ❌
52
+ *
53
+ * 实测(本机真数据,`artifacts/_probe-l0-quality.mjs`):
54
+ * project-notes 16/16 废(100%)、user-notes 46/50 废(92%)、
55
+ * log 449 条 0 废、reflection 21 条 0 废。
56
+ * ⇒ 笔记层(**结论层**)在语义臂里向量彼此几乎相同 ⇒ **几乎不可检索**。
57
+ *
58
+ * 判据:标题若「不携带可检索语义」(纯日期/纯时间/纯符号数字/序号/短代号)
59
+ * ⇒ **不采信,继续下探到规则②**(首个 `- ` 条目首句——那才是笔记的真实内容)。
60
+ *
61
+ * 边界:只拦**明显退化**的形态,绝不拦正常标题(宁可漏判,不可误伤)。
62
+ */
63
+ const DEGENERATE_HEADING_RES = Object.freeze([
64
+ /^\d{4}[-/.]\d{1,2}[-/.]\d{1,2}$/, // 2026-09-17 / 2026/9/17
65
+ /^\d{4}年\d{1,2}月(\d{1,2}日)?$/, // 2026年9月17日
66
+ /^\d{1,2}[-/.]\d{1,2}[-/.]\d{2,4}$/, // 09-17 / 9.17.2026
67
+ /^\d{1,2}:\d{2}(:\d{2})?$/, // 12:02
68
+ /^[\d\s\-/.·、_]+$/, // 纯数字/符号
69
+ /^第?\s*\d+\s*(章|节|部分|阶段|步|次|条|天|周|月|年)?$/, // 第3节 / 3
70
+ /^[A-Za-z]{0,3}\d+(\.\d+)*$/, // v1 / P3 / v1.2.3
71
+ ])
72
+
73
+ /** 标题是否退化(不携带可检索语义)。 */
74
+ function isDegenerateHeading(title) {
75
+ const t = clean(title)
76
+ if (!t) return true
77
+ // ★ 只按**形态**判,不按长度判。
78
+ // 教训(2026-09-18,被既有套件抓出):曾加 `t.length < 4` 作为"过短即退化"的判据,
79
+ // 结果误伤 `主题甲`(3 个 CJK 字符,是合法标题)⇒ 标题被拒 ⇒ 落到规则② ⇒
80
+ // 把正文里的隐私标记当成了 L0(`smoke-test-l0-index.mjs` 的"索引不得含原文"断言真红)。
81
+ // 中文标题短而有效是常态,**长度不是质量信号**。
82
+ // 纪律:宁可漏判(退化标题照旧被当摘要),不可误伤(合法标题被拒而拉入正文)。
83
+ return DEGENERATE_HEADING_RES.some((re) => re.test(t))
84
+ }
85
+
86
+ /** 行级标题正则(与规则① 同一口径,供剥标题用)。 */
87
+ const HEADING_LINE_RE = /^\s{0,3}#{1,6}\s+(.+?)\s*$/
88
+
89
+ export const L0_DEFAULTS = Object.freeze({
90
+ maxChars: 160,
91
+ minChars: 15,
92
+ hardChars: 480,
93
+ })
94
+
95
+ /** 分层取值域(契约 §2 · I4:每条条目必须带 layer)。 */
96
+ export const L0_LAYERS = Object.freeze(['user', 'project', 'log', 'reflection', 'whiteboard'])
97
+
98
+ /** 状态取值域(契约 §2 · I4/I5)。 */
99
+ export const L0_STATUSES = Object.freeze(['current', 'superseded', 'retracted'])
100
+
101
+ /** 无来源信息时的兜底层:`log`。L0 语料主体是日志;且按契约 §2 排序 log 优先级最低——
102
+ * 判错只会**降权**,不会把低价值内容顶到高价值之前(安全侧默认)。 */
103
+ export const L0_DEFAULT_LAYER = 'log'
104
+
105
+ /** 层级契约版本(与 L0_EXTRACT_VERSION 分开:后者是抽取算法版本,已有断言锁定,不动)。 */
106
+ export const L0_LAYER_VERSION = 'l0_layer_v1'
107
+
108
+ const clean = (s) => String(s == null ? '' : s).replace(/\u0000/g, '').trim()
109
+
110
+ /** 契约 §2 提到的来源字段名 / 常用路径键名 → 层(调用方直接手上有 resolvePaths 的键)。 */
111
+ const LAYER_TOKENS = Object.freeze({
112
+ usermemorypath: 'user', userfile: 'user',
113
+ workspacememorypath: 'project', notespath: 'project',
114
+ todaylogpath: 'log', logpath: 'log',
115
+ reflectionpath: 'reflection', reflectdir: 'reflection',
116
+ whiteboardpath: 'whiteboard', planpath: 'whiteboard', handoffpath: 'whiteboard', handoffdir: 'whiteboard',
117
+ })
118
+
119
+ const WS_ROOT_RE = /(^|\/)memory\/workspaces(\/|$)/ // 集中式记忆根:<dshHome>/memory/workspaces/<wsKey>/
120
+ const WS_MARKER_RE = /(^|\/)\.dsh-memory(\/|$)/ // 旧版分散结构:{ws}/.dsh-memory/
121
+ const USER_ROOT_RE = /(^|\/)\.dsh\/memory(\/|$)/ // 用户级记忆根:~/.dsh/memory/
122
+ const DATE_FILE_RE = /^\d{4}-\d{2}-\d{2}\.md$/
123
+
124
+ /**
125
+ * 来源标识 → 分层归属(契约 §2 判定规则)。纯字符串规则、零 IO、永不抛异常。
126
+ *
127
+ * 接受三种入参:① 层名本身(`'project'`);② 契约/路径键名(`'workspaceMemoryPath'`);
128
+ * ③ 文件或目录路径(`'…/workspaces/--x--/reflections/2026-09-13.md'`,Windows 反斜杠同样识别)。
129
+ * 判不出来返回 `null`(调用方回退 L0_DEFAULT_LAYER),不猜。
130
+ *
131
+ * @param {string} source 层名 / 键名 / 路径
132
+ * @returns {'user'|'project'|'log'|'reflection'|'whiteboard'|null}
133
+ */
134
+ export function classifyLayerPre(source) {
135
+ const raw = clean(source)
136
+ if (!raw) return null
137
+ const s = raw.replace(/\\/g, '/').toLowerCase()
138
+ if (L0_LAYERS.includes(s)) return s
139
+ const token = LAYER_TOKENS[s]
140
+ if (token) return token
141
+
142
+ const base = s.slice(s.lastIndexOf('/') + 1)
143
+
144
+ // ① 白板:handoff/ 目录(PLAN.md 与 handoff-<ts>.md 账本都在其下)
145
+ if (/(^|\/)handoff(\/|$)/.test(s) || base === 'plan.md' || /^handoff[-_]\d{8}/.test(base) || /账本|白板/.test(s)) return 'whiteboard'
146
+ // ② 反思:reflections/ 目录。文件名同样是 YYYY-MM-DD.md,必须先于日志判定
147
+ if (/(^|\/)reflections?(\/|$)/.test(s) || /reflection/.test(base) || /反思/.test(s)) return 'reflection'
148
+ // ③ 日志:日期文件名 或 logs/ 目录
149
+ if (DATE_FILE_RE.test(base) || /(^|\/)logs?(\/|$)/.test(s) || /日志/.test(s)) return 'log'
150
+ // ④ 项目笔记:工作区记忆根(集中式 / 旧版)下的 MEMORY.md
151
+ if ((WS_ROOT_RE.test(s) || WS_MARKER_RE.test(s)) && /^(memory|notes)\.md$/.test(base)) return 'project'
152
+ // ⑤ 用户级:用户记忆根下的文件,或带目录的 MEMORY.md(④ 已排除工作区侧)
153
+ if (USER_ROOT_RE.test(s)) return 'user'
154
+ if (s.includes('/') && /^(memory|calendar)\.md$/.test(base)) return 'user'
155
+ return null
156
+ }
157
+
158
+ /** R4-B(2026-09-18):分层**呈现**用的层序(最高优先在前)。
159
+ *
160
+ * 与 `recall-fusion.js:FUSION_LAYER_ORDER_V1` / `tier-layer-inject.js:TIER_LAYER_ORDER_V1`
161
+ * **同序但独立声明** —— 本仓既有约定:跨模块不共享同一常量对象,避免一处改动静默改变另一处语义;
162
+ * 三者相等由断言锁定(见 smoke-test-r4-recall-layers-pre.mjs)。 */
163
+ export const L0_LAYER_DISPLAY_ORDER_V1 = Object.freeze(['project', 'whiteboard', 'user', 'reflection', 'log'])
164
+
165
+ /** R4-B:层 → 呈现标题。措辞刻意让「结论」与「流水」一眼可分 —— 这正是检索区分度问题的靶心:
166
+ * 语义臂内部不分层(`index.js` 纯分数 sort)时,模型看到的 MEMORY.md(结论)与 2026-09-xx.md(流水)
167
+ * 在视觉上完全同级,含金量被数量淹没。 */
168
+ export const L0_LAYER_LABELS_V1 = Object.freeze({
169
+ project: '结论层 · 项目笔记',
170
+ user: '结论层 · 用户级记忆',
171
+ whiteboard: '结论层 · 白板/账本',
172
+ reflection: '反思层 · 每日反思',
173
+ log: '流水层 · 每日日志',
174
+ })
175
+
176
+ /** 判不出层时的兜底标题:**不猜层**,如实说"未分层",且固定排在最后(避免给出错误的层次暗示)。 */
177
+ export const L0_LAYER_UNKNOWN_LABEL_V1 = '未分层'
178
+
179
+ /**
180
+ * R4-B(2026-09-18):把检索命中**按层分组**(**只改呈现,不改排序**)。
181
+ *
182
+ * 硬契约(与本仓"回滚必须逐字节相同"纪律对齐):
183
+ * ① **不增删条目**:输出各组条目总数 = 入参长度,且**层内保持入参原相对顺序**
184
+ * ⇒ 调用方无需改排序;`#finalRank` 标签仍在行内,排序信息可完全还原;
185
+ * ② 组顺序 = `L0_LAYER_DISPLAY_ORDER_V1`;判不出层的固定归**末组**;
186
+ * ③ 空入参 / 非数组 / 取层函数抛错 → **永不抛**(fail-soft:呈现层失败绝不打断检索)。
187
+ *
188
+ * 调用方职责:**只有一组时不打标题** ⇒ 单一层(如纯日志命中)的输出与旧版逐字节相同。
189
+ *
190
+ * @param {Array} items 命中条目
191
+ * @param {(item:any)=>string} layerOf 取层函数(返回值经 classifyLayerPre 归一)
192
+ * @returns {Array<{layer:string,label:string,items:Array}>}
193
+ */
194
+ export function groupL0ByLayerPre(items, layerOf) {
195
+ const groups = []
196
+ try {
197
+ if (!Array.isArray(items) || !items.length) return groups
198
+ const fn = typeof layerOf === 'function' ? layerOf : () => ''
199
+ /** @type {Map<string, Array>} 层 → 条目(插入序即入参相对序) */
200
+ const byLayer = new Map()
201
+ for (const it of items) {
202
+ let layer = null
203
+ try { layer = classifyLayerPre(fn(it)) } catch (_) { layer = null }
204
+ const key = layer || ''
205
+ if (!byLayer.has(key)) byLayer.set(key, [])
206
+ byLayer.get(key).push(it)
207
+ }
208
+ // 已知层按契约层序;未知层紧随其后,保持首次出现序。
209
+ // ★注:`classifyLayerPre` 的返回值域是**闭合词表**(L0_LAYERS 或 null),故下面
210
+ // `l && ...` 分支**当前不可达** —— 保留它是有意的前向兼容:将来 L0_LAYERS 扩容时,
211
+ // 旧版调用方不会把新层误并入"未分层",而是照实单独成组。
212
+ // (变异演示已证实:改这一段不会让任何断言变红 —— 它确实不参与当下语义。)
213
+ const ordered = L0_LAYER_DISPLAY_ORDER_V1.filter((l) => byLayer.has(l))
214
+ for (const l of byLayer.keys()) if (l && ordered.indexOf(l) === -1) ordered.push(l)
215
+ for (const l of ordered) {
216
+ groups.push({ layer: l, label: L0_LAYER_LABELS_V1[l] || l, items: byLayer.get(l) })
217
+ }
218
+ // 未分层固定末组(不猜层)—— **不走通用流程**:它不属层词表,也不该排在结论层之前。
219
+ if (byLayer.has('')) {
220
+ groups.push({ layer: '', label: L0_LAYER_UNKNOWN_LABEL_V1, items: byLayer.get('') })
221
+ }
222
+ } catch (_) { /* fail-soft:分组失败即降级为"无分组",调用方按单组处理,检索不受影响 */ }
223
+ return groups
224
+ }
225
+
226
+ /**
227
+ * 三层契约 I5 的**检索侧过滤谓词**:只有 `status === 'current'` 的记录可进入检索与注入。
228
+ *
229
+ * 缺席 `status` 视为 `current`(向后兼容 C1 之前的记录);`superseded` / `retracted` 一律挡下。
230
+ * 做成导出的小函数是为了能被单测直接断言(验收要求"能失败的断言")。
231
+ *
232
+ * @param {{status?:string}|null|undefined} record 记录(或任何带 status 的对象)
233
+ * @returns {boolean} 是否可作为 current 使用
234
+ */
235
+ export function isCurrentPre(record) {
236
+ if (!record || typeof record !== 'object') return true
237
+ const st = record.status
238
+ if (st === undefined || st === null || st === '') return true
239
+ return st === 'current'
240
+ }
241
+
242
+ /**
243
+ * ★R4-A(2026-09-19 定稿):**检索侧**准入谓词 —— I5 的**收窄**版。
244
+ *
245
+ * ── 变更依据(用户两次修正后定稿)──────────────────────────────────
246
+ * 原 I5(`THREE-LAYER-CONTRACT.md:183`):「非 `current` 的条目在**检索结果与注入内容两处**都被过滤」。
247
+ *
248
+ * ① 用户第一次修正:「**返回但标记是正确的**」⇒ 检索侧改为放行 + 标记。
249
+ * ② 用户第二次修正(**推翻 agent 的"retracted 继续硬挡"方案**):
250
+ * 「这个 retracted **不是过滤掉**……**并不是挡,我感觉是备注**。
251
+ * 因为比如说你之前踩过 3 次的那个坑,如果你不记住这个教训的话,你还会再踩。」
252
+ *
253
+ * ⇒ **三态一律返回、一律标记**。agent 原方案按「检索视角」分(过时的别干扰判断);
254
+ * 用户按「**学习视角**」分(**做错的事恰恰最该被记住**)。
255
+ * 对记忆系统而言后者才是目的:`retracted` 不是垃圾数据,它是**一条教训**——
256
+ * 把它藏起来 = **系统性遗忘自己的错误**,正是「还会再踩」的成因。
257
+ *
258
+ * ⇒ **本谓词不再过滤任何已知 status**,只对**未知值 fail-closed**
259
+ * (防将来新增枚举时静默放行 —— 枚举类常量必须配断言兜底,本仓纪律)。
260
+ *
261
+ * ── 与注入侧的分工(本谓词只用于检索侧)─────────────────────────
262
+ * 注入侧仍用 `isCurrentPre`:注入是**常驻目录**(B0 仅 800 token),
263
+ * 拿常驻预算装过时条目会挤掉现行结论;检索是**按需**的,装一条带警告的过时结论划算。
264
+ * ⇒ **两处判据不同是有意为之**,不是漏改(`tier-layer-inject.js` 继续 import `isCurrentPre`)。
265
+ *
266
+ * @param {{status?:string}|null|undefined} record 记录(或任何带 status 的对象)
267
+ * @returns {boolean} 是否可进入检索结果
268
+ */
269
+ export function isRetrievablePre(record) {
270
+ if (!record || typeof record !== 'object') return true
271
+ const st = record.status
272
+ if (st === undefined || st === null || st === '') return true
273
+ // 已知三态一律放行(含 retracted —— 它是教训,不是垃圾);未知值 fail-closed。
274
+ return L0_STATUSES.includes(st)
275
+ }
276
+
277
+ /** 检索侧标记语的取值域(枚举类常量须配断言兜底 —— 本仓纪律)。 */
278
+ export const L0_SUPERSEDED_MARK_V1 = '⚠已作废'
279
+ export const L0_RETRACTED_MARK_V1 = '⚠已撤回'
280
+
281
+ /**
282
+ * 为「返回但标记」生成**呈现后缀**(R4-A 的可见面)。
283
+ *
284
+ * 契约(**逐字节向后兼容**是硬约束):
285
+ * `current` / 缺 status / 未知 / 非对象 ⇒ 空串(旧行为零变化)
286
+ * `superseded` ⇒ ` ⚠已作废(已被 mem_<32hex> 取代)`
287
+ * `retracted` ⇒ ` ⚠已撤回(原因:<reason>;更正见 mem_<32hex>)`
288
+ *
289
+ * `reason` 才是「教训」的正文,比 status 本身有价值(用户第二次修正的要点)。
290
+ * id 只认 `/^mem_[0-9a-f]{32}$/`,不合法一律丢弃 —— 防止把任意文本拼进检索呈现。
291
+ *
292
+ * @param {{status?:string, supersededBy?:string, reason?:string}} record
293
+ * @returns {string} 追加到条目末尾的后缀(含前导空格;无需标记时为空串)
294
+ */
295
+ export function supersededMarkPre(record) {
296
+ try {
297
+ if (!record || typeof record !== 'object') return ''
298
+ const idOf = (v) => (/^mem_[0-9a-f]{32}$/.test(String(v || '').trim()) ? String(v).trim() : '')
299
+ if (record.status === 'superseded') {
300
+ const safe = idOf(record.supersededBy)
301
+ return ' ' + L0_SUPERSEDED_MARK_V1 + (safe ? '(已被 ' + safe + ' 取代)' : '(已被更新结论取代)')
302
+ }
303
+ if (record.status === 'retracted') {
304
+ const safe = idOf(record.supersededBy)
305
+ const reason = record.reason ? String(record.reason).replace(/[\r\n]+/g, ' ').trim().slice(0, 80) : ''
306
+ const tail = (reason ? '原因:' + reason + ';' : '') + (safe ? '更正见 ' + safe : '已被撤回')
307
+ return ' ' + L0_RETRACTED_MARK_V1 + '(' + tail + ')'
308
+ }
309
+ return ''
310
+ } catch (_) { return '' } // fail-soft:标记失败绝不影响检索
311
+ }
312
+
313
+ /**
314
+ * 按锚点切分记忆条目。
315
+ *
316
+ * 约定:锚点标记**其后**的内容(实测文件结构为 `<!-- A -->内容A<!-- B -->内容B`),
317
+ * 因此第 i 个内容对应第 i 个锚点。锚点之前的游离内容(若有)归入 `preamble`。
318
+ *
319
+ * @param {string} text 文件内容
320
+ * @returns {{items: Array<{id:string, body:string}>, preamble: string, anchors: number}}
321
+ */
322
+ export function parseMemoryItemsPre(text) {
323
+ const src = typeof text === 'string' ? text : ''
324
+ const items = []
325
+ if (!src) return { items, preamble: '', anchors: 0 }
326
+
327
+ MEM_ANCHOR_RE.lastIndex = 0
328
+ const marks = []
329
+ let m
330
+ while ((m = MEM_ANCHOR_RE.exec(src)) !== null) {
331
+ marks.push({ id: m[1], start: m.index, end: m.index + m[0].length })
332
+ if (m.index === MEM_ANCHOR_RE.lastIndex) MEM_ANCHOR_RE.lastIndex++
333
+ }
334
+ if (!marks.length) return { items, preamble: clean(src), anchors: 0 }
335
+
336
+ for (let i = 0; i < marks.length; i++) {
337
+ const from = marks[i].end
338
+ const to = i + 1 < marks.length ? marks[i + 1].start : src.length
339
+ const body = clean(src.slice(from, to))
340
+ if (body) items.push({ id: marks[i].id, body })
341
+ }
342
+ return { items, preamble: clean(src.slice(0, marks[0].start)), anchors: marks.length }
343
+ }
344
+
345
+ /**
346
+ * 抽取单条记忆的 L0。纯函数,永不抛异常。
347
+ *
348
+ * @param {string} body 条目正文
349
+ * @param {{maxChars?:number, minChars?:number}} opts
350
+ * @returns {{l0: string, source: 'heading'|'firstSentence'|'truncate'|'empty'}}
351
+ */
352
+ export function extractL0Pre(body, opts = {}) {
353
+ const maxChars = Math.max(16, Number(opts.maxChars) || L0_DEFAULTS.maxChars)
354
+ const minChars = Math.max(0, Number(opts.minChars) || L0_DEFAULTS.minChars)
355
+ const text = clean(body)
356
+ if (!text) return { l0: '', source: 'empty' }
357
+
358
+ const lines = text.split(/\r?\n/)
359
+
360
+ // ① 主题块标题(★M2.5a:加质量门 —— 退化标题不采信,继续下探)
361
+ for (const line of lines) {
362
+ const h = HEADING_LINE_RE.exec(line)
363
+ if (!h) continue
364
+ const title = clean(h[1]).replace(HEADING_SUFFIX_RE, '')
365
+ if (!title) continue
366
+ // 退化标题(纯日期/纯时间/序号/短代号)不携带可检索语义 ⇒ 跳过,让规则② 取真实内容。
367
+ if (isDegenerateHeading(title)) continue
368
+ return { l0: cut(title, maxChars), source: 'heading' }
369
+ }
370
+
371
+ // ② 首个 `- ` 条目:先取首句,过短再并接(并接时剥列表标记与时间戳)
372
+ for (const line of lines) {
373
+ const b = /^\s*[-*+]\s+(.+?)\s*$/.exec(line)
374
+ if (!b) continue
375
+ let s = clean(b[1]).replace(LEAD_TIME_RE, '')
376
+ if (!s) continue
377
+ const first = clean(s.split(SENTENCE_SPLIT_RE)[0])
378
+ s = growToMin(first || s, text, minChars, maxChars)
379
+ return { l0: cut(s, maxChars), source: 'firstSentence' }
380
+ }
381
+
382
+ // ③ 兜底:正文截断(★M2.5a:先剥标题行 —— 否则退化标题会被压进 L0)
383
+ // 场景:条目既无标题(或标题已退化被跳过)又无 `- ` 列表项时落到此处;
384
+ // 直接压平会把 `## 2026-09-09` 变成 L0 开头(实测 user-notes 修后仍见该形态)。
385
+ const bodyLines = lines.filter((l) => !HEADING_LINE_RE.test(l))
386
+ const flat = clean((bodyLines.length ? bodyLines : lines).join('\n').replace(/\s+/g, ' '))
387
+ return { l0: cut(flat, maxChars), source: 'truncate' }
388
+ }
389
+
390
+ /**
391
+ * 构建文件级 L0 索引(按 id 升序,确定性)。
392
+ *
393
+ * 2026-09-14(三层契约 C1):每条**追加** `layer` + `status` 两个字段——只增不减,
394
+ * 老调用方读 `id / l0 / source / chars / bodyChars` 完全不受影响,签名与调用方式不变。
395
+ * 层归属:`opts.layer`(层名 / 路径 / 键名,经 classifyLayerPre)→ `opts.path` → `L0_DEFAULT_LAYER`。
396
+ *
397
+ * ★R4-A(2026-09-18):新增**可选** `opts.statusOf(id)` 注入解析器 —— 让本模块保持
398
+ * **零 IO 纯函数**(存储格式属写入侧 G3 的决定,本层只透传,不猜存储)。
399
+ * 未注入时行为与从前**逐字节相同**(`status` 恒 `'current'`)⇒ 向后兼容。
400
+ *
401
+ * @param {string} text 文件内容
402
+ * @param {{maxChars?:number, minChars?:number, layer?:string, path?:string,
403
+ * statusOf?:(id:string)=>({status?:string, supersededBy?:string}|null|undefined)}} [opts]
404
+ * @returns {Array<{id:string, l0:string, source:string, chars:number, bodyChars:number, layer:string, status:string, supersededBy?:string}>}
405
+ */
406
+ export function buildL0IndexPre(text, opts = {}) {
407
+ const { items } = parseMemoryItemsPre(text)
408
+ const layer = resolveLayerPre(opts)
409
+ const statusOf = opts && typeof opts.statusOf === 'function' ? opts.statusOf : null
410
+ const out = items.map((it) => {
411
+ const r = extractL0Pre(it.body, opts)
412
+ // C1 写入侧(G3)落盘状态后,由调用方经 `statusOf` 注入;本层只透传。
413
+ // fail-closed 语义在**消费侧**(isRetrievablePre 未知值挡下),此处只做形态净化。
414
+ let status = 'current'
415
+ let supersededBy
416
+ let reason
417
+ if (statusOf) {
418
+ try {
419
+ const st = statusOf(it.id)
420
+ if (st && typeof st === 'object') {
421
+ if (typeof st.status === 'string' && L0_STATUSES.includes(st.status)) status = st.status
422
+ if (typeof st.supersededBy === 'string' && st.supersededBy) supersededBy = st.supersededBy
423
+ // R4-A:撤回原因 —— 这才是「教训」的正文,比 status 本身有价值。
424
+ if (typeof st.reason === 'string' && st.reason) reason = st.reason
425
+ }
426
+ } catch (_) { /* fail-soft:状态解析失败 ⇒ 按 current 处理,绝不影响索引构建 */ }
427
+ }
428
+ const rec = {
429
+ id: it.id, l0: r.l0, source: r.source, chars: r.l0.length, bodyChars: it.body.length,
430
+ layer, status,
431
+ }
432
+ // 只在真的有值时附字段 —— 保证「无状态时」输出与旧版逐字节相同。
433
+ if (supersededBy) rec.supersededBy = supersededBy
434
+ if (reason) rec.reason = reason
435
+ return rec
436
+ })
437
+ out.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0))
438
+ return out
439
+ }
440
+
441
+ /** 本次抽取的层归属:显式层名/路径 > 备用路径键 > 兜底层。判不出不猜,回退 L0_DEFAULT_LAYER。 */
442
+ function resolveLayerPre(opts) {
443
+ const o = opts && typeof opts === 'object' ? opts : {}
444
+ for (const cand of [o.layer, o.path, o.sourcePath]) {
445
+ const hit = classifyLayerPre(cand)
446
+ if (hit) return hit
447
+ }
448
+ return L0_DEFAULT_LAYER
449
+ }
450
+
451
+ // ---------- 内部工具 ----------
452
+
453
+ function cut(s, n) {
454
+ if (s.length <= n) return s
455
+ return s.slice(0, Math.max(1, n - 1)) + '…'
456
+ }
457
+
458
+ /** 首句过短时,并接后续句子直到 minChars 或 maxChars。每句先剥列表标记与时间戳前缀。 */
459
+ function growToMin(first, full, minChars, maxChars) {
460
+ if (first.length >= minChars) return first
461
+ // ★ issue #71 修复(2026-09-19):**保留换行结构**。
462
+ // 旧实现先 `full.replace(/\s+/g, ' ')` 把换行压成空格,**再用含 `\n` 的 SENTENCE_SPLIT_RE 切分**
463
+ // ⇒ `\n` 分支恒不命中(死代码),整段多行正文被当成**一个 part**;
464
+ // 而剥前缀的 `/^\s*[-*+]\s+/` 与 `LEAD_TIME_RE` 都是 `^` 锚定,只剥得掉该 part 的**首个**标记
465
+ // ⇒ 第 2 行起的 `- HH:MM` 原样进入 L0(嵌入/检索输入)。
466
+ // 现改为:只把「行内连续空白」压成单空格,**行界 `\n` 保留** ⇒ 逐行切分、逐 part 剥前缀。
467
+ const flat = clean(full.replace(/[^\S\r\n]+/g, ' ').replace(/\r\n?/g, '\n'))
468
+ if (!flat || flat.length <= first.length) return first
469
+ const parts = flat.split(SENTENCE_SPLIT_RE)
470
+ .map((x) => clean(String(x).replace(/^\s*[-*+]\s+/, '').replace(LEAD_TIME_RE, '')))
471
+ .filter(Boolean)
472
+ let acc = ''
473
+ for (const p of parts) {
474
+ acc = acc ? acc + '。' + p : p
475
+ if (acc.length >= minChars || acc.length >= maxChars) break
476
+ }
477
+ return acc || first
478
+ }