@a9i5k4/dsh-auto-memory 2.5.3 → 3.0.0

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (104) hide show
  1. package/README.md +171 -1
  2. package/README.zh-CN.md +171 -1
  3. package/docs/CONTRIBUTORS.html +471 -0
  4. package/docs/HANDOFF-CRITERIA.md +92 -0
  5. package/docs/INTEGRATION-ANALYSIS.md +350 -348
  6. package/docs/USER-GUIDE.en.md +56 -1
  7. package/docs/USER-GUIDE.zh-CN.md +57 -2
  8. package/docs/internal/ACCEPT-35-LIVE.md +143 -0
  9. package/docs/internal/ACCEPTANCE-20260914.md +90 -0
  10. package/docs/internal/ARCH-REVIEW-BRIEF.md +411 -0
  11. package/docs/internal/ARCH-REVIEW-REQUEST.md +201 -0
  12. package/docs/internal/ARCH-REVIEW-ROUND2.md +169 -0
  13. package/docs/internal/ARCH-REVIEW-ROUND3.md +206 -0
  14. package/docs/internal/AUDIT-WB-GRAPH-FULL-20260916.md +314 -0
  15. package/docs/internal/CONCURRENCY-INVESTIGATION-20260917.md +192 -0
  16. package/docs/internal/CROSS-SESSION-SEARCH-PATH-DECISION.md +72 -0
  17. package/docs/internal/CROSS-SESSION-SEARCH-RESEARCH.md +131 -0
  18. package/docs/internal/DECISIONS-20260914-SESSION.md +269 -0
  19. package/docs/internal/DESIGN-P1-STATE-COMMIT-20260915.md +219 -0
  20. package/docs/internal/DIRECTION-CHECK-WB-GRAPH-20260916.md +132 -0
  21. package/docs/internal/FEEDBACK-TO-DSHAPI-RELAY.md +13 -0
  22. package/docs/internal/GH-DISCUSSION-5732-COMMENT.md +74 -0
  23. package/docs/internal/GPT-ACCEPTANCE-PROMPT-20260916.md +352 -0
  24. package/docs/internal/GPT-REVIEW-PROMPT.md +216 -0
  25. package/docs/internal/GROUP-WEBHOOK-SETUP.md +33 -0
  26. package/docs/internal/KICKOFF-P0.md +254 -0
  27. package/docs/internal/MASTER-PLAN-3.0.md +411 -0
  28. package/docs/internal/MEMORY-MUTATION-AND-INDEX-DESIGN.md +85 -0
  29. package/docs/internal/MERGE-CONFLICT-SCAN-20260914.md +222 -0
  30. package/docs/internal/PENDING-FIXES-20260916.md +289 -0
  31. package/docs/internal/RAG-KARPATHY-PROGRAM.md +229 -0
  32. package/docs/internal/REPORT-P0-NIGHTLY.md +212 -0
  33. package/docs/internal/REPORT-P5-ACCEPTANCE.md +31 -0
  34. package/docs/internal/REPORT-WB-GRAPH-NIGHTLY.md +153 -0
  35. package/docs/internal/REVIEW-WB-GRAPH-SELF.md +81 -0
  36. package/docs/internal/ROADMAP-20260917-WEEK.md +305 -0
  37. package/docs/internal/ROADMAP.md +106 -0
  38. package/docs/internal/RUN-P0-NIGHTLY.md +227 -0
  39. package/docs/internal/S10-CONSTRUCTION-HANDOFF-20260917.md +175 -0
  40. package/docs/internal/S10-GAPS-PLAIN-20260917.md +125 -0
  41. package/docs/internal/SEMANTIC-ARCHITECTURE-SPEC.md +360 -0
  42. package/docs/internal/SESSION-FILE-REPAIR-PROTOCOL.md +90 -0
  43. package/docs/internal/THREE-LAYER-CONTRACT.md +210 -0
  44. package/docs/internal/TODO-BACKLOG.md +263 -142
  45. package/docs/internal/TODO-GRAPH.html +715 -0
  46. package/docs/internal/TODO-GRAPH.html.bak-20260914-v2 +493 -0
  47. package/docs/internal/TODO-GRAPH.html.bak-20260915-alsfix +710 -0
  48. package/docs/internal/TODO-GRAPH.html.bak-20260915-p1 +710 -0
  49. package/docs/internal/TODO-GRAPH.html.bak-20260915-p6a-rev +703 -0
  50. package/docs/internal/TODO-GRAPH.html.bak-20260915-wshint +710 -0
  51. package/docs/internal/TODO-GRAPH.html.bak-20260916-batch +715 -0
  52. package/docs/internal/WB-FORMAT-CONVENTION.md +112 -0
  53. package/docs/internal/WB-GRAPH-DECISIONS-20260914.md +71 -0
  54. package/docs/internal/reviews/CLAIM-VERIFICATION-20260914.md +56 -0
  55. package/docs/internal/reviews/PLAN-gpt6astra-round2-20260914.md +787 -0
  56. package/docs/internal/reviews/REVIEW-gpt6astra-20260914.md +112 -0
  57. package/docs/internal/reviews/ROUND3-REVIEW-INTEGRATION-20260914.md +230 -0
  58. package/docs/prompts/M8-3-enable-verify.md +49 -49
  59. package/lib/acceptance.js +71 -0
  60. package/lib/activation-host.js +90 -9
  61. package/lib/activation-inbox.js +25 -7
  62. package/lib/board-mode.js +30 -0
  63. package/lib/client.js +878 -75
  64. package/lib/context-bridge.js +2 -2
  65. package/lib/context-host.js +70 -6
  66. package/lib/engine-identity.js +149 -0
  67. package/lib/engine-switch.js +247 -0
  68. package/lib/episodic-store.js +11 -10
  69. package/lib/evidence-store.js +2 -2
  70. package/lib/fact-store.js +1 -1
  71. package/lib/fs-retry.js +46 -0
  72. package/lib/index.js +1987 -153
  73. package/lib/intent-clean-safe.js +40 -0
  74. package/lib/intent-clean.js +12 -16
  75. package/lib/l0-extract.js +263 -149
  76. package/lib/l0-index-sync.js +195 -0
  77. package/lib/l0-index.js +349 -239
  78. package/lib/ledger-criteria.js +142 -0
  79. package/lib/m7-index-sync-host.js +65 -4
  80. package/lib/m7-wire.js +3 -3
  81. package/lib/memory-anchor.js +56 -1
  82. package/lib/memory-envelope.js +252 -0
  83. package/lib/memory-hub.js +14 -4
  84. package/lib/memory-mutation.js +246 -0
  85. package/lib/memory-writer.js +204 -24
  86. package/lib/procedure-observation.js +48 -0
  87. package/lib/procedure-store.js +34 -17
  88. package/lib/python-setup.js +1 -1
  89. package/lib/rerank-host.js +160 -0
  90. package/lib/rules-layer.js +261 -0
  91. package/lib/semantic-js.js +15 -0
  92. package/lib/shadow-retrieval.js +3 -3
  93. package/lib/state-commit.js +245 -0
  94. package/lib/subagent-gc.js +4 -8
  95. package/lib/tier-layer-inject.js +650 -0
  96. package/lib/tier0-catalog.js +693 -0
  97. package/lib/water-window.js +263 -186
  98. package/lib/wb-contract.js +495 -0
  99. package/lib/wb-sidecar.js +839 -0
  100. package/lib/ws-overview-rank.js +2 -2
  101. package/package.json +1 -1
  102. package/python/m7_embedding_v1.py +5 -5
  103. package/python/worker_semantic_v1.py +17 -6
  104. package/python/worker_v1.py +38 -4
@@ -0,0 +1,40 @@
1
+ /** Remove runtime envelopes only at unquoted line boundaries; preserve literal examples/code. */
2
+ const tags = new Set(['memory_system', 'system-reminder', 'long_term_memory'])
3
+ export function stripRuntimeIntentPre(text) {
4
+ const lines = String(text == null ? '' : text).split(/\r?\n/)
5
+ const out = [], stack = []
6
+ let fence = null
7
+ for (let line of lines) {
8
+ const trimmed = line.trim()
9
+ if (!stack.length) {
10
+ const f = /^( {0,3})(`{3,}|~{3,})/.exec(line)
11
+ if (f) {
12
+ if (!fence) fence = { char: f[2][0], size: f[2].length }
13
+ else if (f[2][0] === fence.char && f[2].length >= fence.size && /^( {0,3})(`+|~+)\s*$/.test(line)) fence = null
14
+ out.push(line); continue
15
+ }
16
+ if (fence || /^\s*>/.test(line) || /^ {4}/.test(line)) { out.push(line); continue }
17
+ if (/^(?:current runtime context\.|current dsh file policy:)/i.test(trimmed)) continue
18
+ }
19
+ // Consume one or more envelopes at the beginning of an unquoted line.
20
+ // Closing tags can end a prefix split across messages; trailing human text is retained.
21
+ for (;;) {
22
+ if (stack.length) {
23
+ const token = /<(\/?)(memory_system|system-reminder|long_term_memory)>/.exec(line)
24
+ if (!token) { line = ''; break }
25
+ line = line.slice(token.index + token[0].length)
26
+ if (!token[1]) stack.push(token[2])
27
+ else if (stack[stack.length - 1] === token[2]) stack.pop()
28
+ continue
29
+ }
30
+ const open = /^\s*<(memory_system|system-reminder|long_term_memory)>/.exec(line)
31
+ if (open && tags.has(open[1])) { stack.push(open[1]); line = line.slice(open[0].length); continue }
32
+ const close = /^\s*<\/(memory_system|system-reminder|long_term_memory)>/.exec(line)
33
+ if (close) { line = line.slice(close[0].length); continue }
34
+ break
35
+ }
36
+ if (line.trim()) out.push(line)
37
+ else if (!stack.length && !trimmed) out.push('')
38
+ }
39
+ return out.join('\n')
40
+ }
@@ -24,26 +24,22 @@
24
24
  * 命名空间:_pre 隔离。UTF-8 无 BOM。
25
25
  */
26
26
 
27
- /** 完整的注入快照块(开闭标签配对)。 */
28
- const SNAPSHOT_BLOCK_RE = /<memory_system>[\s\S]*?<\/memory_system>/g
29
- /** harness 合成前缀(剥离块后若只剩这段,说明整条都是注入)。 */
30
- const RUNTIME_CONTEXT_RE = /^current runtime context\./i
27
+ import { stripRuntimeIntentPre } from './intent-clean-safe.js'
31
28
 
32
29
  /**
33
- * 剥离注入内容,返回剩余的真人文本:
34
- * (a) 完整的 <memory_system>…</memory_system> 块整体移除(可多块);
35
- * (b) 只剩闭合标签(快照被拆条/半截注入)→ 取其后内容;
36
- * (c) 其余原样。
30
+ * 剥离注入内容,返回剩余的真人文本。
31
+ *
32
+ * 2026-09-16(issue #30):改为复用 `stripRuntimeIntentPre` 的**行边界**剥离器。
33
+ * 旧实现用两条正则(整块配对 + `^current runtime context.`)硬套,漏判面很宽:
34
+ * - 只认 `<memory_system>` 一种标签,新标签(如 `<system-reminder>`/`<long_term_memory>`)整条漏过;
35
+ * - 块被拆条时靠"取最后一个闭合标签之后"这一刀切,前真问题后快照的拼法会**丢掉真问题**;
36
+ * - 导语正则只匹配行首,前缀被拼到行中时不生效。
37
+ * 新剥离器按行处理:仅在**未被引用/未被代码块包裹**的普通行上剥离信封标签,
38
+ * 保留行内尾随的真人文(如 `<memory_system>…</memory_system> 我的问题` 只剩"我的问题"),
39
+ * 并保证代码块/引用块里的字面示例不被误删。
37
40
  */
38
41
  export function stripInjectedBlockPre(text) {
39
- let s = String(text == null ? '' : text)
40
- s = s.replace(SNAPSHOT_BLOCK_RE, '')
41
- const marker = '</memory_system>'
42
- const idx = s.lastIndexOf(marker)
43
- if (idx >= 0) s = s.slice(idx + marker.length)
44
- // harness 的 "Current runtime context. …" 导语是整行噪声,且可能出现在任一行
45
- // (快照在前真问题在后 / 真问题在前快照在后,两种拼法都要能剥干净)
46
- return s.split(/\r?\n/).filter((ln) => !RUNTIME_CONTEXT_RE.test(ln.trim())).join('\n')
42
+ return stripRuntimeIntentPre(text)
47
43
  }
48
44
 
49
45
  /** 合成注入消息识别:剥离注入内容后什么都不剩 → 整条都是注入。 */
package/lib/l0-extract.js CHANGED
@@ -1,149 +1,263 @@
1
- /**
2
- * L0 抽取纯核心(l0_extract_v1)—— 分层语义唤回的地基。
3
- *
4
- * 2026-09-08 建立。目的:为每条记忆生成廉价摘要(L0),使检索可先在小空间
5
- * 收敛候选,再按 id 下钻原文,从而把 token 开销与索引构建成本降约一个数量级
6
- * (实测:条目平均 814 字符 → L0 约 118 字符,压缩比 6.9:1)。
7
- *
8
- * 组成:
9
- * 1) parseMemoryItemsPre —— 按 `<!-- memory:mem_<32hex> -->` 锚点切分记忆条目
10
- * 2) extractL0Pre —— 单条记忆的 L0 抽取(三级 fallback,见下)
11
- * 3) buildL0IndexPre —— 文件级 L0 索引(确定性排序)
12
- *
13
- * L0 抽取优先级(零 LLM,纯解析):
14
- * ① `## 主题块标题` —— 已由写入侧概括,质量最好(去掉 `(HH:MM)` 后缀)
15
- * ② 首个 `- ` 条目首句 —— 去掉 `- HH:MM ` 时间戳前缀后取首句
16
- * ③ 截断兜底 —— 前 maxChars 字符
17
- * 若首句过短(< minChars)则继续并接后续句子,直到达标或触顶。
18
- *
19
- * 边界:纯函数、零 IO、零依赖;对同输入逐字段确定;非法输入 fail closed(返回
20
- * 空串/空数组),不抛异常。所有新增文本 UTF-8 无 BOM。
21
- */
22
-
23
- export const L0_EXTRACT_VERSION = 'l0_extract_v1'
24
-
25
- /** 锚点:`<!-- memory:mem_<32hex> -->`(允许空白浮动)。 */
26
- const MEM_ANCHOR_RE = /<!--\s*memory:(mem_[0-9a-f]{32})\s*-->/g
27
-
28
- /** 行首时间戳:`12:01 ` / `14:5x ` 等(日志条目惯例)。 */
29
- const LEAD_TIME_RE = /^\s*\d{1,2}:\d{2}[a-z]?\s+/
30
-
31
- /** 句末/句读分隔符(含中文)。 */
32
- const SENTENCE_SPLIT_RE = /[。;;!!??\n]/
33
-
34
- /** 主题块标题后缀:`(12:02)` 或 `(12:02)`。 */
35
- const HEADING_SUFFIX_RE = /\s*[((]\s*\d{1,2}:\d{2}\s*[))]\s*$/
36
-
37
- export const L0_DEFAULTS = Object.freeze({
38
- maxChars: 160,
39
- minChars: 15,
40
- hardChars: 480,
41
- })
42
-
43
- const clean = (s) => String(s == null ? '' : s).replace(/\u0000/g, '').trim()
44
-
45
- /**
46
- * 按锚点切分记忆条目。
47
- *
48
- * 约定:锚点标记**其后**的内容(实测文件结构为 `<!-- A -->内容A<!-- B -->内容B`),
49
- * 因此第 i 个内容对应第 i 个锚点。锚点之前的游离内容(若有)归入 `preamble`。
50
- *
51
- * @param {string} text 文件内容
52
- * @returns {{items: Array<{id:string, body:string}>, preamble: string, anchors: number}}
53
- */
54
- export function parseMemoryItemsPre(text) {
55
- const src = typeof text === 'string' ? text : ''
56
- const items = []
57
- if (!src) return { items, preamble: '', anchors: 0 }
58
-
59
- MEM_ANCHOR_RE.lastIndex = 0
60
- const marks = []
61
- let m
62
- while ((m = MEM_ANCHOR_RE.exec(src)) !== null) {
63
- marks.push({ id: m[1], start: m.index, end: m.index + m[0].length })
64
- if (m.index === MEM_ANCHOR_RE.lastIndex) MEM_ANCHOR_RE.lastIndex++
65
- }
66
- if (!marks.length) return { items, preamble: clean(src), anchors: 0 }
67
-
68
- for (let i = 0; i < marks.length; i++) {
69
- const from = marks[i].end
70
- const to = i + 1 < marks.length ? marks[i + 1].start : src.length
71
- const body = clean(src.slice(from, to))
72
- if (body) items.push({ id: marks[i].id, body })
73
- }
74
- return { items, preamble: clean(src.slice(0, marks[0].start)), anchors: marks.length }
75
- }
76
-
77
- /**
78
- * 抽取单条记忆的 L0。纯函数,永不抛异常。
79
- *
80
- * @param {string} body 条目正文
81
- * @param {{maxChars?:number, minChars?:number}} opts
82
- * @returns {{l0: string, source: 'heading'|'firstSentence'|'truncate'|'empty'}}
83
- */
84
- export function extractL0Pre(body, opts = {}) {
85
- const maxChars = Math.max(16, Number(opts.maxChars) || L0_DEFAULTS.maxChars)
86
- const minChars = Math.max(0, Number(opts.minChars) || L0_DEFAULTS.minChars)
87
- const text = clean(body)
88
- if (!text) return { l0: '', source: 'empty' }
89
-
90
- const lines = text.split(/\r?\n/)
91
-
92
- // ① 主题块标题
93
- for (const line of lines) {
94
- const h = /^\s{0,3}#{1,6}\s+(.+?)\s*$/.exec(line)
95
- if (h) {
96
- const title = clean(h[1]).replace(HEADING_SUFFIX_RE, '')
97
- if (title) return { l0: cut(title, maxChars), source: 'heading' }
98
- }
99
- }
100
-
101
- // ② 首个 `- ` 条目:先取首句,过短再并接(并接时剥列表标记与时间戳)
102
- for (const line of lines) {
103
- const b = /^\s*[-*+]\s+(.+?)\s*$/.exec(line)
104
- if (!b) continue
105
- let s = clean(b[1]).replace(LEAD_TIME_RE, '')
106
- if (!s) continue
107
- const first = clean(s.split(SENTENCE_SPLIT_RE)[0])
108
- s = growToMin(first || s, text, minChars, maxChars)
109
- return { l0: cut(s, maxChars), source: 'firstSentence' }
110
- }
111
-
112
- // ③ 兜底:正文截断
113
- const flat = clean(text.replace(/\s+/g, ' '))
114
- return { l0: cut(flat, maxChars), source: 'truncate' }
115
- }
116
-
117
- /** 构建文件级 L0 索引(按 id 升序,确定性)。 */
118
- export function buildL0IndexPre(text, opts = {}) {
119
- const { items } = parseMemoryItemsPre(text)
120
- const out = items.map((it) => {
121
- const r = extractL0Pre(it.body, opts)
122
- return { id: it.id, l0: r.l0, source: r.source, chars: r.l0.length, bodyChars: it.body.length }
123
- })
124
- out.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0))
125
- return out
126
- }
127
-
128
- // ---------- 内部工具 ----------
129
-
130
- function cut(s, n) {
131
- if (s.length <= n) return s
132
- return s.slice(0, Math.max(1, n - 1)) + '…'
133
- }
134
-
135
- /** 首句过短时,并接后续句子直到 minChars 或 maxChars。每句先剥列表标记与时间戳前缀。 */
136
- function growToMin(first, full, minChars, maxChars) {
137
- if (first.length >= minChars) return first
138
- const flat = clean(full.replace(/\s+/g, ' '))
139
- if (!flat || flat.length <= first.length) return first
140
- const parts = flat.split(SENTENCE_SPLIT_RE)
141
- .map((x) => clean(String(x).replace(/^\s*[-*+]\s+/, '').replace(LEAD_TIME_RE, '')))
142
- .filter(Boolean)
143
- let acc = ''
144
- for (const p of parts) {
145
- acc = acc ? acc + '。' + p : p
146
- if (acc.length >= minChars || acc.length >= maxChars) break
147
- }
148
- return acc || first
149
- }
1
+ /**
2
+ * L0 抽取纯核心(l0_extract_v1)—— 分层语义唤回的地基。
3
+ *
4
+ * 2026-09-08 建立。目的:为每条记忆生成廉价摘要(L0),使检索可先在小空间
5
+ * 收敛候选,再按 id 下钻原文,从而把 token 开销与索引构建成本降约一个数量级
6
+ * (实测:条目平均 814 字符 → L0 约 118 字符,压缩比 6.9:1)。
7
+ *
8
+ * 组成:
9
+ * 1) parseMemoryItemsPre —— 按 `<!-- memory:mem_<32hex> -->` 锚点切分记忆条目
10
+ * 2) extractL0Pre —— 单条记忆的 L0 抽取(三级 fallback,见下)
11
+ * 3) buildL0IndexPre —— 文件级 L0 索引(确定性排序)
12
+ * 4) classifyLayerPre —— 来源 → 分层归属(2026-09-14,三层契约 C1,见下)
13
+ *
14
+ * 分层归属与状态(2026-09-14 · THREE-LAYER-CONTRACT §2 / C1):
15
+ * `layer` ∈ { user | project | log | reflection | whiteboard },由**来源标识**判定
16
+ * (userMemoryPath → user;workspaceMemoryPath 与旧版 {ws}/.dsh-memory/MEMORY.md → project;
17
+ * 日志(todayLogPath / 历史日志) → log;reflections/ → reflection;
18
+ * handoff/PLAN.md 与交接账本 → whiteboard)。
19
+ * `status` ∈ { current | superseded | retracted },本轮**只输出 current**(写入侧见
20
+ * buildL0IndexPre 内 TODO)。
21
+ *
22
+ * L0 抽取优先级(零 LLM,纯解析):
23
+ * ① `## 主题块标题` —— 已由写入侧概括,质量最好(去掉 `(HH:MM)` 后缀)
24
+ * ② 首个 `- ` 条目首句 —— 去掉 `- HH:MM ` 时间戳前缀后取首句
25
+ * ③ 截断兜底 —— 前 maxChars 字符
26
+ * 若首句过短(< minChars)则继续并接后续句子,直到达标或触顶。
27
+ *
28
+ * 边界:纯函数、零 IO、零依赖;对同输入逐字段确定;非法输入 fail closed(返回
29
+ * 空串/空数组),不抛异常。所有新增文本 UTF-8 无 BOM。
30
+ */
31
+
32
+ export const L0_EXTRACT_VERSION = 'l0_extract_v1'
33
+
34
+ /** 锚点:`<!-- memory:mem_<32hex> -->`(允许空白浮动)。 */
35
+ const MEM_ANCHOR_RE = /<!--\s*memory:(mem_[0-9a-f]{32})\s*-->/g
36
+
37
+ /** 行首时间戳:`12:01 ` / `14:5x ` 等(日志条目惯例)。 */
38
+ const LEAD_TIME_RE = /^\s*\d{1,2}:\d{2}[a-z]?\s+/
39
+
40
+ /** 句末/句读分隔符(含中文)。 */
41
+ const SENTENCE_SPLIT_RE = /[。;;!!??\n]/
42
+
43
+ /** 主题块标题后缀:`(12:02)` 或 `(12:02)`。 */
44
+ const HEADING_SUFFIX_RE = /\s*[((]\s*\d{1,2}:\d{2}\s*[))]\s*$/
45
+
46
+ export const L0_DEFAULTS = Object.freeze({
47
+ maxChars: 160,
48
+ minChars: 15,
49
+ hardChars: 480,
50
+ })
51
+
52
+ /** 分层取值域(契约 §2 · I4:每条条目必须带 layer)。 */
53
+ export const L0_LAYERS = Object.freeze(['user', 'project', 'log', 'reflection', 'whiteboard'])
54
+
55
+ /** 状态取值域(契约 §2 · I4/I5)。 */
56
+ export const L0_STATUSES = Object.freeze(['current', 'superseded', 'retracted'])
57
+
58
+ /** 无来源信息时的兜底层:`log`。L0 语料主体是日志;且按契约 §2 排序 log 优先级最低——
59
+ * 判错只会**降权**,不会把低价值内容顶到高价值之前(安全侧默认)。 */
60
+ export const L0_DEFAULT_LAYER = 'log'
61
+
62
+ /** 层级契约版本(与 L0_EXTRACT_VERSION 分开:后者是抽取算法版本,已有断言锁定,不动)。 */
63
+ export const L0_LAYER_VERSION = 'l0_layer_v1'
64
+
65
+ const clean = (s) => String(s == null ? '' : s).replace(/\u0000/g, '').trim()
66
+
67
+ /** 契约 §2 提到的来源字段名 / 常用路径键名 → 层(调用方直接手上有 resolvePaths 的键)。 */
68
+ const LAYER_TOKENS = Object.freeze({
69
+ usermemorypath: 'user', userfile: 'user',
70
+ workspacememorypath: 'project', notespath: 'project',
71
+ todaylogpath: 'log', logpath: 'log',
72
+ reflectionpath: 'reflection', reflectdir: 'reflection',
73
+ whiteboardpath: 'whiteboard', planpath: 'whiteboard', handoffpath: 'whiteboard', handoffdir: 'whiteboard',
74
+ })
75
+
76
+ const WS_ROOT_RE = /(^|\/)memory\/workspaces(\/|$)/ // 集中式记忆根:<dshHome>/memory/workspaces/<wsKey>/
77
+ const WS_MARKER_RE = /(^|\/)\.dsh-memory(\/|$)/ // 旧版分散结构:{ws}/.dsh-memory/
78
+ const USER_ROOT_RE = /(^|\/)\.dsh\/memory(\/|$)/ // 用户级记忆根:~/.dsh/memory/
79
+ const DATE_FILE_RE = /^\d{4}-\d{2}-\d{2}\.md$/
80
+
81
+ /**
82
+ * 来源标识 → 分层归属(契约 §2 判定规则)。纯字符串规则、零 IO、永不抛异常。
83
+ *
84
+ * 接受三种入参:① 层名本身(`'project'`);② 契约/路径键名(`'workspaceMemoryPath'`);
85
+ * ③ 文件或目录路径(`'…/workspaces/--x--/reflections/2026-09-13.md'`,Windows 反斜杠同样识别)。
86
+ * 判不出来返回 `null`(调用方回退 L0_DEFAULT_LAYER),不猜。
87
+ *
88
+ * @param {string} source 层名 / 键名 / 路径
89
+ * @returns {'user'|'project'|'log'|'reflection'|'whiteboard'|null}
90
+ */
91
+ export function classifyLayerPre(source) {
92
+ const raw = clean(source)
93
+ if (!raw) return null
94
+ const s = raw.replace(/\\/g, '/').toLowerCase()
95
+ if (L0_LAYERS.includes(s)) return s
96
+ const token = LAYER_TOKENS[s]
97
+ if (token) return token
98
+
99
+ const base = s.slice(s.lastIndexOf('/') + 1)
100
+
101
+ // ① 白板:handoff/ 目录(PLAN.md 与 handoff-<ts>.md 账本都在其下)
102
+ if (/(^|\/)handoff(\/|$)/.test(s) || base === 'plan.md' || /^handoff[-_]\d{8}/.test(base) || /账本|白板/.test(s)) return 'whiteboard'
103
+ // ② 反思:reflections/ 目录。文件名同样是 YYYY-MM-DD.md,必须先于日志判定
104
+ if (/(^|\/)reflections?(\/|$)/.test(s) || /reflection/.test(base) || /反思/.test(s)) return 'reflection'
105
+ // ③ 日志:日期文件名 或 logs/ 目录
106
+ if (DATE_FILE_RE.test(base) || /(^|\/)logs?(\/|$)/.test(s) || /日志/.test(s)) return 'log'
107
+ // ④ 项目笔记:工作区记忆根(集中式 / 旧版)下的 MEMORY.md
108
+ if ((WS_ROOT_RE.test(s) || WS_MARKER_RE.test(s)) && /^(memory|notes)\.md$/.test(base)) return 'project'
109
+ // ⑤ 用户级:用户记忆根下的文件,或带目录的 MEMORY.md(④ 已排除工作区侧)
110
+ if (USER_ROOT_RE.test(s)) return 'user'
111
+ if (s.includes('/') && /^(memory|calendar)\.md$/.test(base)) return 'user'
112
+ return null
113
+ }
114
+
115
+ /**
116
+ * 三层契约 I5 的**检索侧过滤谓词**:只有 `status === 'current'` 的记录可进入检索与注入。
117
+ *
118
+ * 缺席 `status` 视为 `current`(向后兼容 C1 之前的记录);`superseded` / `retracted` 一律挡下。
119
+ * 做成导出的小函数是为了能被单测直接断言(验收要求"能失败的断言")。
120
+ *
121
+ * @param {{status?:string}|null|undefined} record 记录(或任何带 status 的对象)
122
+ * @returns {boolean} 是否可作为 current 使用
123
+ */
124
+ export function isCurrentPre(record) {
125
+ if (!record || typeof record !== 'object') return true
126
+ const st = record.status
127
+ if (st === undefined || st === null || st === '') return true
128
+ return st === 'current'
129
+ }
130
+
131
+ /**
132
+ * 按锚点切分记忆条目。
133
+ *
134
+ * 约定:锚点标记**其后**的内容(实测文件结构为 `<!-- A -->内容A<!-- B -->内容B`),
135
+ * 因此第 i 个内容对应第 i 个锚点。锚点之前的游离内容(若有)归入 `preamble`。
136
+ *
137
+ * @param {string} text 文件内容
138
+ * @returns {{items: Array<{id:string, body:string}>, preamble: string, anchors: number}}
139
+ */
140
+ export function parseMemoryItemsPre(text) {
141
+ const src = typeof text === 'string' ? text : ''
142
+ const items = []
143
+ if (!src) return { items, preamble: '', anchors: 0 }
144
+
145
+ MEM_ANCHOR_RE.lastIndex = 0
146
+ const marks = []
147
+ let m
148
+ while ((m = MEM_ANCHOR_RE.exec(src)) !== null) {
149
+ marks.push({ id: m[1], start: m.index, end: m.index + m[0].length })
150
+ if (m.index === MEM_ANCHOR_RE.lastIndex) MEM_ANCHOR_RE.lastIndex++
151
+ }
152
+ if (!marks.length) return { items, preamble: clean(src), anchors: 0 }
153
+
154
+ for (let i = 0; i < marks.length; i++) {
155
+ const from = marks[i].end
156
+ const to = i + 1 < marks.length ? marks[i + 1].start : src.length
157
+ const body = clean(src.slice(from, to))
158
+ if (body) items.push({ id: marks[i].id, body })
159
+ }
160
+ return { items, preamble: clean(src.slice(0, marks[0].start)), anchors: marks.length }
161
+ }
162
+
163
+ /**
164
+ * 抽取单条记忆的 L0。纯函数,永不抛异常。
165
+ *
166
+ * @param {string} body 条目正文
167
+ * @param {{maxChars?:number, minChars?:number}} opts
168
+ * @returns {{l0: string, source: 'heading'|'firstSentence'|'truncate'|'empty'}}
169
+ */
170
+ export function extractL0Pre(body, opts = {}) {
171
+ const maxChars = Math.max(16, Number(opts.maxChars) || L0_DEFAULTS.maxChars)
172
+ const minChars = Math.max(0, Number(opts.minChars) || L0_DEFAULTS.minChars)
173
+ const text = clean(body)
174
+ if (!text) return { l0: '', source: 'empty' }
175
+
176
+ const lines = text.split(/\r?\n/)
177
+
178
+ // ① 主题块标题
179
+ for (const line of lines) {
180
+ const h = /^\s{0,3}#{1,6}\s+(.+?)\s*$/.exec(line)
181
+ if (h) {
182
+ const title = clean(h[1]).replace(HEADING_SUFFIX_RE, '')
183
+ if (title) return { l0: cut(title, maxChars), source: 'heading' }
184
+ }
185
+ }
186
+
187
+ // ② 首个 `- ` 条目:先取首句,过短再并接(并接时剥列表标记与时间戳)
188
+ for (const line of lines) {
189
+ const b = /^\s*[-*+]\s+(.+?)\s*$/.exec(line)
190
+ if (!b) continue
191
+ let s = clean(b[1]).replace(LEAD_TIME_RE, '')
192
+ if (!s) continue
193
+ const first = clean(s.split(SENTENCE_SPLIT_RE)[0])
194
+ s = growToMin(first || s, text, minChars, maxChars)
195
+ return { l0: cut(s, maxChars), source: 'firstSentence' }
196
+ }
197
+
198
+ // ③ 兜底:正文截断
199
+ const flat = clean(text.replace(/\s+/g, ' '))
200
+ return { l0: cut(flat, maxChars), source: 'truncate' }
201
+ }
202
+
203
+ /**
204
+ * 构建文件级 L0 索引(按 id 升序,确定性)。
205
+ *
206
+ * 2026-09-14(三层契约 C1):每条**追加** `layer` + `status` 两个字段——只增不减,
207
+ * 老调用方读 `id / l0 / source / chars / bodyChars` 完全不受影响,签名与调用方式不变。
208
+ * 层归属:`opts.layer`(层名 / 路径 / 键名,经 classifyLayerPre)→ `opts.path` → `L0_DEFAULT_LAYER`。
209
+ *
210
+ * @param {string} text 文件内容
211
+ * @param {{maxChars?:number, minChars?:number, layer?:string, path?:string}} [opts]
212
+ * @returns {Array<{id:string, l0:string, source:string, chars:number, bodyChars:number, layer:string, status:string}>}
213
+ */
214
+ export function buildL0IndexPre(text, opts = {}) {
215
+ const { items } = parseMemoryItemsPre(text)
216
+ const layer = resolveLayerPre(opts)
217
+ const out = items.map((it) => {
218
+ const r = extractL0Pre(it.body, opts)
219
+ // TODO(C1 写入侧,未实现):status 恒为 current。superseded(被新记录经 supersedes 替代改正)
220
+ // 与 retracted(人工判定作废)需要**写入侧先落盘状态**(fact-store 的 supersedes 边 / 审计视图标记),
221
+ // 本层不做存储、只负责透传;存储与双层过滤(契约 I5)在 C2 接线 + 写入侧一起做。
222
+ const status = 'current'
223
+ return {
224
+ id: it.id, l0: r.l0, source: r.source, chars: r.l0.length, bodyChars: it.body.length,
225
+ layer, status,
226
+ }
227
+ })
228
+ out.sort((a, b) => (a.id < b.id ? -1 : a.id > b.id ? 1 : 0))
229
+ return out
230
+ }
231
+
232
+ /** 本次抽取的层归属:显式层名/路径 > 备用路径键 > 兜底层。判不出不猜,回退 L0_DEFAULT_LAYER。 */
233
+ function resolveLayerPre(opts) {
234
+ const o = opts && typeof opts === 'object' ? opts : {}
235
+ for (const cand of [o.layer, o.path, o.sourcePath]) {
236
+ const hit = classifyLayerPre(cand)
237
+ if (hit) return hit
238
+ }
239
+ return L0_DEFAULT_LAYER
240
+ }
241
+
242
+ // ---------- 内部工具 ----------
243
+
244
+ function cut(s, n) {
245
+ if (s.length <= n) return s
246
+ return s.slice(0, Math.max(1, n - 1)) + '…'
247
+ }
248
+
249
+ /** 首句过短时,并接后续句子直到 minChars 或 maxChars。每句先剥列表标记与时间戳前缀。 */
250
+ function growToMin(first, full, minChars, maxChars) {
251
+ if (first.length >= minChars) return first
252
+ const flat = clean(full.replace(/\s+/g, ' '))
253
+ if (!flat || flat.length <= first.length) return first
254
+ const parts = flat.split(SENTENCE_SPLIT_RE)
255
+ .map((x) => clean(String(x).replace(/^\s*[-*+]\s+/, '').replace(LEAD_TIME_RE, '')))
256
+ .filter(Boolean)
257
+ let acc = ''
258
+ for (const p of parts) {
259
+ acc = acc ? acc + '。' + p : p
260
+ if (acc.length >= minChars || acc.length >= maxChars) break
261
+ }
262
+ return acc || first
263
+ }