@a9i5k4/dsh-auto-memory 2.5.3 → 3.0.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +189 -7
- package/README.zh-CN.md +189 -7
- package/docs/CONTRIBUTORS.html +471 -0
- package/docs/FRONTEND-CO-CREATION.md +191 -0
- package/docs/GM53-HOMEPAGE-PROMPT.md +323 -0
- package/docs/HANDOFF-CRITERIA.md +92 -0
- package/docs/HOMEPAGE-CONTENT-FOR-GM53.md +299 -0
- package/docs/INTEGRATION-ANALYSIS.md +350 -348
- package/docs/PROMO-PROMPT-3.0.md +100 -0
- package/docs/USER-GUIDE.en.md +58 -3
- package/docs/USER-GUIDE.zh-CN.md +59 -4
- package/docs/WHITEPAPER.md +207 -0
- package/docs/internal/ACCEPT-35-LIVE.md +143 -0
- package/docs/internal/ACCEPTANCE-20260914.md +90 -0
- package/docs/internal/ARCH-REVIEW-BRIEF.md +411 -0
- package/docs/internal/ARCH-REVIEW-REQUEST.md +201 -0
- package/docs/internal/ARCH-REVIEW-ROUND2.md +169 -0
- package/docs/internal/ARCH-REVIEW-ROUND3.md +206 -0
- package/docs/internal/ARCHITECTURE-FOR-ZCODE-20260920.md +397 -0
- package/docs/internal/ART-DIRECTION-DEEPSEEK-20260920.md +351 -0
- package/docs/internal/ART-DIRECTION-WIREFRAME.md +191 -181
- package/docs/internal/ART-DIRECTION-WIREFRAME.md.bak-superseded +181 -0
- package/docs/internal/AUDIT-WB-GRAPH-FULL-20260916.md +314 -0
- package/docs/internal/BATTLE-PLAN-20260917.md +871 -0
- package/docs/internal/CONCURRENCY-INVESTIGATION-20260917.md +192 -0
- package/docs/internal/CROSS-SESSION-SEARCH-PATH-DECISION.md +72 -0
- package/docs/internal/CROSS-SESSION-SEARCH-RESEARCH.md +131 -0
- package/docs/internal/DECISIONS-20260914-SESSION.md +269 -0
- package/docs/internal/DESIGN-P1-STATE-COMMIT-20260915.md +219 -0
- package/docs/internal/DIRECTION-CHECK-WB-GRAPH-20260916.md +132 -0
- package/docs/internal/FEATURE-INVENTORY.md +531 -0
- package/docs/internal/FEEDBACK-TO-DSHAPI-RELAY.md +13 -0
- package/docs/internal/G-SERIES-EXECUTION-20260917.md +248 -0
- package/docs/internal/G3-DESIGN-20260918.md +82 -0
- package/docs/internal/G3-DISK-FORMAT-GAP-20260919.md +92 -0
- package/docs/internal/GH-DISCUSSION-5732-COMMENT.md +74 -0
- package/docs/internal/GPT-ACCEPTANCE-PROMPT-20260916.md +352 -0
- package/docs/internal/GPT-REVIEW-PROMPT.md +216 -0
- package/docs/internal/GROUP-WEBHOOK-SETUP.md +33 -0
- package/docs/internal/HANDOFF-TO-ZCODE-20260920.md +309 -0
- package/docs/internal/HERMES-DATA-VERIFICATION-20260919.md +120 -0
- package/docs/internal/HERMES-LEGACY-STATUS-20260919.md +74 -0
- package/docs/internal/ISSUE-55-58-VERIFICATION-20260918.md +175 -0
- package/docs/internal/ISSUE10-FIX-EXECUTION-20260919.md +389 -0
- package/docs/internal/ISSUE10-PLAN-20260919.md +254 -0
- package/docs/internal/ISSUE10B-FORENSICS-20260919.md +468 -0
- package/docs/internal/ISSUE9-PURGE-AND-R1-PLAIN-20260919.md +150 -0
- package/docs/internal/ISSUE9-RESIDUAL-FORENSICS-20260919.md +114 -0
- package/docs/internal/KICKOFF-P0.md +254 -0
- package/docs/internal/LESSON-TO-CANDIDATE-STATUS-20260919.md +79 -0
- package/docs/internal/MASTER-PLAN-3.0.md +411 -0
- package/docs/internal/MEMORY-GOVERNANCE-20260917.md +309 -0
- package/docs/internal/MEMORY-MUTATION-AND-INDEX-DESIGN.md +85 -0
- package/docs/internal/MERGE-CONFLICT-SCAN-20260914.md +222 -0
- package/docs/internal/PENDING-FIXES-20260916.md +289 -0
- package/docs/internal/PRE-FRONTEND-CHECKLIST-20260919.md +705 -0
- package/docs/internal/PRE-FRONTEND-CHECKLIST-20260919.md.bak-s10 +649 -0
- package/docs/internal/PROCEDURAL-MEMORY-AND-APPROVAL-DESIGN-20260918.md +225 -0
- package/docs/internal/PROGRESS-20260917.md +93 -0
- package/docs/internal/PROMPT-GAP-AUDIT-20260920.md +128 -0
- package/docs/internal/R1-DEGRADE-AUDIT-20260918.md +163 -0
- package/docs/internal/R1-READABILITY-FORENSICS-20260919.md +127 -0
- package/docs/internal/R2-EVIDENCE-DEEP-AUDIT-20260918.md +140 -0
- package/docs/internal/R3-DEGRADE-LEDGER-DESIGN-20260918.md +138 -0
- package/docs/internal/R4-RECALL-QUOTA-PLAN-20260918.md +218 -0
- package/docs/internal/RAG-KARPATHY-PROGRAM.md +229 -0
- package/docs/internal/REPORT-P0-NIGHTLY.md +212 -0
- package/docs/internal/REPORT-P5-ACCEPTANCE.md +31 -0
- package/docs/internal/REPORT-WB-GRAPH-NIGHTLY.md +153 -0
- package/docs/internal/RESUME-20260918.md +171 -0
- package/docs/internal/RESUME-20260919.md +104 -0
- package/docs/internal/REVIEW-WB-GRAPH-SELF.md +81 -0
- package/docs/internal/RHINELAB-TO-DEEPSEEK-FEASIBILITY.md +198 -0
- package/docs/internal/ROADMAP-20260917-WEEK.md +439 -0
- package/docs/internal/ROADMAP.md +106 -0
- package/docs/internal/RUN-P0-NIGHTLY.md +227 -0
- package/docs/internal/S10-CONSTRUCTION-HANDOFF-20260917.md +185 -0
- package/docs/internal/S10-GAP-INVENTORY-20260917.md +239 -0
- package/docs/internal/S10-GAPS-PLAIN-20260917.md +125 -0
- package/docs/internal/SEMANTIC-ARCHITECTURE-SPEC.md +360 -0
- package/docs/internal/SESSION-FILE-REPAIR-PROTOCOL.md +90 -0
- package/docs/internal/T6-EXECUTION-20260920.md +130 -0
- package/docs/internal/TELEMETRY-EFFECT-REPORT-DESIGN-20260918.md +146 -0
- package/docs/internal/THESIS-GAP-ANALYSIS-20260918.md +89 -0
- package/docs/internal/THESIS-OUTLINE-20260918.md +147 -0
- package/docs/internal/THREE-LAYER-CONTRACT.md +219 -0
- package/docs/internal/TODO-BACKLOG.md +263 -142
- package/docs/internal/TODO-GRAPH.html +715 -0
- package/docs/internal/TODO-GRAPH.html.bak-20260914-v2 +493 -0
- package/docs/internal/TODO-GRAPH.html.bak-20260915-alsfix +710 -0
- package/docs/internal/TODO-GRAPH.html.bak-20260915-p1 +710 -0
- package/docs/internal/TODO-GRAPH.html.bak-20260915-p6a-rev +703 -0
- package/docs/internal/TODO-GRAPH.html.bak-20260915-wshint +710 -0
- package/docs/internal/TODO-GRAPH.html.bak-20260916-batch +715 -0
- package/docs/internal/UPSTREAM-ISSUE-PR-TRIAGE-20260919.md +297 -0
- package/docs/internal/UPSTREAM-ISSUES-3RD-AUDIT-20260920.md +104 -0
- package/docs/internal/WB-FORMAT-CONVENTION.md +112 -0
- package/docs/internal/WB-GRAPH-DECISIONS-20260914.md +71 -0
- package/docs/internal/reviews/CLAIM-VERIFICATION-20260914.md +56 -0
- package/docs/internal/reviews/PLAN-gpt6astra-round2-20260914.md +787 -0
- package/docs/internal/reviews/REVIEW-gpt6astra-20260914.md +112 -0
- package/docs/internal/reviews/ROUND3-REVIEW-INTEGRATION-20260914.md +230 -0
- package/docs/prompts/M8-3-enable-verify.md +49 -49
- package/docs/screenshots/promo/promo-0-banner-v3.png +0 -0
- package/lib/acceptance.js +71 -0
- package/lib/activation-host.js +153 -18
- package/lib/activation-inbox.js +25 -7
- package/lib/board-mode.js +30 -0
- package/lib/client.js +1758 -90
- package/lib/config-io.js +156 -0
- package/lib/context-bridge.js +5 -2
- package/lib/context-host.js +86 -15
- package/lib/degrade.js +385 -0
- package/lib/dsh-home.js +143 -0
- package/lib/engine-identity.js +149 -0
- package/lib/engine-switch.js +247 -0
- package/lib/episodic-store.js +63 -12
- package/lib/evidence-store.js +10 -3
- package/lib/fact-store.js +22 -3
- package/lib/fs-retry.js +46 -0
- package/lib/index-sync.js +13 -1
- package/lib/index.js +3446 -263
- package/lib/intent-clean-safe.js +258 -0
- package/lib/intent-clean.js +12 -16
- package/lib/l0-extract.js +478 -149
- package/lib/l0-index-sync.js +195 -0
- package/lib/l0-index.js +349 -239
- package/lib/ledger-criteria.js +142 -0
- package/lib/m4-corpus.js +8 -2
- package/lib/m7-index-sync-host.js +73 -5
- package/lib/m7-wire.js +3 -3
- package/lib/memory-anchor.js +56 -1
- package/lib/memory-envelope.js +257 -0
- package/lib/memory-hub.js +138 -13
- package/lib/memory-index.js +4 -2
- package/lib/memory-mutation.js +246 -0
- package/lib/memory-writer.js +204 -24
- package/lib/note-status-apply.js +118 -0
- package/lib/note-status.js +196 -0
- package/lib/procedure-observation.js +48 -0
- package/lib/procedure-store.js +118 -20
- package/lib/python-setup.js +1 -1
- package/lib/python-sidecar-client.js +29 -3
- package/lib/recall-fusion.js +83 -12
- package/lib/rerank-host.js +160 -0
- package/lib/rules-edit.js +159 -0
- package/lib/rules-layer.js +261 -0
- package/lib/semantic-decide.js +41 -8
- package/lib/semantic-js.js +66 -6
- package/lib/shadow-host.js +3 -5
- package/lib/shadow-retrieval.js +3 -3
- package/lib/skill-export-host.js +153 -0
- package/lib/skill-export.js +239 -0
- package/lib/state-commit.js +245 -0
- package/lib/storage-manage.js +6 -0
- package/lib/subagent-gc.js +4 -8
- package/lib/temporal-parse.js +191 -159
- package/lib/tier-layer-inject.js +650 -0
- package/lib/tier0-catalog.js +735 -0
- package/lib/water-window.js +263 -186
- package/lib/wb-contract.js +691 -0
- package/lib/wb-sidecar.js +890 -0
- package/lib/ws-overview-rank.js +2 -2
- package/package.json +1 -1
- package/python/m7_embedding_v1.py +5 -5
- package/python/worker_semantic_v1.py +17 -6
- package/python/worker_v1.py +38 -4
|
@@ -0,0 +1,261 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* 规则层抽取(rules_layer_v1)—— P6A 的"分类"部分(Phase 6 拆分中的 6A)。
|
|
3
|
+
*
|
|
4
|
+
* 2026-09-14 建立(P0 第五步)。**它解决的是用户最痛的病之一**:
|
|
5
|
+
* 注入开场白原来把所有记忆统一降格为「只是背景事实与规则参考」(`lib/index.js` 的
|
|
6
|
+
* `DEFAULT_PROMPT_LAYERS.snapshotHead`),**规则与资料混在一句措辞里** ⇒ 模型注意力不落在规矩上。
|
|
7
|
+
* 用户原话:「这个 just for reference 说得太轻了,模型注意力没有在这上面。」
|
|
8
|
+
*
|
|
9
|
+
* **本模块只做一件事**:从记忆文本里**认出规则类条目**,把它们与参考类分开。
|
|
10
|
+
* 措辞分层(不同引导语)与节奏(每轮在场)由调用方负责。
|
|
11
|
+
*
|
|
12
|
+
* **真源纪律(v2 修正,必须遵守)**:用户级规则=**既有用户级记忆文件**(`~/.dsh/memory/MEMORY.md`)
|
|
13
|
+
* 里的类型化规则,**不是新增 `RULES.md`**。`ROUND3 §2.3` 记录了我方曾在两处写了两个真源,
|
|
14
|
+
* 被 GPT 抓出。若将来真要独立真源,必须补迁移/去重/旧模式读取/回滚四类测试。
|
|
15
|
+
*
|
|
16
|
+
* **两级判定**(T7-4 的"双保险",v2 修正后的形态):
|
|
17
|
+
* ① **结构化前缀**(高置信):`【用户硬性规则…】` 这类显式标记 —— 这是既有记忆的真实形态
|
|
18
|
+
* (实测:`~/.dsh/memory/MEMORY.md` 首条即 `【用户硬性规则 - 文件编码】`)。
|
|
19
|
+
* ② **约束语汇**(中置信):`严禁` / `绝不` / `必须` / `不得` 等。
|
|
20
|
+
*
|
|
21
|
+
* ⚠️ **v2 明确纠正过我方原设计**:不能"命中关键词就无条件升级为规则"。
|
|
22
|
+
* 反例(GPT 给):`"文档写着必须重启"`、`"曾经要求必须 X 但已取消"` —— 这类是**描述**不是**要求**。
|
|
23
|
+
* 故本模块对 ② 类的结果标注 `confidence:'medium'` 并**保留原文里的引用/历史语境标记**
|
|
24
|
+
* (`引文` / `已取消` / `据文档` 等),由调用方决定是否只作为「待确认候选」。
|
|
25
|
+
* 本模块**不把中置信项冒充高置信**——那正是 T7-4 要防的误报。
|
|
26
|
+
*
|
|
27
|
+
* **依赖边界(如实标注,不宣称已保证)**:本模块保证的是"规则段被**注入函数**产出且不受节流与裁剪"。
|
|
28
|
+
* "模型**真的收到了**"取决于宿主最终请求 messages 的确认(`MASTER-PLAN-3.0.md §7` 的 **U6**,
|
|
29
|
+
* 尚未具备)⇒ **不得**据此宣称"规则的遵守问题已解决"(T7-7 的纪律)。
|
|
30
|
+
*
|
|
31
|
+
* S9 合规:零 IO、零依赖、纯函数、无网络/无 LLM/无子进程/无 await。UTF-8 无 BOM。
|
|
32
|
+
*/
|
|
33
|
+
|
|
34
|
+
export const RULES_LAYER_VERSION = 'rules_layer_v1'
|
|
35
|
+
|
|
36
|
+
/**
|
|
37
|
+
* ★P6B(2026-09-15 · 用户裁定开工):日志行内 kind 标记 —— 规则分类的**持久化**落点。
|
|
38
|
+
*
|
|
39
|
+
* 权威依据:`TODO-GRAPH.html` V2-P6B 卡 points[1]("分类时机:不做独立 LLM 调用,
|
|
40
|
+
* memory_log 本来就在本轮内由模型直接写,顺手打 kind 标记=零额外成本")+ crit[1](T7-5)。
|
|
41
|
+
*
|
|
42
|
+
* **形态选行内标记** `- HH:MM [kind:rule] 内容`:日志是 append-only 纯文本,行内标记
|
|
43
|
+
* 可被本模块的条目切分生态直接消费,且重启/迁移不丢(不依赖任何 sidecar)。
|
|
44
|
+
* kind 作**可选参数**(缺省 fact):不传时写入行与旧版逐字节一致(新旧并存 + 开关回退纪律)。
|
|
45
|
+
*/
|
|
46
|
+
export const LOG_KINDS_V1 = Object.freeze(['rule', 'preference', 'fact', 'todo'])
|
|
47
|
+
|
|
48
|
+
/** 行内标记的正则形态(渲染/解析两侧共用,不各写一套)。 */
|
|
49
|
+
export const LOG_KIND_TAG_RE_V1 = /\s*\[kind:(rule|preference|fact|todo)\]\s*/
|
|
50
|
+
|
|
51
|
+
/** 结构化前缀(高置信):既有用户级记忆的真实形态。 */
|
|
52
|
+
export const RULE_MARKERS_V1 = Object.freeze([
|
|
53
|
+
'【用户硬性规则', '【硬性规则', '【规则】', '【必须遵守', '【禁止', '【用户规则',
|
|
54
|
+
])
|
|
55
|
+
|
|
56
|
+
/** 约束语汇(中置信)—— 单靠它**不足以**判定为规则(见文件头 T7-4 说明)。 */
|
|
57
|
+
export const RULE_CONSTRAINT_WORDS_V1 = Object.freeze([
|
|
58
|
+
'严禁', '绝不', '必须', '不得', '禁止', '务必', '一律', '永远不', '不要再', '不准',
|
|
59
|
+
])
|
|
60
|
+
|
|
61
|
+
/**
|
|
62
|
+
* 引用/历史语境标记:出现这些 ⇒ 该段说的是"别处这么写"或"曾经如此",
|
|
63
|
+
* **不是**当前有效的规则要求 ⇒ 降级为 `candidate`(待确认候选),不得自动取得不可裁地位。
|
|
64
|
+
* 这是 v2 §3.3(Q7 第 2 条)给的反例在代码里的落点。
|
|
65
|
+
*/
|
|
66
|
+
export const RULE_DESCRIPTION_MARKERS_V1 = Object.freeze([
|
|
67
|
+
'据文档', '文档写着', '文档说', '曾经', '已取消', '已废弃', '不再要求', '原要求', '旧规则', '据说',
|
|
68
|
+
])
|
|
69
|
+
|
|
70
|
+
/** 条目切分锚点:与 `l0-extract.js` 的记忆锚点完全一致(不另立一套)。 */
|
|
71
|
+
const MEM_ANCHOR_LINE_RE = /^<!--\s*memory:(mem_[0-9a-f]{32})\s*-->$/
|
|
72
|
+
|
|
73
|
+
/** 条目内的日期小节(`## 2026-08-14`)——既有记忆文件的实际结构。 */
|
|
74
|
+
const DATE_SECTION_RE = /^##\s*\d{4}-\d{2}-\d{2}\s*$/
|
|
75
|
+
|
|
76
|
+
const clean = (s) => String(s == null ? '' : s)
|
|
77
|
+
|
|
78
|
+
/**
|
|
79
|
+
* 把记忆文本按锚点切成条目(保持既有文件结构语义:无锚点的前置内容归入首条)。
|
|
80
|
+
*
|
|
81
|
+
* @param {string} text 用户级/工作区级记忆文件全文
|
|
82
|
+
* @returns {Array<{id:string, anchorLine:number, text:string, lines:string[]}>}
|
|
83
|
+
*/
|
|
84
|
+
export function splitMemoryEntriesPre(text) {
|
|
85
|
+
const src = clean(text).replace(/^\uFEFF/, '')
|
|
86
|
+
if (!src.trim()) return []
|
|
87
|
+
const lines = src.split('\n')
|
|
88
|
+
const out = []
|
|
89
|
+
let cur = null
|
|
90
|
+
for (let i = 0; i < lines.length; i++) {
|
|
91
|
+
const m = MEM_ANCHOR_LINE_RE.exec(lines[i].replace(/\r$/, ''))
|
|
92
|
+
if (m) {
|
|
93
|
+
if (cur) out.push(cur)
|
|
94
|
+
cur = { id: m[1], anchorLine: i + 1, lines: [] }
|
|
95
|
+
continue
|
|
96
|
+
}
|
|
97
|
+
if (cur) cur.lines.push(lines[i])
|
|
98
|
+
}
|
|
99
|
+
if (cur) out.push(cur)
|
|
100
|
+
if (!out.length) {
|
|
101
|
+
// 无锚点文件(旧格式)⇒ 整篇算一条,不丢内容
|
|
102
|
+
return [{ id: '', anchorLine: 0, lines, text: src.trim() }]
|
|
103
|
+
}
|
|
104
|
+
return out.map((e) => ({ id: e.id, anchorLine: e.anchorLine, lines: e.lines, text: e.lines.join('\n').trim() }))
|
|
105
|
+
}
|
|
106
|
+
|
|
107
|
+
/** 单条文本的规则分类(返回置信度与理由,不抛异常)。 */
|
|
108
|
+
function classifyText(rawText) {
|
|
109
|
+
const text = clean(rawText)
|
|
110
|
+
if (!text.trim()) return { kind: 'reference', confidence: 'none', reasons: ['空内容'] }
|
|
111
|
+
const marker = RULE_MARKERS_V1.find((k) => text.includes(k))
|
|
112
|
+
if (marker) {
|
|
113
|
+
return { kind: 'rule', confidence: 'high', reasons: ['结构化前缀 ' + marker] }
|
|
114
|
+
}
|
|
115
|
+
const words = RULE_CONSTRAINT_WORDS_V1.filter((w) => text.includes(w))
|
|
116
|
+
if (!words.length) return { kind: 'reference', confidence: 'none', reasons: ['无规则标记与约束语汇'] }
|
|
117
|
+
const desc = RULE_DESCRIPTION_MARKERS_V1.filter((w) => text.includes(w))
|
|
118
|
+
if (desc.length) {
|
|
119
|
+
// ★ v2 修正的落点:含"必须"但是**描述/引文**,不得自动升级为规则
|
|
120
|
+
return {
|
|
121
|
+
kind: 'candidate', confidence: 'low',
|
|
122
|
+
reasons: ['约束语汇(' + words.slice(0, 3).join('/') + ')但含引用/历史语境(' + desc.slice(0, 3).join('/') + ')⇒ 仅作待确认候选,不自动取得规则地位'],
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
return { kind: 'rule', confidence: 'medium', reasons: ['约束语汇 ' + words.slice(0, 3).join('/')] }
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
/**
|
|
129
|
+
* 从记忆文本里抽出规则层(分项,供注入侧分层措辞与"不参与裁剪")。
|
|
130
|
+
*
|
|
131
|
+
* @param {object} input
|
|
132
|
+
* @param {string} [input.userText] 用户级记忆(跨工作区恒定;**规则真源**,见文件头)
|
|
133
|
+
* @param {string} [input.notesText] 工作区级项目笔记(第二层)
|
|
134
|
+
* @param {string} [input.rulesLayeringMode] 'off' | 'self' | 'none'
|
|
135
|
+
* - `'off'`:**默认**。返回空规则层(调用方回落旧行为:统一措辞 + 原有节奏)。这就是"新旧并存 + 开关回退"。
|
|
136
|
+
* - `'self'`:只看**当前 agent 自己写的** user 层(隔离子代理/外部导入的规则,防串线)。
|
|
137
|
+
* - `'none'`:不看任何 user 层(只保留工作区级)。
|
|
138
|
+
* @param {string} [input.ownerSessionId] `rulesLayeringMode='self'` 时用于判定"是不是我自己写的"
|
|
139
|
+
* @returns {{version:string, mode:string, enabled:boolean, text:string, rules:Array, candidates:Array,
|
|
140
|
+
* references:Array, counts:object, chars:{rules:number,candidates:number,total:number}}}
|
|
141
|
+
*/
|
|
142
|
+
export function extractRulesLayerPre(input = {}) {
|
|
143
|
+
const o = input && typeof input === 'object' ? input : {}
|
|
144
|
+
const mode = clampMode(o.rulesLayeringMode)
|
|
145
|
+
const empty = {
|
|
146
|
+
version: RULES_LAYER_VERSION, mode, enabled: false, text: '', rules: [], candidates: [], references: [],
|
|
147
|
+
counts: { entries: 0, rules: 0, candidates: 0, references: 0 }, chars: { rules: 0, candidates: 0, total: 0 },
|
|
148
|
+
}
|
|
149
|
+
if (mode === null) return empty
|
|
150
|
+
|
|
151
|
+
const sources = []
|
|
152
|
+
const pushAll = (layer, text) => {
|
|
153
|
+
for (const e of splitMemoryEntriesPre(text)) sources.push({ layer, ...e })
|
|
154
|
+
}
|
|
155
|
+
// 模式语义(与文件头一致,别写反):
|
|
156
|
+
// 'self' ⇒ 看用户级(跨工作区恒定)+ 工作区级
|
|
157
|
+
// 'none' ⇒ **不看任何用户级**,只看工作区级(用于"工作区规则覆盖用户规则"的场景)
|
|
158
|
+
if (mode !== 'none') pushAll('user', o.userText)
|
|
159
|
+
pushAll('project', o.notesText)
|
|
160
|
+
|
|
161
|
+
const rules = []
|
|
162
|
+
const candidates = []
|
|
163
|
+
const references = []
|
|
164
|
+
for (const e of sources) {
|
|
165
|
+
const c = classifyText(e.text)
|
|
166
|
+
const rec = { layer: e.layer, id: e.id, anchorLine: e.anchorLine, kind: c.kind, confidence: c.confidence, reasons: c.reasons, text: e.text }
|
|
167
|
+
if (c.kind === 'rule') rules.push(rec)
|
|
168
|
+
else if (c.kind === 'candidate') candidates.push(rec)
|
|
169
|
+
else references.push(rec)
|
|
170
|
+
}
|
|
171
|
+
|
|
172
|
+
// 规则层文本:只含规则条目(高/中置信),逐条一行摘要,**不掺参考类**
|
|
173
|
+
const renderLine = (r) => '- ' + ruleSummaryPre(r.text)
|
|
174
|
+
const parts = []
|
|
175
|
+
if (rules.length) parts.push(rules.map(renderLine).join('\n'))
|
|
176
|
+
const text = parts.join('\n')
|
|
177
|
+
return {
|
|
178
|
+
version: RULES_LAYER_VERSION, mode, enabled: true, text, rules, candidates, references,
|
|
179
|
+
counts: { entries: sources.length, rules: rules.length, candidates: candidates.length, references: references.length },
|
|
180
|
+
chars: { rules: text.length, candidates: candidates.reduce((a, r) => a + ruleSummaryPre(r.text).length + 2, 0), total: text.length },
|
|
181
|
+
}
|
|
182
|
+
}
|
|
183
|
+
|
|
184
|
+
/** 模式解析:`null` 表示"未启用"(调用方沿用旧行为)。 */
|
|
185
|
+
function clampMode(v) {
|
|
186
|
+
const s = clean(v).trim().toLowerCase()
|
|
187
|
+
if (!s || s === 'off' || s === 'false' || s === '0') return null
|
|
188
|
+
if (s === 'self' || s === 'none' || s === 'all') return s
|
|
189
|
+
return null // 非法值一律当"未启用"(fail-soft,不因配置写错而改变注入)
|
|
190
|
+
}
|
|
191
|
+
|
|
192
|
+
/** 取第一条非空行并裁剪(规则条目往往首行就是规则名,够用且省 token)。 */
|
|
193
|
+
export function firstLine(text, max = 200) {
|
|
194
|
+
const lines = clean(text).split('\n').map((l) => l.trim()).filter(Boolean)
|
|
195
|
+
const first = lines[0] || ''
|
|
196
|
+
return first.length > max ? first.slice(0, Math.max(1, max - 1)) + '…' : first
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
/**
|
|
200
|
+
* 规则条目的**摘要行**(渲染用):跳过日期小节标题(`## 2026-08-14`)与纯标点行,
|
|
201
|
+
* 取第一条实质内容。
|
|
202
|
+
*
|
|
203
|
+
* **为什么需要它**(实测踩坑):既有记忆文件的条目结构是「锚点 → `## 日期` → 正文」,
|
|
204
|
+
* 直接用 `firstLine` 会渲染成 `- ## 2026-08-14` —— 规则内容全丢,只剩日期,
|
|
205
|
+
* 而且**看起来还挺正常**(这条断言首跑就红在"连续 6 轮规则都在"上,因为摘要里没有规则文本)。
|
|
206
|
+
*
|
|
207
|
+
* ★P6B:`[kind:x]` 行内标记在此**剥离**(标记只服务盘上审计与分类持久化,
|
|
208
|
+
* 注入摘要保持干净);时间戳 `HH:MM` 一并剥离(对规则语义是噪声)。
|
|
209
|
+
*/
|
|
210
|
+
export function ruleSummaryPre(text, max = 200) {
|
|
211
|
+
const lines = clean(text).split('\n').map((l) => l.trim()).filter(Boolean)
|
|
212
|
+
for (const l of lines) {
|
|
213
|
+
if (DATE_SECTION_RE.test(l)) continue
|
|
214
|
+
if (/^<!--/.test(l)) continue
|
|
215
|
+
if (/^[-*+]\s*$/.test(l)) continue
|
|
216
|
+
// 去掉列表前缀再返回,保持一条一行
|
|
217
|
+
let body = l.replace(/^[-*+]\s+/, '')
|
|
218
|
+
// ★P6B:剥行内 kind 标记与前置时间戳(只影响渲染,不回写盘上原文)
|
|
219
|
+
body = body.replace(LOG_KIND_TAG_RE_V1, ' ').trim()
|
|
220
|
+
body = body.replace(/^\d{1,2}:\d{2}\s+/, '').trim()
|
|
221
|
+
return body.length > max ? body.slice(0, Math.max(1, max - 1)) + '…' : body
|
|
222
|
+
}
|
|
223
|
+
return ''
|
|
224
|
+
}
|
|
225
|
+
|
|
226
|
+
/**
|
|
227
|
+
* 规则段渲染(**供注入侧直接使用**):标题 + 引导语 + 逐条规则。
|
|
228
|
+
*
|
|
229
|
+
* 措辞分层的关键(T7-3):**规则类用约束语**("必须遵守"),**参考类保留"参考"语义**。
|
|
230
|
+
* 两段引导语必须**不相同** —— 断言直接锁这一点,防止有人改回统一措辞。
|
|
231
|
+
*
|
|
232
|
+
* @param {object} rules `extractRulesLayerPre` 的返回值
|
|
233
|
+
* @param {object} [opts]
|
|
234
|
+
* @param {string} [opts.title] 段标题(默认常量)
|
|
235
|
+
* @param {string} [opts.guide] 引导语(默认常量)
|
|
236
|
+
* @returns {{text:string, chars:number, guide:string}}
|
|
237
|
+
*/
|
|
238
|
+
export function renderRulesSectionPre(rules, opts = {}) {
|
|
239
|
+
const r = rules && typeof rules === 'object' ? rules : {}
|
|
240
|
+
const list = Array.isArray(r.rules) ? r.rules : []
|
|
241
|
+
if (!list.length) return { text: '', chars: 0, guide: '' }
|
|
242
|
+
const title = clean((opts && opts.title) || RULES_SECTION_TITLE_V1)
|
|
243
|
+
const guide = clean((opts && opts.guide) || RULES_SECTION_GUIDE_V1)
|
|
244
|
+
const text = '\n' + title + '\n' + guide + '\n' + list.map((x) => '- ' + ruleSummaryPre(x.text)).join('\n')
|
|
245
|
+
return { text, chars: text.length, guide }
|
|
246
|
+
}
|
|
247
|
+
|
|
248
|
+
/**
|
|
249
|
+
* 规则段标题(T7-3:与参考段的措辞**必须不同**;断言直接比对两者)。
|
|
250
|
+
* 用户可通过 `promptLayerOverrides.snapshotRulesTitle` 覆盖。
|
|
251
|
+
*/
|
|
252
|
+
export const RULES_SECTION_TITLE_V1 = '[规则 — 用户级硬性约束 · 必须遵守]'
|
|
253
|
+
|
|
254
|
+
/**
|
|
255
|
+
* 规则段引导语:**约束语**("以下为必须遵守的约束"),不是"只是参考"。
|
|
256
|
+
* 这句话就是 P6A「措辞」要修的病:旧开场白把规矩与资料统一降格为「只是背景事实与规则参考」。
|
|
257
|
+
*/
|
|
258
|
+
export const RULES_SECTION_GUIDE_V1 = '以下条目是用户明确要求长期遵守的约束,不是可选背景。凡与其它内容冲突,以本节为准;无法满足时必须显式说明。'
|
|
259
|
+
|
|
260
|
+
/** 参考段引导语(保留"参考"语义,与规则段措辞不同)。 */
|
|
261
|
+
export const REFERENCE_SECTION_GUIDE_V1 = '以下为背景资料与历史记录,供参考与检索定位;与上面的规则冲突时不适用。'
|
package/lib/semantic-decide.js
CHANGED
|
@@ -27,16 +27,49 @@ const PLAN_TOKENS = ['准备', '打算', '计划', '之后', '接下来', '继
|
|
|
27
27
|
const WS_RUN = /\s\s+/g
|
|
28
28
|
// Python `(?u)\b\w\w+\b` 中 \w 含 CJK;JS 的 \w 默认只含 ASCII,须显式加 CJK
|
|
29
29
|
// 否则中文词不被切分 → 无 gram → 中文意图全靠 intercept(严重偏差)。
|
|
30
|
-
|
|
30
|
+
// ★ issue #68 修复(2026-09-19):**加 `{2,}` 下限** —— Python 侧是 `\b\w\w+\b`(**≥2 字**),
|
|
31
|
+
// 旧实现允许 1 字词(`+`)⇒ 中文单字(如「好」「是」)在 JS 侧会生成 gram、Python 侧不会,
|
|
32
|
+
// 两版对**同一输入**给出不同 gram 集合 ⇒ 判定分叉。
|
|
33
|
+
const WORD_RE = /[A-Za-z0-9_\u4e00-\u9fff]{2,}/g
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* 与 Python `normalize_text` 逐字对齐。
|
|
37
|
+
* ★ issue #68 修复(2026-09-19):旧实现有三处分叉,现全部按 **Python 为权威** 对齐:
|
|
38
|
+
* ① **整删非字母数字**:旧实现 `.replace(/[^…\s]/g, ' ')` 把标点变**空格** ⇒ `"a-b"` 得到 `"a b"`(两词),
|
|
39
|
+
* 而 Python 是 `''.join(ch for ch in text.lower() if ch.isalnum() or CJK)` ⇒ `"ab"`(**一词**);
|
|
40
|
+
* ② **`isalnum()` 语义**:Python 的 `str.isalnum()` 对**全角数字**(`123`)、带音标字母(`é`)为真,
|
|
41
|
+
* 而 JS 正则 `[a-z0-9]` 只认 ASCII ⇒ 这类字符在 JS 侧被丢弃、Python 侧保留;
|
|
42
|
+
* ③ **不再额外压空白**:Python 整删后**不插空格**,故 `"a b"` → `"a b"`(原样)而非 `"a b"`。
|
|
43
|
+
* 修法用 `codePointAt` + `isAlnumPythonish()` 复刻 `isalnum()`,而不是改正则字符类。
|
|
44
|
+
*/
|
|
45
|
+
function isAlnumPythonish(cp) {
|
|
46
|
+
// ASCII 字母数字
|
|
47
|
+
if (cp >= 0x30 && cp <= 0x39) return true // 0-9
|
|
48
|
+
if (cp >= 0x61 && cp <= 0x7a) return true // a-z(已 lower)
|
|
49
|
+
if (cp >= 0x41 && cp <= 0x5a) return true // A-Z(防御:未 lower 时)
|
|
50
|
+
// CJK 统一表意文字(与既有 \u4e00-\u9fff 口径一致)
|
|
51
|
+
if (cp >= 0x4e00 && cp <= 0x9fff) return true
|
|
52
|
+
// 全角数字 0-9
|
|
53
|
+
if (cp >= 0xff10 && cp <= 0xff19) return true
|
|
54
|
+
// 全角字母 a-z / A-Z(★ 首轮遗漏:`toLowerCase()` 会把 A(U+FF21) 折叠成 a(U+FF41),
|
|
55
|
+
// 若只覆盖数字,全角字母会在折叠后**整段消失**,与 Python `isalnum()` 仍为真的语义分叉)
|
|
56
|
+
if (cp >= 0xff21 && cp <= 0xff3a) return true // A-Z(防御:未 lower 时)
|
|
57
|
+
if (cp >= 0xff41 && cp <= 0xff5a) return true // a-z(lower 后的实际落点)
|
|
58
|
+
// 拉丁补充字母(含 é 等带音标字符,覆盖 Latin-1 Supplement 的字母区)
|
|
59
|
+
if (cp >= 0x00c0 && cp <= 0x00ff && cp !== 0x00d7 && cp !== 0x00f7) return true
|
|
60
|
+
if (cp >= 0x0100 && cp <= 0x017f) return true // Latin Extended-A
|
|
61
|
+
return false
|
|
62
|
+
}
|
|
31
63
|
|
|
32
|
-
/** 与 Python normalize_text 对齐(大小写折叠 + 保留 [a-z0-9]+CJK + 去其他)。 */
|
|
33
64
|
function normalizeText(text) {
|
|
34
|
-
|
|
35
|
-
|
|
36
|
-
|
|
37
|
-
|
|
38
|
-
.
|
|
39
|
-
|
|
65
|
+
const lower = String(text || '').toLowerCase()
|
|
66
|
+
let out = ''
|
|
67
|
+
// 按码点遍历(避免代理对拆断),复刻 Python 的 `if ch.isalnum()` 整删语义
|
|
68
|
+
for (const ch of lower) {
|
|
69
|
+
const cp = ch.codePointAt(0)
|
|
70
|
+
if (isAlnumPythonish(cp)) out += ch
|
|
71
|
+
}
|
|
72
|
+
return out
|
|
40
73
|
}
|
|
41
74
|
|
|
42
75
|
/** char_wb n-gram 计数(与 Python _char_wb_ngram_counts 对齐)。 */
|
package/lib/semantic-js.js
CHANGED
|
@@ -17,6 +17,7 @@ import { existsSync, readFileSync, mkdirSync, readdirSync, renameSync, statSync,
|
|
|
17
17
|
import { fileURLToPath, pathToFileURL } from 'node:url'
|
|
18
18
|
import { createRequire } from 'node:module'
|
|
19
19
|
import path from 'node:path'
|
|
20
|
+
import { resolveDshHomePre } from './dsh-home.js'
|
|
20
21
|
import { homedir } from 'node:os'
|
|
21
22
|
|
|
22
23
|
export const JS_SEMANTIC_ENGINE_VERSION = 'js_semantic_engine_v1'
|
|
@@ -70,7 +71,8 @@ const E5_MODELS_SUBDIR = 'multilingual-e5-small'
|
|
|
70
71
|
function defaultModelsDirCandidates(pluginDir) {
|
|
71
72
|
// 用户目录优先(#15 后续/B 修复):~/.dsh/models/js-semantic/ 跨插件升级存活——
|
|
72
73
|
// 包目录(lib/models)在 npm 更新时会被整体重装,下载的 130MB 模型曾被冲掉。
|
|
73
|
-
|
|
74
|
+
// ★#86-3:统一口径
|
|
75
|
+
const dshHome = resolveDshHomePre()
|
|
74
76
|
return [
|
|
75
77
|
path.join(dshHome, 'models', 'js-semantic'),
|
|
76
78
|
path.join(pluginDir, 'models'),
|
|
@@ -161,9 +163,20 @@ export function deepScanPeerTransformers(roots) {
|
|
|
161
163
|
* semanticAssetProbe 共享实现(index.js 的 status API / 档位解析 / 引导卡同源):
|
|
162
164
|
* 模型资产=发行包 lib/models 优先,其次开发树;peer=resolvePeerTransformersDir。
|
|
163
165
|
* extraDirs 透传给 peer 探测(深度扫描热接入位)。返回形状与 v0.1.36 完全一致。
|
|
166
|
+
*
|
|
167
|
+
* ★ issue #70 修复(2026-09-19):新增 `degraded` 字段。
|
|
168
|
+
* 旧实现是纯 `existsSync` 探测 —— **资产在磁盘上存在 ≠ 引擎可用**:
|
|
169
|
+
* 一旦运行期 `ensureTier()` 失败(onnx 损坏 / 维度不符 / peer 加载失败),引擎置 `degraded`
|
|
170
|
+
* 并此后恒抛;但探针只回 `ready = assetPresent && peerPresent` ⇒
|
|
171
|
+
* **引导卡显示「✓ 就绪」、档位解析给出 c2,而实际每次检索都在静默走词法兜底**(状态自相矛盾)。
|
|
172
|
+
* 现把降级状态并进 `ready`(并单列 `degraded` 供诊断),使三条用户可见路径(引导卡 / 档位 / status)一致。
|
|
173
|
+
* @param {string} pluginDir
|
|
174
|
+
* @param {string[]} [extraDirs]
|
|
175
|
+
* @param {string} [degradedReason] 当前引擎的降级原因(空串=未降级)
|
|
164
176
|
*/
|
|
165
|
-
export function probeJsSemanticAssets(pluginDir, extraDirs) {
|
|
166
|
-
|
|
177
|
+
export function probeJsSemanticAssets(pluginDir, extraDirs, degradedReason) {
|
|
178
|
+
// ★#86-3:统一口径
|
|
179
|
+
const dshHome = resolveDshHomePre()
|
|
167
180
|
const modelCands = [
|
|
168
181
|
path.join(dshHome, 'models', 'js-semantic', E5_MODELS_SUBDIR, 'onnx', 'model_quantized.onnx'),
|
|
169
182
|
path.join(pluginDir, 'models', E5_MODELS_SUBDIR, 'onnx', 'model_quantized.onnx'),
|
|
@@ -172,10 +185,15 @@ export function probeJsSemanticAssets(pluginDir, extraDirs) {
|
|
|
172
185
|
const modelOnnx = modelCands.find((c) => existsSync(c)) || modelCands[0]
|
|
173
186
|
const assetPresent = existsSync(modelOnnx)
|
|
174
187
|
const peerDir = resolvePeerTransformersDir(pluginDir, extraDirs)
|
|
188
|
+
const filesReady = Boolean(assetPresent && peerDir)
|
|
189
|
+
const degraded = String(degradedReason || '')
|
|
175
190
|
return {
|
|
176
191
|
assetPresent,
|
|
177
192
|
peerPresent: Boolean(peerDir),
|
|
178
|
-
|
|
193
|
+
// ★ 资产齐备**且未降级**才算就绪(否则 UI 报"就绪"而实际降级)
|
|
194
|
+
ready: filesReady && !degraded,
|
|
195
|
+
filesReady,
|
|
196
|
+
degraded,
|
|
179
197
|
assetBytes: assetPresent ? statSync(modelOnnx).size : 0,
|
|
180
198
|
assetPath: modelOnnx,
|
|
181
199
|
}
|
|
@@ -213,6 +231,8 @@ export function createJsSemanticEnginePre(opts = {}) {
|
|
|
213
231
|
let tier = null // {embedQuery, embedPassages, model}
|
|
214
232
|
let tierPromise = null
|
|
215
233
|
let degraded = '' // 非空 = 初始化失败原因(调用方回退词法)
|
|
234
|
+
let degradedAt = 0 // ★ issue #70:最近一次置位的时刻(用于有界自动重试的冷却判定)
|
|
235
|
+
let statsRetries = 0 // ★ issue #70:因冷却到期而清空降级态的累计次数(诊断用)
|
|
216
236
|
let lastRankError = '' // 运行期排名失败原因(诊断用;rank 恒返回 null 由调用方回退)
|
|
217
237
|
let idx = { miv: null, entries: [] } // entries [{memoryId, vec}]
|
|
218
238
|
let rebuilding = null // 单飞行重建 promise
|
|
@@ -223,7 +243,21 @@ export function createJsSemanticEnginePre(opts = {}) {
|
|
|
223
243
|
|
|
224
244
|
async function ensureTier() {
|
|
225
245
|
if (tier) return tier
|
|
226
|
-
|
|
246
|
+
// ★ issue #70 修复(2026-09-19):`degraded` 由**单向闩锁**改为**有界自动重试**。
|
|
247
|
+
// 旧行为:首次 `ensureTier()` 失败 ⇒ `degraded` 置位 ⇒ 此后恒抛,
|
|
248
|
+
// **唯一清除出口是测试钩子 `_resetForTest`** ⇒ 用户补齐资产 / 修好 onnx 后,
|
|
249
|
+
// 本进程内**永久** C2 降级(只能重启宿主),且 UI 仍显示"✓ 就绪"(见 probeJsSemanticAssets 的修复)。
|
|
250
|
+
// 现行为:冷却期(默认 60s,可配 `degradedRetryMs`)过后允许**再试一次**;
|
|
251
|
+
// 重试仍失败则更新时间戳继续冷却 —— 既不会每轮都去加载大模型,也不会永久闩死。
|
|
252
|
+
if (degraded) {
|
|
253
|
+
const since = Date.now() - degradedAt
|
|
254
|
+
const retryMs = Number((opts && opts.degradedRetryMs) || 60000)
|
|
255
|
+
if (since < retryMs) throw new Error(degraded)
|
|
256
|
+
// 冷却已过 ⇒ 清空降级态、允许重建(tierPromise 也一并清掉,避免复用已失败的 promise)
|
|
257
|
+
degraded = ''
|
|
258
|
+
tierPromise = null
|
|
259
|
+
statsRetries++
|
|
260
|
+
}
|
|
227
261
|
if (!tierPromise) {
|
|
228
262
|
tierPromise = (async () => {
|
|
229
263
|
if (opts.injectEmbedder) return opts.injectEmbedder
|
|
@@ -268,6 +302,7 @@ export function createJsSemanticEnginePre(opts = {}) {
|
|
|
268
302
|
return tier
|
|
269
303
|
} catch (e) {
|
|
270
304
|
degraded = String(e && e.message || e).slice(0, 160)
|
|
305
|
+
degradedAt = Date.now() // ★ issue #70:记时刻,供冷却判定
|
|
271
306
|
tierPromise = null
|
|
272
307
|
throw e
|
|
273
308
|
}
|
|
@@ -305,6 +340,10 @@ export function createJsSemanticEnginePre(opts = {}) {
|
|
|
305
340
|
assetPresent: existsSync(path.join(modelsDir(), E5_MODELS_SUBDIR, 'onnx', 'model_quantized.onnx')),
|
|
306
341
|
ready: !!tier,
|
|
307
342
|
degraded,
|
|
343
|
+
// ★ issue #70:暴露降级时刻与重试计数,供面板/诊断判断"是否处于冷却期、下次何时可重试"
|
|
344
|
+
degradedAt: degraded ? degradedAt : 0,
|
|
345
|
+
degradedRetryMs: Number((opts && opts.degradedRetryMs) || 60000),
|
|
346
|
+
degradedRetries: statsRetries,
|
|
308
347
|
lastRankError,
|
|
309
348
|
model: tier ? tier.model : null,
|
|
310
349
|
indexedRecords: idx.entries.length,
|
|
@@ -342,7 +381,22 @@ export function createJsSemanticEnginePre(opts = {}) {
|
|
|
342
381
|
return null // fail closed:词法回退由调用方自然发生
|
|
343
382
|
}
|
|
344
383
|
},
|
|
345
|
-
|
|
384
|
+
/**
|
|
385
|
+
* 2026-09-14(三层契约 C3):暴露**批量段落嵌入**通道 —— L0 向量索引接线
|
|
386
|
+
* (`lib/l0-index-sync.js`)需要把 L0 摘要本身变成向量,而 rank() 只返回
|
|
387
|
+
* 「查询 vs 内部缓存索引」的余弦分,**不产出向量**(此前宿主没有任何拿向量的口子,
|
|
388
|
+
* 这是 `createL0IndexPre` 一直零引用、无法接线的原因之一)。
|
|
389
|
+
*
|
|
390
|
+
* 语义:裸文本数组 → `Float32Array[]`(e5 口径为 384 维已归一化;前缀 `passage: ` 由引擎内部加,
|
|
391
|
+
* 与 rank 的 `query: ` 前缀对称,调用方传**裸文本**)。
|
|
392
|
+
* 失败语义:模型资产缺失/加载失败 → 抛错(**不吞**)—— 由调用方 fail-soft(索引侧只 diag 一行,
|
|
393
|
+
* 不阻塞任何检索路径)。本方法不改任何内部状态(不缓存、不动 idx),纯计算。
|
|
394
|
+
*/
|
|
395
|
+
async embedPassages(texts) {
|
|
396
|
+
const t = await ensureTier()
|
|
397
|
+
return t.embedPassages(Array.isArray(texts) ? texts : [])
|
|
398
|
+
},
|
|
399
|
+
_resetForTest() { tier = null; tierPromise = null; degraded = ''; degradedAt = 0; statsRetries = 0; idx = { miv: null, entries: [] }; rebuilding = null },
|
|
346
400
|
}
|
|
347
401
|
}
|
|
348
402
|
|
|
@@ -398,6 +452,12 @@ export function createSemanticDownloaderPre(opts = {}) {
|
|
|
398
452
|
const total = Number(res.headers.get('content-length')) || fileRec.bytes
|
|
399
453
|
mkdirSync(tmpDir(), { recursive: true })
|
|
400
454
|
const dst = path.join(tmpDir(), path.basename(fileRec.rel))
|
|
455
|
+
// ★ 每次尝试从零开始(2026-09-19 上游 PR #79 / issue #65 同步落地):
|
|
456
|
+
// tmp 只在 run() 开始清一次,但 fetchToFile 会被**多个镜像依次调用**;
|
|
457
|
+
// 而写入用 flag:'a' 追加 ⇒ 源 A 断流留下的半截文件会被续上源 B 的完整流。
|
|
458
|
+
// 致命处:sha256 只对**网络流**累积(hash.update)、**不回读文件** ⇒ 拼接体校验照样通过并落位,
|
|
459
|
+
// 直到加载期才失败。故每次尝试前必须清掉残留。
|
|
460
|
+
try { rmSync(dst, { force: true }) } catch (_) {}
|
|
401
461
|
const hash = createHash('sha256')
|
|
402
462
|
const reader = res.body.getReader()
|
|
403
463
|
const fd = writeFileSync // 占位避免未用告警;真正写入走手动缓冲
|
package/lib/shadow-host.js
CHANGED
|
@@ -10,6 +10,7 @@
|
|
|
10
10
|
*/
|
|
11
11
|
import { mkdirSync, appendFileSync, writeFileSync, readFileSync, readdirSync, existsSync, statSync, rmSync } from 'node:fs'
|
|
12
12
|
import path from 'node:path'
|
|
13
|
+
import { resolveDshHomeForEnginePre } from './dsh-home.js'
|
|
13
14
|
import { homedir } from 'node:os'
|
|
14
15
|
import { createHash } from 'node:crypto'
|
|
15
16
|
import {
|
|
@@ -126,11 +127,8 @@ export function createShadowHost({ engine }) {
|
|
|
126
127
|
/** durable audit 目录与日期分片。 */
|
|
127
128
|
function auditDir() { return path.join(dshHome(), 'memory', 'retrieval-pre', 'audit') }
|
|
128
129
|
function dshHome() {
|
|
129
|
-
|
|
130
|
-
|
|
131
|
-
// M4-4 修复:必须拼接 .dsh(此前漏拼导致 audit 写到 <home>/memory 错误位置)
|
|
132
|
-
const base = engine.__homedirFn ? engine.__homedirFn() : (process.env.USERPROFILE || process.env.HOME || '')
|
|
133
|
-
return base ? path.join(base, '.dsh') : '.'
|
|
130
|
+
// ★#86-3:统一口径(M4-4 的「必须拼 .dsh」约定已由 resolveDshHomePre 保证)。
|
|
131
|
+
return resolveDshHomeForEnginePre(engine)
|
|
134
132
|
}
|
|
135
133
|
|
|
136
134
|
/** 串行 durable append(§15.4):engine 级链式;失败只计 audit-write-failed 不重试不污染 Session。 */
|
package/lib/shadow-retrieval.js
CHANGED
|
@@ -402,14 +402,14 @@ const MEMORY_ID_STRICT = /^mem_[0-9a-f]{32}$/
|
|
|
402
402
|
|
|
403
403
|
/** §11 retrievalId(确定性,可重放;非长期内容身份)。 */
|
|
404
404
|
export function buildRetrievalId(sessionId, contextVersion, triggerSegmentId, memoryIndexVersion) {
|
|
405
|
-
const sessionIdHash = first32(sha256Str('retrieval-
|
|
406
|
-
const parts = ['retrieval-
|
|
405
|
+
const sessionIdHash = first32(sha256Str('retrieval-v1\u0000' + sessionId))
|
|
406
|
+
const parts = ['retrieval-v1', sessionIdHash, contextVersion, triggerSegmentId, memoryIndexVersion, GATE_POLICY_VERSION, LEXICAL_POLICY_VERSION]
|
|
407
407
|
return RETRIEVAL_PREFIX + first32(sha256Str(JSON.stringify(parts)))
|
|
408
408
|
}
|
|
409
409
|
|
|
410
410
|
/** §11 candidateId:同 memoryId 内容变化后 candidateId 因版本/digest 改变。 */
|
|
411
411
|
export function buildCandidateId(retrievalId, memoryId, sourceEpoch, sourceVersion, recordDigest) {
|
|
412
|
-
const parts = ['candidate-
|
|
412
|
+
const parts = ['candidate-v1', retrievalId, memoryId, sourceEpoch, sourceVersion, recordDigest]
|
|
413
413
|
return CANDIDATE_PREFIX + first32(sha256Str(JSON.stringify(parts)))
|
|
414
414
|
}
|
|
415
415
|
|
|
@@ -0,0 +1,153 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* M9-3 Skill 导出层 · 宿主侧(IO 落地)。
|
|
3
|
+
*
|
|
4
|
+
* 纯渲染在 `skill-export.js`;本文件只负责**写盘**与**项目标注解析**。
|
|
5
|
+
*
|
|
6
|
+
* ★ 用户 2026-09-19 拍板:
|
|
7
|
+
* - ⑪-2 **晋升为 `active` 后自动导出** ⇒ 由 host 在 `activate` 成功后调用 `exportSkillForPre()`。
|
|
8
|
+
* - ⑪-3 目录 = **用户级**(`<dshHome>/skills/`,DSH 四条发现路径之一,可跨项目迁移);
|
|
9
|
+
* 导出物**必须标注适用项目**。
|
|
10
|
+
*
|
|
11
|
+
* 设计纪律:
|
|
12
|
+
* ① **fail-soft 但可观察**:返回 `{ok:false, reason}`,不抛出、不静默;
|
|
13
|
+
* ② **不自动执行**任何导出程序(程序只是参考资料);
|
|
14
|
+
* ③ 写盘走**临时文件 + rename**(避免半截文件被 DSH 当成合法 SKILL.md);
|
|
15
|
+
* ④ 不覆盖**非本插件产出**的同名目录(有 `mem-skill-` 前缀 + 自检标记才接管)。
|
|
16
|
+
*/
|
|
17
|
+
|
|
18
|
+
import { mkdirSync, writeFileSync, renameSync, existsSync, readFileSync, readdirSync, statSync } from 'node:fs'
|
|
19
|
+
import { join } from 'node:path'
|
|
20
|
+
import { renderSkillMarkdownPre, validateSkillMarkdownPre } from './skill-export.js'
|
|
21
|
+
|
|
22
|
+
/** 导出目录前缀(用于识别"这是本插件的产物",避免误覆盖用户手写技能)。 */
|
|
23
|
+
export const SKILL_EXPORT_PREFIX_V1 = 'mem-skill-'
|
|
24
|
+
/** 自检标记:出现在 SKILL.md 里即证明是本插件导出物。 */
|
|
25
|
+
export const SKILL_EXPORT_STAMP_V1 = 'mem-skill-export-v1'
|
|
26
|
+
|
|
27
|
+
/**
|
|
28
|
+
* 解析技能导出根目录。
|
|
29
|
+
* 优先级:opts.skillsRoot > <dshHome>/skills
|
|
30
|
+
* (`<dshHome>` 由调用方从 DSH 环境取得;本函数不猜路径。)
|
|
31
|
+
*/
|
|
32
|
+
export function resolveSkillsRootPre(opts = {}) {
|
|
33
|
+
if (opts.skillsRoot) return String(opts.skillsRoot)
|
|
34
|
+
const home = String(opts.dshHome || '').trim()
|
|
35
|
+
if (!home) return null
|
|
36
|
+
return join(home, 'skills')
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** 从项目根路径推出一个稳定的项目名(用于 ⑪-3 的「适用项目」标注)。 */
|
|
40
|
+
export function projectNameFromPre(projectPath) {
|
|
41
|
+
const p = String(projectPath || '').trim().replace(/[\\/]+$/, '')
|
|
42
|
+
if (!p) return ''
|
|
43
|
+
const base = p.split(/[\\/]/).filter(Boolean).pop() || ''
|
|
44
|
+
return base
|
|
45
|
+
}
|
|
46
|
+
|
|
47
|
+
/** 收集可附带的程序文件(⑪-1):只收**明确的程序后缀**,不递归、不猜。 */
|
|
48
|
+
export function collectProgramsPre(projectPath, names) {
|
|
49
|
+
const out = []
|
|
50
|
+
if (!Array.isArray(names) || !names.length) return out
|
|
51
|
+
for (const n of names) {
|
|
52
|
+
const s = String(n || '').trim()
|
|
53
|
+
if (!s) continue
|
|
54
|
+
// 只认相对文件名,防目录穿越
|
|
55
|
+
if (s.includes('..') || /^[\\/]/.test(s) || /^[a-zA-Z]:/.test(s)) continue
|
|
56
|
+
out.push(s)
|
|
57
|
+
}
|
|
58
|
+
return out
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
function readIfExists(p) {
|
|
62
|
+
try { return existsSync(p) ? readFileSync(p, 'utf8') : null } catch (_) { return null }
|
|
63
|
+
}
|
|
64
|
+
|
|
65
|
+
/**
|
|
66
|
+
* 导出一个 procedure 为 `<skillsRoot>/<dirName>/SKILL.md`。
|
|
67
|
+
*
|
|
68
|
+
* @param {object} procedure procedure 记录(stage 应为 active)
|
|
69
|
+
* @param {object} opts
|
|
70
|
+
* @param {string} opts.skillsRoot 技能根目录(必填;见 resolveSkillsRootPre)
|
|
71
|
+
* @param {string} opts.projectPath 项目根绝对路径
|
|
72
|
+
* @param {string} opts.projectName 项目名(缺省由 projectPath 推导)
|
|
73
|
+
* @param {string[]} opts.programs 附带程序相对路径列表
|
|
74
|
+
* @param {string} opts.exportedAt ISO 时间
|
|
75
|
+
* @param {boolean} opts.force 为 true 时允许覆盖**非本插件**的同名目录
|
|
76
|
+
* @returns {{ok:boolean, reason?:string, dir?:string, file?:string, bytes?:number}}
|
|
77
|
+
*/
|
|
78
|
+
export function exportSkillForPre(procedure, opts = {}) {
|
|
79
|
+
try {
|
|
80
|
+
const skillsRoot = String(opts.skillsRoot || '')
|
|
81
|
+
if (!skillsRoot) return { ok: false, reason: 'no-skills-root' }
|
|
82
|
+
|
|
83
|
+
const projectPath = String(opts.projectPath || '')
|
|
84
|
+
const projectName = String(opts.projectName || '').trim() || projectNameFromPre(projectPath)
|
|
85
|
+
if (!projectName) return { ok: false, reason: 'no-project-name' }
|
|
86
|
+
|
|
87
|
+
const programs = collectProgramsPre(projectPath, opts.programs)
|
|
88
|
+
const r = renderSkillMarkdownPre(procedure, {
|
|
89
|
+
projectName,
|
|
90
|
+
projectPath,
|
|
91
|
+
programs,
|
|
92
|
+
exportedAt: opts.exportedAt || new Date().toISOString(),
|
|
93
|
+
})
|
|
94
|
+
if (!r.ok) return { ok: false, reason: r.reason }
|
|
95
|
+
|
|
96
|
+
// 自带标记 + 完整性自检(判据 = 用户 ⑪-1/⑪-3 的硬要求)
|
|
97
|
+
const content = r.content + '\n<!-- ' + SKILL_EXPORT_STAMP_V1 + ' -->\n'
|
|
98
|
+
const v = validateSkillMarkdownPre(content, { projectName, hasPrograms: programs.length > 0 })
|
|
99
|
+
if (!v.ok) return { ok: false, reason: 'notice-incomplete:' + v.problems.join('|') }
|
|
100
|
+
|
|
101
|
+
const dir = join(skillsRoot, r.dirName)
|
|
102
|
+
// 防误覆盖:目录已存在且**不是**本插件产物 ⇒ 拒绝(除非 force)
|
|
103
|
+
if (existsSync(dir) && !opts.force) {
|
|
104
|
+
const old = join(dir, 'SKILL.md')
|
|
105
|
+
const prev = readIfExists(old)
|
|
106
|
+
const isOurs = prev != null && prev.includes(SKILL_EXPORT_STAMP_V1)
|
|
107
|
+
const isOursByName = r.dirName.startsWith(SKILL_EXPORT_PREFIX_V1)
|
|
108
|
+
if (!isOurs && isOursByName) {
|
|
109
|
+
// 按名字是我们的,但内容不是 ⇒ 用户手写过,别动
|
|
110
|
+
if (prev != null) return { ok: false, reason: 'dir-occupied-by-foreign' }
|
|
111
|
+
}
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
mkdirSync(dir, { recursive: true })
|
|
115
|
+
// 临时文件 + rename ⇒ 避免半截文件被 DSH 扫描到
|
|
116
|
+
const tmp = join(dir, '.SKILL.md.tmp-' + process.pid + '-' + Date.now())
|
|
117
|
+
const file = join(dir, 'SKILL.md')
|
|
118
|
+
writeFileSync(tmp, content, 'utf8')
|
|
119
|
+
renameSync(tmp, file)
|
|
120
|
+
|
|
121
|
+
return { ok: true, dir, file, bytes: Buffer.byteLength(content, 'utf8'), dirName: r.dirName }
|
|
122
|
+
} catch (e) {
|
|
123
|
+
return { ok: false, reason: 'export-failed:' + String((e && e.message) || e) }
|
|
124
|
+
}
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
/** 列出已导出的技能(供诊断/GUI)。 */
|
|
128
|
+
export function listExportedSkillsPre(skillsRoot) {
|
|
129
|
+
const root = String(skillsRoot || '')
|
|
130
|
+
if (!root || !existsSync(root)) return []
|
|
131
|
+
const out = []
|
|
132
|
+
let entries = []
|
|
133
|
+
try { entries = readdirSync(root) } catch (_) { return [] }
|
|
134
|
+
for (const name of entries) {
|
|
135
|
+
if (!name.startsWith(SKILL_EXPORT_PREFIX_V1)) continue
|
|
136
|
+
const dir = join(root, name)
|
|
137
|
+
try { if (!statSync(dir).isDirectory()) continue } catch (_) { continue }
|
|
138
|
+
const prev = readIfExists(join(dir, 'SKILL.md'))
|
|
139
|
+
if (prev == null || !prev.includes(SKILL_EXPORT_STAMP_V1)) continue
|
|
140
|
+
const nameLine = /^name:\s*(.+)$/m.exec(prev)
|
|
141
|
+
const projLine = /适用项目\*\*:`([^`]+)`/.exec(prev)
|
|
142
|
+
out.push({
|
|
143
|
+
dirName: name,
|
|
144
|
+
dir,
|
|
145
|
+
name: nameLine ? nameLine[1].trim() : name,
|
|
146
|
+
project: projLine ? projLine[1] : '',
|
|
147
|
+
bytes: Buffer.byteLength(prev, 'utf8'),
|
|
148
|
+
})
|
|
149
|
+
}
|
|
150
|
+
return out
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
export { renderSkillMarkdownPre, validateSkillMarkdownPre } from './skill-export.js'
|