@yolk_vat-y/dsh-project-memory 0.5.6 → 0.5.8
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +153 -0
- package/README.md +37 -34
- package/README.zh-CN.md +35 -34
- package/client/client.js +458 -107
- package/client/client.js.map +1 -1
- package/package.json +3 -6
- package/src/audit.js +63 -0
- package/src/auto-inject.js +447 -286
- package/src/chunker.js +14 -6
- package/src/client/MemoryView.tsx +12 -4
- package/src/client/TaskCommandNode.tsx +1 -1
- package/src/client/TaskComponents.tsx +3 -2
- package/src/client/TaskPanel.tsx +13 -13
- package/src/client/client.ts +93 -21
- package/src/client/icons.ts +63 -0
- package/src/client/locales.ts +11 -0
- package/src/client/session-id.js +88 -0
- package/src/client/slash.ts +184 -0
- package/src/commands/insight-actions.js +31 -16
- package/src/commands/invocation.js +36 -0
- package/src/commands/task-actions.js +36 -40
- package/src/commands/tasks.js +18 -13
- package/src/commands/workflow.js +54 -0
- package/src/enhancer.js +11 -2
- package/src/index-pipeline.js +119 -0
- package/src/index.js +24 -8
- package/src/insight-store.js +96 -28
- package/src/lazy.js +15 -46
- package/src/link.js +10 -1
- package/src/parsers/pdfjs-parser.js +2 -4
- package/src/project-profile.js +12 -5
- package/src/readiness.js +27 -3
- package/src/recall.js +28 -8
- package/src/setup/taskbridge.js +11 -9
- package/src/store.js +56 -17
- package/src/symbols.js +58 -19
- package/src/tools/forget.js +2 -1
- package/src/tools/index-doc.js +4 -1
- package/src/tools/index-repo.js +33 -86
- package/src/tools/lesson-tools.js +2 -1
- package/src/tools/query-memory.js +29 -4
- package/src/tools/remember.js +3 -1
- package/src/tools/task-tools.js +4 -1
- package/src/tools/watch-repo.js +9 -5
- package/src/util/fs.js +14 -0
- package/src/util/session-cache.js +72 -0
- package/src/watch.js +39 -81
|
@@ -0,0 +1,119 @@
|
|
|
1
|
+
// 索引一个文件的**唯一**实现。
|
|
2
|
+
//
|
|
3
|
+
// 这段逻辑以前在 index_repo(批量)、watch(轮询)、lazy(fs/observed 懒索引)里各抄一份,
|
|
4
|
+
// 三份随后各自打补丁、已经漂移:index_repo 多一道冗余的 dump 预检、体积超限时删旧记录,
|
|
5
|
+
// 而 watch/lazy 保留旧记录;terms 回填与单次读盘的写法也各不相同。
|
|
6
|
+
// 现在「这个文件该不该读、读出来该写什么条目」只在这里定义一次;
|
|
7
|
+
// 调用方只保留自己的编排(批量计数、watch 的 mtime 快照、lazy 的根解析与副作用)。
|
|
8
|
+
import { statSync } from 'node:fs'
|
|
9
|
+
import { isSupportedCode, isSupportedDoc, readFileForIndex } from './util/fs.js'
|
|
10
|
+
import { buildDocEntries } from './doc-pipeline.js'
|
|
11
|
+
import { docEntriesNeedBackfill } from './doc-index.js'
|
|
12
|
+
import { scanSymbols } from './symbols.js'
|
|
13
|
+
import { linkEntries } from './link.js'
|
|
14
|
+
|
|
15
|
+
/** 不索引的后缀。 */
|
|
16
|
+
export const UNSUPPORTED = 'unsupported'
|
|
17
|
+
/** 代码文件超过 maxFileSizeMb:读盘成本不抵符号表收益。 */
|
|
18
|
+
export const OVERSIZE = 'oversize'
|
|
19
|
+
/** 内容哈希未变,且无需回填词项。 */
|
|
20
|
+
export const UNCHANGED = 'unchanged'
|
|
21
|
+
/** 文档不可索引(dump / 抽取返回 null)→ 从 store 移除。 */
|
|
22
|
+
export const DROP = 'drop'
|
|
23
|
+
/** 有内容要写入 store。 */
|
|
24
|
+
export const WRITE = 'write'
|
|
25
|
+
|
|
26
|
+
/** 后缀 → 'code' | 'doc' | null。 */
|
|
27
|
+
export function fileKind(ext) {
|
|
28
|
+
if (isSupportedCode(ext)) return 'code'
|
|
29
|
+
if (isSupportedDoc(ext)) return 'doc'
|
|
30
|
+
return null
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function exceedsSizeCap(kind, size, config) {
|
|
34
|
+
return kind === 'code' && !!config.maxFileSizeMb && size > config.maxFileSizeMb * 1024 * 1024
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
/**
|
|
38
|
+
* 读取 + 判重 + 抽条目。抛出的异常(读盘失败、PDF 解析失败……)由调用方按各自策略处理:
|
|
39
|
+
* index_repo 收进 failures 汇总,watch 去重打印并可重试,lazy 记一行日志。
|
|
40
|
+
*
|
|
41
|
+
* @returns {{key: string, hash?: string, size?: number, type?: string, entries?: object[]}}
|
|
42
|
+
* key 为 UNSUPPORTED / OVERSIZE / UNCHANGED / DROP / WRITE 之一。
|
|
43
|
+
* @param {object} args
|
|
44
|
+
* @param {number} [args.size] 调用方已 statSync 时传入,避免重复 stat。
|
|
45
|
+
* @param {boolean} [args.force] index_repo 的 reindex:忽略哈希跳过。
|
|
46
|
+
*/
|
|
47
|
+
export async function planFileIndex({ rel, filePath, kind, config, record, existingEntries = [], size, force = false }) {
|
|
48
|
+
if (!kind) return { key: UNSUPPORTED }
|
|
49
|
+
const fileSize = typeof size === 'number' ? size : statSync(filePath).size
|
|
50
|
+
if (exceedsSizeCap(kind, fileSize, config)) return { key: OVERSIZE }
|
|
51
|
+
|
|
52
|
+
// 单次读盘:同一 buffer 供内容哈希与正文解码使用(未变更的文件不解码)。
|
|
53
|
+
const { hash, buffer } = readFileForIndex(filePath)
|
|
54
|
+
// 旧 store 的 doc 条目缺 terms → 即使哈希未变也重抽一次(一次性回填)。
|
|
55
|
+
const backfill = kind === 'doc' && docEntriesNeedBackfill(existingEntries)
|
|
56
|
+
if (!force && record && record.sha256 === hash && !backfill) return { key: UNCHANGED }
|
|
57
|
+
|
|
58
|
+
if (kind === 'code') {
|
|
59
|
+
return { key: WRITE, hash, size: fileSize, type: 'code', entries: scanSymbols(rel, filePath, buffer.toString('utf8')) }
|
|
60
|
+
}
|
|
61
|
+
// dump 判定在 buildDocEntries 内部(唯一真源):不可索引的文档返回 null。
|
|
62
|
+
const entries = await buildDocEntries(rel, filePath, {
|
|
63
|
+
chunkChars: config.chunkChars,
|
|
64
|
+
maxChunks: config.maxChunksPerFile,
|
|
65
|
+
maxFileSizeMb: config.maxFileSizeMb,
|
|
66
|
+
maxPdfPages: config.maxPdfPages,
|
|
67
|
+
})
|
|
68
|
+
return entries === null ? { key: DROP } : { key: WRITE, hash, size: fileSize, type: 'doc', entries }
|
|
69
|
+
}
|
|
70
|
+
|
|
71
|
+
/**
|
|
72
|
+
* plan → 落盘用的更新记录。只接受 DROP / WRITE:UNCHANGED / OVERSIZE 由调用方自己跳过,
|
|
73
|
+
* 误传会直接抛错——静默写进一条 entries=undefined 的记录才是真正的坑。
|
|
74
|
+
* `expectedHash` 是 CAS 基准:扫描之后文件又被改动时,commit 会拒绝这次写入,
|
|
75
|
+
* 调用方据此回滚自己的快照(见 commitFileUpdates 的返回值)。
|
|
76
|
+
*/
|
|
77
|
+
export function toFileUpdate(rel, plan, record) {
|
|
78
|
+
if (plan.key === DROP) return { rel, drop: true }
|
|
79
|
+
if (plan.key !== WRITE) throw new Error(`toFileUpdate: unexpected plan key ${plan.key}`)
|
|
80
|
+
return {
|
|
81
|
+
rel,
|
|
82
|
+
expectedHash: record?.sha256 ?? null,
|
|
83
|
+
hash: plan.hash,
|
|
84
|
+
size: plan.size,
|
|
85
|
+
type: plan.type,
|
|
86
|
+
entries: plan.entries,
|
|
87
|
+
}
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
/**
|
|
91
|
+
* 一批更新一次性落盘:写入 / 移除 → 清掉本轮未见到的旧条目 → 重建链接。
|
|
92
|
+
* 单事务的好处是 store 只 save 一次,watch 每轮不会反复重写。
|
|
93
|
+
*
|
|
94
|
+
* @returns {{stale: string[], removed: number}} stale 是 CAS 失败(并发改动)的 rel,
|
|
95
|
+
* 调用方应让它们保持「未落快照」状态,下一轮重试。
|
|
96
|
+
*/
|
|
97
|
+
export function commitFileUpdates(store, { updates, unseen = null, link = true }) {
|
|
98
|
+
const stale = []
|
|
99
|
+
let removed = 0
|
|
100
|
+
store.commit((s) => {
|
|
101
|
+
for (const update of updates) {
|
|
102
|
+
if (update.drop) {
|
|
103
|
+
s.removeFile(update.rel)
|
|
104
|
+
continue
|
|
105
|
+
}
|
|
106
|
+
if (!s.applyFileUpdate(update.rel, update)) stale.push(update.rel)
|
|
107
|
+
}
|
|
108
|
+
if (unseen) {
|
|
109
|
+
for (const rel of Object.keys(s.files)) {
|
|
110
|
+
if (!unseen.has(rel)) {
|
|
111
|
+
s.removeFile(rel)
|
|
112
|
+
removed++
|
|
113
|
+
}
|
|
114
|
+
}
|
|
115
|
+
}
|
|
116
|
+
if (link) linkEntries(s)
|
|
117
|
+
})
|
|
118
|
+
return { stale, removed }
|
|
119
|
+
}
|
package/src/index.js
CHANGED
|
@@ -14,9 +14,7 @@ import { listTasksTool, selectTaskTool, archiveTaskTool, showTaskPanelTool } fro
|
|
|
14
14
|
import { lessonTool } from './tools/lesson-tools.js'
|
|
15
15
|
import { installAutoInject } from './auto-inject.js'
|
|
16
16
|
import { rememberRoute } from './llm-route.js'
|
|
17
|
-
import {
|
|
18
|
-
import { taskCommandDefinition } from './commands/task-actions.js'
|
|
19
|
-
import { insightCommandDefinition } from './commands/insight-actions.js'
|
|
17
|
+
import { workflowCommandDefinition } from './commands/workflow.js'
|
|
20
18
|
|
|
21
19
|
export const name = 'dsh-project-memory'
|
|
22
20
|
export const inject = ['llm', 'tools']
|
|
@@ -90,6 +88,24 @@ export const Config = Schema.object({
|
|
|
90
88
|
// 同一条 insight 在本会话里重复注入的冷却(pre-step 步数)。0(默认)= 正文没变就不再注入:
|
|
91
89
|
// 注入消息留在会话历史里(宿主只追加不压缩),整块重发只是重复占位。>0 用于外部裁剪历史的场景。
|
|
92
90
|
reinjectItemsAfter: Schema.number().default(0),
|
|
91
|
+
// --- 准入旋钮(0.5.8 起补声明)---
|
|
92
|
+
// 下面这些键自 S2/S4 起就在 cfgEngine / cfgAudit 里生效、README 也一直写着,但**从未**在
|
|
93
|
+
// Schema 里声明过:走 cordis.patch.yml 配它们会被宿主按「not a declared property」拒掉,
|
|
94
|
+
// 等于文档里的旋钮是假的。默认值以 cfgEngine 的兜底值为准(那里是权威,这里只负责暴露)。
|
|
95
|
+
gateCooldownSteps: Schema.number().default(2),
|
|
96
|
+
maxItemsPerSession: Schema.number().default(12),
|
|
97
|
+
maxItemCharsPerSession: Schema.number().default(4000),
|
|
98
|
+
hintMinCoverage: Schema.number().default(0.45),
|
|
99
|
+
hintMinMatched: Schema.number().default(2),
|
|
100
|
+
hintMinSupport: Schema.number().default(0.15),
|
|
101
|
+
legacyScope: Schema.union(['filter', 'ignore']).default('filter'),
|
|
102
|
+
auditLog: Schema.boolean().default(true),
|
|
103
|
+
auditMaxBytes: Schema.number().default(262144),
|
|
104
|
+
// 影子记录(admission-shadow.jsonl):**每步**一行,含全部候选的判据特征与场景。
|
|
105
|
+
// 主审计只在真的注入时写,静默步零痕迹 → 无法离线重放"换个阈值会怎样",也攒不出样本。
|
|
106
|
+
// 只写盘、不进 prompt、不花 token,所以默认开。
|
|
107
|
+
shadowLog: Schema.boolean().default(true),
|
|
108
|
+
shadowMaxBytes: Schema.number().default(2097152),
|
|
93
109
|
}).default({}),
|
|
94
110
|
})
|
|
95
111
|
|
|
@@ -123,15 +139,15 @@ export function apply(ctx, config) {
|
|
|
123
139
|
ctx.tools.register(archiveTaskTool(config, { llm: ctx.llm, ctx }))
|
|
124
140
|
ctx.tools.register(showTaskPanelTool(config))
|
|
125
141
|
|
|
126
|
-
// /tasks、/task、/insight
|
|
142
|
+
// /tasks:唯一的用户命令(合并自原 /tasks、/task、/insight —— 宿主命令只要注册就会
|
|
143
|
+
// 出现在 `/` 菜单的「指令」组里且无法隐藏,三条命令就是三行去不掉的原始行)。
|
|
144
|
+
// 切换/归档/审核等动作作为子动词,由面板按钮经 remote.commands.execute 驱动。
|
|
127
145
|
try {
|
|
128
146
|
ctx.inject(['commands'], (commandsCtx) => {
|
|
129
|
-
commandsCtx.commands.register(
|
|
130
|
-
commandsCtx.commands.register(taskCommandDefinition(config, ctx))
|
|
131
|
-
commandsCtx.commands.register(insightCommandDefinition(config, ctx))
|
|
147
|
+
commandsCtx.commands.register(workflowCommandDefinition(config, ctx))
|
|
132
148
|
})
|
|
133
149
|
} catch (err) {
|
|
134
|
-
console.error(`[dsh-project-memory] /tasks
|
|
150
|
+
console.error(`[dsh-project-memory] /tasks registration skipped: ${err.message}`)
|
|
135
151
|
}
|
|
136
152
|
|
|
137
153
|
ctx.tools.register(indexDocTool(ctx, config))
|
package/src/insight-store.js
CHANGED
|
@@ -10,6 +10,7 @@ import { mkdirSync, readFileSync, renameSync, writeFileSync } from 'node:fs'
|
|
|
10
10
|
import path from 'node:path'
|
|
11
11
|
import os from 'node:os'
|
|
12
12
|
import { normalizedTokenOverlap, findBestOverlapMatch, insightMatchText } from './similarity.js'
|
|
13
|
+
import { backfillDerivedTriggers } from './readiness.js'
|
|
13
14
|
|
|
14
15
|
export const INSIGHT_KINDS = ['lesson', 'decision', 'procedure', 'experience']
|
|
15
16
|
export const INSIGHT_SCOPES = ['task', 'project', 'global']
|
|
@@ -134,6 +135,23 @@ export function unionStrings(base = [], add = []) {
|
|
|
134
135
|
return [...out]
|
|
135
136
|
}
|
|
136
137
|
|
|
138
|
+
/**
|
|
139
|
+
* 跨 scope 移动(promote/demote)时保留"使用痕迹"。
|
|
140
|
+
*
|
|
141
|
+
* 提升/降级只改 scope,不该清零命中数、创建时间,也不该丢掉 `triggerDerived`
|
|
142
|
+
* (提示通道拿它当检索词)。这些字段不在 {@link normalizeInsight} 的白名单里,
|
|
143
|
+
* 必须显式带回——否则每次移动都会让条目"看起来从没被用过",global 层更是每次都丢派生词。
|
|
144
|
+
* @param {object} copy - 已归一化到新 scope 的副本
|
|
145
|
+
* @param {object} src - 原条目
|
|
146
|
+
*/
|
|
147
|
+
export function carryUsageFields(copy, src) {
|
|
148
|
+
copy.hitCount = num(src.hitCount, 0)
|
|
149
|
+
if (src.lastHitAt) copy.lastHitAt = src.lastHitAt
|
|
150
|
+
if (src.createdAt) copy.createdAt = src.createdAt
|
|
151
|
+
if (src.triggerDerived) copy.triggerDerived = src.triggerDerived
|
|
152
|
+
return copy
|
|
153
|
+
}
|
|
154
|
+
|
|
137
155
|
/** 把 base 合并进既有条目(content 加固、成员/文件/符号并集、置信取高、记命中)。 */
|
|
138
156
|
export function mergeInto(existing, base, cfg, nowIso) {
|
|
139
157
|
const now = nowIso || new Date().toISOString()
|
|
@@ -162,6 +180,45 @@ export function mergeInto(existing, base, cfg, nowIso) {
|
|
|
162
180
|
return existing
|
|
163
181
|
}
|
|
164
182
|
|
|
183
|
+
/**
|
|
184
|
+
* 使用记账:条目被**真正用到**时更新活跃度——注入进了上下文,或被 `query_memory` 命中。
|
|
185
|
+
*
|
|
186
|
+
* 为什么必须有:`applyDecay` / `pruneItems` 判活跃度只看 `lastHitAt || updatedAt || createdAt`,
|
|
187
|
+
* 而 `lastHitAt` 此前只由 merge/reinforce 写(= 模型又写了一条相近的知识)。于是一条天天被注入、
|
|
188
|
+
* 但从没人重写它的教训,`decayDays`(默认 90)之后会被自动归档、再也不会被推送——**用得最多的
|
|
189
|
+
* 反而等于没人用过**。这是行为缺陷,不是调优问题。
|
|
190
|
+
*
|
|
191
|
+
* 记账**不参与任何注入判据**:它只影响活跃度/衰减,以及离线训练样本的标签。
|
|
192
|
+
* 语义是"曝光次数"而非"被采纳次数"——更强的信号(模型是否真的照着做了)需要另外的回路。
|
|
193
|
+
* 只记非归档条目(归档件本来就召回不到)。返回实际加一的条数,调用方据此决定是否落盘。
|
|
194
|
+
* @returns {number} 被加一的条目数
|
|
195
|
+
*/
|
|
196
|
+
export function recordHit({ store, globalStore, ids, nowIso } = {}) {
|
|
197
|
+
const want = new Set((ids || []).filter(Boolean))
|
|
198
|
+
if (!want.size) return 0
|
|
199
|
+
const now = nowIso || new Date().toISOString()
|
|
200
|
+
let n = 0
|
|
201
|
+
const touch = (items) => {
|
|
202
|
+
let c = 0
|
|
203
|
+
for (const it of items || []) {
|
|
204
|
+
if (!it || it.archived || !want.has(it.id)) continue
|
|
205
|
+
it.hitCount = num(it.hitCount, 0) + 1
|
|
206
|
+
it.lastHitAt = now
|
|
207
|
+
c++
|
|
208
|
+
}
|
|
209
|
+
n += c
|
|
210
|
+
return c
|
|
211
|
+
}
|
|
212
|
+
if (store && typeof store.insightItems === 'function') {
|
|
213
|
+
const items = store.insightItems()
|
|
214
|
+
if (touch(items)) store.replaceInsightItems(items) // 标脏;落盘由调用方决定
|
|
215
|
+
}
|
|
216
|
+
if (globalStore && typeof globalStore.items === 'function') {
|
|
217
|
+
if (touch(globalStore.items())) globalStore.markDirty()
|
|
218
|
+
}
|
|
219
|
+
return n
|
|
220
|
+
}
|
|
221
|
+
|
|
165
222
|
export function reinforceOnly(existing, base, nowIso) {
|
|
166
223
|
const now = nowIso || new Date().toISOString()
|
|
167
224
|
existing.sourceTaskIds = unionStrings(existing.sourceTaskIds, base.sourceTaskIds)
|
|
@@ -210,9 +267,12 @@ export function applyDecay(items, cfg, nowIso) {
|
|
|
210
267
|
const limit = Date.parse(now) - cfg.decayDays * DAY_MS
|
|
211
268
|
let archived = 0
|
|
212
269
|
for (const it of items) {
|
|
213
|
-
if (it.archived) continue
|
|
214
|
-
|
|
215
|
-
|
|
270
|
+
if (!it || it.archived) continue
|
|
271
|
+
// 用"最近活动"而不是 lastHitAt:lastHitAt 只由 merge/reinforce 写,而那两处同时把
|
|
272
|
+
// hitCount 加一。旧实现先 `hitCount > 0 → continue`,于是 lastHitAt 分支永远不可达,
|
|
273
|
+
// decayDays 实际是个死开关。activityOf 退到 updatedAt/createdAt,没命中过的条目也能衰减。
|
|
274
|
+
const last = activityOf(it)
|
|
275
|
+
if (last && last <= limit) {
|
|
216
276
|
it.archived = true
|
|
217
277
|
it.updatedAt = now
|
|
218
278
|
archived++
|
|
@@ -221,7 +281,11 @@ export function applyDecay(items, cfg, nowIso) {
|
|
|
221
281
|
return archived
|
|
222
282
|
}
|
|
223
283
|
|
|
224
|
-
/**
|
|
284
|
+
/**
|
|
285
|
+
* 超限按"最不活跃"物理删除,直到回落到上限。已有 archived 条目优先被删。
|
|
286
|
+
* 注意:没有归档可删时,本轮归档的条目会在同一次调用的后续循环里被删掉——
|
|
287
|
+
* `archived` 计数表示"先归档、随即被删",不是"留下了软删副本"。
|
|
288
|
+
*/
|
|
225
289
|
export function pruneItems(items, max, nowIso) {
|
|
226
290
|
const now = nowIso || new Date().toISOString()
|
|
227
291
|
let removed = 0
|
|
@@ -300,7 +364,11 @@ export class GlobalStore {
|
|
|
300
364
|
}
|
|
301
365
|
|
|
302
366
|
load() {
|
|
303
|
-
if (!this.doc)
|
|
367
|
+
if (!this.doc) {
|
|
368
|
+
this.doc = normalizeDoc(readJson(this.file, null))
|
|
369
|
+
// 与 project 级一致:global 条目也要补派生 trigger(提示通道用它提升召回;确定性、幂等)。
|
|
370
|
+
if (backfillDerivedTriggers(this.doc)) this.markDirty()
|
|
371
|
+
}
|
|
304
372
|
return this
|
|
305
373
|
}
|
|
306
374
|
|
|
@@ -475,16 +543,14 @@ export function promoteAllTasksToProject(store, globalStore, cfg, now) {
|
|
|
475
543
|
promoted++
|
|
476
544
|
continue
|
|
477
545
|
}
|
|
478
|
-
const moved =
|
|
479
|
-
|
|
480
|
-
|
|
481
|
-
|
|
482
|
-
|
|
483
|
-
|
|
484
|
-
|
|
485
|
-
|
|
486
|
-
movedFrom: { scope: 'task', id: ins.id, at: now },
|
|
487
|
-
}
|
|
546
|
+
const moved = carryUsageFields(
|
|
547
|
+
normalizeInsight(ins, { scope: 'project', source: ins.source, nowIso: now }),
|
|
548
|
+
ins,
|
|
549
|
+
)
|
|
550
|
+
moved.id = ins.id
|
|
551
|
+
moved.draft = false
|
|
552
|
+
moved.confidence = num(ins.confidence, cfg.promoteConfidence)
|
|
553
|
+
moved.movedFrom = { scope: 'task', id: ins.id, at: now }
|
|
488
554
|
pjItems.push(moved)
|
|
489
555
|
store.replaceInsightItems(pjItems)
|
|
490
556
|
removeFromTask(store, ins.id)
|
|
@@ -514,16 +580,14 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
|
|
|
514
580
|
if (glHit && glHit.score >= cfg.dedupOverlap) {
|
|
515
581
|
mergeInto(glHit.item, { sourceTaskIds: ins.sourceTaskIds, files: ins.files, symbols: ins.symbols, tags: ins.tags, confidence: ins.confidence }, cfg, now)
|
|
516
582
|
} else {
|
|
517
|
-
const movedIns =
|
|
518
|
-
|
|
519
|
-
|
|
520
|
-
|
|
521
|
-
|
|
522
|
-
|
|
523
|
-
|
|
524
|
-
|
|
525
|
-
movedFrom: { scope: 'project', id: ins.id, at: now },
|
|
526
|
-
}
|
|
583
|
+
const movedIns = carryUsageFields(
|
|
584
|
+
normalizeInsight(ins, { scope: 'global', source: ins.source, nowIso: now }),
|
|
585
|
+
ins,
|
|
586
|
+
)
|
|
587
|
+
movedIns.id = ins.id
|
|
588
|
+
movedIns.draft = false
|
|
589
|
+
movedIns.confidence = num(ins.confidence, cfg.promoteConfidence)
|
|
590
|
+
movedIns.movedFrom = { scope: 'project', id: ins.id, at: now }
|
|
527
591
|
glItems.push(movedIns)
|
|
528
592
|
}
|
|
529
593
|
items.splice(items.indexOf(ins), 1)
|
|
@@ -531,6 +595,8 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
|
|
|
531
595
|
}
|
|
532
596
|
if (moved) {
|
|
533
597
|
store.replaceInsightItems(items)
|
|
598
|
+
// 提升只增不减 global:不在这里收口,maxGlobalProcedures 就形同虚设。
|
|
599
|
+
pruneItems(globalStore.items(), cfg.maxGlobalProcedures, now)
|
|
534
600
|
globalStore.markDirty()
|
|
535
601
|
}
|
|
536
602
|
return moved
|
|
@@ -538,17 +604,19 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
|
|
|
538
604
|
|
|
539
605
|
// ---- 降级(反向,PR3 UI 使用;现在提供最小实现) ----
|
|
540
606
|
|
|
541
|
-
export function demoteToProject(store, globalStore, id, nowIso) {
|
|
607
|
+
export function demoteToProject(store, globalStore, id, nowIso, cfg = cfgInsight({})) {
|
|
542
608
|
if (!globalStore) return { ok: false, error: 'global 存储未初始化' }
|
|
543
609
|
const now = nowIso || new Date().toISOString()
|
|
544
610
|
const idx = globalStore.items().findIndex((i) => i.id === id)
|
|
545
611
|
if (idx === -1) return { ok: false, error: `global 无此条目: ${id}` }
|
|
546
612
|
const ins = globalStore.items()[idx]
|
|
547
613
|
const items = store.insightItems()
|
|
548
|
-
const copy = normalizeInsight(ins, { scope: 'project', nowIso: now })
|
|
614
|
+
const copy = carryUsageFields(normalizeInsight(ins, { scope: 'project', nowIso: now }), ins)
|
|
549
615
|
copy.id = `${ins.id}_d${Date.now().toString(36)}`
|
|
550
616
|
copy.movedFrom = { scope: 'global', id: ins.id, at: now }
|
|
551
617
|
items.push(copy)
|
|
618
|
+
// 降级只增不减 project:同样要收口 maxProject。
|
|
619
|
+
pruneItems(items, cfg.maxProject, now)
|
|
552
620
|
store.replaceInsightItems(items)
|
|
553
621
|
globalStore.items().splice(idx, 1)
|
|
554
622
|
globalStore.markDirty()
|
|
@@ -564,7 +632,7 @@ export function demoteToTask(store, taskId, id, nowIso) {
|
|
|
564
632
|
if (idx === -1) return { ok: false, error: `project 无此条目: ${id}` }
|
|
565
633
|
const ins = items[idx]
|
|
566
634
|
if (!Array.isArray(task.insights)) task.insights = []
|
|
567
|
-
const copy = normalizeInsight(ins, { scope: 'task', nowIso: now })
|
|
635
|
+
const copy = carryUsageFields(normalizeInsight(ins, { scope: 'task', nowIso: now }), ins)
|
|
568
636
|
copy.id = `${ins.id}_d${Date.now().toString(36)}`
|
|
569
637
|
copy.movedFrom = { scope: 'project', id: ins.id, at: now }
|
|
570
638
|
task.insights.push(copy)
|
package/src/lazy.js
CHANGED
|
@@ -1,11 +1,8 @@
|
|
|
1
1
|
import path from 'node:path'
|
|
2
2
|
import { tmpdir } from 'node:os'
|
|
3
3
|
import { existsSync, readdirSync, statSync } from 'node:fs'
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import { docEntriesNeedBackfill } from './doc-index.js'
|
|
7
|
-
import { scanSymbols } from './symbols.js'
|
|
8
|
-
import { linkEntries } from './link.js'
|
|
4
|
+
import { memoryRootFor, relativePath, storeKey } from './util/fs.js'
|
|
5
|
+
import { DROP, OVERSIZE, UNCHANGED, commitFileUpdates, fileKind, planFileIndex, toFileUpdate } from './index-pipeline.js'
|
|
9
6
|
import { ProjectMemoryStore } from './store.js'
|
|
10
7
|
import { onFileObserved } from './enhancer.js'
|
|
11
8
|
|
|
@@ -78,70 +75,42 @@ export function findProjectRoot(filePath, ceiling = path.resolve(tmpdir())) {
|
|
|
78
75
|
}
|
|
79
76
|
|
|
80
77
|
export async function indexFile(ctx, config, filePath, watchManager = null) {
|
|
81
|
-
const
|
|
82
|
-
if (!
|
|
78
|
+
const kind = fileKind(path.extname(filePath).toLowerCase())
|
|
79
|
+
if (!kind) return false
|
|
83
80
|
const root = findProjectRoot(filePath)
|
|
84
81
|
if (!root) return false
|
|
85
82
|
|
|
86
83
|
const memoryDir = memoryRootFor(root, config.memoryDir)
|
|
87
84
|
const store = new ProjectMemoryStore(memoryDir).load()
|
|
88
85
|
const rel = storeKey(relativePath(root, filePath))
|
|
89
|
-
const
|
|
90
|
-
|
|
86
|
+
const record = store.fileRecord(rel)
|
|
87
|
+
|
|
91
88
|
let size
|
|
92
|
-
let
|
|
93
|
-
let entries
|
|
89
|
+
let plan
|
|
94
90
|
try {
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
}
|
|
98
|
-
;({ hash, size, buffer } = readFileForIndex(filePath))
|
|
91
|
+
size = statSync(filePath).size
|
|
92
|
+
plan = await planFileIndex({ rel, filePath, kind, config, record, existingEntries: store.entries[rel], size })
|
|
99
93
|
} catch {
|
|
100
94
|
return false
|
|
101
95
|
}
|
|
102
|
-
|
|
103
|
-
if (existing && existing.sha256 === hash && !(isSupportedDoc(ext) && docEntriesNeedBackfill(store.entries[rel]))) return false
|
|
96
|
+
if (plan.key === UNCHANGED || plan.key === OVERSIZE) return false
|
|
104
97
|
|
|
105
98
|
if (watchManager) {
|
|
106
99
|
watchManager.addRoot(root)
|
|
107
100
|
store.addWatch(root)
|
|
108
101
|
}
|
|
109
102
|
|
|
110
|
-
if (
|
|
111
|
-
entries = scanSymbols(rel, filePath, buffer.toString('utf8'))
|
|
103
|
+
if (plan.key !== DROP && plan.type === 'code') {
|
|
112
104
|
onFileObserved(store, rel, filePath, config, root)
|
|
113
|
-
return store.commit((s) => {
|
|
114
|
-
s.markFile(rel, { sha256: hash, size, type: 'code', indexedAt: new Date().toISOString() })
|
|
115
|
-
s.setEntries(rel, entries)
|
|
116
|
-
linkEntries(s)
|
|
117
|
-
return true
|
|
118
|
-
})
|
|
119
|
-
} else {
|
|
120
|
-
entries = await buildDocEntries(rel, filePath, {
|
|
121
|
-
chunkChars: config.chunkChars,
|
|
122
|
-
maxChunks: config.maxChunksPerFile,
|
|
123
|
-
maxFileSizeMb: config.maxFileSizeMb,
|
|
124
|
-
maxPdfPages: config.maxPdfPages,
|
|
125
|
-
})
|
|
126
|
-
if (entries === null) {
|
|
127
|
-
return store.commit((s) => {
|
|
128
|
-
s.removeFile(rel)
|
|
129
|
-
return false
|
|
130
|
-
})
|
|
131
|
-
}
|
|
132
|
-
return store.commit((s) => {
|
|
133
|
-
s.markFile(rel, { sha256: hash, size, type: 'doc', indexedAt: new Date().toISOString() })
|
|
134
|
-
s.setEntries(rel, entries)
|
|
135
|
-
linkEntries(s)
|
|
136
|
-
return true
|
|
137
|
-
})
|
|
138
105
|
}
|
|
106
|
+
commitFileUpdates(store, { updates: [toFileUpdate(rel, plan, record)] })
|
|
107
|
+
return plan.key !== DROP
|
|
139
108
|
}
|
|
140
109
|
|
|
141
110
|
export function codeFirst(paths) {
|
|
142
111
|
return [...paths].sort((a, b) => {
|
|
143
|
-
const aCode =
|
|
144
|
-
const bCode =
|
|
112
|
+
const aCode = fileKind(path.extname(a).toLowerCase()) === 'code' ? 0 : 1
|
|
113
|
+
const bCode = fileKind(path.extname(b).toLowerCase()) === 'code' ? 0 : 1
|
|
145
114
|
return aCode - bCode
|
|
146
115
|
})
|
|
147
116
|
}
|
package/src/link.js
CHANGED
|
@@ -39,7 +39,16 @@ export function linkEntries(store) {
|
|
|
39
39
|
if (linked.size > before) links++
|
|
40
40
|
}
|
|
41
41
|
}
|
|
42
|
-
|
|
42
|
+
const before = Array.isArray(doc.linkedSymbols) ? doc.linkedSymbols.join('\u0000') : ''
|
|
43
|
+
const next = [...linked]
|
|
44
|
+
if (before === next.join('\u0000')) continue
|
|
45
|
+
doc.linkedSymbols = next.length ? next : undefined
|
|
46
|
+
// 链接是在**符号**落盘那一刻算出来的,此时 doc 的 shard 往往不是脏的;不标脏就只存在于内存,
|
|
47
|
+
// 下次进程启动重新加载后链接全部丢失("文档先索引、符号后到"的正常顺序)。
|
|
48
|
+
if (typeof store.markFile === 'function' && typeof store.fileRecord === 'function' && doc.sourcePath) {
|
|
49
|
+
const record = store.fileRecord(doc.sourcePath)
|
|
50
|
+
if (record) store.markFile(doc.sourcePath, record)
|
|
51
|
+
}
|
|
43
52
|
}
|
|
44
53
|
return links
|
|
45
54
|
}
|
|
@@ -60,9 +60,8 @@ export async function parsePdf(filePath, { pages = null, maxPages = 1000, backen
|
|
|
60
60
|
const data = new Uint8Array(await readFile(filePath))
|
|
61
61
|
const { getDocument } = await loadPdfjs()
|
|
62
62
|
const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
|
|
63
|
-
const doc = await loadingTask.promise
|
|
64
|
-
|
|
65
63
|
try {
|
|
64
|
+
const doc = await loadingTask.promise
|
|
66
65
|
const total = doc.numPages
|
|
67
66
|
if (total > maxPages) {
|
|
68
67
|
throw new Error(`PDF has ${total} pages, over the maxPages limit of ${maxPages}`)
|
|
@@ -112,9 +111,8 @@ export async function parsePdfInfo(filePath, maxPages = 1000) {
|
|
|
112
111
|
const data = new Uint8Array(await readFile(filePath))
|
|
113
112
|
const { getDocument } = await loadPdfjs()
|
|
114
113
|
const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
|
|
115
|
-
const doc = await loadingTask.promise
|
|
116
|
-
|
|
117
114
|
try {
|
|
115
|
+
const doc = await loadingTask.promise
|
|
118
116
|
if (doc.numPages > maxPages) {
|
|
119
117
|
throw new Error(`PDF has ${doc.numPages} pages, over the maxPages limit of ${maxPages}`)
|
|
120
118
|
}
|
package/src/project-profile.js
CHANGED
|
@@ -17,11 +17,18 @@ function depsOf(manifest, fields) {
|
|
|
17
17
|
const tags = new Set()
|
|
18
18
|
for (const f of fields) {
|
|
19
19
|
const deps = manifest && manifest[f]
|
|
20
|
-
if (deps
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
20
|
+
if (!deps || typeof deps !== 'object') continue
|
|
21
|
+
for (const name of Object.keys(deps)) {
|
|
22
|
+
if (typeof name !== 'string' || !name) continue
|
|
23
|
+
tags.add(name.toLowerCase())
|
|
24
|
+
if (!name.startsWith('@')) continue
|
|
25
|
+
// scoped 包名 `@scope/pkg`:要加的是 **scope**(`vue`),不是包名段(`compiler-sfc`)。
|
|
26
|
+
// 旧写法 `name.split('/')[1]` 让 `@vue/*` 项目永远拿不到 `vue` tag,
|
|
27
|
+
// trigger.scope:['vue'] / legacy scope 过滤会把最相关的那条 procedure 静默判死;
|
|
28
|
+
// 而 `@malformed`(没有 `/`)会在这里抛错,异常被 collectTags 整个吞掉 → 全项目 tags 清零。
|
|
29
|
+
const parts = name.slice(1).split('/')
|
|
30
|
+
if (parts[0]) tags.add(parts[0].toLowerCase())
|
|
31
|
+
if (parts[1]) tags.add(parts[1].toLowerCase())
|
|
25
32
|
}
|
|
26
33
|
}
|
|
27
34
|
return tags
|
package/src/readiness.js
CHANGED
|
@@ -82,7 +82,16 @@ export function buildReadinessContext(input = {}) {
|
|
|
82
82
|
const humanText = String(input.humanText ?? input.query ?? '')
|
|
83
83
|
const actionText = String(input.actionText || '')
|
|
84
84
|
const actions = new Set([...(input.actions || []), ...detectActions(humanText), ...detectActions(actionText)])
|
|
85
|
-
|
|
85
|
+
// 人类消息里提到的路径 = **动手前的写意图**("帮我改 src/x.js");动作文本里的路径不区分读写,
|
|
86
|
+
// 读一个文件不该算写目标(否则 `read README.md` 会命中 writes:['README.md'])。
|
|
87
|
+
// 两者对外仍合并成 `paths`(兼容既有语义),写通道单看 humanPaths。
|
|
88
|
+
// 调用方若已带 humanPaths(injectForStep 构建后再传给 buildInjection),以它为准,
|
|
89
|
+
// 否则退回 paths —— 避免把上一次算出来的并集(含读路径)再当写意图。
|
|
90
|
+
const humanPaths = new Set([
|
|
91
|
+
...(Array.isArray(input.humanPaths) ? input.humanPaths : (input.paths || [])),
|
|
92
|
+
...extractPaths(humanText),
|
|
93
|
+
])
|
|
94
|
+
const paths = new Set([...humanPaths, ...extractPaths(actionText)])
|
|
86
95
|
// 动作平面(S1):结构化调用优先;没有结构化调用时对 actionText 跑一遍 shell 规则兜底。
|
|
87
96
|
const fromText = activityFromText(actionText)
|
|
88
97
|
const ops = new Set([...(input.ops || []), ...fromText.ops])
|
|
@@ -97,6 +106,7 @@ export function buildReadinessContext(input = {}) {
|
|
|
97
106
|
actionText,
|
|
98
107
|
actions: [...actions],
|
|
99
108
|
paths: [...paths],
|
|
109
|
+
humanPaths: [...humanPaths],
|
|
100
110
|
ops: [...ops],
|
|
101
111
|
targets: [...targets],
|
|
102
112
|
hosts: [...hosts],
|
|
@@ -209,6 +219,10 @@ export function intentText(text) {
|
|
|
209
219
|
.replace(/“[^”\n]*”/g, ' ')
|
|
210
220
|
.replace(/[A-Za-z]:\\[^\s"']*/g, ' ')
|
|
211
221
|
.replace(/(?:[A-Za-z0-9_.@-]+\/)+[A-Za-z0-9_.@-]+/g, ' ')
|
|
222
|
+
// 带扩展名的**非 ASCII 文件名**(`石啸天-记忆方向调研.pptx` / `docs/面试演示-王金鹏.md`):
|
|
223
|
+
// 只挡 ASCII 路径不够——中文文件名整块留下,"调研""面试"这类子串照样触发 when.intents,
|
|
224
|
+
// 正是引号剥离要防的那类假阳性。ASCII 裸名(如 de-TODO.md)保持原语义,不在这里动。
|
|
225
|
+
.replace(/\S*[^\x00-\x7F]\S*\.[A-Za-z][A-Za-z0-9]{0,7}\b/g, ' ')
|
|
212
226
|
// 仓库里的裸文件名(README / CHANGELOG …)也是语料,不是意图
|
|
213
227
|
.replace(/\b(?:README|CHANGELOG|LICENSE|AGENTS|CONTRIBUTING|Dockerfile|Makefile)\b/g, ' ')
|
|
214
228
|
.replace(/\s+/g, ' ')
|
|
@@ -301,10 +315,14 @@ export function matchTrigger(trigger, ctx) {
|
|
|
301
315
|
for (const op of when.ops || []) {
|
|
302
316
|
if (op && (ctx?.ops || []).includes(String(op))) return `op:${op}`
|
|
303
317
|
}
|
|
318
|
+
// 写目标 = 已观察到的结构化写调用 ∪ 人类消息里提到的路径(动手前的写意图)。
|
|
319
|
+
// DSH 没有工具执行前拦截钩子,动手前唯一能拿到的写意图就是人类消息里的路径;
|
|
320
|
+
// 但动作文本里的**读**路径不算(`read README.md` 不是"即将写 README.md")。
|
|
321
|
+
const writeTargets = [...(ctx?.targets || []), ...(ctx?.humanPaths || [])]
|
|
304
322
|
for (const w of when.writes || []) {
|
|
305
323
|
// 扩展名/泛名 glob 在这里被硬性忽略:`*.pptx` 这类条件只能撒谎,不能收窄。
|
|
306
324
|
if (!w || isDroppableGlob(w)) continue
|
|
307
|
-
if (matchAnyPath(w,
|
|
325
|
+
if (matchAnyPath(w, writeTargets)) return `write:${w}`
|
|
308
326
|
}
|
|
309
327
|
const intent = ctx?.intent ?? ctx?.humanText ?? ''
|
|
310
328
|
for (const it of when.intents || []) {
|
|
@@ -355,6 +373,9 @@ export function normalizeTrigger(it, opts = {}) {
|
|
|
355
373
|
// (实测:那样会让 D 场景一次多出 6 条假阳性)。
|
|
356
374
|
const intents = []
|
|
357
375
|
for (const k of t.keywords || []) if (isIntentWord(k)) intents.push(String(k))
|
|
376
|
+
// 旧 `symbols` 也走文本平面:不归一就等于静默丢掉一个作者写下的触发面
|
|
377
|
+
// (README 承诺 keywords/symbols/actions/paths/scope 都会迁移)。与 keywords 同一把准入尺子。
|
|
378
|
+
for (const s of t.symbols || []) if (isIntentWord(s)) intents.push(String(s))
|
|
358
379
|
const when = {}
|
|
359
380
|
if (ops.size) when.ops = [...ops].sort()
|
|
360
381
|
if (writes.length) when.writes = [...new Set(writes)]
|
|
@@ -463,6 +484,9 @@ export function relativeHits(scored, { ratioMin = 0.5 } = {}) {
|
|
|
463
484
|
const list = (scored || []).filter((r) => r && Number.isFinite(r.score) && r.score > 0)
|
|
464
485
|
if (!list.length) return []
|
|
465
486
|
const top = list[0].score
|
|
466
|
-
|
|
487
|
+
// 显式的 0 表示"关掉相对门槛"(只剩 score>0);未给/非法值才回退 0.5。
|
|
488
|
+
// 旧写法 `ratioMin > 0 ? ratioMin : 0.5` 把 0 当成"没配",配置上无法关闭。
|
|
489
|
+
const ratio = typeof ratioMin === 'number' && Number.isFinite(ratioMin) ? ratioMin : 0.5
|
|
490
|
+
const floor = top * Math.max(0, Math.min(1, ratio))
|
|
467
491
|
return list.filter((r) => r.score >= floor)
|
|
468
492
|
}
|