@yolk_vat-y/dsh-project-memory 0.5.5 → 0.5.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +71 -2
- package/README.md +27 -4
- package/README.zh-CN.md +26 -3
- package/package.json +4 -2
- package/src/audit.js +87 -0
- package/src/auto-inject.js +383 -197
- package/src/chunker.js +14 -6
- package/src/commands/insight-actions.js +18 -14
- package/src/commands/invocation.js +36 -0
- package/src/commands/task-actions.js +24 -39
- package/src/commands/tasks.js +18 -13
- package/src/enhancer.js +11 -2
- package/src/index-pipeline.js +119 -0
- package/src/insight-store.js +78 -29
- package/src/lazy.js +15 -46
- package/src/link.js +10 -1
- package/src/ops.js +210 -0
- package/src/parsers/pdfjs-parser.js +2 -4
- package/src/project-profile.js +12 -5
- package/src/readiness.js +305 -37
- package/src/recall.js +28 -8
- package/src/setup/taskbridge.js +11 -9
- package/src/store.js +56 -17
- package/src/symbols.js +58 -19
- package/src/tools/forget.js +2 -1
- package/src/tools/index-doc.js +4 -1
- package/src/tools/index-repo.js +33 -86
- package/src/tools/lesson-tools.js +31 -8
- package/src/tools/query-memory.js +17 -3
- package/src/tools/remember.js +3 -1
- package/src/tools/task-tools.js +4 -1
- package/src/tools/watch-repo.js +9 -5
- package/src/util/fs.js +14 -0
- package/src/util/session-cache.js +72 -0
- package/src/watch.js +39 -81
package/src/insight-store.js
CHANGED
|
@@ -10,6 +10,7 @@ import { mkdirSync, readFileSync, renameSync, writeFileSync } from 'node:fs'
|
|
|
10
10
|
import path from 'node:path'
|
|
11
11
|
import os from 'node:os'
|
|
12
12
|
import { normalizedTokenOverlap, findBestOverlapMatch, insightMatchText } from './similarity.js'
|
|
13
|
+
import { backfillDerivedTriggers } from './readiness.js'
|
|
13
14
|
|
|
14
15
|
export const INSIGHT_KINDS = ['lesson', 'decision', 'procedure', 'experience']
|
|
15
16
|
export const INSIGHT_SCOPES = ['task', 'project', 'global']
|
|
@@ -98,11 +99,31 @@ export function normalizeInsight(raw, extra = {}) {
|
|
|
98
99
|
if (Array.isArray(raw[f]) && raw[f].length) ins[f] = [...new Set(raw[f].map((x) => String(x)))]
|
|
99
100
|
}
|
|
100
101
|
if (raw.trigger && typeof raw.trigger === 'object') {
|
|
101
|
-
// 所有 kind
|
|
102
|
+
// 所有 kind 通用。准入化 schema:when(唯一触发面)/ guard(只收窄)/ prevents(准入条件);
|
|
103
|
+
// 旧字段 keywords / symbols / actions / paths / scope 继续接受(readiness 会内存内迁移)。
|
|
102
104
|
const tr = {}
|
|
103
105
|
for (const f of ['keywords', 'symbols', 'actions', 'paths', 'scope']) {
|
|
104
106
|
if (Array.isArray(raw.trigger[f]) && raw.trigger[f].length) tr[f] = raw.trigger[f].map(String)
|
|
105
107
|
}
|
|
108
|
+
const when = raw.trigger.when
|
|
109
|
+
if (when && typeof when === 'object') {
|
|
110
|
+
const w = {}
|
|
111
|
+
for (const f of ['ops', 'writes', 'intents']) {
|
|
112
|
+
if (Array.isArray(when[f]) && when[f].length) w[f] = when[f].map(String)
|
|
113
|
+
}
|
|
114
|
+
if (Object.keys(w).length) tr.when = w
|
|
115
|
+
}
|
|
116
|
+
const guard = raw.trigger.guard
|
|
117
|
+
if (guard && typeof guard === 'object') {
|
|
118
|
+
const g = {}
|
|
119
|
+
for (const f of ['paths', 'not_paths', 'hosts', 'tags']) {
|
|
120
|
+
if (Array.isArray(guard[f]) && guard[f].length) g[f] = guard[f].map(String)
|
|
121
|
+
}
|
|
122
|
+
if (Object.keys(g).length) tr.guard = g
|
|
123
|
+
}
|
|
124
|
+
if (typeof raw.trigger.prevents === 'string' && raw.trigger.prevents.trim()) {
|
|
125
|
+
tr.prevents = raw.trigger.prevents.trim()
|
|
126
|
+
}
|
|
106
127
|
if (Object.keys(tr).length) ins.trigger = tr
|
|
107
128
|
}
|
|
108
129
|
return ins
|
|
@@ -114,6 +135,23 @@ export function unionStrings(base = [], add = []) {
|
|
|
114
135
|
return [...out]
|
|
115
136
|
}
|
|
116
137
|
|
|
138
|
+
/**
|
|
139
|
+
* 跨 scope 移动(promote/demote)时保留"使用痕迹"。
|
|
140
|
+
*
|
|
141
|
+
* 提升/降级只改 scope,不该清零命中数、创建时间,也不该丢掉 `triggerDerived`
|
|
142
|
+
* (提示通道拿它当检索词)。这些字段不在 {@link normalizeInsight} 的白名单里,
|
|
143
|
+
* 必须显式带回——否则每次移动都会让条目"看起来从没被用过",global 层更是每次都丢派生词。
|
|
144
|
+
* @param {object} copy - 已归一化到新 scope 的副本
|
|
145
|
+
* @param {object} src - 原条目
|
|
146
|
+
*/
|
|
147
|
+
export function carryUsageFields(copy, src) {
|
|
148
|
+
copy.hitCount = num(src.hitCount, 0)
|
|
149
|
+
if (src.lastHitAt) copy.lastHitAt = src.lastHitAt
|
|
150
|
+
if (src.createdAt) copy.createdAt = src.createdAt
|
|
151
|
+
if (src.triggerDerived) copy.triggerDerived = src.triggerDerived
|
|
152
|
+
return copy
|
|
153
|
+
}
|
|
154
|
+
|
|
117
155
|
/** 把 base 合并进既有条目(content 加固、成员/文件/符号并集、置信取高、记命中)。 */
|
|
118
156
|
export function mergeInto(existing, base, cfg, nowIso) {
|
|
119
157
|
const now = nowIso || new Date().toISOString()
|
|
@@ -190,9 +228,12 @@ export function applyDecay(items, cfg, nowIso) {
|
|
|
190
228
|
const limit = Date.parse(now) - cfg.decayDays * DAY_MS
|
|
191
229
|
let archived = 0
|
|
192
230
|
for (const it of items) {
|
|
193
|
-
if (it.archived) continue
|
|
194
|
-
|
|
195
|
-
|
|
231
|
+
if (!it || it.archived) continue
|
|
232
|
+
// 用"最近活动"而不是 lastHitAt:lastHitAt 只由 merge/reinforce 写,而那两处同时把
|
|
233
|
+
// hitCount 加一。旧实现先 `hitCount > 0 → continue`,于是 lastHitAt 分支永远不可达,
|
|
234
|
+
// decayDays 实际是个死开关。activityOf 退到 updatedAt/createdAt,没命中过的条目也能衰减。
|
|
235
|
+
const last = activityOf(it)
|
|
236
|
+
if (last && last <= limit) {
|
|
196
237
|
it.archived = true
|
|
197
238
|
it.updatedAt = now
|
|
198
239
|
archived++
|
|
@@ -201,7 +242,11 @@ export function applyDecay(items, cfg, nowIso) {
|
|
|
201
242
|
return archived
|
|
202
243
|
}
|
|
203
244
|
|
|
204
|
-
/**
|
|
245
|
+
/**
|
|
246
|
+
* 超限按"最不活跃"物理删除,直到回落到上限。已有 archived 条目优先被删。
|
|
247
|
+
* 注意:没有归档可删时,本轮归档的条目会在同一次调用的后续循环里被删掉——
|
|
248
|
+
* `archived` 计数表示"先归档、随即被删",不是"留下了软删副本"。
|
|
249
|
+
*/
|
|
205
250
|
export function pruneItems(items, max, nowIso) {
|
|
206
251
|
const now = nowIso || new Date().toISOString()
|
|
207
252
|
let removed = 0
|
|
@@ -280,7 +325,11 @@ export class GlobalStore {
|
|
|
280
325
|
}
|
|
281
326
|
|
|
282
327
|
load() {
|
|
283
|
-
if (!this.doc)
|
|
328
|
+
if (!this.doc) {
|
|
329
|
+
this.doc = normalizeDoc(readJson(this.file, null))
|
|
330
|
+
// 与 project 级一致:global 条目也要补派生 trigger(提示通道用它提升召回;确定性、幂等)。
|
|
331
|
+
if (backfillDerivedTriggers(this.doc)) this.markDirty()
|
|
332
|
+
}
|
|
284
333
|
return this
|
|
285
334
|
}
|
|
286
335
|
|
|
@@ -455,16 +504,14 @@ export function promoteAllTasksToProject(store, globalStore, cfg, now) {
|
|
|
455
504
|
promoted++
|
|
456
505
|
continue
|
|
457
506
|
}
|
|
458
|
-
const moved =
|
|
459
|
-
|
|
460
|
-
|
|
461
|
-
|
|
462
|
-
|
|
463
|
-
|
|
464
|
-
|
|
465
|
-
|
|
466
|
-
movedFrom: { scope: 'task', id: ins.id, at: now },
|
|
467
|
-
}
|
|
507
|
+
const moved = carryUsageFields(
|
|
508
|
+
normalizeInsight(ins, { scope: 'project', source: ins.source, nowIso: now }),
|
|
509
|
+
ins,
|
|
510
|
+
)
|
|
511
|
+
moved.id = ins.id
|
|
512
|
+
moved.draft = false
|
|
513
|
+
moved.confidence = num(ins.confidence, cfg.promoteConfidence)
|
|
514
|
+
moved.movedFrom = { scope: 'task', id: ins.id, at: now }
|
|
468
515
|
pjItems.push(moved)
|
|
469
516
|
store.replaceInsightItems(pjItems)
|
|
470
517
|
removeFromTask(store, ins.id)
|
|
@@ -494,16 +541,14 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
|
|
|
494
541
|
if (glHit && glHit.score >= cfg.dedupOverlap) {
|
|
495
542
|
mergeInto(glHit.item, { sourceTaskIds: ins.sourceTaskIds, files: ins.files, symbols: ins.symbols, tags: ins.tags, confidence: ins.confidence }, cfg, now)
|
|
496
543
|
} else {
|
|
497
|
-
const movedIns =
|
|
498
|
-
|
|
499
|
-
|
|
500
|
-
|
|
501
|
-
|
|
502
|
-
|
|
503
|
-
|
|
504
|
-
|
|
505
|
-
movedFrom: { scope: 'project', id: ins.id, at: now },
|
|
506
|
-
}
|
|
544
|
+
const movedIns = carryUsageFields(
|
|
545
|
+
normalizeInsight(ins, { scope: 'global', source: ins.source, nowIso: now }),
|
|
546
|
+
ins,
|
|
547
|
+
)
|
|
548
|
+
movedIns.id = ins.id
|
|
549
|
+
movedIns.draft = false
|
|
550
|
+
movedIns.confidence = num(ins.confidence, cfg.promoteConfidence)
|
|
551
|
+
movedIns.movedFrom = { scope: 'project', id: ins.id, at: now }
|
|
507
552
|
glItems.push(movedIns)
|
|
508
553
|
}
|
|
509
554
|
items.splice(items.indexOf(ins), 1)
|
|
@@ -511,6 +556,8 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
|
|
|
511
556
|
}
|
|
512
557
|
if (moved) {
|
|
513
558
|
store.replaceInsightItems(items)
|
|
559
|
+
// 提升只增不减 global:不在这里收口,maxGlobalProcedures 就形同虚设。
|
|
560
|
+
pruneItems(globalStore.items(), cfg.maxGlobalProcedures, now)
|
|
514
561
|
globalStore.markDirty()
|
|
515
562
|
}
|
|
516
563
|
return moved
|
|
@@ -518,17 +565,19 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
|
|
|
518
565
|
|
|
519
566
|
// ---- 降级(反向,PR3 UI 使用;现在提供最小实现) ----
|
|
520
567
|
|
|
521
|
-
export function demoteToProject(store, globalStore, id, nowIso) {
|
|
568
|
+
export function demoteToProject(store, globalStore, id, nowIso, cfg = cfgInsight({})) {
|
|
522
569
|
if (!globalStore) return { ok: false, error: 'global 存储未初始化' }
|
|
523
570
|
const now = nowIso || new Date().toISOString()
|
|
524
571
|
const idx = globalStore.items().findIndex((i) => i.id === id)
|
|
525
572
|
if (idx === -1) return { ok: false, error: `global 无此条目: ${id}` }
|
|
526
573
|
const ins = globalStore.items()[idx]
|
|
527
574
|
const items = store.insightItems()
|
|
528
|
-
const copy = normalizeInsight(ins, { scope: 'project', nowIso: now })
|
|
575
|
+
const copy = carryUsageFields(normalizeInsight(ins, { scope: 'project', nowIso: now }), ins)
|
|
529
576
|
copy.id = `${ins.id}_d${Date.now().toString(36)}`
|
|
530
577
|
copy.movedFrom = { scope: 'global', id: ins.id, at: now }
|
|
531
578
|
items.push(copy)
|
|
579
|
+
// 降级只增不减 project:同样要收口 maxProject。
|
|
580
|
+
pruneItems(items, cfg.maxProject, now)
|
|
532
581
|
store.replaceInsightItems(items)
|
|
533
582
|
globalStore.items().splice(idx, 1)
|
|
534
583
|
globalStore.markDirty()
|
|
@@ -544,7 +593,7 @@ export function demoteToTask(store, taskId, id, nowIso) {
|
|
|
544
593
|
if (idx === -1) return { ok: false, error: `project 无此条目: ${id}` }
|
|
545
594
|
const ins = items[idx]
|
|
546
595
|
if (!Array.isArray(task.insights)) task.insights = []
|
|
547
|
-
const copy = normalizeInsight(ins, { scope: 'task', nowIso: now })
|
|
596
|
+
const copy = carryUsageFields(normalizeInsight(ins, { scope: 'task', nowIso: now }), ins)
|
|
548
597
|
copy.id = `${ins.id}_d${Date.now().toString(36)}`
|
|
549
598
|
copy.movedFrom = { scope: 'project', id: ins.id, at: now }
|
|
550
599
|
task.insights.push(copy)
|
package/src/lazy.js
CHANGED
|
@@ -1,11 +1,8 @@
|
|
|
1
1
|
import path from 'node:path'
|
|
2
2
|
import { tmpdir } from 'node:os'
|
|
3
3
|
import { existsSync, readdirSync, statSync } from 'node:fs'
|
|
4
|
-
import {
|
|
5
|
-
import {
|
|
6
|
-
import { docEntriesNeedBackfill } from './doc-index.js'
|
|
7
|
-
import { scanSymbols } from './symbols.js'
|
|
8
|
-
import { linkEntries } from './link.js'
|
|
4
|
+
import { memoryRootFor, relativePath, storeKey } from './util/fs.js'
|
|
5
|
+
import { DROP, OVERSIZE, UNCHANGED, commitFileUpdates, fileKind, planFileIndex, toFileUpdate } from './index-pipeline.js'
|
|
9
6
|
import { ProjectMemoryStore } from './store.js'
|
|
10
7
|
import { onFileObserved } from './enhancer.js'
|
|
11
8
|
|
|
@@ -78,70 +75,42 @@ export function findProjectRoot(filePath, ceiling = path.resolve(tmpdir())) {
|
|
|
78
75
|
}
|
|
79
76
|
|
|
80
77
|
export async function indexFile(ctx, config, filePath, watchManager = null) {
|
|
81
|
-
const
|
|
82
|
-
if (!
|
|
78
|
+
const kind = fileKind(path.extname(filePath).toLowerCase())
|
|
79
|
+
if (!kind) return false
|
|
83
80
|
const root = findProjectRoot(filePath)
|
|
84
81
|
if (!root) return false
|
|
85
82
|
|
|
86
83
|
const memoryDir = memoryRootFor(root, config.memoryDir)
|
|
87
84
|
const store = new ProjectMemoryStore(memoryDir).load()
|
|
88
85
|
const rel = storeKey(relativePath(root, filePath))
|
|
89
|
-
const
|
|
90
|
-
|
|
86
|
+
const record = store.fileRecord(rel)
|
|
87
|
+
|
|
91
88
|
let size
|
|
92
|
-
let
|
|
93
|
-
let entries
|
|
89
|
+
let plan
|
|
94
90
|
try {
|
|
95
|
-
|
|
96
|
-
|
|
97
|
-
}
|
|
98
|
-
;({ hash, size, buffer } = readFileForIndex(filePath))
|
|
91
|
+
size = statSync(filePath).size
|
|
92
|
+
plan = await planFileIndex({ rel, filePath, kind, config, record, existingEntries: store.entries[rel], size })
|
|
99
93
|
} catch {
|
|
100
94
|
return false
|
|
101
95
|
}
|
|
102
|
-
|
|
103
|
-
if (existing && existing.sha256 === hash && !(isSupportedDoc(ext) && docEntriesNeedBackfill(store.entries[rel]))) return false
|
|
96
|
+
if (plan.key === UNCHANGED || plan.key === OVERSIZE) return false
|
|
104
97
|
|
|
105
98
|
if (watchManager) {
|
|
106
99
|
watchManager.addRoot(root)
|
|
107
100
|
store.addWatch(root)
|
|
108
101
|
}
|
|
109
102
|
|
|
110
|
-
if (
|
|
111
|
-
entries = scanSymbols(rel, filePath, buffer.toString('utf8'))
|
|
103
|
+
if (plan.key !== DROP && plan.type === 'code') {
|
|
112
104
|
onFileObserved(store, rel, filePath, config, root)
|
|
113
|
-
return store.commit((s) => {
|
|
114
|
-
s.markFile(rel, { sha256: hash, size, type: 'code', indexedAt: new Date().toISOString() })
|
|
115
|
-
s.setEntries(rel, entries)
|
|
116
|
-
linkEntries(s)
|
|
117
|
-
return true
|
|
118
|
-
})
|
|
119
|
-
} else {
|
|
120
|
-
entries = await buildDocEntries(rel, filePath, {
|
|
121
|
-
chunkChars: config.chunkChars,
|
|
122
|
-
maxChunks: config.maxChunksPerFile,
|
|
123
|
-
maxFileSizeMb: config.maxFileSizeMb,
|
|
124
|
-
maxPdfPages: config.maxPdfPages,
|
|
125
|
-
})
|
|
126
|
-
if (entries === null) {
|
|
127
|
-
return store.commit((s) => {
|
|
128
|
-
s.removeFile(rel)
|
|
129
|
-
return false
|
|
130
|
-
})
|
|
131
|
-
}
|
|
132
|
-
return store.commit((s) => {
|
|
133
|
-
s.markFile(rel, { sha256: hash, size, type: 'doc', indexedAt: new Date().toISOString() })
|
|
134
|
-
s.setEntries(rel, entries)
|
|
135
|
-
linkEntries(s)
|
|
136
|
-
return true
|
|
137
|
-
})
|
|
138
105
|
}
|
|
106
|
+
commitFileUpdates(store, { updates: [toFileUpdate(rel, plan, record)] })
|
|
107
|
+
return plan.key !== DROP
|
|
139
108
|
}
|
|
140
109
|
|
|
141
110
|
export function codeFirst(paths) {
|
|
142
111
|
return [...paths].sort((a, b) => {
|
|
143
|
-
const aCode =
|
|
144
|
-
const bCode =
|
|
112
|
+
const aCode = fileKind(path.extname(a).toLowerCase()) === 'code' ? 0 : 1
|
|
113
|
+
const bCode = fileKind(path.extname(b).toLowerCase()) === 'code' ? 0 : 1
|
|
145
114
|
return aCode - bCode
|
|
146
115
|
})
|
|
147
116
|
}
|
package/src/link.js
CHANGED
|
@@ -39,7 +39,16 @@ export function linkEntries(store) {
|
|
|
39
39
|
if (linked.size > before) links++
|
|
40
40
|
}
|
|
41
41
|
}
|
|
42
|
-
|
|
42
|
+
const before = Array.isArray(doc.linkedSymbols) ? doc.linkedSymbols.join('\u0000') : ''
|
|
43
|
+
const next = [...linked]
|
|
44
|
+
if (before === next.join('\u0000')) continue
|
|
45
|
+
doc.linkedSymbols = next.length ? next : undefined
|
|
46
|
+
// 链接是在**符号**落盘那一刻算出来的,此时 doc 的 shard 往往不是脏的;不标脏就只存在于内存,
|
|
47
|
+
// 下次进程启动重新加载后链接全部丢失("文档先索引、符号后到"的正常顺序)。
|
|
48
|
+
if (typeof store.markFile === 'function' && typeof store.fileRecord === 'function' && doc.sourcePath) {
|
|
49
|
+
const record = store.fileRecord(doc.sourcePath)
|
|
50
|
+
if (record) store.markFile(doc.sourcePath, record)
|
|
51
|
+
}
|
|
43
52
|
}
|
|
44
53
|
return links
|
|
45
54
|
}
|
package/src/ops.js
ADDED
|
@@ -0,0 +1,210 @@
|
|
|
1
|
+
// 动作平面(PLAN S1):把"这一步在做什么"从工具调用解析成有限的 op + 写目标 + 主机环境。
|
|
2
|
+
//
|
|
3
|
+
// 为什么不能继续用文本匹配:语料(路径、命令、被读文件正文)里的字符串可以冒充动作。
|
|
4
|
+
// 实测病征是 `keyword:rm` 命中 `dcterms`、`keyword:pptx` 命中一个文件扩展名。
|
|
5
|
+
// op 与 target 是低维、可枚举、可审计的:不随文本长度变化,也不会被正文污染。
|
|
6
|
+
//
|
|
7
|
+
// 分工:ops.js 只负责"把调用读成什么动作",不负责决定注入什么(那是 readiness/auto-inject)。
|
|
8
|
+
// 刻意不 import readiness:readiness 要 import 本模块的 op 闭集,反向依赖会造成循环。
|
|
9
|
+
|
|
10
|
+
/** 带扩展名的路径 token(与 readiness.extractPaths 同规则;这里就地实现以免循环依赖)。 */
|
|
11
|
+
const PATH_TOKEN = /(?:[A-Za-z0-9_.@-]+\/)*[A-Za-z0-9_.@-]+\.[A-Za-z][A-Za-z0-9]{0,7}\b/g
|
|
12
|
+
|
|
13
|
+
function extractPaths(text) {
|
|
14
|
+
const s = String(text || '')
|
|
15
|
+
if (!s) return []
|
|
16
|
+
return [...new Set([...s.matchAll(PATH_TOKEN)].map((m) => m[0]))]
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
/** op 闭集。新增一个 op 必须同时想清楚:它是动作,不是话题。 */
|
|
20
|
+
export const OP_IDS = [
|
|
21
|
+
'file-write',
|
|
22
|
+
'file-delete',
|
|
23
|
+
'shell-run',
|
|
24
|
+
'git-commit',
|
|
25
|
+
'git-push',
|
|
26
|
+
'release',
|
|
27
|
+
'npm-publish',
|
|
28
|
+
'deploy',
|
|
29
|
+
'migrate',
|
|
30
|
+
'go-public',
|
|
31
|
+
'render-doc',
|
|
32
|
+
'index-doc',
|
|
33
|
+
'read-image',
|
|
34
|
+
'fetch-web',
|
|
35
|
+
'run-bench',
|
|
36
|
+
'query-memory',
|
|
37
|
+
]
|
|
38
|
+
|
|
39
|
+
/**
|
|
40
|
+
* 兜底 op:它们描述的是"发生了某类操作",信息量低、假阳性高。
|
|
41
|
+
* 允许写进 `when.ops`,但自检会 warn——兜底 op 只该命中兜底级记忆。
|
|
42
|
+
*/
|
|
43
|
+
export const FALLBACK_OPS = new Set(['shell-run', 'file-write', 'query-memory'])
|
|
44
|
+
|
|
45
|
+
/**
|
|
46
|
+
* 弱 op:日常动作。它本身不构成边界(每天都在提交),所以当条目**已经有更精确的 writes**
|
|
47
|
+
* 时,迁移会把弱 op 丢掉——留着它只会把"改这个文件时"扩大成"任何一次提交时"。
|
|
48
|
+
* 实测:`git-commit` 让一条"别删内部文档"的教训在发版场景里被注入。
|
|
49
|
+
*/
|
|
50
|
+
export const WEAK_OPS = new Set([...FALLBACK_OPS, 'git-commit', 'git-push'])
|
|
51
|
+
|
|
52
|
+
/** 旧 action id → op。旧值里有一批从来没进过 ACTION_LEXICON 的死值,这里给它们一个归宿。 */
|
|
53
|
+
export const LEGACY_ACTION_ALIAS = {
|
|
54
|
+
'git-add': 'git-commit',
|
|
55
|
+
'npm-pack': 'npm-publish',
|
|
56
|
+
'release': 'release',
|
|
57
|
+
'generate-pptx': 'render-doc',
|
|
58
|
+
'render': 'render-doc',
|
|
59
|
+
'read-image': 'read-image',
|
|
60
|
+
'web-fetch': 'fetch-web',
|
|
61
|
+
'benchmark': 'run-bench',
|
|
62
|
+
'cleanup': 'file-delete',
|
|
63
|
+
'delete': 'file-delete',
|
|
64
|
+
'recover': null, // 没有对应动作:恢复是意图,不是工具动作
|
|
65
|
+
'research': null,
|
|
66
|
+
'survey': null,
|
|
67
|
+
'interview-prep': null,
|
|
68
|
+
'write-resume': null,
|
|
69
|
+
'report-metrics': null,
|
|
70
|
+
'write-docs': null,
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const WRITE_TOOLS = new Set(['write', 'edit', 'multi_edit', 'apply_patch', 'str_replace', 'create_file', 'notebook_edit'])
|
|
74
|
+
const READ_IMAGE_TOOLS = new Set(['read_image'])
|
|
75
|
+
const INDEX_TOOLS = new Set(['index_doc', 'index_repo', 'watch_repo'])
|
|
76
|
+
const FETCH_TOOLS = new Set(['web_fetch', 'web_search'])
|
|
77
|
+
const SHELL_TOOLS = new Set(['bash', 'shell', 'pwsh', 'powershell', 'run', 'exec', 'job_run'])
|
|
78
|
+
const QUERY_TOOLS = new Set(['query_memory', 'recall'])
|
|
79
|
+
|
|
80
|
+
/** bash 命令 → op。顺序有意义:先具体后兜底。 */
|
|
81
|
+
const SHELL_OP_RULES = [
|
|
82
|
+
[/\bnpm\s+(publish|pack)\b/, 'npm-publish'],
|
|
83
|
+
[/\bnpm\s+version\b/, 'release'],
|
|
84
|
+
[/\bgit\s+tag\b/, 'release'],
|
|
85
|
+
[/\bgit\s+push\b/, 'git-push'],
|
|
86
|
+
[/\bgit\s+(commit|add)\b/, 'git-commit'],
|
|
87
|
+
[/\bgit\s+(checkout|restore|reset)\b/, 'file-write'],
|
|
88
|
+
// `\brm\b` 而不是 `rm\s+-[rf]`:后者被 `\b` 卡在 'rm -rf' 的 r 后面(f 还是词字符),永远匹配不上。
|
|
89
|
+
[/\b(rm|git\s+rm|unlink|rmdir|shred)\b/, 'file-delete'],
|
|
90
|
+
[/\b(kubectl|docker\s+push|systemctl\s+restart|pm2\s+(deploy|restart)|vercel|netlify\s+deploy)\b/, 'deploy'],
|
|
91
|
+
[/\b(migrate|alembic|prisma\s+migrate|db:migrate)\b/, 'migrate'],
|
|
92
|
+
// 注意:只认"渲染器",不认 .pptx 扩展名——否则给 pptx 改个时间戳也会命中"做 PPT"
|
|
93
|
+
[/(pptxgenjs|soffice|libreoffice|render\.sh|marp|pandoc|slidev)/, 'render-doc'],
|
|
94
|
+
[/(bench\.mjs|\bbench(mark)?\b)/, 'run-bench'],
|
|
95
|
+
[/\b(curl|wget|web_fetch)\b/, 'fetch-web'],
|
|
96
|
+
]
|
|
97
|
+
|
|
98
|
+
/** 会改动文件系统的 shell 动作前缀:只有这些命令的路径才算"写目标"。 */
|
|
99
|
+
const SHELL_WRITE_RE = /\b(rm|rmdir|unlink|shred|mv|cp|touch|tee|truncate|mkdir)\b|>>?\s*\S|\bsed\s+-i\b/
|
|
100
|
+
|
|
101
|
+
const WSL_HINT_RE = /\/mnt\/[a-z]\//i
|
|
102
|
+
|
|
103
|
+
function safeJson(text) {
|
|
104
|
+
if (typeof text !== 'string' || !text.trim()) return null
|
|
105
|
+
try {
|
|
106
|
+
const v = JSON.parse(text)
|
|
107
|
+
return v && typeof v === 'object' ? v : null
|
|
108
|
+
} catch {
|
|
109
|
+
return null
|
|
110
|
+
}
|
|
111
|
+
}
|
|
112
|
+
|
|
113
|
+
function argsFields(argsText) {
|
|
114
|
+
const obj = safeJson(argsText)
|
|
115
|
+
if (!obj) return {}
|
|
116
|
+
const pick = (...keys) => keys
|
|
117
|
+
.map((k) => obj[k])
|
|
118
|
+
.filter((v) => typeof v === 'string' && v)
|
|
119
|
+
.join(' ')
|
|
120
|
+
return {
|
|
121
|
+
command: pick('command', 'cmd', 'script', 'code'),
|
|
122
|
+
path: pick('file_path', 'path', 'file', 'target', 'notebook_path'),
|
|
123
|
+
}
|
|
124
|
+
}
|
|
125
|
+
|
|
126
|
+
/** bash 命令 → { ops, targets, hosts }。 */
|
|
127
|
+
export function classifyShellCommand(command) {
|
|
128
|
+
const cmd = String(command || '')
|
|
129
|
+
const ops = []
|
|
130
|
+
if (cmd) {
|
|
131
|
+
for (const [re, op] of SHELL_OP_RULES) {
|
|
132
|
+
if (re.test(cmd) && !ops.includes(op)) ops.push(op)
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
if (!ops.length) ops.push('shell-run')
|
|
136
|
+
const targets = SHELL_WRITE_RE.test(cmd) ? extractPaths(cmd) : []
|
|
137
|
+
const hosts = WSL_HINT_RE.test(cmd) || /\b(powershell|pwsh|cmd)\.exe\b/i.test(cmd) ? ['wsl'] : []
|
|
138
|
+
return { ops, targets, hosts }
|
|
139
|
+
}
|
|
140
|
+
|
|
141
|
+
/**
|
|
142
|
+
* 一次工具调用 → { ops, targets, hosts }。
|
|
143
|
+
* @param {string} name 工具名
|
|
144
|
+
* @param {string} argsText 工具参数(JSON 字符串或裸文本)
|
|
145
|
+
*/
|
|
146
|
+
export function classifyToolCall(name, argsText) {
|
|
147
|
+
const tool = String(name || '').toLowerCase()
|
|
148
|
+
const { command, path } = argsFields(argsText)
|
|
149
|
+
if (SHELL_TOOLS.has(tool)) return classifyShellCommand(command || String(argsText || ''))
|
|
150
|
+
if (READ_IMAGE_TOOLS.has(tool)) return { ops: ['read-image'], targets: [], hosts: [] }
|
|
151
|
+
if (INDEX_TOOLS.has(tool)) return { ops: ['index-doc'], targets: [], hosts: [] }
|
|
152
|
+
if (FETCH_TOOLS.has(tool)) return { ops: ['fetch-web'], targets: [], hosts: [] }
|
|
153
|
+
if (QUERY_TOOLS.has(tool)) return { ops: ['query-memory'], targets: [], hosts: [] }
|
|
154
|
+
if (WRITE_TOOLS.has(tool)) {
|
|
155
|
+
const targets = path ? [path, ...extractPaths(path)] : []
|
|
156
|
+
const hosts = targets.some((t) => WSL_HINT_RE.test(t)) ? ['wsl'] : []
|
|
157
|
+
return { ops: ['file-write'], targets: [...new Set(targets)], hosts }
|
|
158
|
+
}
|
|
159
|
+
return { ops: [], targets: [], hosts: [] }
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
/**
|
|
163
|
+
* 把一组已观察到的工具调用合成这一个线程的动态面。
|
|
164
|
+
* @param {{name?: string, arguments?: string}[]} calls
|
|
165
|
+
*/
|
|
166
|
+
export function activityFromCalls(calls) {
|
|
167
|
+
const ops = new Set()
|
|
168
|
+
const targets = new Set()
|
|
169
|
+
const hosts = new Set()
|
|
170
|
+
for (const call of calls || []) {
|
|
171
|
+
const r = classifyToolCall(call?.name, call?.arguments)
|
|
172
|
+
for (const op of r.ops) ops.add(op)
|
|
173
|
+
for (const t of r.targets) targets.add(t)
|
|
174
|
+
for (const h of r.hosts) hosts.add(h)
|
|
175
|
+
}
|
|
176
|
+
return { ops: [...ops], targets: [...targets], hosts: [...hosts] }
|
|
177
|
+
}
|
|
178
|
+
|
|
179
|
+
/**
|
|
180
|
+
* 没拿到结构化调用时的兜底:直接对一段文本跑 shell 规则。
|
|
181
|
+
* 用于 `actionText` 形式的就绪上下文(测试与老宿主),以及人类消息里的动作词。
|
|
182
|
+
*/
|
|
183
|
+
export function activityFromText(text) {
|
|
184
|
+
const s = String(text || '')
|
|
185
|
+
if (!s) return { ops: [], targets: [], hosts: [] }
|
|
186
|
+
const lines = s.split('\n').filter(Boolean)
|
|
187
|
+
const ops = new Set()
|
|
188
|
+
const targets = new Set()
|
|
189
|
+
const hosts = new Set()
|
|
190
|
+
for (const line of lines) {
|
|
191
|
+
// 先按"工具调用"解析(`edit {"file_path":"..."}`):结构化参数比正则猜命令可靠得多。
|
|
192
|
+
const call = /^\s*([A-Za-z_][\w-]*)\s+(\{[\s\S]*\})\s*$/.exec(line)
|
|
193
|
+
const r = call ? classifyToolCall(call[1], call[2]) : classifyShellCommand(line)
|
|
194
|
+
// 纯文本行里没有 shell 动作时不要退化成 shell-run(那是噪音源)
|
|
195
|
+
if (r.ops.length === 1 && r.ops[0] === 'shell-run' && !/\b\w+\s+-/.test(line)) continue
|
|
196
|
+
for (const op of r.ops) ops.add(op)
|
|
197
|
+
for (const t of r.targets) targets.add(t)
|
|
198
|
+
for (const h of r.hosts) hosts.add(h)
|
|
199
|
+
}
|
|
200
|
+
return { ops: [...ops], targets: [...targets], hosts: [...hosts] }
|
|
201
|
+
}
|
|
202
|
+
|
|
203
|
+
/** 旧 action id(含死值)→ op;无法映射返回 null。 */
|
|
204
|
+
export function opForLegacyAction(action) {
|
|
205
|
+
const a = String(action || '')
|
|
206
|
+
if (!a) return null
|
|
207
|
+
if (OP_IDS.includes(a)) return a
|
|
208
|
+
if (a in LEGACY_ACTION_ALIAS) return LEGACY_ACTION_ALIAS[a]
|
|
209
|
+
return null
|
|
210
|
+
}
|
|
@@ -60,9 +60,8 @@ export async function parsePdf(filePath, { pages = null, maxPages = 1000, backen
|
|
|
60
60
|
const data = new Uint8Array(await readFile(filePath))
|
|
61
61
|
const { getDocument } = await loadPdfjs()
|
|
62
62
|
const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
|
|
63
|
-
const doc = await loadingTask.promise
|
|
64
|
-
|
|
65
63
|
try {
|
|
64
|
+
const doc = await loadingTask.promise
|
|
66
65
|
const total = doc.numPages
|
|
67
66
|
if (total > maxPages) {
|
|
68
67
|
throw new Error(`PDF has ${total} pages, over the maxPages limit of ${maxPages}`)
|
|
@@ -112,9 +111,8 @@ export async function parsePdfInfo(filePath, maxPages = 1000) {
|
|
|
112
111
|
const data = new Uint8Array(await readFile(filePath))
|
|
113
112
|
const { getDocument } = await loadPdfjs()
|
|
114
113
|
const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
|
|
115
|
-
const doc = await loadingTask.promise
|
|
116
|
-
|
|
117
114
|
try {
|
|
115
|
+
const doc = await loadingTask.promise
|
|
118
116
|
if (doc.numPages > maxPages) {
|
|
119
117
|
throw new Error(`PDF has ${doc.numPages} pages, over the maxPages limit of ${maxPages}`)
|
|
120
118
|
}
|
package/src/project-profile.js
CHANGED
|
@@ -17,11 +17,18 @@ function depsOf(manifest, fields) {
|
|
|
17
17
|
const tags = new Set()
|
|
18
18
|
for (const f of fields) {
|
|
19
19
|
const deps = manifest && manifest[f]
|
|
20
|
-
if (deps
|
|
21
|
-
|
|
22
|
-
|
|
23
|
-
|
|
24
|
-
|
|
20
|
+
if (!deps || typeof deps !== 'object') continue
|
|
21
|
+
for (const name of Object.keys(deps)) {
|
|
22
|
+
if (typeof name !== 'string' || !name) continue
|
|
23
|
+
tags.add(name.toLowerCase())
|
|
24
|
+
if (!name.startsWith('@')) continue
|
|
25
|
+
// scoped 包名 `@scope/pkg`:要加的是 **scope**(`vue`),不是包名段(`compiler-sfc`)。
|
|
26
|
+
// 旧写法 `name.split('/')[1]` 让 `@vue/*` 项目永远拿不到 `vue` tag,
|
|
27
|
+
// trigger.scope:['vue'] / legacy scope 过滤会把最相关的那条 procedure 静默判死;
|
|
28
|
+
// 而 `@malformed`(没有 `/`)会在这里抛错,异常被 collectTags 整个吞掉 → 全项目 tags 清零。
|
|
29
|
+
const parts = name.slice(1).split('/')
|
|
30
|
+
if (parts[0]) tags.add(parts[0].toLowerCase())
|
|
31
|
+
if (parts[1]) tags.add(parts[1].toLowerCase())
|
|
25
32
|
}
|
|
26
33
|
}
|
|
27
34
|
return tags
|