@yolk_vat-y/dsh-project-memory 0.5.5 → 0.5.7

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -10,6 +10,7 @@ import { mkdirSync, readFileSync, renameSync, writeFileSync } from 'node:fs'
10
10
  import path from 'node:path'
11
11
  import os from 'node:os'
12
12
  import { normalizedTokenOverlap, findBestOverlapMatch, insightMatchText } from './similarity.js'
13
+ import { backfillDerivedTriggers } from './readiness.js'
13
14
 
14
15
  export const INSIGHT_KINDS = ['lesson', 'decision', 'procedure', 'experience']
15
16
  export const INSIGHT_SCOPES = ['task', 'project', 'global']
@@ -98,11 +99,31 @@ export function normalizeInsight(raw, extra = {}) {
98
99
  if (Array.isArray(raw[f]) && raw[f].length) ins[f] = [...new Set(raw[f].map((x) => String(x)))]
99
100
  }
100
101
  if (raw.trigger && typeof raw.trigger === 'object') {
101
- // 所有 kind 通用:keywords / symbols / actions / paths / scope(actions/paths 见 src/readiness.js)
102
+ // 所有 kind 通用。准入化 schema:when(唯一触发面)/ guard(只收窄)/ prevents(准入条件);
103
+ // 旧字段 keywords / symbols / actions / paths / scope 继续接受(readiness 会内存内迁移)。
102
104
  const tr = {}
103
105
  for (const f of ['keywords', 'symbols', 'actions', 'paths', 'scope']) {
104
106
  if (Array.isArray(raw.trigger[f]) && raw.trigger[f].length) tr[f] = raw.trigger[f].map(String)
105
107
  }
108
+ const when = raw.trigger.when
109
+ if (when && typeof when === 'object') {
110
+ const w = {}
111
+ for (const f of ['ops', 'writes', 'intents']) {
112
+ if (Array.isArray(when[f]) && when[f].length) w[f] = when[f].map(String)
113
+ }
114
+ if (Object.keys(w).length) tr.when = w
115
+ }
116
+ const guard = raw.trigger.guard
117
+ if (guard && typeof guard === 'object') {
118
+ const g = {}
119
+ for (const f of ['paths', 'not_paths', 'hosts', 'tags']) {
120
+ if (Array.isArray(guard[f]) && guard[f].length) g[f] = guard[f].map(String)
121
+ }
122
+ if (Object.keys(g).length) tr.guard = g
123
+ }
124
+ if (typeof raw.trigger.prevents === 'string' && raw.trigger.prevents.trim()) {
125
+ tr.prevents = raw.trigger.prevents.trim()
126
+ }
106
127
  if (Object.keys(tr).length) ins.trigger = tr
107
128
  }
108
129
  return ins
@@ -114,6 +135,23 @@ export function unionStrings(base = [], add = []) {
114
135
  return [...out]
115
136
  }
116
137
 
138
+ /**
139
+ * 跨 scope 移动(promote/demote)时保留"使用痕迹"。
140
+ *
141
+ * 提升/降级只改 scope,不该清零命中数、创建时间,也不该丢掉 `triggerDerived`
142
+ * (提示通道拿它当检索词)。这些字段不在 {@link normalizeInsight} 的白名单里,
143
+ * 必须显式带回——否则每次移动都会让条目"看起来从没被用过",global 层更是每次都丢派生词。
144
+ * @param {object} copy - 已归一化到新 scope 的副本
145
+ * @param {object} src - 原条目
146
+ */
147
+ export function carryUsageFields(copy, src) {
148
+ copy.hitCount = num(src.hitCount, 0)
149
+ if (src.lastHitAt) copy.lastHitAt = src.lastHitAt
150
+ if (src.createdAt) copy.createdAt = src.createdAt
151
+ if (src.triggerDerived) copy.triggerDerived = src.triggerDerived
152
+ return copy
153
+ }
154
+
117
155
  /** 把 base 合并进既有条目(content 加固、成员/文件/符号并集、置信取高、记命中)。 */
118
156
  export function mergeInto(existing, base, cfg, nowIso) {
119
157
  const now = nowIso || new Date().toISOString()
@@ -190,9 +228,12 @@ export function applyDecay(items, cfg, nowIso) {
190
228
  const limit = Date.parse(now) - cfg.decayDays * DAY_MS
191
229
  let archived = 0
192
230
  for (const it of items) {
193
- if (it.archived) continue
194
- if ((it.hitCount || 0) > 0) continue
195
- if (it.lastHitAt && Date.parse(it.lastHitAt) <= limit) {
231
+ if (!it || it.archived) continue
232
+ // 用"最近活动"而不是 lastHitAt:lastHitAt 只由 merge/reinforce 写,而那两处同时把
233
+ // hitCount 加一。旧实现先 `hitCount > 0 → continue`,于是 lastHitAt 分支永远不可达,
234
+ // decayDays 实际是个死开关。activityOf 退到 updatedAt/createdAt,没命中过的条目也能衰减。
235
+ const last = activityOf(it)
236
+ if (last && last <= limit) {
196
237
  it.archived = true
197
238
  it.updatedAt = now
198
239
  archived++
@@ -201,7 +242,11 @@ export function applyDecay(items, cfg, nowIso) {
201
242
  return archived
202
243
  }
203
244
 
204
- /** 超限先物理删归档里最不活跃的,再归档最不活跃的(下一轮被删),直到回落到上限。 */
245
+ /**
246
+ * 超限按"最不活跃"物理删除,直到回落到上限。已有 archived 条目优先被删。
247
+ * 注意:没有归档可删时,本轮归档的条目会在同一次调用的后续循环里被删掉——
248
+ * `archived` 计数表示"先归档、随即被删",不是"留下了软删副本"。
249
+ */
205
250
  export function pruneItems(items, max, nowIso) {
206
251
  const now = nowIso || new Date().toISOString()
207
252
  let removed = 0
@@ -280,7 +325,11 @@ export class GlobalStore {
280
325
  }
281
326
 
282
327
  load() {
283
- if (!this.doc) this.doc = normalizeDoc(readJson(this.file, null))
328
+ if (!this.doc) {
329
+ this.doc = normalizeDoc(readJson(this.file, null))
330
+ // 与 project 级一致:global 条目也要补派生 trigger(提示通道用它提升召回;确定性、幂等)。
331
+ if (backfillDerivedTriggers(this.doc)) this.markDirty()
332
+ }
284
333
  return this
285
334
  }
286
335
 
@@ -455,16 +504,14 @@ export function promoteAllTasksToProject(store, globalStore, cfg, now) {
455
504
  promoted++
456
505
  continue
457
506
  }
458
- const moved = {
459
- ...normalizeInsight(ins, { scope: 'project', source: ins.source, nowIso: now }),
460
- id: ins.id,
461
- draft: false,
462
- confidence: num(ins.confidence, cfg.promoteConfidence),
463
- hitCount: ins.hitCount || 0,
464
- lastHitAt: ins.lastHitAt,
465
- createdAt: ins.createdAt || now,
466
- movedFrom: { scope: 'task', id: ins.id, at: now },
467
- }
507
+ const moved = carryUsageFields(
508
+ normalizeInsight(ins, { scope: 'project', source: ins.source, nowIso: now }),
509
+ ins,
510
+ )
511
+ moved.id = ins.id
512
+ moved.draft = false
513
+ moved.confidence = num(ins.confidence, cfg.promoteConfidence)
514
+ moved.movedFrom = { scope: 'task', id: ins.id, at: now }
468
515
  pjItems.push(moved)
469
516
  store.replaceInsightItems(pjItems)
470
517
  removeFromTask(store, ins.id)
@@ -494,16 +541,14 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
494
541
  if (glHit && glHit.score >= cfg.dedupOverlap) {
495
542
  mergeInto(glHit.item, { sourceTaskIds: ins.sourceTaskIds, files: ins.files, symbols: ins.symbols, tags: ins.tags, confidence: ins.confidence }, cfg, now)
496
543
  } else {
497
- const movedIns = {
498
- ...normalizeInsight(ins, { scope: 'global', source: ins.source, nowIso: now }),
499
- id: ins.id,
500
- draft: false,
501
- confidence: num(ins.confidence, cfg.promoteConfidence),
502
- hitCount: ins.hitCount || 0,
503
- lastHitAt: ins.lastHitAt,
504
- createdAt: ins.createdAt || now,
505
- movedFrom: { scope: 'project', id: ins.id, at: now },
506
- }
544
+ const movedIns = carryUsageFields(
545
+ normalizeInsight(ins, { scope: 'global', source: ins.source, nowIso: now }),
546
+ ins,
547
+ )
548
+ movedIns.id = ins.id
549
+ movedIns.draft = false
550
+ movedIns.confidence = num(ins.confidence, cfg.promoteConfidence)
551
+ movedIns.movedFrom = { scope: 'project', id: ins.id, at: now }
507
552
  glItems.push(movedIns)
508
553
  }
509
554
  items.splice(items.indexOf(ins), 1)
@@ -511,6 +556,8 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
511
556
  }
512
557
  if (moved) {
513
558
  store.replaceInsightItems(items)
559
+ // 提升只增不减 global:不在这里收口,maxGlobalProcedures 就形同虚设。
560
+ pruneItems(globalStore.items(), cfg.maxGlobalProcedures, now)
514
561
  globalStore.markDirty()
515
562
  }
516
563
  return moved
@@ -518,17 +565,19 @@ export function promoteProjectToGlobal(store, globalStore, cfg, now) {
518
565
 
519
566
  // ---- 降级(反向,PR3 UI 使用;现在提供最小实现) ----
520
567
 
521
- export function demoteToProject(store, globalStore, id, nowIso) {
568
+ export function demoteToProject(store, globalStore, id, nowIso, cfg = cfgInsight({})) {
522
569
  if (!globalStore) return { ok: false, error: 'global 存储未初始化' }
523
570
  const now = nowIso || new Date().toISOString()
524
571
  const idx = globalStore.items().findIndex((i) => i.id === id)
525
572
  if (idx === -1) return { ok: false, error: `global 无此条目: ${id}` }
526
573
  const ins = globalStore.items()[idx]
527
574
  const items = store.insightItems()
528
- const copy = normalizeInsight(ins, { scope: 'project', nowIso: now })
575
+ const copy = carryUsageFields(normalizeInsight(ins, { scope: 'project', nowIso: now }), ins)
529
576
  copy.id = `${ins.id}_d${Date.now().toString(36)}`
530
577
  copy.movedFrom = { scope: 'global', id: ins.id, at: now }
531
578
  items.push(copy)
579
+ // 降级只增不减 project:同样要收口 maxProject。
580
+ pruneItems(items, cfg.maxProject, now)
532
581
  store.replaceInsightItems(items)
533
582
  globalStore.items().splice(idx, 1)
534
583
  globalStore.markDirty()
@@ -544,7 +593,7 @@ export function demoteToTask(store, taskId, id, nowIso) {
544
593
  if (idx === -1) return { ok: false, error: `project 无此条目: ${id}` }
545
594
  const ins = items[idx]
546
595
  if (!Array.isArray(task.insights)) task.insights = []
547
- const copy = normalizeInsight(ins, { scope: 'task', nowIso: now })
596
+ const copy = carryUsageFields(normalizeInsight(ins, { scope: 'task', nowIso: now }), ins)
548
597
  copy.id = `${ins.id}_d${Date.now().toString(36)}`
549
598
  copy.movedFrom = { scope: 'project', id: ins.id, at: now }
550
599
  task.insights.push(copy)
package/src/lazy.js CHANGED
@@ -1,11 +1,8 @@
1
1
  import path from 'node:path'
2
2
  import { tmpdir } from 'node:os'
3
3
  import { existsSync, readdirSync, statSync } from 'node:fs'
4
- import { isSupportedCode, isSupportedDoc, memoryRootFor, readFileForIndex, relativePath, storeKey } from './util/fs.js'
5
- import { buildDocEntries } from './doc-pipeline.js'
6
- import { docEntriesNeedBackfill } from './doc-index.js'
7
- import { scanSymbols } from './symbols.js'
8
- import { linkEntries } from './link.js'
4
+ import { memoryRootFor, relativePath, storeKey } from './util/fs.js'
5
+ import { DROP, OVERSIZE, UNCHANGED, commitFileUpdates, fileKind, planFileIndex, toFileUpdate } from './index-pipeline.js'
9
6
  import { ProjectMemoryStore } from './store.js'
10
7
  import { onFileObserved } from './enhancer.js'
11
8
 
@@ -78,70 +75,42 @@ export function findProjectRoot(filePath, ceiling = path.resolve(tmpdir())) {
78
75
  }
79
76
 
80
77
  export async function indexFile(ctx, config, filePath, watchManager = null) {
81
- const ext = path.extname(filePath).toLowerCase()
82
- if (!isSupportedDoc(ext) && !isSupportedCode(ext)) return false
78
+ const kind = fileKind(path.extname(filePath).toLowerCase())
79
+ if (!kind) return false
83
80
  const root = findProjectRoot(filePath)
84
81
  if (!root) return false
85
82
 
86
83
  const memoryDir = memoryRootFor(root, config.memoryDir)
87
84
  const store = new ProjectMemoryStore(memoryDir).load()
88
85
  const rel = storeKey(relativePath(root, filePath))
89
- const existing = store.fileRecord(rel)
90
- let hash
86
+ const record = store.fileRecord(rel)
87
+
91
88
  let size
92
- let buffer
93
- let entries
89
+ let plan
94
90
  try {
95
- if (isSupportedCode(ext) && config.maxFileSizeMb && statSync(filePath).size > config.maxFileSizeMb * 1024 * 1024) {
96
- return false
97
- }
98
- ;({ hash, size, buffer } = readFileForIndex(filePath))
91
+ size = statSync(filePath).size
92
+ plan = await planFileIndex({ rel, filePath, kind, config, record, existingEntries: store.entries[rel], size })
99
93
  } catch {
100
94
  return false
101
95
  }
102
- // 旧 store 的 doc 条目缺 terms → 一次性回填(即使哈希未变)
103
- if (existing && existing.sha256 === hash && !(isSupportedDoc(ext) && docEntriesNeedBackfill(store.entries[rel]))) return false
96
+ if (plan.key === UNCHANGED || plan.key === OVERSIZE) return false
104
97
 
105
98
  if (watchManager) {
106
99
  watchManager.addRoot(root)
107
100
  store.addWatch(root)
108
101
  }
109
102
 
110
- if (isSupportedCode(ext)) {
111
- entries = scanSymbols(rel, filePath, buffer.toString('utf8'))
103
+ if (plan.key !== DROP && plan.type === 'code') {
112
104
  onFileObserved(store, rel, filePath, config, root)
113
- return store.commit((s) => {
114
- s.markFile(rel, { sha256: hash, size, type: 'code', indexedAt: new Date().toISOString() })
115
- s.setEntries(rel, entries)
116
- linkEntries(s)
117
- return true
118
- })
119
- } else {
120
- entries = await buildDocEntries(rel, filePath, {
121
- chunkChars: config.chunkChars,
122
- maxChunks: config.maxChunksPerFile,
123
- maxFileSizeMb: config.maxFileSizeMb,
124
- maxPdfPages: config.maxPdfPages,
125
- })
126
- if (entries === null) {
127
- return store.commit((s) => {
128
- s.removeFile(rel)
129
- return false
130
- })
131
- }
132
- return store.commit((s) => {
133
- s.markFile(rel, { sha256: hash, size, type: 'doc', indexedAt: new Date().toISOString() })
134
- s.setEntries(rel, entries)
135
- linkEntries(s)
136
- return true
137
- })
138
105
  }
106
+ commitFileUpdates(store, { updates: [toFileUpdate(rel, plan, record)] })
107
+ return plan.key !== DROP
139
108
  }
140
109
 
141
110
  export function codeFirst(paths) {
142
111
  return [...paths].sort((a, b) => {
143
- const aCode = isSupportedCode(path.extname(a).toLowerCase()) ? 0 : 1
144
- const bCode = isSupportedCode(path.extname(b).toLowerCase()) ? 0 : 1
112
+ const aCode = fileKind(path.extname(a).toLowerCase()) === 'code' ? 0 : 1
113
+ const bCode = fileKind(path.extname(b).toLowerCase()) === 'code' ? 0 : 1
145
114
  return aCode - bCode
146
115
  })
147
116
  }
package/src/link.js CHANGED
@@ -39,7 +39,16 @@ export function linkEntries(store) {
39
39
  if (linked.size > before) links++
40
40
  }
41
41
  }
42
- doc.linkedSymbols = linked.size ? [...linked] : undefined
42
+ const before = Array.isArray(doc.linkedSymbols) ? doc.linkedSymbols.join('\u0000') : ''
43
+ const next = [...linked]
44
+ if (before === next.join('\u0000')) continue
45
+ doc.linkedSymbols = next.length ? next : undefined
46
+ // 链接是在**符号**落盘那一刻算出来的,此时 doc 的 shard 往往不是脏的;不标脏就只存在于内存,
47
+ // 下次进程启动重新加载后链接全部丢失("文档先索引、符号后到"的正常顺序)。
48
+ if (typeof store.markFile === 'function' && typeof store.fileRecord === 'function' && doc.sourcePath) {
49
+ const record = store.fileRecord(doc.sourcePath)
50
+ if (record) store.markFile(doc.sourcePath, record)
51
+ }
43
52
  }
44
53
  return links
45
54
  }
package/src/ops.js ADDED
@@ -0,0 +1,210 @@
1
+ // 动作平面(PLAN S1):把"这一步在做什么"从工具调用解析成有限的 op + 写目标 + 主机环境。
2
+ //
3
+ // 为什么不能继续用文本匹配:语料(路径、命令、被读文件正文)里的字符串可以冒充动作。
4
+ // 实测病征是 `keyword:rm` 命中 `dcterms`、`keyword:pptx` 命中一个文件扩展名。
5
+ // op 与 target 是低维、可枚举、可审计的:不随文本长度变化,也不会被正文污染。
6
+ //
7
+ // 分工:ops.js 只负责"把调用读成什么动作",不负责决定注入什么(那是 readiness/auto-inject)。
8
+ // 刻意不 import readiness:readiness 要 import 本模块的 op 闭集,反向依赖会造成循环。
9
+
10
+ /** 带扩展名的路径 token(与 readiness.extractPaths 同规则;这里就地实现以免循环依赖)。 */
11
+ const PATH_TOKEN = /(?:[A-Za-z0-9_.@-]+\/)*[A-Za-z0-9_.@-]+\.[A-Za-z][A-Za-z0-9]{0,7}\b/g
12
+
13
+ function extractPaths(text) {
14
+ const s = String(text || '')
15
+ if (!s) return []
16
+ return [...new Set([...s.matchAll(PATH_TOKEN)].map((m) => m[0]))]
17
+ }
18
+
19
+ /** op 闭集。新增一个 op 必须同时想清楚:它是动作,不是话题。 */
20
+ export const OP_IDS = [
21
+ 'file-write',
22
+ 'file-delete',
23
+ 'shell-run',
24
+ 'git-commit',
25
+ 'git-push',
26
+ 'release',
27
+ 'npm-publish',
28
+ 'deploy',
29
+ 'migrate',
30
+ 'go-public',
31
+ 'render-doc',
32
+ 'index-doc',
33
+ 'read-image',
34
+ 'fetch-web',
35
+ 'run-bench',
36
+ 'query-memory',
37
+ ]
38
+
39
+ /**
40
+ * 兜底 op:它们描述的是"发生了某类操作",信息量低、假阳性高。
41
+ * 允许写进 `when.ops`,但自检会 warn——兜底 op 只该命中兜底级记忆。
42
+ */
43
+ export const FALLBACK_OPS = new Set(['shell-run', 'file-write', 'query-memory'])
44
+
45
+ /**
46
+ * 弱 op:日常动作。它本身不构成边界(每天都在提交),所以当条目**已经有更精确的 writes**
47
+ * 时,迁移会把弱 op 丢掉——留着它只会把"改这个文件时"扩大成"任何一次提交时"。
48
+ * 实测:`git-commit` 让一条"别删内部文档"的教训在发版场景里被注入。
49
+ */
50
+ export const WEAK_OPS = new Set([...FALLBACK_OPS, 'git-commit', 'git-push'])
51
+
52
+ /** 旧 action id → op。旧值里有一批从来没进过 ACTION_LEXICON 的死值,这里给它们一个归宿。 */
53
+ export const LEGACY_ACTION_ALIAS = {
54
+ 'git-add': 'git-commit',
55
+ 'npm-pack': 'npm-publish',
56
+ 'release': 'release',
57
+ 'generate-pptx': 'render-doc',
58
+ 'render': 'render-doc',
59
+ 'read-image': 'read-image',
60
+ 'web-fetch': 'fetch-web',
61
+ 'benchmark': 'run-bench',
62
+ 'cleanup': 'file-delete',
63
+ 'delete': 'file-delete',
64
+ 'recover': null, // 没有对应动作:恢复是意图,不是工具动作
65
+ 'research': null,
66
+ 'survey': null,
67
+ 'interview-prep': null,
68
+ 'write-resume': null,
69
+ 'report-metrics': null,
70
+ 'write-docs': null,
71
+ }
72
+
73
+ const WRITE_TOOLS = new Set(['write', 'edit', 'multi_edit', 'apply_patch', 'str_replace', 'create_file', 'notebook_edit'])
74
+ const READ_IMAGE_TOOLS = new Set(['read_image'])
75
+ const INDEX_TOOLS = new Set(['index_doc', 'index_repo', 'watch_repo'])
76
+ const FETCH_TOOLS = new Set(['web_fetch', 'web_search'])
77
+ const SHELL_TOOLS = new Set(['bash', 'shell', 'pwsh', 'powershell', 'run', 'exec', 'job_run'])
78
+ const QUERY_TOOLS = new Set(['query_memory', 'recall'])
79
+
80
+ /** bash 命令 → op。顺序有意义:先具体后兜底。 */
81
+ const SHELL_OP_RULES = [
82
+ [/\bnpm\s+(publish|pack)\b/, 'npm-publish'],
83
+ [/\bnpm\s+version\b/, 'release'],
84
+ [/\bgit\s+tag\b/, 'release'],
85
+ [/\bgit\s+push\b/, 'git-push'],
86
+ [/\bgit\s+(commit|add)\b/, 'git-commit'],
87
+ [/\bgit\s+(checkout|restore|reset)\b/, 'file-write'],
88
+ // `\brm\b` 而不是 `rm\s+-[rf]`:后者被 `\b` 卡在 'rm -rf' 的 r 后面(f 还是词字符),永远匹配不上。
89
+ [/\b(rm|git\s+rm|unlink|rmdir|shred)\b/, 'file-delete'],
90
+ [/\b(kubectl|docker\s+push|systemctl\s+restart|pm2\s+(deploy|restart)|vercel|netlify\s+deploy)\b/, 'deploy'],
91
+ [/\b(migrate|alembic|prisma\s+migrate|db:migrate)\b/, 'migrate'],
92
+ // 注意:只认"渲染器",不认 .pptx 扩展名——否则给 pptx 改个时间戳也会命中"做 PPT"
93
+ [/(pptxgenjs|soffice|libreoffice|render\.sh|marp|pandoc|slidev)/, 'render-doc'],
94
+ [/(bench\.mjs|\bbench(mark)?\b)/, 'run-bench'],
95
+ [/\b(curl|wget|web_fetch)\b/, 'fetch-web'],
96
+ ]
97
+
98
+ /** 会改动文件系统的 shell 动作前缀:只有这些命令的路径才算"写目标"。 */
99
+ const SHELL_WRITE_RE = /\b(rm|rmdir|unlink|shred|mv|cp|touch|tee|truncate|mkdir)\b|>>?\s*\S|\bsed\s+-i\b/
100
+
101
+ const WSL_HINT_RE = /\/mnt\/[a-z]\//i
102
+
103
+ function safeJson(text) {
104
+ if (typeof text !== 'string' || !text.trim()) return null
105
+ try {
106
+ const v = JSON.parse(text)
107
+ return v && typeof v === 'object' ? v : null
108
+ } catch {
109
+ return null
110
+ }
111
+ }
112
+
113
+ function argsFields(argsText) {
114
+ const obj = safeJson(argsText)
115
+ if (!obj) return {}
116
+ const pick = (...keys) => keys
117
+ .map((k) => obj[k])
118
+ .filter((v) => typeof v === 'string' && v)
119
+ .join(' ')
120
+ return {
121
+ command: pick('command', 'cmd', 'script', 'code'),
122
+ path: pick('file_path', 'path', 'file', 'target', 'notebook_path'),
123
+ }
124
+ }
125
+
126
+ /** bash 命令 → { ops, targets, hosts }。 */
127
+ export function classifyShellCommand(command) {
128
+ const cmd = String(command || '')
129
+ const ops = []
130
+ if (cmd) {
131
+ for (const [re, op] of SHELL_OP_RULES) {
132
+ if (re.test(cmd) && !ops.includes(op)) ops.push(op)
133
+ }
134
+ }
135
+ if (!ops.length) ops.push('shell-run')
136
+ const targets = SHELL_WRITE_RE.test(cmd) ? extractPaths(cmd) : []
137
+ const hosts = WSL_HINT_RE.test(cmd) || /\b(powershell|pwsh|cmd)\.exe\b/i.test(cmd) ? ['wsl'] : []
138
+ return { ops, targets, hosts }
139
+ }
140
+
141
+ /**
142
+ * 一次工具调用 → { ops, targets, hosts }。
143
+ * @param {string} name 工具名
144
+ * @param {string} argsText 工具参数(JSON 字符串或裸文本)
145
+ */
146
+ export function classifyToolCall(name, argsText) {
147
+ const tool = String(name || '').toLowerCase()
148
+ const { command, path } = argsFields(argsText)
149
+ if (SHELL_TOOLS.has(tool)) return classifyShellCommand(command || String(argsText || ''))
150
+ if (READ_IMAGE_TOOLS.has(tool)) return { ops: ['read-image'], targets: [], hosts: [] }
151
+ if (INDEX_TOOLS.has(tool)) return { ops: ['index-doc'], targets: [], hosts: [] }
152
+ if (FETCH_TOOLS.has(tool)) return { ops: ['fetch-web'], targets: [], hosts: [] }
153
+ if (QUERY_TOOLS.has(tool)) return { ops: ['query-memory'], targets: [], hosts: [] }
154
+ if (WRITE_TOOLS.has(tool)) {
155
+ const targets = path ? [path, ...extractPaths(path)] : []
156
+ const hosts = targets.some((t) => WSL_HINT_RE.test(t)) ? ['wsl'] : []
157
+ return { ops: ['file-write'], targets: [...new Set(targets)], hosts }
158
+ }
159
+ return { ops: [], targets: [], hosts: [] }
160
+ }
161
+
162
+ /**
163
+ * 把一组已观察到的工具调用合成这一个线程的动态面。
164
+ * @param {{name?: string, arguments?: string}[]} calls
165
+ */
166
+ export function activityFromCalls(calls) {
167
+ const ops = new Set()
168
+ const targets = new Set()
169
+ const hosts = new Set()
170
+ for (const call of calls || []) {
171
+ const r = classifyToolCall(call?.name, call?.arguments)
172
+ for (const op of r.ops) ops.add(op)
173
+ for (const t of r.targets) targets.add(t)
174
+ for (const h of r.hosts) hosts.add(h)
175
+ }
176
+ return { ops: [...ops], targets: [...targets], hosts: [...hosts] }
177
+ }
178
+
179
+ /**
180
+ * 没拿到结构化调用时的兜底:直接对一段文本跑 shell 规则。
181
+ * 用于 `actionText` 形式的就绪上下文(测试与老宿主),以及人类消息里的动作词。
182
+ */
183
+ export function activityFromText(text) {
184
+ const s = String(text || '')
185
+ if (!s) return { ops: [], targets: [], hosts: [] }
186
+ const lines = s.split('\n').filter(Boolean)
187
+ const ops = new Set()
188
+ const targets = new Set()
189
+ const hosts = new Set()
190
+ for (const line of lines) {
191
+ // 先按"工具调用"解析(`edit {"file_path":"..."}`):结构化参数比正则猜命令可靠得多。
192
+ const call = /^\s*([A-Za-z_][\w-]*)\s+(\{[\s\S]*\})\s*$/.exec(line)
193
+ const r = call ? classifyToolCall(call[1], call[2]) : classifyShellCommand(line)
194
+ // 纯文本行里没有 shell 动作时不要退化成 shell-run(那是噪音源)
195
+ if (r.ops.length === 1 && r.ops[0] === 'shell-run' && !/\b\w+\s+-/.test(line)) continue
196
+ for (const op of r.ops) ops.add(op)
197
+ for (const t of r.targets) targets.add(t)
198
+ for (const h of r.hosts) hosts.add(h)
199
+ }
200
+ return { ops: [...ops], targets: [...targets], hosts: [...hosts] }
201
+ }
202
+
203
+ /** 旧 action id(含死值)→ op;无法映射返回 null。 */
204
+ export function opForLegacyAction(action) {
205
+ const a = String(action || '')
206
+ if (!a) return null
207
+ if (OP_IDS.includes(a)) return a
208
+ if (a in LEGACY_ACTION_ALIAS) return LEGACY_ACTION_ALIAS[a]
209
+ return null
210
+ }
@@ -60,9 +60,8 @@ export async function parsePdf(filePath, { pages = null, maxPages = 1000, backen
60
60
  const data = new Uint8Array(await readFile(filePath))
61
61
  const { getDocument } = await loadPdfjs()
62
62
  const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
63
- const doc = await loadingTask.promise
64
-
65
63
  try {
64
+ const doc = await loadingTask.promise
66
65
  const total = doc.numPages
67
66
  if (total > maxPages) {
68
67
  throw new Error(`PDF has ${total} pages, over the maxPages limit of ${maxPages}`)
@@ -112,9 +111,8 @@ export async function parsePdfInfo(filePath, maxPages = 1000) {
112
111
  const data = new Uint8Array(await readFile(filePath))
113
112
  const { getDocument } = await loadPdfjs()
114
113
  const loadingTask = getDocument({ data, ...PDFJS_OPTIONS })
115
- const doc = await loadingTask.promise
116
-
117
114
  try {
115
+ const doc = await loadingTask.promise
118
116
  if (doc.numPages > maxPages) {
119
117
  throw new Error(`PDF has ${doc.numPages} pages, over the maxPages limit of ${maxPages}`)
120
118
  }
@@ -17,11 +17,18 @@ function depsOf(manifest, fields) {
17
17
  const tags = new Set()
18
18
  for (const f of fields) {
19
19
  const deps = manifest && manifest[f]
20
- if (deps && typeof deps === 'object') {
21
- for (const name of Object.keys(deps)) {
22
- tags.add(name.toLowerCase())
23
- if (name.startsWith('@')) tags.add(name.split('/')[1].toLowerCase())
24
- }
20
+ if (!deps || typeof deps !== 'object') continue
21
+ for (const name of Object.keys(deps)) {
22
+ if (typeof name !== 'string' || !name) continue
23
+ tags.add(name.toLowerCase())
24
+ if (!name.startsWith('@')) continue
25
+ // scoped 包名 `@scope/pkg`:要加的是 **scope**(`vue`),不是包名段(`compiler-sfc`)。
26
+ // 旧写法 `name.split('/')[1]` 让 `@vue/*` 项目永远拿不到 `vue` tag,
27
+ // trigger.scope:['vue'] / legacy scope 过滤会把最相关的那条 procedure 静默判死;
28
+ // 而 `@malformed`(没有 `/`)会在这里抛错,异常被 collectTags 整个吞掉 → 全项目 tags 清零。
29
+ const parts = name.slice(1).split('/')
30
+ if (parts[0]) tags.add(parts[0].toLowerCase())
31
+ if (parts[1]) tags.add(parts[1].toLowerCase())
25
32
  }
26
33
  }
27
34
  return tags