@yolk_vat-y/dsh-project-memory 0.5.3 → 0.5.5

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,71 @@
1
+ // 文档索引的结构化词项(纯规则、确定性;索引期不调用任何模型)。
2
+ //
3
+ // 动机:检索只吃 title/keywords/summary/sourcePath(util/search.js 的 weightedFieldText),
4
+ // 而 summary 为了注入预算被压到 300 字符 —— 一个 3000 字符的 chunk 有九成内容检索不到。
5
+ // 这里把「注入用的短摘要」和「检索用的词项」拆开:summary 仍然短,terms 覆盖整个 chunk。
6
+ //
7
+ // 纯规则、确定性、可重放:tokenizeRaw(拉丁词 + CJK bigram)→ 去停用词/纯数字 → 词频排序 → 截断。
8
+ import { tokenizeRaw } from './util/search.js'
9
+
10
+ export const MAX_TERMS = 160
11
+ export const MAX_KEYWORDS = 8
12
+
13
+ const STOPWORDS = new Set([
14
+ 'the', 'and', 'for', 'are', 'but', 'not', 'you', 'all', 'any', 'can', 'had', 'her', 'was', 'one',
15
+ 'our', 'out', 'day', 'get', 'has', 'him', 'his', 'how', 'its', 'new', 'now', 'old', 'see', 'two',
16
+ 'way', 'who', 'boy', 'did', 'use', 'that', 'this', 'with', 'from', 'they', 'will', 'would', 'there',
17
+ 'their', 'what', 'about', 'which', 'when', 'make', 'like', 'time', 'just', 'know', 'take', 'into',
18
+ 'your', 'some', 'them', 'than', 'then', 'only', 'come', 'over', 'also', 'back', 'after', 'other',
19
+ 'many', 'most', 'such', 'even', 'much', 'more', 'been', 'were', 'have', 'each', 'does', 'doing',
20
+ 'should', 'could', 'these', 'those', 'being', 'where', 'while', 'because', 'before', 'between',
21
+ 'under', 'again', 'further', 'once', 'here', 'both', 'few', 'same', 'too', 'very', 'own', 'off',
22
+ 'per', 'via', 'etc', 'see', 'note', 'used', 'using', 'uses', 'may', 'must', 'shall',
23
+ ])
24
+
25
+ /** 一个 token 是否值得作为检索词项。 */
26
+ function isUseful(token) {
27
+ if (token.length < 2) return false
28
+ if (STOPWORDS.has(token)) return false
29
+ if (/^\d+$/.test(token)) return false
30
+ return true
31
+ }
32
+
33
+ /**
34
+ * 整个 chunk 的字面词项(唯一、按词频排序、有上限)。
35
+ * @returns {string[]} 词项数组(已去重)
36
+ */
37
+ export function extractTerms(text, { max = MAX_TERMS } = {}) {
38
+ const counts = new Map()
39
+ for (const token of tokenizeRaw(text)) {
40
+ if (!isUseful(token)) continue
41
+ counts.set(token, (counts.get(token) || 0) + 1)
42
+ }
43
+ return [...counts.entries()]
44
+ .sort((a, b) => b[1] - a[1] || b[0].length - a[0].length || a[0].localeCompare(b[0]))
45
+ .slice(0, max)
46
+ .map(([token]) => token)
47
+ }
48
+
49
+ /**
50
+ * 结构化的检索词项串(用于 entry.terms,并入 BM25 的检索文本)。
51
+ */
52
+ export function extractTermText(text, { max = MAX_TERMS } = {}) {
53
+ return extractTerms(text, { max }).join(' ')
54
+ }
55
+
56
+ /**
57
+ * 规则化关键词:标题重复一次以取得小幅加权(替代此前依赖 LLM 的 5–10 个关键词)。
58
+ */
59
+ export function extractKeywords(title, text, { max = MAX_KEYWORDS } = {}) {
60
+ const source = title ? `${title} ${title} ${text}` : text
61
+ return extractTerms(source, { max })
62
+ }
63
+
64
+ /**
65
+ * 旧 store 的 doc 条目没有 `terms`:需要一次定向回填,否则内容哈希未变的文件会被跳过,
66
+ * 新的检索覆盖不会生效。只对 doc 条目判断,code 条目(符号表)不需要 terms。
67
+ */
68
+ export function docEntriesNeedBackfill(entries) {
69
+ if (!Array.isArray(entries) || entries.length === 0) return true
70
+ return entries.some((e) => e && e.type === 'doc' && typeof e.terms !== 'string')
71
+ }
@@ -2,9 +2,10 @@ import { createHash } from 'node:crypto'
2
2
  import path from 'node:path'
3
3
  import { stat } from 'node:fs/promises'
4
4
  import { looksLikeDump, readTextFile } from './util/fs.js'
5
+ import { summarizeText } from './util/text.js'
5
6
  import { parsePdf } from './parsers/pdfjs-parser.js'
6
7
  import { chunkText } from './chunker.js'
7
- import { extractDocEntry } from './llm.js'
8
+ import { extractKeywords, extractTermText } from './doc-index.js'
8
9
 
9
10
  export async function extractTextFromFile(filePath, { maxFileSizeMb = 50, maxPdfPages = 1000 } = {}) {
10
11
  const ext = path.extname(filePath).toLowerCase()
@@ -21,43 +22,37 @@ export async function extractTextFromFile(filePath, { maxFileSizeMb = 50, maxPdf
21
22
  return readTextFile(filePath, maxFileSizeMb ? maxFileSizeMb * 1024 * 1024 : Infinity)
22
23
  }
23
24
 
24
- const DOC_CONCURRENCY = 4
25
-
26
- export async function buildDocEntries(llm, a, b, c) {
27
- // Backward compatible: old signature (llm, filePath, opts) or new (llm, relPath, filePath, opts)
28
- const [relPath, filePath, opts] = c === undefined ? [a, a, b] : [a, b, c]
25
+ /**
26
+ * 文档分片 → 记忆条目(索引期不调用任何模型:纯规则、确定性、可重放)。
27
+ *
28
+ * 注入用 summary 与检索用 terms 分离:
29
+ * - summary:≤300 字符,进上下文,保持小预算;
30
+ * - terms:整个 chunk 的字面词项,只进 BM25 检索文本,不进注入。
31
+ * 于是「chunk 只有前 300 字符可检索」的旧限制被移除,且没有任何 LLM 调用。
32
+ * 函数签名里刻意没有 llm —— 索引期零 LLM 由构造保证,而不是靠 catch。
33
+ */
34
+ export async function buildDocEntries(relPath, filePath, opts = {}) {
29
35
  const text = await extractTextFromFile(filePath, opts)
30
36
  if (looksLikeDump(text)) return null
31
-
37
+
32
38
  // Compute content hash for update detection
33
39
  const hash = createHash('sha256').update(text).digest('hex').slice(0, 16)
34
-
40
+
35
41
  const chunks = chunkText(text, opts.chunkChars, opts.maxChunks)
36
- const metas = new Array(chunks.length)
37
- let cursor = 0
38
- await Promise.all(
39
- Array.from({ length: Math.min(DOC_CONCURRENCY, chunks.length) }, () =>
40
- (async () => {
41
- while (cursor < chunks.length) {
42
- const i = cursor++
43
- metas[i] = await extractDocEntry(llm, chunks[i], filePath)
44
- }
45
- })(),
46
- ),
47
- )
48
- return metas.map((meta, i) => ({
42
+ return chunks.map((chunk, i) => ({
49
43
  id: `${relativeId(relPath)}#${i}`,
50
44
  sourcePath: relPath,
51
- sourceLine: chunks[i].line,
45
+ sourceLine: chunk.line,
52
46
  type: 'doc',
53
- title: meta.title,
54
- summary: meta.summary,
55
- blindSpots: meta.blindSpots || '',
56
- keywords: meta.keywords,
47
+ title: chunk.title || relPath,
48
+ summary: summarizeText(chunk.text),
49
+ blindSpots: '',
50
+ keywords: extractKeywords(chunk.title, chunk.text),
51
+ terms: extractTermText(chunk.text),
57
52
  hash,
58
53
  }))
59
54
  }
60
55
 
61
56
  function relativeId(filePath) {
62
57
  return String(filePath).replace(/[\\/:\s]/g, '_')
63
- }
58
+ }
package/src/index.js CHANGED
@@ -13,6 +13,7 @@ import { setupTaskbridge } from './setup/taskbridge.js'
13
13
  import { listTasksTool, selectTaskTool, archiveTaskTool, showTaskPanelTool } from './tools/task-tools.js'
14
14
  import { lessonTool } from './tools/lesson-tools.js'
15
15
  import { installAutoInject } from './auto-inject.js'
16
+ import { rememberRoute } from './llm-route.js'
16
17
  import { tasksCommandDefinition } from './commands/tasks.js'
17
18
  import { taskCommandDefinition } from './commands/task-actions.js'
18
19
  import { insightCommandDefinition } from './commands/insight-actions.js'
@@ -29,6 +30,14 @@ export const Config = Schema.object({
29
30
  maxPdfPages: Schema.number().default(1000),
30
31
  llmQueryExpansion: Schema.boolean().default(false),
31
32
  expansionCount: Schema.number().default(6),
33
+ // 辅助 LLM 调用的路由覆写 —— **索引期零 LLM**,这里的路由只服务于召回期可选项
34
+ // (llmQueryExpansion 查询扩展)与反思期(reflection.enabled)。
35
+ // 未设置时:工具调用取当前会话 header 的 provider/model;后台路径取最近一次会话路由(request/header)。
36
+ // 两者都拿不到时明确走回退并记录 degraded(不静默)。
37
+ llm: Schema.object({
38
+ provider: Schema.string(),
39
+ model: Schema.string(),
40
+ }).default({}),
32
41
  lazyIndexing: Schema.boolean().default(true),
33
42
  autoIndexOnFirstUse: Schema.boolean().default(false),
34
43
  watch: Schema.boolean().default(true),
@@ -40,7 +49,7 @@ export const Config = Schema.object({
40
49
  // 任务成为会话绑定(select_task / /task switch)时,把任务步骤推成宿主 todo/write 快照
41
50
  syncHostOnAdopt: Schema.boolean().default(true),
42
51
  }).default({}),
43
- // v0.5 单一 insight 实体:分层去重/强化/提升/容量/归档(详见 PLAN-v0.5.0.md)
52
+ // v0.5 单一 insight 实体:分层去重/强化/提升/容量/归档
44
53
  insight: Schema.object({
45
54
  dedupOverlap: Schema.number().default(0.7),
46
55
  reinforceBand: Schema.number().default(0.65),
@@ -51,7 +60,7 @@ export const Config = Schema.object({
51
60
  decayDays: Schema.number().default(90),
52
61
  globalFile: Schema.string(),
53
62
  }).default({}),
54
- // PR 1b:反思管线(LLM 消费点,默认关)。产出只写 task 级草稿,见 PLAN §3
63
+ // 反思管线(召回期可选 LLM 消费点,默认关)。产出只写 task 级草稿。
55
64
  reflection: Schema.object({
56
65
  enabled: Schema.boolean().default(false),
57
66
  cooldownMs: Schema.number().default(1800000),
@@ -63,7 +72,24 @@ export const Config = Schema.object({
63
72
  enabled: Schema.boolean().default(true),
64
73
  entryOn: Schema.boolean().default(true),
65
74
  maxTokens: Schema.number().default(400),
66
- relevanceMin: Schema.number().default(0.25),
75
+ entryMaxInsights: Schema.number().default(6),
76
+ // 提示通道的相对阈值(层内最高分的比例)——出厂判据
77
+ signalMinRatio: Schema.number().default(0.5),
78
+ // resident 任务卡最多显示几个「编辑中」文件
79
+ editedMax: Schema.number().default(3),
80
+ // 模型自己维护清单且之后没有新人类消息时,不把任务卡回声给模型
81
+ skipEchoSelfTodo: Schema.boolean().default(true),
82
+ // ⚠️ 旧兼容闸门,**刻意不给默认值**:cfgEngine 以「relevanceMin 是否为 number」选分支,
83
+ // 一旦有默认值就会永远走旧的绝对 overlap 判据——而它对长消息实测只有 0.014~0.057,
84
+ // 必然过不了,相对阈值 signalMinRatio 就变成死代码。只有用户显式配置才保留旧行为。
85
+ relevanceMin: Schema.number(),
86
+ // 预算审计日志(stderr)**默认 off**:终端是用户可见面,而"预算挤掉低优先级条目"是
87
+ // 正常降级、不是故障——默认打印会让一次 dsh web 启动刷出多行,用户的第一反应是卸载插件。
88
+ // off(默认,永不打印)/ once(每个会话最多一行,首次出现丢弃时)/ all(丢弃组合每变化一次一行,作者排查)
89
+ budgetLog: Schema.union(['off', 'once', 'all']).default('off'),
90
+ // 同一条 insight 在本会话里重复注入的冷却(pre-step 步数)。0(默认)= 正文没变就不再注入:
91
+ // 注入消息留在会话历史里(宿主只追加不压缩),整块重发只是重复占位。>0 用于外部裁剪历史的场景。
92
+ reinjectItemsAfter: Schema.number().default(0),
67
93
  }).default({}),
68
94
  })
69
95
 
@@ -88,6 +114,10 @@ export function apply(ctx, config) {
88
114
 
89
115
  // TaskBridge:任务实体 + 宿主 todo 同步
90
116
  setupTaskbridge(ctx, config)
117
+ // 跟踪会话路由(request/header 事件),为无会话上下文的召回期可选 LLM(查询扩展 / 反思)兜底
118
+ ctx.on('session/event', (session, event) => {
119
+ if (event?.type === 'request/header') rememberRoute(session)
120
+ })
91
121
  ctx.tools.register(listTasksTool(config))
92
122
  ctx.tools.register(selectTaskTool(config, { llm: ctx.llm, ctx }))
93
123
  ctx.tools.register(archiveTaskTool(config, { llm: ctx.llm, ctx }))
@@ -98,10 +98,12 @@ export function normalizeInsight(raw, extra = {}) {
98
98
  if (Array.isArray(raw[f]) && raw[f].length) ins[f] = [...new Set(raw[f].map((x) => String(x)))]
99
99
  }
100
100
  if (raw.trigger && typeof raw.trigger === 'object') {
101
- const tr = { keywords: Array.isArray(raw.trigger.keywords) ? raw.trigger.keywords.map(String) : [] }
102
- if (Array.isArray(raw.trigger.symbols)) tr.symbols = raw.trigger.symbols.map(String)
103
- if (Array.isArray(raw.trigger.scope)) tr.scope = raw.trigger.scope.map(String)
104
- if (tr.keywords.length) ins.trigger = tr
101
+ // 所有 kind 通用:keywords / symbols / actions / paths / scope(actions/paths 见 src/readiness.js)
102
+ const tr = {}
103
+ for (const f of ['keywords', 'symbols', 'actions', 'paths', 'scope']) {
104
+ if (Array.isArray(raw.trigger[f]) && raw.trigger[f].length) tr[f] = raw.trigger[f].map(String)
105
+ }
106
+ if (Object.keys(tr).length) ins.trigger = tr
105
107
  }
106
108
  return ins
107
109
  }
package/src/lazy.js CHANGED
@@ -3,6 +3,7 @@ import { tmpdir } from 'node:os'
3
3
  import { existsSync, readdirSync, statSync } from 'node:fs'
4
4
  import { isSupportedCode, isSupportedDoc, memoryRootFor, readFileForIndex, relativePath, storeKey } from './util/fs.js'
5
5
  import { buildDocEntries } from './doc-pipeline.js'
6
+ import { docEntriesNeedBackfill } from './doc-index.js'
6
7
  import { scanSymbols } from './symbols.js'
7
8
  import { linkEntries } from './link.js'
8
9
  import { ProjectMemoryStore } from './store.js'
@@ -98,7 +99,8 @@ export async function indexFile(ctx, config, filePath, watchManager = null) {
98
99
  } catch {
99
100
  return false
100
101
  }
101
- if (existing && existing.sha256 === hash) return false
102
+ // 旧 store 的 doc 条目缺 terms → 一次性回填(即使哈希未变)
103
+ if (existing && existing.sha256 === hash && !(isSupportedDoc(ext) && docEntriesNeedBackfill(store.entries[rel]))) return false
102
104
 
103
105
  if (watchManager) {
104
106
  watchManager.addRoot(root)
@@ -115,7 +117,7 @@ export async function indexFile(ctx, config, filePath, watchManager = null) {
115
117
  return true
116
118
  })
117
119
  } else {
118
- entries = await buildDocEntries(ctx.llm, rel, filePath, {
120
+ entries = await buildDocEntries(rel, filePath, {
119
121
  chunkChars: config.chunkChars,
120
122
  maxChunks: config.maxChunksPerFile,
121
123
  maxFileSizeMb: config.maxFileSizeMb,
package/src/link.js CHANGED
@@ -28,7 +28,8 @@ export function linkEntries(store) {
28
28
  let links = 0
29
29
  for (const doc of docs) {
30
30
  const linked = new Set()
31
- const haystack = `${doc.title || ''} ${doc.summary || ''} ${doc.keywords ? doc.keywords.join(' ') : ''}`.toLowerCase()
31
+ // 结构词项也参与链接:terms 覆盖整个 chunk,符号在后半段被提及时同样能链上(与检索同源)
32
+ const haystack = `${doc.title || ''} ${doc.summary || ''} ${doc.terms || ''} ${doc.keywords ? doc.keywords.join(' ') : ''}`.toLowerCase()
32
33
  for (const [, entry] of symbolByName) {
33
34
  const hit = entry.re ? entry.re.test(haystack) : haystack.includes(entry.lower)
34
35
  for (const s of entry.syms) {
@@ -0,0 +1,95 @@
1
+ // 辅助 LLM 调用的路由解析(仅召回期可选:查询扩展 / 反思)。
2
+ //
3
+ // 背景:宿主 LlmRuntime.stream 要求 provider/model 必填,缺失即抛 NO_ADAPTER。
4
+ // 本模块负责:
5
+ // 1) 从工具 exec / 会话 / 配置解析出 (provider, model);
6
+ // 2) 把「无法路由 / 调用失败」记成可见的 degraded 记录,而不是静默降级。
7
+ //
8
+ // 解析优先级(与宿主 compaction summarizer 同序,再补一条后台 hint):
9
+ // config.llm 显式覆写 > exec 会话 requestHeader().config > exec agent options >
10
+ // 最近一次 request/header 事件记下的路由(后台 watch/lazy 兜底)。
11
+ //
12
+ // 登记规则:路由只影响辅助调用走哪个模型,不改变检索/排序行为。
13
+
14
+ /** 最近一次在会话里观察到的路由,供无 exec 的后台索引(watch/lazy)兜底。 */
15
+ let routeHint = null
16
+
17
+ /** 无法路由 / 调用失败留下的可见痕迹:code -> { code, reason, at, count }。 */
18
+ const degraded = new Map()
19
+
20
+ function normalize(provider, model, source) {
21
+ if (typeof provider !== 'string' || !provider) return null
22
+ if (typeof model !== 'string' || !model) return null
23
+ return { provider, model, source }
24
+ }
25
+
26
+ function routeFromSession(session) {
27
+ try {
28
+ const config = session?.requestHeader?.()?.config
29
+ return normalize(config?.provider, config?.model, 'session')
30
+ } catch {
31
+ return null
32
+ }
33
+ }
34
+
35
+ /** 记下会话当前路由,作为后台索引的兜底。由 request/header 事件驱动。 */
36
+ export function rememberRoute(session) {
37
+ const route = routeFromSession(session)
38
+ if (route) routeHint = { provider: route.provider, model: route.model }
39
+ }
40
+
41
+ /** 显式配置的路由(config.llm.provider + config.llm.model 同时存在才算)。 */
42
+ function routeFromConfig(config) {
43
+ return normalize(config?.llm?.provider, config?.llm?.model, 'config')
44
+ }
45
+
46
+ /**
47
+ * 解析一次辅助 LLM 调用的路由。
48
+ * @param exec - 工具执行上下文(可含 agent.session / agent.options),可为空。
49
+ * @param config - 插件配置(可含 llm 覆写)。
50
+ * @returns {{provider: string, model: string, source: string}|null}
51
+ */
52
+ export function resolveRoute(exec, config) {
53
+ const configured = routeFromConfig(config)
54
+ if (configured) return configured
55
+
56
+ const fromExecSession = routeFromSession(exec?.agent?.session)
57
+ if (fromExecSession) return fromExecSession
58
+
59
+ const options = exec?.agent?.options
60
+ const fromOptions = normalize(options?.provider, options?.model, 'agent')
61
+ if (fromOptions) return fromOptions
62
+
63
+ if (routeHint) return { ...routeHint, source: 'session-hint' }
64
+ return null
65
+ }
66
+
67
+ /**
68
+ * 记录一次降级(允许降级,不允许静默)。同一 code 只打印一次,避免逐 chunk 刷屏;
69
+ * 计数与最近原因保留下来,供 memory_stats / 测试读取。
70
+ */
71
+ export function noteDegraded(code, reason) {
72
+ const at = new Date().toISOString()
73
+ const prev = degraded.get(code)
74
+ if (prev) {
75
+ prev.count += 1
76
+ prev.at = at
77
+ prev.reason = reason
78
+ return prev
79
+ }
80
+ const record = { code, reason, at, count: 1 }
81
+ degraded.set(code, record)
82
+ console.warn(`[dsh-project-memory] degraded ${code}: ${reason}`)
83
+ return record
84
+ }
85
+
86
+ /** 当前累计的降级记录(快照,按 code 去重)。 */
87
+ export function degradedList() {
88
+ return [...degraded.values()].map((d) => ({ ...d }))
89
+ }
90
+
91
+ /** 测试用:清空路由 hint 与降级记录。 */
92
+ export function resetRouteState() {
93
+ routeHint = null
94
+ degraded.clear()
95
+ }
package/src/llm.js CHANGED
@@ -1,5 +1,7 @@
1
+ // 辅助 LLM 调用(**仅召回期 / 反思期,按需可选**;索引期不调用任何模型)。
2
+ // chatText 是唯一的宿主调用出口:provider/model 必填,缺失时显式抛错,由调用方决定回退并记 degraded。
1
3
  import { BlockAssembler, createUserMessage } from '@deepseek-ai/dsh-llm'
2
- import { tokenize } from './util/search.js'
4
+ import { noteDegraded } from './llm-route.js'
3
5
 
4
6
  function systemMessage(text) {
5
7
  return { role: 'system', content: [{ type: 'text', text }] }
@@ -13,24 +15,19 @@ function textOf(message) {
13
15
  .join('\n')
14
16
  }
15
17
 
16
- const MAX_SUMMARY = 300
17
-
18
- export function summarizeText(text, max = MAX_SUMMARY) {
19
- const flat = String(text || '').replace(/\s+/g, ' ').trim()
20
- if (!flat) return ''
21
- if (flat.length <= max) return flat
22
- const clip = max - 1
23
- const clipped = flat.slice(0, clip)
24
- const lastBreak = Math.max(clipped.lastIndexOf('。'), clipped.lastIndexOf('.'), clipped.lastIndexOf(';'))
25
- return lastBreak > clip * 0.4 ? clipped.slice(0, lastBreak + 1) : clipped + '…'
26
- }
27
-
28
- export async function chatText(llm, system, user, { timeoutMs = 120000 } = {}) {
18
+ export async function chatText(llm, system, user, { timeoutMs = 120000, route } = {}) {
19
+ // provider/model 是宿主 GenerateOptions 的必填项,缺失时 LlmRuntime 抛 NO_ADAPTER。
20
+ // 这里显式失败(由调用方决定回退并记 degraded),不让异常悄悄消失。
21
+ if (!route?.provider || !route?.model) {
22
+ throw new Error('auxiliary LLM call requires an explicit provider/model route')
23
+ }
29
24
  const assembler = new BlockAssembler()
30
25
  const controller = new AbortController()
31
26
  const timer = setTimeout(() => controller.abort(), timeoutMs)
32
27
  try {
33
28
  for await (const chunk of llm.stream({
29
+ provider: route.provider,
30
+ model: route.model,
34
31
  messages: [systemMessage(system), createUserMessage({ content: [{ type: 'text', text: user }] })],
35
32
  signal: controller.signal,
36
33
  })) {
@@ -72,8 +69,13 @@ function parseJson(text, validate) {
72
69
  return null
73
70
  }
74
71
 
75
- export async function expandQuery(llm, query, count = 6) {
72
+ /** 召回期可选的查询扩展(默认关;开启时才需要 provider/model 路由)。 */
73
+ export async function expandQuery(llm, query, count = 6, { route } = {}) {
76
74
  if (!llm) return [query]
75
+ if (!route) {
76
+ noteDegraded('llm.expand.no-route', 'no provider/model route for query expansion; search uses the raw query')
77
+ return [query]
78
+ }
77
79
  const system =
78
80
  'You are a search-query expander for a codebase/document memory search engine. ' +
79
81
  'Given a user query, return a STRICT JSON array of alternative search queries that ' +
@@ -81,57 +83,14 @@ export async function expandQuery(llm, query, count = 6) {
81
83
  'code identifier guesses, and narrower/longer phrasings. Include the original query first. ' +
82
84
  'Output only the JSON array of strings, no fences, no commentary.'
83
85
  try {
84
- const raw = await chatText(llm, system, `Query: "${query}"\n\nReturn the JSON array.`)
86
+ const raw = await chatText(llm, system, `Query: "${query}"\n\nReturn the JSON array.`, { route })
85
87
  const parsed = parseJsonArray(raw)
86
88
  if (Array.isArray(parsed) && parsed.length) {
87
89
  const variants = parsed.map(String).filter((s) => s.trim()).slice(0, count)
88
90
  if (variants.length) return variants
89
91
  }
90
- } catch {
91
- // fall through to the raw query
92
+ } catch (err) {
93
+ noteDegraded('llm.expand.failed', `query expansion LLM call failed: ${err?.message || err}`)
92
94
  }
93
95
  return [query]
94
96
  }
95
-
96
- export async function extractDocEntry(llm, chunk, sourcePath) {
97
- const system =
98
- 'You are a project-documentation indexer. Given a chunk of a project document, ' +
99
- 'return a STRICT JSON object with exactly four fields: ' +
100
- '"title" (short section title, string), ' +
101
- '"summary" (3-5 sentence dense summary of what this section covers, ' +
102
- 'mentioning concrete names, decisions, constraints, and key technical details), ' +
103
- '"blindSpots" (string describing what this summary does NOT cover, ' +
104
- 'e.g. "未覆盖:部署细节、性能基准、v0.2 前 API", empty string if none), ' +
105
- '"keywords" (array of 5-10 searchable strings: ' +
106
- 'cover the document\'s own language AND English equivalents, ' +
107
- 'so a query in either language can match). ' +
108
- 'Do not include markdown fences, do not add commentary, output only the JSON object.'
109
-
110
- const user =
111
- `Document: ${sourcePath}\nSection: ${chunk.title || '(untitled)'}\n\n` +
112
- `Content:\n${chunk.text.slice(0, 6000)}\n\nReturn the JSON object.`
113
-
114
- const fallback = () => ({
115
- title: chunk.title || sourcePath,
116
- summary: summarizeText(chunk.text),
117
- blindSpots: '',
118
- keywords: tokenize(chunk.title).slice(0, 5),
119
- })
120
-
121
- if (!llm) return fallback()
122
-
123
- try {
124
- const raw = await chatText(llm, system, user)
125
- const parsed = parseStructuredJson(raw)
126
- if (!parsed || typeof parsed.summary !== 'string' || !parsed.summary.trim()) return fallback()
127
- const kw = Array.isArray(parsed.keywords) ? parsed.keywords.map(String).filter((k) => k).slice(0, 8) : []
128
- return {
129
- title: typeof parsed.title === 'string' && parsed.title.trim() ? parsed.title.trim() : chunk.title || sourcePath,
130
- summary: summarizeText(parsed.summary.trim()),
131
- blindSpots: typeof parsed.blindSpots === 'string' ? parsed.blindSpots.trim() : '',
132
- keywords: kw.length ? kw : tokenize(chunk.title).slice(0, 5),
133
- }
134
- } catch {
135
- return fallback()
136
- }
137
- }
@@ -25,6 +25,9 @@ const PDFJS_OPTIONS = {
25
25
  isEvalSupported: false,
26
26
  useWorkerFetch: false,
27
27
  useWorker: false,
28
+ // 0 = ERRORS。损坏/线性化缺失的 PDF 会让 pdf.js 打 "Warning: Indexing all PDF objects"
29
+ // 之类的告警——那是它自己的恢复路径,对使用者没有可操作性,静音。
30
+ verbosity: 0,
28
31
  }
29
32
 
30
33
  function buildMarkdown(pages) {