@customize-agent/knowledge 4.0.9 → 4.0.10

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -46,6 +46,7 @@ export declare class TextChunker {
46
46
  private extractMarkdownTableRowRange;
47
47
  private splitMarkdownTable;
48
48
  private enforceCandidateLimit;
49
+ private buildTitlePaths;
49
50
  private withHeader;
50
51
  private createChunk;
51
52
  private kindForCategory;
@@ -60,6 +60,7 @@ export class TextChunker {
60
60
  createTextCandidates(text, category, config) {
61
61
  const sections = this.splitIntoSections(text, category);
62
62
  const candidates = [];
63
+ const titlePaths = this.buildTitlePaths(sections);
63
64
  sections.forEach((section, parentIndex) => {
64
65
  const parentId = `p${parentIndex}`;
65
66
  const parts = this.mergeLeadingHeader(this.recursiveSplit(section.text, config.maxChunkSize));
@@ -72,6 +73,7 @@ export class TextChunker {
72
73
  startChar,
73
74
  endChar: startChar + part.length,
74
75
  sectionTitle: section.title,
76
+ titlePath: titlePaths[parentIndex],
75
77
  kind: this.kindForCategory(category),
76
78
  parentId,
77
79
  parentIndex,
@@ -90,6 +92,7 @@ export class TextChunker {
90
92
  startChar,
91
93
  endChar: startChar + part.length,
92
94
  sectionTitle: this.extractSectionTitle(part) ?? '表格数据',
95
+ titlePath: this.extractSectionTitle(part) ?? '表格数据',
93
96
  kind: 'table',
94
97
  parentId: `table-${index}`,
95
98
  parentIndex: index,
@@ -108,6 +111,7 @@ export class TextChunker {
108
111
  startChar: Math.max(0, text.indexOf(part.slice(0, 40))),
109
112
  endChar: Math.max(0, text.indexOf(part.slice(0, 40))) + part.length,
110
113
  sectionTitle: this.extractSectionTitle(part),
114
+ titlePath: this.extractSectionTitle(part),
111
115
  kind: 'data',
112
116
  parentId: `data-${index}`,
113
117
  parentIndex: index,
@@ -128,6 +132,7 @@ export class TextChunker {
128
132
  startChar,
129
133
  endChar: startChar + part.length,
130
134
  sectionTitle: this.extractSectionTitle(part),
135
+ titlePath: this.extractSectionTitle(part),
131
136
  kind: 'code',
132
137
  parentId: `code-${language}-${index}`,
133
138
  parentIndex: index,
@@ -321,6 +326,24 @@ export class TextChunker {
321
326
  }));
322
327
  });
323
328
  }
329
+ buildTitlePaths(sections) {
330
+ const stack = [];
331
+ return sections.map(section => {
332
+ const firstLine = section.text.trim().split(/\r?\n/u)[0] ?? '';
333
+ const heading = firstLine.match(/^(#{1,6})\s+(.+)$/u);
334
+ if (heading?.[1] && heading[2]) {
335
+ const level = heading[1].length;
336
+ const title = heading[2].trim();
337
+ while (stack.length > 0 && stack[stack.length - 1].level >= level)
338
+ stack.pop();
339
+ stack.push({ level, title });
340
+ }
341
+ else if (section.title && stack.length === 0) {
342
+ stack.push({ level: 1, title: section.title });
343
+ }
344
+ return stack.map(item => item.title).join(' > ') || section.title;
345
+ });
346
+ }
324
347
  withHeader(text, file, config) {
325
348
  if (!config.headerInjection)
326
349
  return text;
@@ -344,9 +367,10 @@ export class TextChunker {
344
367
  childIndex: candidate.childIndex,
345
368
  rowRange: candidate.rowRange,
346
369
  sectionTitle: candidate.sectionTitle ?? this.extractSectionTitle(text),
370
+ titlePath: candidate.titlePath ?? candidate.sectionTitle ?? this.extractSectionTitle(text),
347
371
  startChar: candidate.startChar,
348
372
  endChar: candidate.endChar,
349
- splitStrategy: 'recursive_parent_child_v1',
373
+ splitStrategy: 'recursive_parent_child_v2',
350
374
  },
351
375
  };
352
376
  }
@@ -17,11 +17,18 @@ export interface StoredChunk {
17
17
  relativePath: string;
18
18
  chunkIndex: number;
19
19
  content: string;
20
+ searchContent?: string;
20
21
  category: FileCategory;
21
22
  format: string;
22
23
  collectionName: string;
23
24
  tokenCount: number;
24
25
  sectionTitle?: string;
26
+ titlePath?: string;
27
+ parentId?: string;
28
+ chunkKind?: string;
29
+ rowRange?: string;
30
+ startChar?: number;
31
+ endChar?: number;
25
32
  metadataJson?: string;
26
33
  createdAt: number;
27
34
  }
@@ -183,6 +190,7 @@ export declare class IndexStateStore {
183
190
  relativePath: string;
184
191
  }>;
185
192
  close(): void;
193
+ private resetLegacyChunkSchemaIfNeeded;
186
194
  private initTables;
187
195
  private initFts;
188
196
  private rebuildFtsIfNeeded;
@@ -194,6 +202,8 @@ export declare class IndexStateStore {
194
202
  private splitParentGroups;
195
203
  private rowToJob;
196
204
  private metadataString;
205
+ private buildChunkSearchContent;
206
+ private searchableRowText;
197
207
  private rowToChunk;
198
208
  private expandSearchTerms;
199
209
  private toFtsQuery;
@@ -128,13 +128,14 @@ export class IndexStateStore {
128
128
  this.db.prepare('DELETE FROM kb_chunks_fts WHERE relative_path = ?').run(relativePath);
129
129
  const insert = this.db.prepare(`
130
130
  INSERT INTO kb_chunks (
131
- id, relative_path, chunk_index, content, category, format,
132
- collection_name, token_count, section_title, metadata_json, created_at
133
- ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
131
+ id, relative_path, chunk_index, content, search_content, category, format,
132
+ collection_name, token_count, section_title, title_path, parent_id, chunk_kind,
133
+ row_range, start_char, end_char, metadata_json, created_at
134
+ ) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)
134
135
  `);
135
136
  const insertFts = this.ftsEnabled ? this.db.prepare(`
136
- INSERT INTO kb_chunks_fts (id, relative_path, category, format, section_title, content)
137
- VALUES (?, ?, ?, ?, ?, ?)
137
+ INSERT INTO kb_chunks_fts (id, relative_path, category, format, section_title, title_path, chunk_kind, content)
138
+ VALUES (?, ?, ?, ?, ?, ?, ?, ?)
138
139
  `) : undefined;
139
140
  const insertParent = this.db.prepare(`
140
141
  INSERT INTO kb_parent_chunks (
@@ -160,13 +161,18 @@ export class IndexStateStore {
160
161
  for (const [parentId, group] of parentGroups.entries()) {
161
162
  const parentContent = group.map(chunk => chunk.text).join('\n\n---\n\n');
162
163
  parentContents.push(parentContent);
163
- insertParent.run(parentId, relativePath, parentId, parentContent, file.category, file.format, file.collectionName, group.find(chunk => chunk.sectionTitle)?.sectionTitle ?? null, group.length, JSON.stringify({ parentId, splitStrategy: this.metadataString(group[0]?.metadata.splitStrategy), chunkKind: this.metadataString(group[0]?.metadata.chunkKind) }), now);
164
+ insertParent.run(parentId, relativePath, parentId, parentContent, file.category, file.format, file.collectionName, group.find(chunk => chunk.sectionTitle)?.sectionTitle ?? null, group.length, JSON.stringify({ parentId, splitStrategy: this.metadataString(group[0]?.metadata.splitStrategy), chunkKind: this.metadataString(group[0]?.metadata.chunkKind), titlePath: this.metadataString(group[0]?.metadata.titlePath) }), now);
164
165
  }
165
166
  insertDocument.run(`${relativePath}#document`, relativePath, parentContents.join('\n\n=== SECTION ===\n\n'), file.category, file.format, file.collectionName, parentGroups.size, groupedChunks.length, JSON.stringify({ parentType: 'document', splitStrategy: 'document_section_child_v1' }), now);
166
167
  for (const chunk of groupedChunks) {
167
168
  const chunkId = `${relativePath}#${chunk.index}`;
168
- insert.run(chunkId, relativePath, chunk.index, chunk.text, file.category, file.format, file.collectionName, chunk.tokenCount, chunk.sectionTitle ?? null, JSON.stringify(chunk.metadata), now);
169
- insertFts?.run(chunkId, relativePath, file.category, file.format, chunk.sectionTitle ?? '', chunk.text);
169
+ const titlePath = this.metadataString(chunk.metadata.titlePath) ?? chunk.sectionTitle ?? '';
170
+ const parentId = this.metadataString(chunk.metadata.parentId) ?? null;
171
+ const chunkKind = this.metadataString(chunk.metadata.chunkKind) ?? null;
172
+ const rowRange = this.metadataString(chunk.metadata.rowRange) ?? null;
173
+ const searchContent = this.buildChunkSearchContent(relativePath, file, chunk, titlePath);
174
+ insert.run(chunkId, relativePath, chunk.index, chunk.text, searchContent, file.category, file.format, file.collectionName, chunk.tokenCount, chunk.sectionTitle ?? null, titlePath || null, parentId, chunkKind, rowRange, Number(chunk.metadata.startChar ?? chunk.startChar), Number(chunk.metadata.endChar ?? chunk.endChar), JSON.stringify(chunk.metadata), now);
175
+ insertFts?.run(chunkId, relativePath, file.category, file.format, chunk.sectionTitle ?? '', titlePath, chunkKind ?? '', searchContent);
170
176
  }
171
177
  });
172
178
  transaction();
@@ -278,7 +284,7 @@ export class IndexStateStore {
278
284
  `).all(matchQuery, limit * 8);
279
285
  return rows
280
286
  .map(row => {
281
- const keyword = this.scoreChunkDetailed(`${String(row.section_title ?? '')}\n${String(row.category)}\n${String(row.format)}\n${String(row.content)}`, terms);
287
+ const keyword = this.scoreChunkDetailed(this.searchableRowText(row), terms);
282
288
  const bm25Score = this.bm25ToPositiveScore(Number(row.bm25_score));
283
289
  return this.rowToChunk(row, keyword.keywordScore + bm25Score, { ...keyword, bm25Score });
284
290
  })
@@ -293,13 +299,13 @@ export class IndexStateStore {
293
299
  searchChunksLike(terms, limit) {
294
300
  const rows = this.db.prepare(`
295
301
  SELECT rowid, * FROM kb_chunks
296
- WHERE ${terms.map(() => '(LOWER(content) LIKE ? OR LOWER(relative_path) LIKE ? OR LOWER(category) LIKE ? OR LOWER(format) LIKE ?)').join(' OR ')}
302
+ WHERE ${terms.map(() => '(LOWER(search_content) LIKE ? OR LOWER(content) LIKE ? OR LOWER(relative_path) LIKE ? OR LOWER(category) LIKE ? OR LOWER(format) LIKE ? OR LOWER(COALESCE(title_path, \'\')) LIKE ? OR LOWER(COALESCE(chunk_kind, \'\')) LIKE ?)').join(' OR ')}
297
303
  ORDER BY created_at DESC
298
304
  LIMIT ?
299
- `).all(...terms.flatMap(term => [`%${term}%`, `%${term}%`, `%${term}%`, `%${term}%`]), limit * 6);
305
+ `).all(...terms.flatMap(term => [`%${term}%`, `%${term}%`, `%${term}%`, `%${term}%`, `%${term}%`, `%${term}%`, `%${term}%`]), limit * 6);
300
306
  return rows
301
307
  .map(row => {
302
- const keyword = this.scoreChunkDetailed(`${String(row.section_title ?? '')}\n${String(row.category)}\n${String(row.format)}\n${String(row.content)}`, terms);
308
+ const keyword = this.scoreChunkDetailed(this.searchableRowText(row), terms);
303
309
  return this.rowToChunk(row, keyword.keywordScore, keyword);
304
310
  })
305
311
  .filter(row => row.score > 0)
@@ -479,7 +485,23 @@ export class IndexStateStore {
479
485
  close() {
480
486
  this.db.close();
481
487
  }
488
+ resetLegacyChunkSchemaIfNeeded() {
489
+ const table = this.db.prepare("SELECT name FROM sqlite_master WHERE type = 'table' AND name = 'kb_chunks'").get();
490
+ if (!table)
491
+ return;
492
+ const columns = this.db.prepare('PRAGMA table_info(kb_chunks)').all();
493
+ const names = new Set(columns.map(column => column.name));
494
+ if (names.has('search_content') && names.has('title_path') && names.has('chunk_kind'))
495
+ return;
496
+ this.db.exec(`
497
+ DROP TABLE IF EXISTS kb_chunks_fts;
498
+ DROP TABLE IF EXISTS kb_chunks;
499
+ DROP TABLE IF EXISTS kb_parent_chunks;
500
+ DROP TABLE IF EXISTS kb_document_chunks;
501
+ `);
502
+ }
482
503
  initTables() {
504
+ this.resetLegacyChunkSchemaIfNeeded();
483
505
  this.db.exec(`
484
506
  CREATE TABLE IF NOT EXISTS kb_index_state (
485
507
  relative_path TEXT PRIMARY KEY,
@@ -506,17 +528,27 @@ export class IndexStateStore {
506
528
  relative_path TEXT NOT NULL,
507
529
  chunk_index INTEGER NOT NULL,
508
530
  content TEXT NOT NULL,
531
+ search_content TEXT NOT NULL,
509
532
  category TEXT NOT NULL,
510
533
  format TEXT NOT NULL,
511
534
  collection_name TEXT NOT NULL,
512
535
  token_count INTEGER NOT NULL,
513
536
  section_title TEXT,
537
+ title_path TEXT,
538
+ parent_id TEXT,
539
+ chunk_kind TEXT,
540
+ row_range TEXT,
541
+ start_char INTEGER,
542
+ end_char INTEGER,
514
543
  metadata_json TEXT,
515
544
  created_at INTEGER NOT NULL
516
545
  );
517
546
  CREATE INDEX IF NOT EXISTS idx_kb_chunks_path ON kb_chunks(relative_path);
518
547
  CREATE INDEX IF NOT EXISTS idx_kb_chunks_category ON kb_chunks(category);
519
548
  CREATE INDEX IF NOT EXISTS idx_kb_chunks_collection ON kb_chunks(collection_name);
549
+ CREATE INDEX IF NOT EXISTS idx_kb_chunks_parent ON kb_chunks(relative_path, parent_id);
550
+ CREATE INDEX IF NOT EXISTS idx_kb_chunks_kind ON kb_chunks(chunk_kind);
551
+ CREATE INDEX IF NOT EXISTS idx_kb_chunks_title_path ON kb_chunks(title_path);
520
552
 
521
553
  CREATE TABLE IF NOT EXISTS kb_parent_chunks (
522
554
  id TEXT PRIMARY KEY,
@@ -626,8 +658,8 @@ export class IndexStateStore {
626
658
  `);
627
659
  this.initFts();
628
660
  try {
629
- if (this.getMetadata('schema_version') !== '2')
630
- this.setMetadata('schema_version', '2');
661
+ if (this.getMetadata('schema_version') !== '3')
662
+ this.setMetadata('schema_version', '3');
631
663
  }
632
664
  catch {
633
665
  // 受限环境中已有索引库可能以只读方式挂载;运行时元数据不是必需项。
@@ -642,6 +674,8 @@ export class IndexStateStore {
642
674
  category,
643
675
  format,
644
676
  section_title,
677
+ title_path,
678
+ chunk_kind,
645
679
  content,
646
680
  tokenize = 'unicode61 remove_diacritics 2'
647
681
  );
@@ -660,8 +694,8 @@ export class IndexStateStore {
660
694
  if (Number(row.count ?? 0) > 0)
661
695
  return;
662
696
  this.db.prepare(`
663
- INSERT INTO kb_chunks_fts (id, relative_path, category, format, section_title, content)
664
- SELECT id, relative_path, category, format, COALESCE(section_title, ''), content FROM kb_chunks
697
+ INSERT INTO kb_chunks_fts (id, relative_path, category, format, section_title, title_path, chunk_kind, content)
698
+ SELECT id, relative_path, category, format, COALESCE(section_title, ''), COALESCE(title_path, ''), COALESCE(chunk_kind, ''), COALESCE(search_content, content) FROM kb_chunks
665
699
  `).run();
666
700
  }
667
701
  rowToMinHash(row) {
@@ -747,6 +781,32 @@ export class IndexStateStore {
747
781
  metadataString(value) {
748
782
  return typeof value === 'string' ? value : undefined;
749
783
  }
784
+ buildChunkSearchContent(relativePath, file, chunk, titlePath) {
785
+ const metadata = chunk.metadata;
786
+ const fields = [
787
+ `文件路径: ${relativePath}`,
788
+ `资料类型: ${file.category}/${file.format}`,
789
+ chunk.sectionTitle ? `章节标题: ${chunk.sectionTitle}` : '',
790
+ titlePath ? `标题路径: ${titlePath}` : '',
791
+ this.metadataString(metadata.chunkKind) ? `切片类型: ${this.metadataString(metadata.chunkKind)}` : '',
792
+ this.metadataString(metadata.rowRange) ? `表格行范围: ${this.metadataString(metadata.rowRange)}` : '',
793
+ chunk.text,
794
+ ];
795
+ return fields.filter(Boolean).join('\n');
796
+ }
797
+ searchableRowText(row) {
798
+ return [
799
+ row.relative_path,
800
+ row.category,
801
+ row.format,
802
+ row.section_title,
803
+ row.title_path,
804
+ row.chunk_kind,
805
+ row.row_range,
806
+ row.search_content,
807
+ row.content,
808
+ ].map(value => value == null ? '' : String(value)).join('\n');
809
+ }
750
810
  rowToChunk(row, score, scoreDetails) {
751
811
  return {
752
812
  rowid: Number(row.rowid ?? 0),
@@ -754,11 +814,18 @@ export class IndexStateStore {
754
814
  relativePath: String(row.relative_path),
755
815
  chunkIndex: Number(row.chunk_index),
756
816
  content: String(row.content),
817
+ searchContent: row.search_content == null ? undefined : String(row.search_content),
757
818
  category: String(row.category),
758
819
  format: String(row.format),
759
820
  collectionName: String(row.collection_name),
760
821
  tokenCount: Number(row.token_count),
761
822
  sectionTitle: row.section_title == null ? undefined : String(row.section_title),
823
+ titlePath: row.title_path == null ? undefined : String(row.title_path),
824
+ parentId: row.parent_id == null ? undefined : String(row.parent_id),
825
+ chunkKind: row.chunk_kind == null ? undefined : String(row.chunk_kind),
826
+ rowRange: row.row_range == null ? undefined : String(row.row_range),
827
+ startChar: row.start_char == null ? undefined : Number(row.start_char),
828
+ endChar: row.end_char == null ? undefined : Number(row.end_char),
762
829
  metadataJson: row.metadata_json == null ? undefined : String(row.metadata_json),
763
830
  createdAt: Number(row.created_at),
764
831
  score,
@@ -315,6 +315,7 @@ export class KnowledgeBaseManager {
315
315
  chunkIndex,
316
316
  parentId: parent.parentId,
317
317
  sectionTitle: parent.sectionTitle ?? item.sectionTitle,
318
+ titlePath: item.titlePath ?? this.parseMetadataString(parent.metadataJson, 'titlePath'),
318
319
  };
319
320
  }
320
321
  const parentChunks = item.parentId ? this.store.getChunksByParent(item.filePath, item.parentId, 6) : [];
@@ -328,6 +329,7 @@ export class KnowledgeBaseManager {
328
329
  content: chunks.map(chunk => chunk.content).join('\n\n---\n\n'),
329
330
  chunkIndex,
330
331
  parentId: item.parentId ?? this.parseMetadataString(chunks[0]?.metadataJson, 'parentId'),
332
+ titlePath: item.titlePath ?? chunks.map(chunk => this.parseMetadataString(chunk.metadataJson, 'titlePath')).find(Boolean),
331
333
  sectionTitle: item.sectionTitle ?? chunks.find(chunk => chunk.sectionTitle)?.sectionTitle,
332
334
  };
333
335
  }
@@ -341,7 +343,7 @@ export class KnowledgeBaseManager {
341
343
  rankedLists.push({ source: 'keyword', items: this.keywordSearchItems(rewritten, limit * 3), queryIndex });
342
344
  if (queryIndex < 3) {
343
345
  try {
344
- rankedLists.push({ source: 'vector', items: (await this.semanticSearch(rewritten, { ...options, limit: limit * 3 })).results, queryIndex });
346
+ rankedLists.push({ source: 'vector', items: (await this.semanticSearch(rewritten, { ...options, limit: limit * 6 })).results.slice(0, limit * 3), queryIndex });
345
347
  }
346
348
  catch { /* 向量搜索在混合搜索中是可选的 */ }
347
349
  }
@@ -664,15 +666,21 @@ export class KnowledgeBaseManager {
664
666
  const terms = query.toLowerCase().split(/[\s,,。;;::、]+/u).filter(Boolean);
665
667
  const phrase = query.toLowerCase().trim();
666
668
  return items.map(item => {
667
- const content = `${item.filePath}\n${item.sectionTitle ?? ''}\n${item.content}`.toLowerCase();
669
+ const content = `${item.filePath}\n${item.titlePath ?? ''}\n${item.sectionTitle ?? ''}\n${item.chunkKind ?? ''}\n${item.content}`.toLowerCase();
668
670
  let rerankBoost = 0;
669
671
  if (phrase && content.includes(phrase))
670
672
  rerankBoost += 120;
671
- for (const term of terms)
672
- if (term && content.includes(term))
673
+ const titleText = `${item.titlePath ?? ''}\n${item.sectionTitle ?? ''}`.toLowerCase();
674
+ for (const term of terms) {
675
+ if (!term)
676
+ continue;
677
+ if (content.includes(term))
673
678
  rerankBoost += 8;
674
- if (item.chunkKind === 'table' && /表|行|列|金额|数量|报价|评分/u.test(query))
675
- rerankBoost += 30;
679
+ if (titleText.includes(term))
680
+ rerankBoost += 18;
681
+ }
682
+ if (item.chunkKind === 'table' && /表|行|列|金额|数量|报价|评分|清单|明细|统计|数据/u.test(query))
683
+ rerankBoost += 40;
676
684
  if (item.chunkKind === 'metadata' && /图纸|图层|轴网|标注|块|实体|cad|dxf|step|iges|模型/u.test(query))
677
685
  rerankBoost += 60;
678
686
  if (item.chunkKind === 'data' && /json|xml|yaml|字段|配置|数据|路径|price|id|name/u.test(query))
@@ -695,7 +703,7 @@ export class KnowledgeBaseManager {
695
703
  const candidates = items.slice(0, 20);
696
704
  const resultsText = candidates.map((item, index) => {
697
705
  const contentPreview = item.content.slice(0, 300).replace(/[\n\r]+/g, ' ');
698
- return `[DOC_${index}] 路径: ${item.filePath} | 类型: ${item.chunkKind ?? 'text'}\n 内容: ${contentPreview}`;
706
+ return `[DOC_${index}] 路径: ${item.filePath} | 标题路径: ${item.titlePath ?? item.sectionTitle ?? ''} | 类型: ${item.chunkKind ?? 'text'}\n 内容: ${contentPreview}`;
699
707
  }).join('\n\n');
700
708
  const prompt = `你是一个文档相关性评估器。根据用户查询,为以下文档片段打分(1-10)。
701
709
  1=完全不相关,10=高度相关。
@@ -767,8 +775,11 @@ ${resultsText}
767
775
  parentId: this.metadataString(metadata.parentId),
768
776
  source,
769
777
  sectionTitle: result.sectionTitle,
770
- rowRange: this.metadataString(metadata.rowRange),
771
- chunkKind: this.metadataString(metadata.chunkKind),
778
+ titlePath: result.titlePath ?? this.metadataString(metadata.titlePath),
779
+ rowRange: result.rowRange ?? this.metadataString(metadata.rowRange),
780
+ chunkKind: result.chunkKind ?? this.metadataString(metadata.chunkKind),
781
+ startChar: result.startChar,
782
+ endChar: result.endChar,
772
783
  scoreDetails: result.scoreDetails,
773
784
  facets: this.metadataFacets(metadata),
774
785
  };
@@ -800,8 +811,11 @@ ${resultsText}
800
811
  if (item.score > (existing.scoreDetails?.keywordScore ?? existing.scoreDetails?.vectorScore ?? 0)) {
801
812
  existing.content = item.content;
802
813
  existing.sectionTitle = item.sectionTitle ?? existing.sectionTitle;
814
+ existing.titlePath = item.titlePath ?? existing.titlePath;
803
815
  existing.chunkIndex = item.chunkIndex ?? existing.chunkIndex;
804
816
  existing.parentId = item.parentId ?? existing.parentId;
817
+ existing.chunkKind = item.chunkKind ?? existing.chunkKind;
818
+ existing.rowRange = item.rowRange ?? existing.rowRange;
805
819
  }
806
820
  });
807
821
  }
@@ -858,7 +872,7 @@ ${resultsText}
858
872
  return typeof value === 'string' ? value : undefined;
859
873
  }
860
874
  metadataFacets(metadata) {
861
- const keys = ['sheetNames', 'columnNames', 'rowCount', 'columnCount', 'dataPaths', 'layerNames', 'blockNames', 'entityTypes', 'productNames', 'materialNames', 'ocrRecommended', 'ocrReason'];
875
+ const keys = ['titlePath', 'sectionTitle', 'chunkKind', 'rowRange', 'sheetNames', 'columnNames', 'rowCount', 'columnCount', 'dataPaths', 'layerNames', 'blockNames', 'entityTypes', 'productNames', 'materialNames', 'ocrRecommended', 'ocrReason'];
862
876
  const facets = {};
863
877
  for (const key of keys) {
864
878
  const value = metadata[key];
@@ -60,8 +60,10 @@ export declare class ContentExtractor {
60
60
  private extractPdf;
61
61
  private extractScannedPdfOcr;
62
62
  private extractPdfText;
63
+ private normalizedTextLength;
63
64
  private toPdfTextItem;
64
65
  private layoutPdfTextItems;
66
+ private joinPdfRowText;
65
67
  private groupPdfItemsIntoRows;
66
68
  private detectPdfColumnSplit;
67
69
  private rowsToPdfMarkdownWithTables;
@@ -1173,6 +1173,7 @@ process.stdout.write(JSON.stringify({ pageCount: doc.numPages, pageLimit, text:
1173
1173
  }
1174
1174
  }
1175
1175
  async extractPdfText(buffer) {
1176
+ let pdfjsText = '';
1176
1177
  // 第一层:pdfjs-dist 文本提取(处理压缩内容流、CJK 字体、现代 PDF)
1177
1178
  try {
1178
1179
  const mod = await resolveAndImport('pdfjs-dist/legacy/build/pdf.mjs');
@@ -1182,7 +1183,7 @@ process.stdout.write(JSON.stringify({ pageCount: doc.numPages, pageLimit, text:
1182
1183
  const pageLimit = doc.numPages;
1183
1184
  for (let i = 1; i <= pageLimit; i++) {
1184
1185
  const page = await doc.getPage(i);
1185
- const content = await page.getTextContent();
1186
+ const content = await page.getTextContent({ normalizeWhitespace: false, disableCombineTextItems: true });
1186
1187
  const items = content.items
1187
1188
  .map((item) => this.toPdfTextItem(item))
1188
1189
  .filter((item) => !!item && item.str.trim().length > 0);
@@ -1191,28 +1192,30 @@ process.stdout.write(JSON.stringify({ pageCount: doc.numPages, pageLimit, text:
1191
1192
  pages.push(pageText.trim());
1192
1193
  }
1193
1194
  await doc.destroy();
1194
- if (pages.length > 0) {
1195
- const combined = pages.join('\n\n');
1196
- if (combined.trim())
1197
- return combined;
1198
- }
1195
+ pdfjsText = pages.join('\n\n').trim();
1199
1196
  }
1200
1197
  catch (e) {
1201
1198
  if (process.env.KB_DEBUG === '1')
1202
1199
  console.warn('[kb] pdfjs-dist extraction failed:', e.message);
1203
1200
  }
1204
- // 第二层:pdf-parse(兼容旧版 PDF
1201
+ // 第二层:pdf-parse(兼容旧版 PDF),与 pdfjs 结果互补,避免单一解析器漏字
1205
1202
  try {
1206
1203
  const mod = await resolveAndImport('pdf-parse');
1207
1204
  const pdfParse = mod.default;
1208
1205
  if (pdfParse) {
1209
1206
  const result = await pdfParse(buffer);
1210
- if (result.text.trim())
1211
- return result.text;
1207
+ const parseText = result.text.trim();
1208
+ if (pdfjsText && parseText && this.normalizedTextLength(parseText) > this.normalizedTextLength(pdfjsText) * 1.08)
1209
+ return [pdfjsText, '## PDF 备用解析文本', parseText].join('\n\n');
1210
+ if (pdfjsText)
1211
+ return pdfjsText;
1212
+ if (parseText)
1213
+ return parseText;
1212
1214
  }
1213
1215
  }
1214
1216
  catch {
1215
- // 降级到下方纯正则提取
1217
+ if (pdfjsText)
1218
+ return pdfjsText;
1216
1219
  }
1217
1220
  // 第三层:raw regex 回退(未压缩的古老 PDF)
1218
1221
  const raw = buffer.toString('latin1');
@@ -1229,6 +1232,9 @@ process.stdout.write(JSON.stringify({ pageCount: doc.numPages, pageLimit, text:
1229
1232
  .join('')
1230
1233
  .trim();
1231
1234
  }
1235
+ normalizedTextLength(value) {
1236
+ return value.replace(/\s+/gu, '').length;
1237
+ }
1232
1238
  toPdfTextItem(item) {
1233
1239
  if (!item || typeof item !== 'object' || !('str' in item))
1234
1240
  return undefined;
@@ -1257,6 +1263,25 @@ process.stdout.write(JSON.stringify({ pageCount: doc.numPages, pageLimit, text:
1257
1263
  const markdown = this.rowsToPdfMarkdownWithTables(orderedRows);
1258
1264
  return [`## PDF 第 ${pageNumber} 页`, markdown].join('\n\n');
1259
1265
  }
1266
+ joinPdfRowText(items) {
1267
+ let output = '';
1268
+ let previous;
1269
+ for (const item of items) {
1270
+ const text = item.str.trim();
1271
+ if (!text)
1272
+ continue;
1273
+ if (!previous) {
1274
+ output += text;
1275
+ }
1276
+ else {
1277
+ const gap = item.x - (previous.x + previous.width);
1278
+ const cjkJoin = /[\p{Script=Han}((《“‘]$/u.test(output) || /^[\p{Script=Han}))》”’、,。;:!?]/u.test(text);
1279
+ output += gap > Math.max(3, previous.height * 0.35) && !cjkJoin ? ` ${text}` : text;
1280
+ }
1281
+ previous = item;
1282
+ }
1283
+ return output.replace(/\s+/gu, ' ').trim();
1284
+ }
1260
1285
  groupPdfItemsIntoRows(items) {
1261
1286
  const sorted = [...items].sort((a, b) => b.y - a.y || a.x - b.x);
1262
1287
  const rows = [];
@@ -1270,7 +1295,7 @@ process.stdout.write(JSON.stringify({ pageCount: doc.numPages, pageLimit, text:
1270
1295
  return rows.map(row => {
1271
1296
  const rowItems = row.items.sort((a, b) => a.x - b.x);
1272
1297
  return {
1273
- text: rowItems.map(item => item.str.trim()).filter(Boolean).join(' ').replace(/\s+/gu, ' '),
1298
+ text: this.joinPdfRowText(rowItems),
1274
1299
  x: Math.min(...rowItems.map(item => item.x)),
1275
1300
  y: row.y,
1276
1301
  height: Math.max(...rowItems.map(item => item.height || 0)),
@@ -15,8 +15,11 @@ export interface FederatedSearchItem {
15
15
  parentId?: string;
16
16
  source?: 'keyword' | 'vector' | 'hybrid';
17
17
  sectionTitle?: string;
18
+ titlePath?: string;
18
19
  rowRange?: string;
19
20
  chunkKind?: string;
21
+ startChar?: number;
22
+ endChar?: number;
20
23
  scoreDetails?: {
21
24
  keywordScore?: number;
22
25
  bm25Score?: number;
@@ -85,8 +85,11 @@ export class FederationSearch {
85
85
  parentId: typeof result.document.metadata.parent_id === 'string' ? result.document.metadata.parent_id : undefined,
86
86
  source: 'vector',
87
87
  sectionTitle: typeof result.document.metadata.section_title === 'string' ? result.document.metadata.section_title : undefined,
88
+ titlePath: typeof result.document.metadata.title_path === 'string' ? result.document.metadata.title_path : undefined,
88
89
  rowRange: typeof result.document.metadata.row_range === 'string' ? result.document.metadata.row_range : undefined,
89
90
  chunkKind: typeof result.document.metadata.chunk_kind === 'string' ? result.document.metadata.chunk_kind : undefined,
91
+ startChar: typeof result.document.metadata.start_char === 'number' ? result.document.metadata.start_char : undefined,
92
+ endChar: typeof result.document.metadata.end_char === 'number' ? result.document.metadata.end_char : undefined,
90
93
  scoreDetails: { vectorScore: result.score },
91
94
  };
92
95
  }
@@ -17,6 +17,7 @@ export declare class HNSWVectorStore implements VectorStoreInterface {
17
17
  flush(): Promise<void>;
18
18
  needsRebuild(): boolean;
19
19
  search(query: VectorSearchQuery): Promise<VectorSearchResult[]>;
20
+ private matchesWhere;
20
21
  private persist;
21
22
  private toStoredDocument;
22
23
  private loadDocuments;
@@ -81,17 +81,22 @@ export class HNSWVectorStore {
81
81
  }
82
82
  async search(query) {
83
83
  await this.ensureCollection();
84
- const result = this.index.searchKnn(query.queryEmbedding, query.topK);
84
+ const hasFilter = !!query.where && Object.keys(query.where).length > 0;
85
+ const candidateK = hasFilter ? Math.min(this.documents.size, Math.max(query.topK * 10, query.topK + 50)) : query.topK;
86
+ const result = this.index.searchKnn(query.queryEmbedding, candidateK);
85
87
  return result.neighbors.flatMap((rowid, index) => {
86
88
  const document = this.documents.get(rowid);
87
- if (!document)
88
- return [];
89
- if (typeof query.where?.file_path === 'string' && document.metadata.file_path !== query.where.file_path)
89
+ if (!document || !this.matchesWhere(document, query.where))
90
90
  return [];
91
91
  const { embedding: _embedding, ...stored } = document;
92
92
  const distance = result.distances[index] ?? 0;
93
93
  return [{ collection: this.collectionName, document: stored, score: 1 / (1 + distance) }];
94
- });
94
+ }).slice(0, query.topK);
95
+ }
96
+ matchesWhere(document, where) {
97
+ if (!where)
98
+ return true;
99
+ return Object.entries(where).every(([key, value]) => document.metadata[key] === value);
95
100
  }
96
101
  persist() {
97
102
  this.index.writeIndexSync(this.indexPath);
@@ -29,6 +29,7 @@ export declare class VectorIndexer {
29
29
  private embedDocuments;
30
30
  private isValidEmbeddings;
31
31
  private groupByCollection;
32
+ private embeddingText;
32
33
  private toVectorDocument;
33
34
  private parseMetadata;
34
35
  private metadataString;
@@ -29,7 +29,7 @@ export class VectorIndexer {
29
29
  let processedChunks = 0;
30
30
  for (let offset = 0; offset < collectionChunks.length; offset += batchSize) {
31
31
  const batchChunks = collectionChunks.slice(offset, offset + batchSize);
32
- const texts = batchChunks.map(chunk => chunk.content);
32
+ const texts = batchChunks.map(chunk => this.embeddingText(chunk));
33
33
  const embeddings = await this.embedDocuments(texts);
34
34
  const documents = batchChunks.map((chunk, index) => this.toVectorDocument(chunk, embeddings[index] ?? []));
35
35
  await store.upsert(documents, { persist: options.persistEachBatch === true });
@@ -72,6 +72,17 @@ export class VectorIndexer {
72
72
  }
73
73
  return grouped;
74
74
  }
75
+ embeddingText(chunk) {
76
+ return chunk.searchContent ?? [
77
+ `文件路径: ${chunk.relativePath}`,
78
+ `资料类型: ${chunk.category}/${chunk.format}`,
79
+ chunk.titlePath ? `标题路径: ${chunk.titlePath}` : '',
80
+ chunk.sectionTitle ? `章节标题: ${chunk.sectionTitle}` : '',
81
+ chunk.chunkKind ? `切片类型: ${chunk.chunkKind}` : '',
82
+ chunk.rowRange ? `表格行范围: ${chunk.rowRange}` : '',
83
+ chunk.content,
84
+ ].filter(Boolean).join('\n');
85
+ }
75
86
  toVectorDocument(chunk, embedding) {
76
87
  const chunkMetadata = this.parseMetadata(chunk.metadataJson);
77
88
  return {
@@ -86,11 +97,14 @@ export class VectorIndexer {
86
97
  format: chunk.format,
87
98
  token_count: chunk.tokenCount,
88
99
  section_title: chunk.sectionTitle ?? null,
89
- parent_id: this.metadataString(chunkMetadata.parentId),
100
+ title_path: chunk.titlePath ?? this.metadataString(chunkMetadata.titlePath),
101
+ parent_id: chunk.parentId ?? this.metadataString(chunkMetadata.parentId),
90
102
  parent_index: this.metadataNumber(chunkMetadata.parentIndex),
91
103
  child_index: this.metadataNumber(chunkMetadata.childIndex),
92
- chunk_kind: this.metadataString(chunkMetadata.chunkKind),
93
- row_range: this.metadataString(chunkMetadata.rowRange),
104
+ chunk_kind: chunk.chunkKind ?? this.metadataString(chunkMetadata.chunkKind),
105
+ row_range: chunk.rowRange ?? this.metadataString(chunkMetadata.rowRange),
106
+ start_char: chunk.startChar ?? this.metadataNumber(chunkMetadata.startChar),
107
+ end_char: chunk.endChar ?? this.metadataNumber(chunkMetadata.endChar),
94
108
  split_strategy: this.metadataString(chunkMetadata.splitStrategy),
95
109
  },
96
110
  };
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@customize-agent/knowledge",
3
- "version": "4.0.9",
3
+ "version": "4.0.10",
4
4
  "description": "Local knowledge base infrastructure for customize-agent",
5
5
  "type": "module",
6
6
  "main": "./dist/index.js",