@wei840222/qmd 2026.8.28 → 2026.9.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/README.md +84 -2
- package/dist/cli/build-info.json +2 -2
- package/dist/cli/qmd.js +96 -18
- package/dist/collections.js +9 -4
- package/dist/db.d.ts +16 -29
- package/dist/db.js +63 -40
- package/dist/index.d.ts +10 -0
- package/dist/index.js +14 -2
- package/dist/llm.d.ts +7 -1
- package/dist/llm.js +23 -4
- package/dist/mcp/server.js +70 -6
- package/dist/metadata-filter.d.ts +74 -0
- package/dist/metadata-filter.js +279 -0
- package/dist/metadata-store.d.ts +45 -0
- package/dist/metadata-store.js +173 -0
- package/dist/metadata.d.ts +61 -0
- package/dist/metadata.js +215 -0
- package/dist/search/zh-dict.txt +3 -0
- package/dist/store.d.ts +23 -16
- package/dist/store.js +354 -166
- package/package.json +3 -5
- package/scripts/sync-zh-dict.mjs +4 -1
- package/skills/qmd/SKILL.md +11 -2
- package/skills/qmd/references/query-syntax.md +1 -8
- package/skills/release/SKILL.md +0 -141
- package/skills/release/scripts/install-hooks.sh +0 -38
- package/skills/release/scripts/release-context.sh +0 -129
package/dist/store.js
CHANGED
|
@@ -25,6 +25,9 @@ import { analyzeCjkSync, containsCjk } from "./search/cjk-analyzer.js";
|
|
|
25
25
|
import { ExpansionPolicyError, parseExpansionDirective, resolveExpansionPolicy, } from "./search/query-expansion.js";
|
|
26
26
|
import { LlamaCpp, getDefaultLlamaCpp, formatQueryForEmbedding, formatDocForEmbedding, withLLMSessionForLlm, DEFAULT_EMBED_MODEL_URI, DEFAULT_RERANK_MODEL_URI, DEFAULT_GENERATE_MODEL_URI, } from "./llm.js";
|
|
27
27
|
const readOnlyDatabases = new WeakSet();
|
|
28
|
+
import { METADATA_EXTRACTION_VERSION } from "./metadata.js";
|
|
29
|
+
import { compileMetadataFilter } from "./metadata-filter.js";
|
|
30
|
+
import { initializeMetadataSchema, syncDocumentMetadata, countDocumentsPendingMetadata, getMetadataByFilepath, parseMetadataJson, } from "./metadata-store.js";
|
|
28
31
|
// =============================================================================
|
|
29
32
|
// Configuration
|
|
30
33
|
// =============================================================================
|
|
@@ -925,18 +928,15 @@ export function normalizeCjkForFTS(text) {
|
|
|
925
928
|
return text.replace(CJK_RUN_PATTERN, run => ` ${Array.from(run).join(' ')} `);
|
|
926
929
|
}
|
|
927
930
|
function sanitizeFTS5Phrase(phrase) {
|
|
928
|
-
//
|
|
929
|
-
//
|
|
930
|
-
//
|
|
931
|
+
// A quoted phrase is matched against tokens the porter unicode61 tokenizer
|
|
932
|
+
// produced, and that tokenizer splits document text on every separator.
|
|
933
|
+
// Deleting the separators here instead would collapse "1.0.21" to "1021" and
|
|
934
|
+
// "PIO-1384" to "pio1384", tokens no document holds, so the query returns
|
|
935
|
+
// nothing with no error (#757 for dots, #916 for the rest). Split on the same
|
|
936
|
+
// separators the tokenizer does and emit the parts as adjacent phrase terms.
|
|
931
937
|
return normalizeCjkForFTS(phrase)
|
|
932
938
|
.split(/\s+/)
|
|
933
|
-
.flatMap(t =>
|
|
934
|
-
if (isDottedToken(t)) {
|
|
935
|
-
return t.split('.').map(p => sanitizeFTS5Term(p)).filter(p => p);
|
|
936
|
-
}
|
|
937
|
-
const sanitized = sanitizeFTS5Term(t);
|
|
938
|
-
return sanitized ? [sanitized] : [];
|
|
939
|
-
})
|
|
939
|
+
.flatMap(t => splitFTS5CompoundTerm(t))
|
|
940
940
|
.join(' ');
|
|
941
941
|
}
|
|
942
942
|
function getUserVersion(db) {
|
|
@@ -1208,7 +1208,7 @@ function rebuildFTSForCjkNormalization(db) {
|
|
|
1208
1208
|
}
|
|
1209
1209
|
});
|
|
1210
1210
|
// Pull bounded batches and finalize each SELECT before starting the insert
|
|
1211
|
-
// transaction. Holding
|
|
1211
|
+
// transaction. Holding an SQLite iterator open while beginning a
|
|
1212
1212
|
// transaction on the same connection raises "database is busy".
|
|
1213
1213
|
const selectBatch = db.prepare(`
|
|
1214
1214
|
SELECT d.id, d.collection, d.path, d.title, content.doc as body
|
|
@@ -1402,6 +1402,9 @@ function initializeDatabase(db) {
|
|
|
1402
1402
|
`);
|
|
1403
1403
|
ensureEmbeddingIdentitySchema(db);
|
|
1404
1404
|
ensureContentVectorsStatusIndex(db);
|
|
1405
|
+
// Document metadata — extraction state plus normalized value rows for
|
|
1406
|
+
// metadata filtering. Keyed by document identity, not content hash.
|
|
1407
|
+
initializeMetadataSchema(db);
|
|
1405
1408
|
// Store collections — makes the DB self-contained (no external config needed)
|
|
1406
1409
|
db.exec(`
|
|
1407
1410
|
CREATE TABLE IF NOT EXISTS store_collections (
|
|
@@ -1532,6 +1535,12 @@ export function setStoreGlobalContext(db, value) {
|
|
|
1532
1535
|
db.prepare(`INSERT INTO store_config (key, value) VALUES ('global_context', ?) ON CONFLICT(key) DO UPDATE SET value = excluded.value`).run(value);
|
|
1533
1536
|
}
|
|
1534
1537
|
}
|
|
1538
|
+
function writeConfigSyncDiagnostic(db, diagnostic) {
|
|
1539
|
+
db.prepare(`
|
|
1540
|
+
INSERT INTO store_config (key, value) VALUES ('config_sync_diagnostic', ?)
|
|
1541
|
+
ON CONFLICT(key) DO UPDATE SET value = excluded.value
|
|
1542
|
+
`).run(JSON.stringify(diagnostic));
|
|
1543
|
+
}
|
|
1535
1544
|
function storeCollectionMatchesConfig(row, collection) {
|
|
1536
1545
|
return row.path === collection.path
|
|
1537
1546
|
&& row.pattern === (collection.pattern || '**/*.md')
|
|
@@ -1556,12 +1565,14 @@ export function syncConfigToDb(db, config) {
|
|
|
1556
1565
|
});
|
|
1557
1566
|
const currentGlobalContext = getStoreGlobalContext(db);
|
|
1558
1567
|
if (allMatch && currentGlobalContext === config.global_context) {
|
|
1559
|
-
|
|
1568
|
+
const diagnostic = {
|
|
1560
1569
|
configHashChanged: false,
|
|
1561
1570
|
reconciled: false,
|
|
1562
1571
|
collections: { added: [], updated: [], removed: [] },
|
|
1563
1572
|
globalContextUpdated: false,
|
|
1564
1573
|
};
|
|
1574
|
+
writeConfigSyncDiagnostic(db, diagnostic);
|
|
1575
|
+
return diagnostic;
|
|
1565
1576
|
}
|
|
1566
1577
|
}
|
|
1567
1578
|
}
|
|
@@ -1722,7 +1733,7 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1722
1733
|
return !parts.some(part => part.startsWith("."));
|
|
1723
1734
|
});
|
|
1724
1735
|
const total = files.length;
|
|
1725
|
-
let indexed = 0, updated = 0, unchanged = 0, processed = 0;
|
|
1736
|
+
let indexed = 0, updated = 0, unchanged = 0, processed = 0, metadataErrors = 0;
|
|
1726
1737
|
const skippedFiles = [];
|
|
1727
1738
|
const seenPaths = new Set();
|
|
1728
1739
|
// Literal paths of every file in this scan. Passed to the legacy-path
|
|
@@ -1764,8 +1775,12 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1764
1775
|
const hash = await hashContent(content);
|
|
1765
1776
|
const title = extractTitle(content, relativeFile);
|
|
1766
1777
|
const existing = findOrMigrateLegacyDocument(db, collectionName, path, livePaths);
|
|
1778
|
+
let documentId;
|
|
1779
|
+
let contentChanged = true;
|
|
1767
1780
|
if (existing) {
|
|
1781
|
+
documentId = existing.id;
|
|
1768
1782
|
if (existing.hash === hash) {
|
|
1783
|
+
contentChanged = false;
|
|
1769
1784
|
if (existing.title !== title) {
|
|
1770
1785
|
updateDocumentTitle(db, existing.id, title, now);
|
|
1771
1786
|
updated++;
|
|
@@ -1783,8 +1798,12 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1783
1798
|
else {
|
|
1784
1799
|
indexed++;
|
|
1785
1800
|
const stat = statSync(filepath);
|
|
1786
|
-
insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1801
|
+
documentId = insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1787
1802
|
}
|
|
1803
|
+
// Unchanged content still backfills missing or stale extraction state.
|
|
1804
|
+
const extraction = syncDocumentMetadata(db, documentId, content, path, contentChanged ? undefined : { onlyIfStale: true });
|
|
1805
|
+
if (extraction?.error)
|
|
1806
|
+
metadataErrors++;
|
|
1788
1807
|
processed++;
|
|
1789
1808
|
options?.onProgress?.({ file: relativeFile, current: processed, total });
|
|
1790
1809
|
}
|
|
@@ -1798,7 +1817,7 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1798
1817
|
}
|
|
1799
1818
|
}
|
|
1800
1819
|
const orphanedCleaned = cleanupOrphanedContent(db);
|
|
1801
|
-
return { indexed, updated, unchanged, removed, orphanedCleaned, skipped: skippedFiles.length, skippedFiles };
|
|
1820
|
+
return { indexed, updated, unchanged, removed, orphanedCleaned, skipped: skippedFiles.length, skippedFiles, metadataErrors };
|
|
1802
1821
|
}
|
|
1803
1822
|
function validatePositiveIntegerOption(name, value, fallback) {
|
|
1804
1823
|
if (value === undefined)
|
|
@@ -2456,7 +2475,7 @@ export async function generateEmbeddings(store, options) {
|
|
|
2456
2475
|
pos: chunk.pos,
|
|
2457
2476
|
tokens: chunk.tokenUpperBound,
|
|
2458
2477
|
}))
|
|
2459
|
-
: await
|
|
2478
|
+
: await chunkDocumentByTokensWithLlm((llm ?? getLlm(store) ?? getDefaultLlamaCpp()), doc.body, undefined, undefined, undefined, doc.path, options?.chunkStrategy, session.signal);
|
|
2460
2479
|
if (embeddingLease) {
|
|
2461
2480
|
pruneEmbeddingRowsOutsideLayout(db, doc.hash, model, fingerprint, chunks.length, embeddingLease);
|
|
2462
2481
|
}
|
|
@@ -2633,11 +2652,11 @@ export function createStore(dbPath, options = {}) {
|
|
|
2633
2652
|
const commitPendingDocumentInsert = (collectionName, path, title, hash, createdAt, modifiedAt) => {
|
|
2634
2653
|
const pending = pendingContent.get(hash);
|
|
2635
2654
|
if (!pending) {
|
|
2636
|
-
insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt);
|
|
2637
|
-
return;
|
|
2655
|
+
return insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt);
|
|
2638
2656
|
}
|
|
2639
|
-
insertDocumentWithContent(db, hash, pending.content, pending.createdAt, collectionName, path, title, createdAt, modifiedAt);
|
|
2657
|
+
const docId = insertDocumentWithContent(db, hash, pending.content, pending.createdAt, collectionName, path, title, createdAt, modifiedAt);
|
|
2640
2658
|
pendingContent.delete(hash);
|
|
2659
|
+
return docId;
|
|
2641
2660
|
};
|
|
2642
2661
|
const commitPendingDocumentUpdate = (documentId, title, hash, modifiedAt) => {
|
|
2643
2662
|
const pending = pendingContent.get(hash);
|
|
@@ -2685,9 +2704,9 @@ export function createStore(dbPath, options = {}) {
|
|
|
2685
2704
|
resolveVirtualPath: (virtualPath) => resolveVirtualPath(db, virtualPath),
|
|
2686
2705
|
toVirtualPath: (absolutePath) => toVirtualPath(db, absolutePath),
|
|
2687
2706
|
// Search
|
|
2688
|
-
searchCharFTS: (query, limit, collectionName) => searchCharFTS(db, query, limit, collectionName),
|
|
2689
|
-
searchFTS: (query, limit, collectionName) => searchFTS(db, query, limit, collectionName),
|
|
2690
|
-
searchVec: (query, model, limit, collectionFilter, session, precomputedEmbedding) => searchVec(db, query, model, limit, collectionFilter, session, precomputedEmbedding, store.embeddingProvider, store.authorizeRemoteRequest, store.llm),
|
|
2707
|
+
searchCharFTS: (query, limit, collectionName, filter) => searchCharFTS(db, query, limit, collectionName, filter),
|
|
2708
|
+
searchFTS: (query, limit, collectionName, filter) => searchFTS(db, query, limit, collectionName, filter),
|
|
2709
|
+
searchVec: (query, model, limit, collectionFilter, session, precomputedEmbedding, filter) => searchVec(db, query, model, limit, collectionFilter, session, precomputedEmbedding, store.embeddingProvider, store.authorizeRemoteRequest, store.llm, filter),
|
|
2691
2710
|
// Query expansion & reranking
|
|
2692
2711
|
expandQuery: (query, model, expansionContext, options) => expandQuery(query, model ?? store.localLlm?.generateModelName ?? store.llm?.generateModelName ?? DEFAULT_QUERY_MODEL, db, expansionContext, store.llm, options),
|
|
2693
2712
|
invalidateExpansionCache: (query, expansionContext, options) => deleteExpansionCacheEntry(db, query, store.localLlm?.generateModelName ?? store.llm?.generateModelName ?? DEFAULT_QUERY_MODEL, expansionContext, options),
|
|
@@ -2716,6 +2735,30 @@ export function createStore(dbPath, options = {}) {
|
|
|
2716
2735
|
getActiveDocumentPaths: (collectionName) => getActiveDocumentPaths(db, collectionName),
|
|
2717
2736
|
// Vector/embedding operations
|
|
2718
2737
|
getHashesForEmbedding: () => getHashesForEmbedding(db),
|
|
2738
|
+
insertEmbedding: (hash, seq, pos, embedding, model, embeddedAt, totalChunks, fingerprint, lease) => {
|
|
2739
|
+
const state = db.prepare(`
|
|
2740
|
+
SELECT status, fingerprint, model, generation, lease_owner, lease_expires_at
|
|
2741
|
+
FROM embedding_index_state
|
|
2742
|
+
WHERE singleton = 1
|
|
2743
|
+
`).get();
|
|
2744
|
+
if (!state && !lease) {
|
|
2745
|
+
withLazyContentVectorMigration(db, () => {
|
|
2746
|
+
db.prepare(`
|
|
2747
|
+
INSERT OR REPLACE INTO content_vectors (hash, seq, pos, model, embed_fingerprint, total_chunks, embedded_at)
|
|
2748
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
2749
|
+
`).run(hash, seq, pos, model, fingerprint ?? "", totalChunks ?? 1, embeddedAt);
|
|
2750
|
+
const hasCollectionCol = db.prepare(`SELECT sql FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get();
|
|
2751
|
+
if (hasCollectionCol?.sql?.includes("collection")) {
|
|
2752
|
+
db.prepare(`INSERT OR REPLACE INTO vectors_vec (hash_seq, collection, embedding) VALUES (?, ?, ?)`).run(`${hash}_${seq}`, "", embedding);
|
|
2753
|
+
}
|
|
2754
|
+
else {
|
|
2755
|
+
db.prepare(`INSERT OR REPLACE INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_${seq}`, embedding);
|
|
2756
|
+
}
|
|
2757
|
+
});
|
|
2758
|
+
return;
|
|
2759
|
+
}
|
|
2760
|
+
insertEmbedding(db, hash, seq, pos, embedding, model, embeddedAt, totalChunks, fingerprint, lease);
|
|
2761
|
+
},
|
|
2719
2762
|
};
|
|
2720
2763
|
return store;
|
|
2721
2764
|
}
|
|
@@ -2883,7 +2926,7 @@ export async function maybeAdoptLegacyEmbeddingFingerprint(store, model = DEFAUL
|
|
|
2883
2926
|
const title = extractTitle(sample.body, sample.path);
|
|
2884
2927
|
const llm = getLlm(store);
|
|
2885
2928
|
return await withLLMSessionForLlm(llm, async (session) => {
|
|
2886
|
-
const chunks = await
|
|
2929
|
+
const chunks = await chunkDocumentByTokensWithLlm(llm, sample.body, undefined, undefined, undefined, sample.path, undefined, session.signal);
|
|
2887
2930
|
const chunk = chunks[sample.seq];
|
|
2888
2931
|
if (!chunk) {
|
|
2889
2932
|
return { checked: true, adopted: 0, reason: `sample chunk ${expectedHashSeq} no longer exists` };
|
|
@@ -2905,7 +2948,7 @@ export async function maybeAdoptLegacyEmbeddingFingerprint(store, model = DEFAUL
|
|
|
2905
2948
|
return { checked: true, adopted: 0, reason: `legacy sample differs from current fingerprint (nearest ${nearest.hash_seq}, distance ${nearest.distance.toFixed(6)})` };
|
|
2906
2949
|
}
|
|
2907
2950
|
const update = withLazyContentVectorMigration(db, () => db.prepare(`UPDATE content_vectors SET embed_fingerprint = ? WHERE model = ? AND embed_fingerprint = ''`).run(fingerprint, model));
|
|
2908
|
-
return { checked: true, adopted: update.changes, reason: `sample ${expectedHashSeq} matched current fingerprint at distance ${nearest.distance.toFixed(6)}` };
|
|
2951
|
+
return { checked: true, adopted: Number(update.changes), reason: `sample ${expectedHashSeq} matched current fingerprint at distance ${nearest.distance.toFixed(6)}` };
|
|
2909
2952
|
});
|
|
2910
2953
|
}
|
|
2911
2954
|
export function getIndexHealth(db, model = DEFAULT_EMBED_MODEL) {
|
|
@@ -2952,7 +2995,7 @@ export function clearCache(db) {
|
|
|
2952
2995
|
*/
|
|
2953
2996
|
export function deleteLLMCache(db) {
|
|
2954
2997
|
const result = db.prepare(`DELETE FROM llm_cache`).run();
|
|
2955
|
-
return result.changes;
|
|
2998
|
+
return Number(result.changes);
|
|
2956
2999
|
}
|
|
2957
3000
|
/**
|
|
2958
3001
|
* Remove inactive document records (active = 0).
|
|
@@ -2976,7 +3019,7 @@ export function cleanupOrphanedContent(db) {
|
|
|
2976
3019
|
DELETE FROM content
|
|
2977
3020
|
WHERE hash NOT IN (SELECT DISTINCT hash FROM documents)
|
|
2978
3021
|
`).run();
|
|
2979
|
-
return result.changes;
|
|
3022
|
+
return Number(result.changes);
|
|
2980
3023
|
}
|
|
2981
3024
|
/**
|
|
2982
3025
|
* Count content hashes that would be unreferenced after inactive documents
|
|
@@ -3189,9 +3232,10 @@ function rebuildDocumentFTS(db, documentId) {
|
|
|
3189
3232
|
}
|
|
3190
3233
|
/**
|
|
3191
3234
|
* Insert a new document into the documents table.
|
|
3235
|
+
* Returns the document's id so callers can attach document-scoped state.
|
|
3192
3236
|
*/
|
|
3193
3237
|
export function insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt) {
|
|
3194
|
-
runCjkSynchronizedMutation(db, () => {
|
|
3238
|
+
return runCjkSynchronizedMutation(db, () => {
|
|
3195
3239
|
db.prepare(`
|
|
3196
3240
|
INSERT INTO documents (collection, path, title, hash, created_at, modified_at, active)
|
|
3197
3241
|
VALUES (?, ?, ?, ?, ?, ?, 1)
|
|
@@ -3202,15 +3246,17 @@ export function insertDocument(db, collectionName, path, title, hash, createdAt,
|
|
|
3202
3246
|
active = 1
|
|
3203
3247
|
`).run(collectionName, path, title, hash, createdAt, modifiedAt);
|
|
3204
3248
|
const row = db.prepare(`SELECT id FROM documents WHERE collection = ? AND path = ?`).get(collectionName, path);
|
|
3205
|
-
if (row)
|
|
3206
|
-
|
|
3249
|
+
if (!row)
|
|
3250
|
+
throw new Error(`Document row missing after insert: ${collectionName}/${path}`);
|
|
3251
|
+
rebuildDocumentFTS(db, row.id);
|
|
3252
|
+
return row.id;
|
|
3207
3253
|
});
|
|
3208
3254
|
}
|
|
3209
3255
|
/** Insert immutable content and its document row in the same lexical transaction. */
|
|
3210
3256
|
export function insertDocumentWithContent(db, hash, content, contentCreatedAt, collectionName, path, title, documentCreatedAt, modifiedAt) {
|
|
3211
|
-
runCjkSynchronizedMutation(db, () => {
|
|
3257
|
+
return runCjkSynchronizedMutation(db, () => {
|
|
3212
3258
|
insertContent(db, hash, content, contentCreatedAt);
|
|
3213
|
-
insertDocument(db, collectionName, path, title, hash, documentCreatedAt, modifiedAt);
|
|
3259
|
+
return insertDocument(db, collectionName, path, title, hash, documentCreatedAt, modifiedAt);
|
|
3214
3260
|
});
|
|
3215
3261
|
}
|
|
3216
3262
|
/**
|
|
@@ -3417,8 +3463,7 @@ function stripUnpairedSurrogates(text) {
|
|
|
3417
3463
|
* When filepath and chunkStrategy are provided, uses AST-aware break points
|
|
3418
3464
|
* for supported code files.
|
|
3419
3465
|
*/
|
|
3420
|
-
|
|
3421
|
-
const llm = getDefaultLlamaCpp();
|
|
3466
|
+
async function chunkDocumentByTokensWithLlm(llm, content, maxTokens = CHUNK_SIZE_TOKENS, overlapTokens = CHUNK_OVERLAP_TOKENS, windowTokens = CHUNK_WINDOW_TOKENS, filepath, chunkStrategy = "regex", signal) {
|
|
3422
3467
|
// Use moderate chars/token estimate (prose ~4, code ~2, mixed ~3)
|
|
3423
3468
|
// If chunks exceed limit, they'll be re-split with actual ratio
|
|
3424
3469
|
const avgCharsPerToken = 3;
|
|
@@ -3492,6 +3537,9 @@ export async function chunkDocumentByTokens(content, maxTokens = CHUNK_SIZE_TOKE
|
|
|
3492
3537
|
}
|
|
3493
3538
|
return results;
|
|
3494
3539
|
}
|
|
3540
|
+
export async function chunkDocumentByTokens(content, maxTokens = CHUNK_SIZE_TOKENS, overlapTokens = CHUNK_OVERLAP_TOKENS, windowTokens = CHUNK_WINDOW_TOKENS, filepath, chunkStrategy = "regex", signal) {
|
|
3541
|
+
return await chunkDocumentByTokensWithLlm(getDefaultLlamaCpp(), content, maxTokens, overlapTokens, windowTokens, filepath, chunkStrategy, signal);
|
|
3542
|
+
}
|
|
3495
3543
|
// =============================================================================
|
|
3496
3544
|
// Fuzzy matching
|
|
3497
3545
|
// =============================================================================
|
|
@@ -3954,38 +4002,29 @@ export function sanitizeFTS5Term(term) {
|
|
|
3954
4002
|
return term.replace(/[^\p{L}\p{N}'_]/gu, '').toLowerCase();
|
|
3955
4003
|
}
|
|
3956
4004
|
/**
|
|
3957
|
-
*
|
|
3958
|
-
*
|
|
3959
|
-
|
|
3960
|
-
|
|
3961
|
-
|
|
3962
|
-
|
|
3963
|
-
|
|
3964
|
-
*
|
|
3965
|
-
* and sanitizing each part. Returns the parts joined by spaces for use
|
|
3966
|
-
* inside FTS5 quotes: "multi agent" matches "multi-agent" in porter tokenizer.
|
|
3967
|
-
*/
|
|
3968
|
-
function sanitizeHyphenatedTerm(term) {
|
|
3969
|
-
return term.split('-').map(t => sanitizeFTS5Term(t)).filter(t => t).join(' ');
|
|
3970
|
-
}
|
|
3971
|
-
/**
|
|
3972
|
-
* Check if a token is a dotted version/version-like string (e.g., 2026.4.10, 3.14.0).
|
|
3973
|
-
* Returns true if splitting on dots yields at least 2 non-empty parts consisting of
|
|
3974
|
-
* word/digit characters only. This avoids incorrectly splitting tokens with leading/
|
|
3975
|
-
* trailing dots. Version strings like "2026.4.10" split into ["2026","4","10"] (3 parts).
|
|
4005
|
+
* A run of characters the FTS tokenizer treats as a separator.
|
|
4006
|
+
*
|
|
4007
|
+
* `documents_fts` is tokenized with `porter unicode61`, which starts a new
|
|
4008
|
+
* token at every character that is not a letter or a digit. Underscore is one
|
|
4009
|
+
* of those, but it is deliberately kept here rather than split on: FTS5 applies
|
|
4010
|
+
* the same tokenizer to a quoted phrase, so leaving `apply_secrets` intact lets
|
|
4011
|
+
* it split symmetrically into `apply secrets` on both sides, and that is the
|
|
4012
|
+
* behaviour #305 shipped. The apostrophe is kept for the same reason.
|
|
3976
4013
|
*/
|
|
3977
|
-
|
|
3978
|
-
const parts = token.split('.');
|
|
3979
|
-
return parts.length >= 2 && parts.every(p => p.length > 0 && /^[\p{L}\p{N}_]+$/u.test(p));
|
|
3980
|
-
}
|
|
4014
|
+
const FTS5_SEPARATOR_RUN = /[^\p{L}\p{N}'_]+/u;
|
|
3981
4015
|
/**
|
|
3982
|
-
*
|
|
3983
|
-
*
|
|
3984
|
-
*
|
|
3985
|
-
*
|
|
4016
|
+
* Split one query term the way the tokenizer split the document text, and
|
|
4017
|
+
* sanitize each part.
|
|
4018
|
+
*
|
|
4019
|
+
* `PIO-1384` becomes ["pio", "1384"], `src/lib/i18n.ts` becomes
|
|
4020
|
+
* ["src", "lib", "i18n", "ts"], and a term with no separator in it comes back
|
|
4021
|
+
* as a single part. Callers join the parts into an FTS5 phrase, which is what
|
|
4022
|
+
* makes the parts have to be adjacent in the document rather than merely all
|
|
4023
|
+
* present. Parts that sanitize to nothing are dropped, so a term that is all
|
|
4024
|
+
* punctuation yields an empty list and the caller skips it.
|
|
3986
4025
|
*/
|
|
3987
|
-
function
|
|
3988
|
-
return term.split(
|
|
4026
|
+
function splitFTS5CompoundTerm(term) {
|
|
4027
|
+
return term.split(FTS5_SEPARATOR_RUN).map(p => sanitizeFTS5Term(p)).filter(p => p);
|
|
3989
4028
|
}
|
|
3990
4029
|
/**
|
|
3991
4030
|
* Parse lex query syntax into FTS5 query.
|
|
@@ -3993,7 +4032,8 @@ function sanitizeDottedTerm(term) {
|
|
|
3993
4032
|
* Supports:
|
|
3994
4033
|
* - Quoted phrases: "exact phrase" → "exact phrase" (exact match)
|
|
3995
4034
|
* - Negation: -term or -"phrase" → uses FTS5 NOT operator
|
|
3996
|
-
* -
|
|
4035
|
+
* - Terms holding a separator: multi-agent, DEC-0054, gpt-4, 2026.4.10,
|
|
4036
|
+
* src/lib/i18n.ts, @tobilu/qmd → treated as phrases over their parts
|
|
3997
4037
|
* - Plain terms: term → "term"* (prefix match)
|
|
3998
4038
|
*
|
|
3999
4039
|
* FTS5 NOT is a binary operator: `term1 NOT term2` means "match term1 but not term2".
|
|
@@ -4010,6 +4050,8 @@ function sanitizeDottedTerm(term) {
|
|
|
4010
4050
|
* multi-agent memory → "multi agent" AND "memory"*
|
|
4011
4051
|
* DEC-0054 → "dec 0054"
|
|
4012
4052
|
* -multi-agent → NOT "multi agent"
|
|
4053
|
+
* "DEC-0054" → "dec 0054"
|
|
4054
|
+
* src/lib/i18n.ts → "src lib i18n ts"
|
|
4013
4055
|
*/
|
|
4014
4056
|
function buildFTS5Query(query) {
|
|
4015
4057
|
const positive = [];
|
|
@@ -4053,41 +4095,7 @@ function buildFTS5Query(query) {
|
|
|
4053
4095
|
while (i < s.length && !/[\s"]/.test(s[i]))
|
|
4054
4096
|
i++;
|
|
4055
4097
|
const term = s.slice(start, i);
|
|
4056
|
-
|
|
4057
|
-
// These get split into phrase queries so FTS5 porter tokenizer matches them.
|
|
4058
|
-
if (isHyphenatedToken(term)) {
|
|
4059
|
-
const sanitized = sanitizeHyphenatedTerm(term);
|
|
4060
|
-
if (sanitized) {
|
|
4061
|
-
const ftsPhrase = `"${sanitized}"`; // Phrase match (no prefix)
|
|
4062
|
-
if (negated) {
|
|
4063
|
-
negative.push(ftsPhrase);
|
|
4064
|
-
}
|
|
4065
|
-
else {
|
|
4066
|
-
positive.push(ftsPhrase);
|
|
4067
|
-
}
|
|
4068
|
-
}
|
|
4069
|
-
}
|
|
4070
|
-
else if (isDottedToken(term)) {
|
|
4071
|
-
// Handle dotted version strings: 2026.4.10, 3.14.0, v1.2.3
|
|
4072
|
-
// The porter tokenizer splits on dots, so the index has individual tokens.
|
|
4073
|
-
// We AND all parts together so the query matches documents containing all parts.
|
|
4074
|
-
const sanitized = sanitizeDottedTerm(term);
|
|
4075
|
-
if (sanitized) {
|
|
4076
|
-
// sanitizeDottedTerm already wraps each part in quotes with prefix match
|
|
4077
|
-
if (negated) {
|
|
4078
|
-
// Wrap multi-token AND expression in parens for NOT negation
|
|
4079
|
-
negative.push(`(${sanitized})`);
|
|
4080
|
-
}
|
|
4081
|
-
else {
|
|
4082
|
-
// Flatten individual AND'd terms into the positive list so they combine
|
|
4083
|
-
// correctly with other terms (avoids double-wrapping in outer AND).
|
|
4084
|
-
for (const part of sanitized.split(' AND ')) {
|
|
4085
|
-
positive.push(part.trim());
|
|
4086
|
-
}
|
|
4087
|
-
}
|
|
4088
|
-
}
|
|
4089
|
-
}
|
|
4090
|
-
else if (containsCjk(term)) {
|
|
4098
|
+
if (containsCjk(term)) {
|
|
4091
4099
|
const sanitized = sanitizeFTS5Phrase(term);
|
|
4092
4100
|
if (sanitized) {
|
|
4093
4101
|
const ftsPhrase = `"${sanitized}"`; // CJK phrase over character tokens
|
|
@@ -4100,9 +4108,16 @@ function buildFTS5Query(query) {
|
|
|
4100
4108
|
}
|
|
4101
4109
|
}
|
|
4102
4110
|
else {
|
|
4103
|
-
|
|
4104
|
-
|
|
4105
|
-
|
|
4111
|
+
// Any separator inside the term (multi-agent, DEC-0054, 2026.4.10,
|
|
4112
|
+
// src/lib/i18n.ts, @tobilu/qmd) split it at index time too, so the term
|
|
4113
|
+
// has to be matched as the phrase those parts form. A term with no
|
|
4114
|
+
// separator is one part and keeps its prefix match, which is what makes
|
|
4115
|
+
// a plain word still match longer words that start with it.
|
|
4116
|
+
const parts = splitFTS5CompoundTerm(term);
|
|
4117
|
+
if (parts.length > 0) {
|
|
4118
|
+
const ftsTerm = parts.length > 1
|
|
4119
|
+
? `"${parts.join(' ')}"` // Phrase match (no prefix)
|
|
4120
|
+
: `"${parts[0]}"*`; // Prefix match
|
|
4106
4121
|
if (negated) {
|
|
4107
4122
|
negative.push(ftsTerm);
|
|
4108
4123
|
}
|
|
@@ -4162,7 +4177,28 @@ function normalizeCollectionFilter(filter) {
|
|
|
4162
4177
|
const names = typeof filter === "string" ? [filter] : filter;
|
|
4163
4178
|
return [...new Set(names.filter(name => name.length > 0))];
|
|
4164
4179
|
}
|
|
4165
|
-
function
|
|
4180
|
+
function scopedCollectionNames(scope) {
|
|
4181
|
+
if (scope == null)
|
|
4182
|
+
return undefined;
|
|
4183
|
+
const names = (typeof scope === "string" ? [scope] : Array.from(scope))
|
|
4184
|
+
.map(n => n.trim())
|
|
4185
|
+
.filter(n => n.length > 0);
|
|
4186
|
+
return names.length > 0 ? names : undefined;
|
|
4187
|
+
}
|
|
4188
|
+
function mergeSearchResultsByScore(lists, limit) {
|
|
4189
|
+
const best = new Map();
|
|
4190
|
+
for (const list of lists) {
|
|
4191
|
+
for (const r of list) {
|
|
4192
|
+
const prev = best.get(r.filepath);
|
|
4193
|
+
if (!prev || r.score > prev.score)
|
|
4194
|
+
best.set(r.filepath, r);
|
|
4195
|
+
}
|
|
4196
|
+
}
|
|
4197
|
+
return Array.from(best.values())
|
|
4198
|
+
.sort((a, b) => b.score - a.score)
|
|
4199
|
+
.slice(0, limit);
|
|
4200
|
+
}
|
|
4201
|
+
function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter, filter) {
|
|
4166
4202
|
// Keep collection membership inside the ranked FTS candidate set. Filtering
|
|
4167
4203
|
// after LIMIT can lose every matching row from a smaller collection.
|
|
4168
4204
|
const params = [ftsQuery];
|
|
@@ -4174,7 +4210,10 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4174
4210
|
? `AND filtered_d.active = 1 AND filtered_d.collection IN (${collections.map(() => "?").join(", ")})`
|
|
4175
4211
|
: "";
|
|
4176
4212
|
params.push(...collections);
|
|
4177
|
-
|
|
4213
|
+
// When filtering by metadata, fetch extra candidates from the FTS index
|
|
4214
|
+
// since some will be filtered out. Without a filter we can fetch exactly the requested limit.
|
|
4215
|
+
const ftsLimit = filter ? limit * 10 : limit;
|
|
4216
|
+
params.push(ftsLimit);
|
|
4178
4217
|
let sql = `
|
|
4179
4218
|
WITH fts_matches AS (
|
|
4180
4219
|
SELECT ${table}.rowid AS rowid, bm25(${table}, 1.5, 4.0, 1.0) as bm25_score
|
|
@@ -4191,12 +4230,21 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4191
4230
|
d.title,
|
|
4192
4231
|
content.doc as body,
|
|
4193
4232
|
d.hash,
|
|
4194
|
-
fm.bm25_score
|
|
4233
|
+
fm.bm25_score,
|
|
4234
|
+
dm.metadata_json
|
|
4195
4235
|
FROM fts_matches fm
|
|
4196
4236
|
JOIN documents d ON d.id = fm.rowid
|
|
4197
4237
|
JOIN content ON content.hash = d.hash
|
|
4238
|
+
LEFT JOIN document_metadata dm ON dm.document_id = d.id
|
|
4198
4239
|
WHERE d.active = 1
|
|
4199
4240
|
`;
|
|
4241
|
+
if (filter) {
|
|
4242
|
+
// Only documents with current, error-free extraction can match — an
|
|
4243
|
+
// unprocessed document must not accidentally satisfy `exists: false`.
|
|
4244
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4245
|
+
sql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`;
|
|
4246
|
+
params.push(...compiledFilter.params);
|
|
4247
|
+
}
|
|
4200
4248
|
// bm25 lower is better; sort ascending.
|
|
4201
4249
|
sql += ` ORDER BY fm.bm25_score ASC LIMIT ?`;
|
|
4202
4250
|
params.push(limit);
|
|
@@ -4219,6 +4267,7 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4219
4267
|
bodyLength: row.body.length,
|
|
4220
4268
|
body: row.body,
|
|
4221
4269
|
context: getContextForFile(db, row.filepath),
|
|
4270
|
+
metadata: parseMetadataJson(row.metadata_json),
|
|
4222
4271
|
score,
|
|
4223
4272
|
source: "fts",
|
|
4224
4273
|
};
|
|
@@ -4230,17 +4279,21 @@ function buildCjkSignalQuery(signal) {
|
|
|
4230
4279
|
return null;
|
|
4231
4280
|
return terms.map(term => `"${term.replace(/"/g, '""')}"`).join(" AND ");
|
|
4232
4281
|
}
|
|
4233
|
-
export function searchCharFTS(db, query, limit = 20, collectionFilter) {
|
|
4282
|
+
export function searchCharFTS(db, query, limit = 20, collectionFilter, filter) {
|
|
4234
4283
|
if (!readOnlyDatabases.has(db))
|
|
4235
4284
|
repairDirtyCjkCharFallback(db);
|
|
4236
4285
|
const charQuery = buildFTS5Query(query);
|
|
4237
4286
|
if (!charQuery)
|
|
4238
4287
|
return [];
|
|
4239
|
-
return searchFtsChannel(db, "documents_fts", charQuery, limit, collectionFilter);
|
|
4288
|
+
return searchFtsChannel(db, "documents_fts", charQuery, limit, collectionFilter, filter);
|
|
4240
4289
|
}
|
|
4241
|
-
export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
4290
|
+
export function searchFTS(db, query, limit = 20, collectionFilter, filter) {
|
|
4291
|
+
const names = scopedCollectionNames(collectionFilter);
|
|
4292
|
+
if (names && names.length > 1) {
|
|
4293
|
+
return mergeSearchResultsByScore(names.map(name => searchFTS(db, query, limit, name, filter)), limit);
|
|
4294
|
+
}
|
|
4242
4295
|
if (!containsCjk(query))
|
|
4243
|
-
return searchCharFTS(db, query, limit, collectionFilter);
|
|
4296
|
+
return searchCharFTS(db, query, limit, collectionFilter, filter);
|
|
4244
4297
|
const charQuery = buildFTS5Query(query);
|
|
4245
4298
|
if (!charQuery)
|
|
4246
4299
|
return [];
|
|
@@ -4249,7 +4302,7 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4249
4302
|
const candidateDepth = Math.max(limit, CJK_LEXICAL_CANDIDATE_DEPTH);
|
|
4250
4303
|
const rankedChannels = [{
|
|
4251
4304
|
channel: "char",
|
|
4252
|
-
results: searchCharFTS(db, query, candidateDepth, collectionFilter),
|
|
4305
|
+
results: searchCharFTS(db, query, candidateDepth, collectionFilter, filter),
|
|
4253
4306
|
}];
|
|
4254
4307
|
let secondaryReason = null;
|
|
4255
4308
|
try {
|
|
@@ -4285,7 +4338,7 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4285
4338
|
}
|
|
4286
4339
|
rankedChannels.push({
|
|
4287
4340
|
channel: entry.channel,
|
|
4288
|
-
results: searchFtsChannel(db, entry.table, ftsQuery, candidateDepth, collectionFilter),
|
|
4341
|
+
results: searchFtsChannel(db, entry.table, ftsQuery, candidateDepth, collectionFilter, filter),
|
|
4289
4342
|
});
|
|
4290
4343
|
channels.push({ channel: entry.channel, status: "used" });
|
|
4291
4344
|
}
|
|
@@ -4325,6 +4378,49 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4325
4378
|
// =============================================================================
|
|
4326
4379
|
// Vector Search
|
|
4327
4380
|
// =============================================================================
|
|
4381
|
+
/** sqlite-vec rejects k above this in MATCH queries (v0.1.9). */
|
|
4382
|
+
const SQLITE_VEC_MAX_K = 4096;
|
|
4383
|
+
/**
|
|
4384
|
+
* Max filter-eligible vectors for an exact cosine scan. Above this we fall
|
|
4385
|
+
* back to global ANN with a capped over-fetch. Exact scan avoids the
|
|
4386
|
+
* post-filter starvation of small eligible sets — originally small
|
|
4387
|
+
* collections (#791, #803), now also selective metadata filters; ANN remains
|
|
4388
|
+
* for very large eligible sets where a full scan would be expensive.
|
|
4389
|
+
*/
|
|
4390
|
+
const FILTERED_VEC_EXACT_SCAN_MAX = 20_000;
|
|
4391
|
+
const VEC_HASH_SEQ_IN_CHUNK = 400;
|
|
4392
|
+
/**
|
|
4393
|
+
* Exact cosine-distance scan over a known set of hash_seq keys.
|
|
4394
|
+
* Uses vec_distance_cosine with chunked IN lists (no JOIN with vectors_vec).
|
|
4395
|
+
*/
|
|
4396
|
+
function exactVecScanByHashSeq(db, embedding, hashSeqs, limit) {
|
|
4397
|
+
if (hashSeqs.length === 0 || limit <= 0)
|
|
4398
|
+
return [];
|
|
4399
|
+
const queryVec = new Float32Array(embedding);
|
|
4400
|
+
// Over-fetch a bit so multi-chunk docs can still yield `limit` unique files.
|
|
4401
|
+
const fetchLimit = Math.max(limit * 3, limit);
|
|
4402
|
+
const scored = [];
|
|
4403
|
+
for (let i = 0; i < hashSeqs.length; i += VEC_HASH_SEQ_IN_CHUNK) {
|
|
4404
|
+
const chunk = hashSeqs.slice(i, i + VEC_HASH_SEQ_IN_CHUNK);
|
|
4405
|
+
const placeholders = chunk.map(() => "?").join(",");
|
|
4406
|
+
const rows = db.prepare(`
|
|
4407
|
+
SELECT hash_seq, vec_distance_cosine(embedding, ?) AS distance
|
|
4408
|
+
FROM vectors_vec
|
|
4409
|
+
WHERE hash_seq IN (${placeholders})
|
|
4410
|
+
`).all(queryVec, ...chunk);
|
|
4411
|
+
scored.push(...rows);
|
|
4412
|
+
}
|
|
4413
|
+
scored.sort((a, b) => a.distance - b.distance);
|
|
4414
|
+
return scored.slice(0, fetchLimit);
|
|
4415
|
+
}
|
|
4416
|
+
function annVecScan(db, embedding, k) {
|
|
4417
|
+
const vecK = Math.max(1, Math.min(SQLITE_VEC_MAX_K, k));
|
|
4418
|
+
return db.prepare(`
|
|
4419
|
+
SELECT hash_seq, distance
|
|
4420
|
+
FROM vectors_vec
|
|
4421
|
+
WHERE embedding MATCH ? AND k = ?
|
|
4422
|
+
`).all(new Float32Array(embedding), vecK);
|
|
4423
|
+
}
|
|
4328
4424
|
function resolveReadyProviderEmbeddingIdentity(db, provider) {
|
|
4329
4425
|
const storedIdentity = readStoredEmbeddingIdentity(db);
|
|
4330
4426
|
if (!storedIdentity || storedIdentity.providerId !== provider.providerId
|
|
@@ -4365,10 +4461,29 @@ function hasSearchableVectorIndex(store) {
|
|
|
4365
4461
|
return storedIdentity !== undefined
|
|
4366
4462
|
&& inspectEmbeddingIndexState(store.db, storedIdentity).status === "ready";
|
|
4367
4463
|
}
|
|
4368
|
-
export async function searchVec(db, query, model, limit = 20, collectionFilter, session, precomputedEmbedding,
|
|
4464
|
+
export async function searchVec(db, query, model, limit = 20, collectionFilter, session, precomputedEmbedding, providerOrLlm, authorizeRemoteRequestOrFilter, llmOverride, metadataFilter) {
|
|
4369
4465
|
const tableExists = db.prepare(`SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get();
|
|
4370
4466
|
if (!tableExists)
|
|
4371
4467
|
return [];
|
|
4468
|
+
// Disambiguate overloaded arguments
|
|
4469
|
+
let provider;
|
|
4470
|
+
let authorizeRemoteRequest;
|
|
4471
|
+
let filter = metadataFilter;
|
|
4472
|
+
let effectiveLlmOverride = llmOverride;
|
|
4473
|
+
if (providerOrLlm && typeof providerOrLlm === "object") {
|
|
4474
|
+
if ("providerId" in providerOrLlm) {
|
|
4475
|
+
provider = providerOrLlm;
|
|
4476
|
+
}
|
|
4477
|
+
else {
|
|
4478
|
+
effectiveLlmOverride = providerOrLlm;
|
|
4479
|
+
}
|
|
4480
|
+
}
|
|
4481
|
+
if (typeof authorizeRemoteRequestOrFilter === "function") {
|
|
4482
|
+
authorizeRemoteRequest = authorizeRemoteRequestOrFilter;
|
|
4483
|
+
}
|
|
4484
|
+
else if (typeof authorizeRemoteRequestOrFilter === "object" && authorizeRemoteRequestOrFilter !== null && !filter) {
|
|
4485
|
+
filter = authorizeRemoteRequestOrFilter;
|
|
4486
|
+
}
|
|
4372
4487
|
if (provider && model !== provider.model) {
|
|
4373
4488
|
throw new Error(`Embedding model ${model} does not match borrowed provider model ${provider.model}.`);
|
|
4374
4489
|
}
|
|
@@ -4397,13 +4512,19 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4397
4512
|
})).vector;
|
|
4398
4513
|
}
|
|
4399
4514
|
else if (!embedding) {
|
|
4400
|
-
embedding = await getEmbedding(query, model, true, session,
|
|
4515
|
+
embedding = await getEmbedding(query, model, true, session, effectiveLlmOverride);
|
|
4401
4516
|
}
|
|
4402
4517
|
if (!embedding)
|
|
4403
4518
|
return [];
|
|
4519
|
+
// Multi-collection union: search each collection separately to avoid starvation
|
|
4520
|
+
const names = scopedCollectionNames(collectionFilter);
|
|
4521
|
+
if (names && names.length > 1) {
|
|
4522
|
+
const lists = await Promise.all(names.map(name => searchVec(db, query, model, limit, name, session, embedding, provider ?? effectiveLlmOverride, authorizeRemoteRequest ?? filter, effectiveLlmOverride, filter)));
|
|
4523
|
+
return mergeSearchResultsByScore(lists, limit);
|
|
4524
|
+
}
|
|
4404
4525
|
const activeFingerprint = providerIdentity
|
|
4405
4526
|
? providerIdentity.fingerprint
|
|
4406
|
-
: storedIdentity
|
|
4527
|
+
: storedIdentity?.fingerprint;
|
|
4407
4528
|
const toSearchResult = (row, body, distance) => {
|
|
4408
4529
|
return {
|
|
4409
4530
|
filepath: row.filepath,
|
|
@@ -4416,56 +4537,87 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4416
4537
|
bodyLength: body.length,
|
|
4417
4538
|
body,
|
|
4418
4539
|
context: getContextForFile(db, row.filepath),
|
|
4540
|
+
metadata: parseMetadataJson(row.metadata_json),
|
|
4419
4541
|
score: 1 - distance, // Cosine similarity = 1 - cosine distance
|
|
4420
4542
|
source: "vec",
|
|
4421
4543
|
chunkPos: row.pos,
|
|
4422
4544
|
};
|
|
4423
4545
|
};
|
|
4424
4546
|
const collections = normalizeCollectionFilter(collectionFilter);
|
|
4425
|
-
const queryVec = embedding instanceof Float32Array ? embedding : new Float32Array(embedding);
|
|
4426
|
-
const SQLITE_VEC_MAX_K = 4096;
|
|
4427
|
-
let collectionFilterSql = "";
|
|
4428
|
-
const queryParams = [queryVec];
|
|
4429
|
-
if (collections.length === 1) {
|
|
4430
|
-
collectionFilterSql = "AND collection = ?";
|
|
4431
|
-
queryParams.push(collections[0]);
|
|
4432
|
-
}
|
|
4433
|
-
else if (collections.length > 1) {
|
|
4434
|
-
const placeholders = collections.map(() => "?").join(", ");
|
|
4435
|
-
collectionFilterSql = `AND collection IN (${placeholders})`;
|
|
4436
|
-
queryParams.push(...collections);
|
|
4437
|
-
}
|
|
4438
|
-
// Request generous k to account for multi-chunk deduplication per document
|
|
4439
|
-
const fetchK = Math.min(SQLITE_VEC_MAX_K, Math.max(limit * 15, 60));
|
|
4440
|
-
queryParams.push(fetchK);
|
|
4441
4547
|
let vecRows;
|
|
4442
|
-
|
|
4443
|
-
|
|
4444
|
-
|
|
4445
|
-
|
|
4446
|
-
|
|
4447
|
-
|
|
4448
|
-
|
|
4449
|
-
|
|
4548
|
+
if (filter) {
|
|
4549
|
+
let eligibleSql = `
|
|
4550
|
+
SELECT DISTINCT cv.hash || '_' || cv.seq AS hash_seq
|
|
4551
|
+
FROM content_vectors cv
|
|
4552
|
+
JOIN documents d ON d.hash = cv.hash AND d.active = 1
|
|
4553
|
+
JOIN document_metadata dm ON dm.document_id = d.id
|
|
4554
|
+
`;
|
|
4555
|
+
const eligibleConditions = [];
|
|
4556
|
+
const eligibleParams = [];
|
|
4557
|
+
if (activeFingerprint) {
|
|
4558
|
+
eligibleConditions.push(`cv.model = ? AND cv.embed_fingerprint = ?`);
|
|
4559
|
+
eligibleParams.push(model, activeFingerprint);
|
|
4560
|
+
}
|
|
4561
|
+
if (collections.length === 1) {
|
|
4562
|
+
eligibleConditions.push(`d.collection = ?`);
|
|
4563
|
+
eligibleParams.push(collections[0]);
|
|
4564
|
+
}
|
|
4565
|
+
else if (collections.length > 1) {
|
|
4566
|
+
eligibleConditions.push(`d.collection IN (${collections.map(() => "?").join(", ")})`);
|
|
4567
|
+
eligibleParams.push(...collections);
|
|
4568
|
+
}
|
|
4569
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4570
|
+
eligibleConditions.push(`dm.extraction_version = ${METADATA_EXTRACTION_VERSION}`);
|
|
4571
|
+
eligibleConditions.push(`dm.extraction_error IS NULL`);
|
|
4572
|
+
eligibleConditions.push(compiledFilter.sql);
|
|
4573
|
+
eligibleParams.push(...compiledFilter.params);
|
|
4574
|
+
eligibleSql += ` WHERE ${eligibleConditions.join(" AND ")}`;
|
|
4575
|
+
const eligibleHashSeqs = withLazyContentVectorMigration(db, () => db.prepare(eligibleSql).all(...eligibleParams)).map((r) => r.hash_seq);
|
|
4576
|
+
if (eligibleHashSeqs.length === 0)
|
|
4577
|
+
return [];
|
|
4578
|
+
if (eligibleHashSeqs.length <= FILTERED_VEC_EXACT_SCAN_MAX) {
|
|
4579
|
+
vecRows = exactVecScanByHashSeq(db, embedding, eligibleHashSeqs, limit);
|
|
4580
|
+
}
|
|
4581
|
+
else {
|
|
4582
|
+
vecRows = annVecScan(db, embedding, Math.max(limit * 30, limit * 3));
|
|
4583
|
+
}
|
|
4450
4584
|
}
|
|
4451
|
-
|
|
4452
|
-
//
|
|
4453
|
-
|
|
4454
|
-
|
|
4455
|
-
|
|
4456
|
-
|
|
4457
|
-
|
|
4458
|
-
|
|
4585
|
+
else {
|
|
4586
|
+
// No metadata filter: use fast ANN scan (with collection pushdown if supported)
|
|
4587
|
+
const queryVec = embedding instanceof Float32Array ? embedding : new Float32Array(embedding);
|
|
4588
|
+
let collectionFilterSql = "";
|
|
4589
|
+
const queryParams = [queryVec];
|
|
4590
|
+
if (collections.length === 1) {
|
|
4591
|
+
collectionFilterSql = "AND collection = ?";
|
|
4592
|
+
queryParams.push(collections[0]);
|
|
4593
|
+
}
|
|
4594
|
+
else if (collections.length > 1) {
|
|
4595
|
+
const placeholders = collections.map(() => "?").join(", ");
|
|
4596
|
+
collectionFilterSql = `AND collection IN (${placeholders})`;
|
|
4597
|
+
queryParams.push(...collections);
|
|
4598
|
+
}
|
|
4599
|
+
const fetchK = Math.min(SQLITE_VEC_MAX_K, Math.max(limit * 15, 60));
|
|
4600
|
+
queryParams.push(fetchK);
|
|
4601
|
+
try {
|
|
4602
|
+
vecRows = withLazyContentVectorMigration(db, () => db.prepare(`
|
|
4603
|
+
SELECT hash_seq, distance
|
|
4604
|
+
FROM vectors_vec
|
|
4605
|
+
WHERE embedding MATCH ?
|
|
4606
|
+
${collectionFilterSql}
|
|
4607
|
+
AND k = ?
|
|
4608
|
+
`).all(...queryParams));
|
|
4609
|
+
}
|
|
4610
|
+
catch {
|
|
4611
|
+
// Fallback if table lacks collection column before migration
|
|
4612
|
+
vecRows = annVecScan(db, embedding, fetchK);
|
|
4613
|
+
}
|
|
4459
4614
|
}
|
|
4460
4615
|
if (vecRows.length === 0)
|
|
4461
4616
|
return [];
|
|
4462
4617
|
const keys = vecRows.map(r => r.hash_seq);
|
|
4463
4618
|
const distMap = new Map(vecRows.map(r => [r.hash_seq, r.distance]));
|
|
4464
4619
|
const placeholders = keys.map(() => "?").join(", ");
|
|
4465
|
-
|
|
4466
|
-
? `AND d.collection IN (${collections.map(() => "?").join(", ")})`
|
|
4467
|
-
: "";
|
|
4468
|
-
const metaRows = withLazyContentVectorMigration(db, () => db.prepare(`
|
|
4620
|
+
let metaSql = `
|
|
4469
4621
|
SELECT
|
|
4470
4622
|
cv.hash || '_' || cv.seq AS hash_seq,
|
|
4471
4623
|
cv.hash,
|
|
@@ -4473,14 +4625,32 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4473
4625
|
'qmd://' || d.collection || '/' || d.path AS filepath,
|
|
4474
4626
|
d.collection || '/' || d.path AS display_path,
|
|
4475
4627
|
d.title,
|
|
4476
|
-
d.collection
|
|
4628
|
+
d.collection,
|
|
4629
|
+
dm.metadata_json
|
|
4477
4630
|
FROM content_vectors cv
|
|
4478
4631
|
JOIN documents d ON d.hash = cv.hash AND d.active = 1
|
|
4479
|
-
|
|
4480
|
-
|
|
4481
|
-
|
|
4482
|
-
|
|
4483
|
-
|
|
4632
|
+
LEFT JOIN document_metadata dm ON dm.document_id = d.id
|
|
4633
|
+
WHERE (cv.hash || '_' || cv.seq) IN (${placeholders})
|
|
4634
|
+
`;
|
|
4635
|
+
const metaParams = [...keys];
|
|
4636
|
+
if (activeFingerprint) {
|
|
4637
|
+
metaSql += ` AND cv.model = ? AND cv.embed_fingerprint = ?`;
|
|
4638
|
+
metaParams.push(model, activeFingerprint);
|
|
4639
|
+
}
|
|
4640
|
+
if (collections.length === 1) {
|
|
4641
|
+
metaSql += ` AND d.collection = ?`;
|
|
4642
|
+
metaParams.push(collections[0]);
|
|
4643
|
+
}
|
|
4644
|
+
else if (collections.length > 1) {
|
|
4645
|
+
metaSql += ` AND d.collection IN (${collections.map(() => "?").join(", ")})`;
|
|
4646
|
+
metaParams.push(...collections);
|
|
4647
|
+
}
|
|
4648
|
+
if (filter) {
|
|
4649
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4650
|
+
metaSql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`;
|
|
4651
|
+
metaParams.push(...compiledFilter.params);
|
|
4652
|
+
}
|
|
4653
|
+
const metaRows = withLazyContentVectorMigration(db, () => db.prepare(metaSql).all(...metaParams));
|
|
4484
4654
|
const byFile = new Map();
|
|
4485
4655
|
for (const m of metaRows) {
|
|
4486
4656
|
const dist = distMap.get(m.hash_seq);
|
|
@@ -5310,6 +5480,7 @@ export function getStatusReadOnly(db, needsEmbedding) {
|
|
|
5310
5480
|
totalDocuments: totalDocs,
|
|
5311
5481
|
needsEmbedding,
|
|
5312
5482
|
hasVectorIndex: hasVectors,
|
|
5483
|
+
pendingMetadata: countDocumentsPendingMetadata(db),
|
|
5313
5484
|
collections,
|
|
5314
5485
|
};
|
|
5315
5486
|
}
|
|
@@ -5435,6 +5606,15 @@ export function addLineNumbers(text, startLine = 1) {
|
|
|
5435
5606
|
const lines = text.split('\n');
|
|
5436
5607
|
return lines.map((line, i) => `${startLine + i}: ${line}`).join('\n');
|
|
5437
5608
|
}
|
|
5609
|
+
/**
|
|
5610
|
+
* Attach canonical metadata to final search results with one batch query.
|
|
5611
|
+
* Runs after RRF/reranking so metadata is never duplicated through the
|
|
5612
|
+
* intermediate ranked lists.
|
|
5613
|
+
*/
|
|
5614
|
+
function attachResultMetadata(db, results) {
|
|
5615
|
+
const metadataByFilepath = getMetadataByFilepath(db, results.map(r => r.file));
|
|
5616
|
+
return results.map(r => ({ ...r, metadata: metadataByFilepath.get(r.file) ?? {} }));
|
|
5617
|
+
}
|
|
5438
5618
|
/**
|
|
5439
5619
|
* RRF list weights for hybridQuery.
|
|
5440
5620
|
*
|
|
@@ -5469,6 +5649,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5469
5649
|
const collectionFilter = options?.collections && options.collections.length > 0
|
|
5470
5650
|
? options.collections
|
|
5471
5651
|
: options?.collection;
|
|
5652
|
+
const filter = options?.filter;
|
|
5472
5653
|
const explain = options?.explain ?? false;
|
|
5473
5654
|
const expansionContext = options?.expansionContext;
|
|
5474
5655
|
const rerankContext = options?.rerankContext;
|
|
@@ -5483,9 +5664,9 @@ export async function hybridQuery(store, query, options) {
|
|
|
5483
5664
|
// When either context is provided, disable strong-signal bypass — the obvious BM25
|
|
5484
5665
|
// match may not be what the caller wants (e.g. "performance" with context
|
|
5485
5666
|
// "web page load times" should NOT shortcut to a sports-performance doc).
|
|
5486
|
-
// Pass collection directly into FTS query (filter at SQL level, not post-hoc)
|
|
5667
|
+
// Pass collection and metadata filter directly into FTS query (filter at SQL level, not post-hoc)
|
|
5487
5668
|
const parsedDirective = parseExpansionDirective(query);
|
|
5488
|
-
const initialFts = store.searchFTS(parsedDirective.query, 20, collectionFilter);
|
|
5669
|
+
const initialFts = store.searchFTS(parsedDirective.query, 20, collectionFilter, filter);
|
|
5489
5670
|
const strongSignal = getLexicalStrongSignal(initialFts);
|
|
5490
5671
|
const topScore = strongSignal.topScore;
|
|
5491
5672
|
let expansionDecision;
|
|
@@ -5553,7 +5734,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5553
5734
|
// 3a: Run FTS for all lex expansions right away (no LLM needed)
|
|
5554
5735
|
for (const q of expanded) {
|
|
5555
5736
|
if (q.type === 'lex') {
|
|
5556
|
-
const ftsResults = store.searchFTS(q.query, 20, collectionFilter);
|
|
5737
|
+
const ftsResults = store.searchFTS(q.query, 20, collectionFilter, filter);
|
|
5557
5738
|
if (ftsResults.length > 0) {
|
|
5558
5739
|
for (const r of ftsResults)
|
|
5559
5740
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5585,7 +5766,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5585
5766
|
const embedding = embeddings[i];
|
|
5586
5767
|
if (!embedding || embedding.length === 0)
|
|
5587
5768
|
continue;
|
|
5588
|
-
const vecResults = await store.searchVec(vecQueries[i].text, embedModel, 20, collectionFilter, undefined, embedding);
|
|
5769
|
+
const vecResults = await store.searchVec(vecQueries[i].text, embedModel, 20, collectionFilter, undefined, embedding, filter);
|
|
5589
5770
|
if (vecResults.length > 0) {
|
|
5590
5771
|
for (const r of vecResults)
|
|
5591
5772
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5652,7 +5833,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5652
5833
|
if (skipRerank) {
|
|
5653
5834
|
// Skip LLM reranking — return candidates scored by RRF only
|
|
5654
5835
|
const seenFiles = new Set();
|
|
5655
|
-
|
|
5836
|
+
const rrfResults = candidates
|
|
5656
5837
|
.map((cand, i) => {
|
|
5657
5838
|
const chunkInfo = docChunkMap.get(cand.file);
|
|
5658
5839
|
const bestIdx = chunkInfo?.bestIdx ?? 0;
|
|
@@ -5697,6 +5878,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5697
5878
|
})
|
|
5698
5879
|
.filter(r => r.score >= minScore)
|
|
5699
5880
|
.slice(0, limit);
|
|
5881
|
+
return attachResultMetadata(store.db, rrfResults);
|
|
5700
5882
|
}
|
|
5701
5883
|
// Step 6: Rerank chunks (NOT full bodies)
|
|
5702
5884
|
const chunksToRerank = [];
|
|
@@ -5763,7 +5945,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5763
5945
|
}).sort((a, b) => b.score - a.score);
|
|
5764
5946
|
// Step 8: Dedup by file (safety net — prevents duplicate output)
|
|
5765
5947
|
const seenFiles = new Set();
|
|
5766
|
-
|
|
5948
|
+
const finalResults = blended
|
|
5767
5949
|
.filter(r => {
|
|
5768
5950
|
if (seenFiles.has(r.file))
|
|
5769
5951
|
return false;
|
|
@@ -5772,6 +5954,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5772
5954
|
})
|
|
5773
5955
|
.filter(r => r.score >= minScore)
|
|
5774
5956
|
.slice(0, limit);
|
|
5957
|
+
return attachResultMetadata(store.db, finalResults);
|
|
5775
5958
|
}
|
|
5776
5959
|
/**
|
|
5777
5960
|
* Vector-only semantic search with query expansion.
|
|
@@ -5786,6 +5969,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5786
5969
|
const limit = options?.limit ?? 10;
|
|
5787
5970
|
const minScore = options?.minScore ?? 0.3;
|
|
5788
5971
|
const collection = options?.collection;
|
|
5972
|
+
const filter = options?.filter;
|
|
5789
5973
|
const expansionContext = options?.expansionContext;
|
|
5790
5974
|
const includeHyde = options?.includeHyde ?? true;
|
|
5791
5975
|
if (!hasSearchableVectorIndex(store))
|
|
@@ -5801,7 +5985,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5801
5985
|
const queryTexts = [query, ...vecExpanded.map(q => q.query)];
|
|
5802
5986
|
const allResults = new Map();
|
|
5803
5987
|
for (const q of queryTexts) {
|
|
5804
|
-
const vecResults = await store.searchVec(q, embedModel, limit, collection);
|
|
5988
|
+
const vecResults = await store.searchVec(q, embedModel, limit, collection, undefined, undefined, filter);
|
|
5805
5989
|
for (const r of vecResults) {
|
|
5806
5990
|
const existing = allResults.get(r.filepath);
|
|
5807
5991
|
if (!existing || r.score > existing.score) {
|
|
@@ -5813,6 +5997,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5813
5997
|
score: r.score,
|
|
5814
5998
|
context: store.getContextForFile(r.filepath),
|
|
5815
5999
|
docid: r.docid,
|
|
6000
|
+
metadata: r.metadata,
|
|
5816
6001
|
});
|
|
5817
6002
|
}
|
|
5818
6003
|
}
|
|
@@ -5849,6 +6034,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5849
6034
|
const skipRerank = options?.skipRerank ?? false;
|
|
5850
6035
|
const hooks = options?.hooks;
|
|
5851
6036
|
const collections = options?.collections;
|
|
6037
|
+
const filter = options?.filter;
|
|
5852
6038
|
if (searches.length === 0)
|
|
5853
6039
|
return [];
|
|
5854
6040
|
// Validate queries before executing
|
|
@@ -5881,7 +6067,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5881
6067
|
for (const [searchIndex, search] of searches.entries()) {
|
|
5882
6068
|
if (search.type === 'lex') {
|
|
5883
6069
|
for (const coll of collectionList) {
|
|
5884
|
-
const ftsResults = store.searchFTS(search.query, 20, coll);
|
|
6070
|
+
const ftsResults = store.searchFTS(search.query, 20, coll, filter);
|
|
5885
6071
|
if (ftsResults.length > 0) {
|
|
5886
6072
|
for (const r of ftsResults)
|
|
5887
6073
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5914,7 +6100,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5914
6100
|
if (!embedding || embedding.length === 0)
|
|
5915
6101
|
continue;
|
|
5916
6102
|
for (const coll of collectionList) {
|
|
5917
|
-
const vecResults = await store.searchVec(vecSearches[i].search.query, embedModel, 20, coll, undefined, embedding);
|
|
6103
|
+
const vecResults = await store.searchVec(vecSearches[i].search.query, embedModel, 20, coll, undefined, embedding, filter);
|
|
5918
6104
|
if (vecResults.length > 0) {
|
|
5919
6105
|
for (const r of vecResults)
|
|
5920
6106
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5977,7 +6163,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5977
6163
|
if (skipRerank) {
|
|
5978
6164
|
// Skip LLM reranking — return candidates scored by RRF only
|
|
5979
6165
|
const seenFiles = new Set();
|
|
5980
|
-
|
|
6166
|
+
const rrfResults = candidates
|
|
5981
6167
|
.map((cand, i) => {
|
|
5982
6168
|
const chunkInfo = docChunkMap.get(cand.file);
|
|
5983
6169
|
const bestIdx = chunkInfo?.bestIdx ?? 0;
|
|
@@ -6022,6 +6208,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6022
6208
|
})
|
|
6023
6209
|
.filter(r => r.score >= minScore)
|
|
6024
6210
|
.slice(0, limit);
|
|
6211
|
+
return attachResultMetadata(store.db, rrfResults);
|
|
6025
6212
|
}
|
|
6026
6213
|
// Step 5: Rerank chunks
|
|
6027
6214
|
const chunksToRerank = [];
|
|
@@ -6087,7 +6274,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6087
6274
|
}).sort((a, b) => b.score - a.score);
|
|
6088
6275
|
// Step 7: Dedup by file
|
|
6089
6276
|
const seenFiles = new Set();
|
|
6090
|
-
|
|
6277
|
+
const finalResults = blended
|
|
6091
6278
|
.filter(r => {
|
|
6092
6279
|
if (seenFiles.has(r.file))
|
|
6093
6280
|
return false;
|
|
@@ -6096,4 +6283,5 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6096
6283
|
})
|
|
6097
6284
|
.filter(r => r.score >= minScore)
|
|
6098
6285
|
.slice(0, limit);
|
|
6286
|
+
return attachResultMetadata(store.db, finalResults);
|
|
6099
6287
|
}
|