@wei840222/qmd 2026.9.6 → 2026.9.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +84 -1
- package/dist/cli/build-info.json +2 -2
- package/dist/cli/qmd.js +141 -11
- package/dist/collections.d.ts +3 -0
- package/dist/collections.js +9 -4
- package/dist/{hybrid-llm.d.ts → hybrid.d.ts} +9 -3
- package/dist/hybrid.js +97 -0
- package/dist/index.d.ts +10 -0
- package/dist/index.js +28 -4
- package/dist/llm.d.ts +19 -1
- package/dist/llm.js +28 -5
- package/dist/mcp/server.js +70 -6
- package/dist/metadata-filter.d.ts +74 -0
- package/dist/metadata-filter.js +279 -0
- package/dist/metadata-store.d.ts +54 -0
- package/dist/metadata-store.js +194 -0
- package/dist/metadata.d.ts +61 -0
- package/dist/metadata.js +221 -0
- package/dist/remote-jev.d.ts +38 -0
- package/dist/remote-jev.js +167 -0
- package/dist/remote-llm.d.ts +3 -1
- package/dist/remote-llm.js +21 -4
- package/dist/search/query-expansion.d.ts +1 -0
- package/dist/search/query-expansion.js +4 -1
- package/dist/search/zh-dict.txt +3 -0
- package/dist/store.d.ts +29 -16
- package/dist/store.js +349 -166
- package/package.json +3 -2
- package/scripts/sync-zh-dict.mjs +4 -1
- package/skills/qmd/SKILL.md +11 -0
- package/dist/hybrid-llm.js +0 -53
- package/skills/release/SKILL.md +0 -141
- package/skills/release/scripts/install-hooks.sh +0 -38
- package/skills/release/scripts/release-context.sh +0 -129
package/dist/store.js
CHANGED
|
@@ -22,9 +22,12 @@ import fastGlob from "fast-glob";
|
|
|
22
22
|
import { qmdHomedir } from "./paths.js";
|
|
23
23
|
import { cleanupExpiredCjkIndexBuilds, getCjkAnalyzerFingerprint, getCjkLexicalIndexState, initializeCjkLexicalIndexSchema, repairDirtyCjkCharFallback, runCjkSynchronizedMutation, } from "./search/cjk-index.js";
|
|
24
24
|
import { analyzeCjkSync, containsCjk } from "./search/cjk-analyzer.js";
|
|
25
|
-
import { ExpansionPolicyError, parseExpansionDirective, resolveExpansionPolicy, } from "./search/query-expansion.js";
|
|
25
|
+
import { ExpansionPolicyError, parseExpansionDirective, resolveExpansionPolicy, containsRelativeTemporalTerms, } from "./search/query-expansion.js";
|
|
26
26
|
import { LlamaCpp, getDefaultLlamaCpp, formatQueryForEmbedding, formatDocForEmbedding, withLLMSessionForLlm, DEFAULT_EMBED_MODEL_URI, DEFAULT_RERANK_MODEL_URI, DEFAULT_GENERATE_MODEL_URI, } from "./llm.js";
|
|
27
27
|
const readOnlyDatabases = new WeakSet();
|
|
28
|
+
import { METADATA_EXTRACTION_VERSION } from "./metadata.js";
|
|
29
|
+
import { compileMetadataFilter } from "./metadata-filter.js";
|
|
30
|
+
import { initializeMetadataSchema, syncDocumentMetadata, countDocumentsPendingMetadata, getMetadataByFilepath, parseMetadataJson, } from "./metadata-store.js";
|
|
28
31
|
// =============================================================================
|
|
29
32
|
// Configuration
|
|
30
33
|
// =============================================================================
|
|
@@ -925,18 +928,15 @@ export function normalizeCjkForFTS(text) {
|
|
|
925
928
|
return text.replace(CJK_RUN_PATTERN, run => ` ${Array.from(run).join(' ')} `);
|
|
926
929
|
}
|
|
927
930
|
function sanitizeFTS5Phrase(phrase) {
|
|
928
|
-
//
|
|
929
|
-
//
|
|
930
|
-
//
|
|
931
|
+
// A quoted phrase is matched against tokens the porter unicode61 tokenizer
|
|
932
|
+
// produced, and that tokenizer splits document text on every separator.
|
|
933
|
+
// Deleting the separators here instead would collapse "1.0.21" to "1021" and
|
|
934
|
+
// "PIO-1384" to "pio1384", tokens no document holds, so the query returns
|
|
935
|
+
// nothing with no error (#757 for dots, #916 for the rest). Split on the same
|
|
936
|
+
// separators the tokenizer does and emit the parts as adjacent phrase terms.
|
|
931
937
|
return normalizeCjkForFTS(phrase)
|
|
932
938
|
.split(/\s+/)
|
|
933
|
-
.flatMap(t =>
|
|
934
|
-
if (isDottedToken(t)) {
|
|
935
|
-
return t.split('.').map(p => sanitizeFTS5Term(p)).filter(p => p);
|
|
936
|
-
}
|
|
937
|
-
const sanitized = sanitizeFTS5Term(t);
|
|
938
|
-
return sanitized ? [sanitized] : [];
|
|
939
|
-
})
|
|
939
|
+
.flatMap(t => splitFTS5CompoundTerm(t))
|
|
940
940
|
.join(' ');
|
|
941
941
|
}
|
|
942
942
|
function getUserVersion(db) {
|
|
@@ -1402,6 +1402,9 @@ function initializeDatabase(db) {
|
|
|
1402
1402
|
`);
|
|
1403
1403
|
ensureEmbeddingIdentitySchema(db);
|
|
1404
1404
|
ensureContentVectorsStatusIndex(db);
|
|
1405
|
+
// Document metadata — extraction state plus normalized value rows for
|
|
1406
|
+
// metadata filtering. Keyed by document identity, not content hash.
|
|
1407
|
+
initializeMetadataSchema(db);
|
|
1405
1408
|
// Store collections — makes the DB self-contained (no external config needed)
|
|
1406
1409
|
db.exec(`
|
|
1407
1410
|
CREATE TABLE IF NOT EXISTS store_collections (
|
|
@@ -1730,8 +1733,9 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1730
1733
|
return !parts.some(part => part.startsWith("."));
|
|
1731
1734
|
});
|
|
1732
1735
|
const total = files.length;
|
|
1733
|
-
let indexed = 0, updated = 0, unchanged = 0, processed = 0;
|
|
1736
|
+
let indexed = 0, updated = 0, unchanged = 0, processed = 0, metadataErrors = 0;
|
|
1734
1737
|
const skippedFiles = [];
|
|
1738
|
+
const metadataErrorFiles = [];
|
|
1735
1739
|
const seenPaths = new Set();
|
|
1736
1740
|
// Literal paths of every file in this scan. Passed to the legacy-path
|
|
1737
1741
|
// migration so it never adopts a row that still belongs to a live file.
|
|
@@ -1772,8 +1776,12 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1772
1776
|
const hash = await hashContent(content);
|
|
1773
1777
|
const title = extractTitle(content, relativeFile);
|
|
1774
1778
|
const existing = findOrMigrateLegacyDocument(db, collectionName, path, livePaths);
|
|
1779
|
+
let documentId;
|
|
1780
|
+
let contentChanged = true;
|
|
1775
1781
|
if (existing) {
|
|
1782
|
+
documentId = existing.id;
|
|
1776
1783
|
if (existing.hash === hash) {
|
|
1784
|
+
contentChanged = false;
|
|
1777
1785
|
if (existing.title !== title) {
|
|
1778
1786
|
updateDocumentTitle(db, existing.id, title, now);
|
|
1779
1787
|
updated++;
|
|
@@ -1791,7 +1799,13 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1791
1799
|
else {
|
|
1792
1800
|
indexed++;
|
|
1793
1801
|
const stat = statSync(filepath);
|
|
1794
|
-
insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1802
|
+
documentId = insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1803
|
+
}
|
|
1804
|
+
// Unchanged content still backfills missing or stale extraction state.
|
|
1805
|
+
const extraction = syncDocumentMetadata(db, documentId, content, path, contentChanged ? undefined : { onlyIfStale: true });
|
|
1806
|
+
if (extraction?.error) {
|
|
1807
|
+
metadataErrors++;
|
|
1808
|
+
metadataErrorFiles.push({ file: relativeFile, error: extraction.error });
|
|
1795
1809
|
}
|
|
1796
1810
|
processed++;
|
|
1797
1811
|
options?.onProgress?.({ file: relativeFile, current: processed, total });
|
|
@@ -1806,7 +1820,7 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1806
1820
|
}
|
|
1807
1821
|
}
|
|
1808
1822
|
const orphanedCleaned = cleanupOrphanedContent(db);
|
|
1809
|
-
return { indexed, updated, unchanged, removed, orphanedCleaned, skipped: skippedFiles.length, skippedFiles };
|
|
1823
|
+
return { indexed, updated, unchanged, removed, orphanedCleaned, skipped: skippedFiles.length, skippedFiles, metadataErrors, metadataErrorFiles };
|
|
1810
1824
|
}
|
|
1811
1825
|
function validatePositiveIntegerOption(name, value, fallback) {
|
|
1812
1826
|
if (value === undefined)
|
|
@@ -2464,7 +2478,7 @@ export async function generateEmbeddings(store, options) {
|
|
|
2464
2478
|
pos: chunk.pos,
|
|
2465
2479
|
tokens: chunk.tokenUpperBound,
|
|
2466
2480
|
}))
|
|
2467
|
-
: await
|
|
2481
|
+
: await chunkDocumentByTokensWithLlm((llm ?? getLlm(store) ?? getDefaultLlamaCpp()), doc.body, undefined, undefined, undefined, doc.path, options?.chunkStrategy, session.signal);
|
|
2468
2482
|
if (embeddingLease) {
|
|
2469
2483
|
pruneEmbeddingRowsOutsideLayout(db, doc.hash, model, fingerprint, chunks.length, embeddingLease);
|
|
2470
2484
|
}
|
|
@@ -2641,11 +2655,11 @@ export function createStore(dbPath, options = {}) {
|
|
|
2641
2655
|
const commitPendingDocumentInsert = (collectionName, path, title, hash, createdAt, modifiedAt) => {
|
|
2642
2656
|
const pending = pendingContent.get(hash);
|
|
2643
2657
|
if (!pending) {
|
|
2644
|
-
insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt);
|
|
2645
|
-
return;
|
|
2658
|
+
return insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt);
|
|
2646
2659
|
}
|
|
2647
|
-
insertDocumentWithContent(db, hash, pending.content, pending.createdAt, collectionName, path, title, createdAt, modifiedAt);
|
|
2660
|
+
const docId = insertDocumentWithContent(db, hash, pending.content, pending.createdAt, collectionName, path, title, createdAt, modifiedAt);
|
|
2648
2661
|
pendingContent.delete(hash);
|
|
2662
|
+
return docId;
|
|
2649
2663
|
};
|
|
2650
2664
|
const commitPendingDocumentUpdate = (documentId, title, hash, modifiedAt) => {
|
|
2651
2665
|
const pending = pendingContent.get(hash);
|
|
@@ -2693,9 +2707,9 @@ export function createStore(dbPath, options = {}) {
|
|
|
2693
2707
|
resolveVirtualPath: (virtualPath) => resolveVirtualPath(db, virtualPath),
|
|
2694
2708
|
toVirtualPath: (absolutePath) => toVirtualPath(db, absolutePath),
|
|
2695
2709
|
// Search
|
|
2696
|
-
searchCharFTS: (query, limit, collectionName) => searchCharFTS(db, query, limit, collectionName),
|
|
2697
|
-
searchFTS: (query, limit, collectionName) => searchFTS(db, query, limit, collectionName),
|
|
2698
|
-
searchVec: (query, model, limit, collectionFilter, session, precomputedEmbedding) => searchVec(db, query, model, limit, collectionFilter, session, precomputedEmbedding, store.embeddingProvider, store.authorizeRemoteRequest, store.llm),
|
|
2710
|
+
searchCharFTS: (query, limit, collectionName, filter) => searchCharFTS(db, query, limit, collectionName, filter),
|
|
2711
|
+
searchFTS: (query, limit, collectionName, filter) => searchFTS(db, query, limit, collectionName, filter),
|
|
2712
|
+
searchVec: (query, model, limit, collectionFilter, session, precomputedEmbedding, filter) => searchVec(db, query, model, limit, collectionFilter, session, precomputedEmbedding, store.embeddingProvider, store.authorizeRemoteRequest, store.llm, filter),
|
|
2699
2713
|
// Query expansion & reranking
|
|
2700
2714
|
expandQuery: (query, model, expansionContext, options) => expandQuery(query, model ?? store.localLlm?.generateModelName ?? store.llm?.generateModelName ?? DEFAULT_QUERY_MODEL, db, expansionContext, store.llm, options),
|
|
2701
2715
|
invalidateExpansionCache: (query, expansionContext, options) => deleteExpansionCacheEntry(db, query, store.localLlm?.generateModelName ?? store.llm?.generateModelName ?? DEFAULT_QUERY_MODEL, expansionContext, options),
|
|
@@ -2724,6 +2738,30 @@ export function createStore(dbPath, options = {}) {
|
|
|
2724
2738
|
getActiveDocumentPaths: (collectionName) => getActiveDocumentPaths(db, collectionName),
|
|
2725
2739
|
// Vector/embedding operations
|
|
2726
2740
|
getHashesForEmbedding: () => getHashesForEmbedding(db),
|
|
2741
|
+
insertEmbedding: (hash, seq, pos, embedding, model, embeddedAt, totalChunks, fingerprint, lease) => {
|
|
2742
|
+
const state = db.prepare(`
|
|
2743
|
+
SELECT status, fingerprint, model, generation, lease_owner, lease_expires_at
|
|
2744
|
+
FROM embedding_index_state
|
|
2745
|
+
WHERE singleton = 1
|
|
2746
|
+
`).get();
|
|
2747
|
+
if (!state && !lease) {
|
|
2748
|
+
withLazyContentVectorMigration(db, () => {
|
|
2749
|
+
db.prepare(`
|
|
2750
|
+
INSERT OR REPLACE INTO content_vectors (hash, seq, pos, model, embed_fingerprint, total_chunks, embedded_at)
|
|
2751
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
2752
|
+
`).run(hash, seq, pos, model, fingerprint ?? "", totalChunks ?? 1, embeddedAt);
|
|
2753
|
+
const hasCollectionCol = db.prepare(`SELECT sql FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get();
|
|
2754
|
+
if (hasCollectionCol?.sql?.includes("collection")) {
|
|
2755
|
+
db.prepare(`INSERT OR REPLACE INTO vectors_vec (hash_seq, collection, embedding) VALUES (?, ?, ?)`).run(`${hash}_${seq}`, "", embedding);
|
|
2756
|
+
}
|
|
2757
|
+
else {
|
|
2758
|
+
db.prepare(`INSERT OR REPLACE INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_${seq}`, embedding);
|
|
2759
|
+
}
|
|
2760
|
+
});
|
|
2761
|
+
return;
|
|
2762
|
+
}
|
|
2763
|
+
insertEmbedding(db, hash, seq, pos, embedding, model, embeddedAt, totalChunks, fingerprint, lease);
|
|
2764
|
+
},
|
|
2727
2765
|
};
|
|
2728
2766
|
return store;
|
|
2729
2767
|
}
|
|
@@ -2891,7 +2929,7 @@ export async function maybeAdoptLegacyEmbeddingFingerprint(store, model = DEFAUL
|
|
|
2891
2929
|
const title = extractTitle(sample.body, sample.path);
|
|
2892
2930
|
const llm = getLlm(store);
|
|
2893
2931
|
return await withLLMSessionForLlm(llm, async (session) => {
|
|
2894
|
-
const chunks = await
|
|
2932
|
+
const chunks = await chunkDocumentByTokensWithLlm(llm, sample.body, undefined, undefined, undefined, sample.path, undefined, session.signal);
|
|
2895
2933
|
const chunk = chunks[sample.seq];
|
|
2896
2934
|
if (!chunk) {
|
|
2897
2935
|
return { checked: true, adopted: 0, reason: `sample chunk ${expectedHashSeq} no longer exists` };
|
|
@@ -3197,9 +3235,10 @@ function rebuildDocumentFTS(db, documentId) {
|
|
|
3197
3235
|
}
|
|
3198
3236
|
/**
|
|
3199
3237
|
* Insert a new document into the documents table.
|
|
3238
|
+
* Returns the document's id so callers can attach document-scoped state.
|
|
3200
3239
|
*/
|
|
3201
3240
|
export function insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt) {
|
|
3202
|
-
runCjkSynchronizedMutation(db, () => {
|
|
3241
|
+
return runCjkSynchronizedMutation(db, () => {
|
|
3203
3242
|
db.prepare(`
|
|
3204
3243
|
INSERT INTO documents (collection, path, title, hash, created_at, modified_at, active)
|
|
3205
3244
|
VALUES (?, ?, ?, ?, ?, ?, 1)
|
|
@@ -3210,15 +3249,17 @@ export function insertDocument(db, collectionName, path, title, hash, createdAt,
|
|
|
3210
3249
|
active = 1
|
|
3211
3250
|
`).run(collectionName, path, title, hash, createdAt, modifiedAt);
|
|
3212
3251
|
const row = db.prepare(`SELECT id FROM documents WHERE collection = ? AND path = ?`).get(collectionName, path);
|
|
3213
|
-
if (row)
|
|
3214
|
-
|
|
3252
|
+
if (!row)
|
|
3253
|
+
throw new Error(`Document row missing after insert: ${collectionName}/${path}`);
|
|
3254
|
+
rebuildDocumentFTS(db, row.id);
|
|
3255
|
+
return row.id;
|
|
3215
3256
|
});
|
|
3216
3257
|
}
|
|
3217
3258
|
/** Insert immutable content and its document row in the same lexical transaction. */
|
|
3218
3259
|
export function insertDocumentWithContent(db, hash, content, contentCreatedAt, collectionName, path, title, documentCreatedAt, modifiedAt) {
|
|
3219
|
-
runCjkSynchronizedMutation(db, () => {
|
|
3260
|
+
return runCjkSynchronizedMutation(db, () => {
|
|
3220
3261
|
insertContent(db, hash, content, contentCreatedAt);
|
|
3221
|
-
insertDocument(db, collectionName, path, title, hash, documentCreatedAt, modifiedAt);
|
|
3262
|
+
return insertDocument(db, collectionName, path, title, hash, documentCreatedAt, modifiedAt);
|
|
3222
3263
|
});
|
|
3223
3264
|
}
|
|
3224
3265
|
/**
|
|
@@ -3425,8 +3466,7 @@ function stripUnpairedSurrogates(text) {
|
|
|
3425
3466
|
* When filepath and chunkStrategy are provided, uses AST-aware break points
|
|
3426
3467
|
* for supported code files.
|
|
3427
3468
|
*/
|
|
3428
|
-
|
|
3429
|
-
const llm = getDefaultLlamaCpp();
|
|
3469
|
+
async function chunkDocumentByTokensWithLlm(llm, content, maxTokens = CHUNK_SIZE_TOKENS, overlapTokens = CHUNK_OVERLAP_TOKENS, windowTokens = CHUNK_WINDOW_TOKENS, filepath, chunkStrategy = "regex", signal) {
|
|
3430
3470
|
// Use moderate chars/token estimate (prose ~4, code ~2, mixed ~3)
|
|
3431
3471
|
// If chunks exceed limit, they'll be re-split with actual ratio
|
|
3432
3472
|
const avgCharsPerToken = 3;
|
|
@@ -3500,6 +3540,9 @@ export async function chunkDocumentByTokens(content, maxTokens = CHUNK_SIZE_TOKE
|
|
|
3500
3540
|
}
|
|
3501
3541
|
return results;
|
|
3502
3542
|
}
|
|
3543
|
+
export async function chunkDocumentByTokens(content, maxTokens = CHUNK_SIZE_TOKENS, overlapTokens = CHUNK_OVERLAP_TOKENS, windowTokens = CHUNK_WINDOW_TOKENS, filepath, chunkStrategy = "regex", signal) {
|
|
3544
|
+
return await chunkDocumentByTokensWithLlm(getDefaultLlamaCpp(), content, maxTokens, overlapTokens, windowTokens, filepath, chunkStrategy, signal);
|
|
3545
|
+
}
|
|
3503
3546
|
// =============================================================================
|
|
3504
3547
|
// Fuzzy matching
|
|
3505
3548
|
// =============================================================================
|
|
@@ -3962,38 +4005,29 @@ export function sanitizeFTS5Term(term) {
|
|
|
3962
4005
|
return term.replace(/[^\p{L}\p{N}'_]/gu, '').toLowerCase();
|
|
3963
4006
|
}
|
|
3964
4007
|
/**
|
|
3965
|
-
*
|
|
3966
|
-
*
|
|
3967
|
-
|
|
3968
|
-
|
|
3969
|
-
|
|
3970
|
-
|
|
3971
|
-
|
|
3972
|
-
*
|
|
3973
|
-
* and sanitizing each part. Returns the parts joined by spaces for use
|
|
3974
|
-
* inside FTS5 quotes: "multi agent" matches "multi-agent" in porter tokenizer.
|
|
4008
|
+
* A run of characters the FTS tokenizer treats as a separator.
|
|
4009
|
+
*
|
|
4010
|
+
* `documents_fts` is tokenized with `porter unicode61`, which starts a new
|
|
4011
|
+
* token at every character that is not a letter or a digit. Underscore is one
|
|
4012
|
+
* of those, but it is deliberately kept here rather than split on: FTS5 applies
|
|
4013
|
+
* the same tokenizer to a quoted phrase, so leaving `apply_secrets` intact lets
|
|
4014
|
+
* it split symmetrically into `apply secrets` on both sides, and that is the
|
|
4015
|
+
* behaviour #305 shipped. The apostrophe is kept for the same reason.
|
|
3975
4016
|
*/
|
|
3976
|
-
|
|
3977
|
-
return term.split('-').map(t => sanitizeFTS5Term(t)).filter(t => t).join(' ');
|
|
3978
|
-
}
|
|
4017
|
+
const FTS5_SEPARATOR_RUN = /[^\p{L}\p{N}'_]+/u;
|
|
3979
4018
|
/**
|
|
3980
|
-
*
|
|
3981
|
-
*
|
|
3982
|
-
*
|
|
3983
|
-
*
|
|
3984
|
-
|
|
3985
|
-
|
|
3986
|
-
|
|
3987
|
-
|
|
3988
|
-
|
|
3989
|
-
/**
|
|
3990
|
-
* Sanitize a dotted term into individual FTS5 tokens joined with AND.
|
|
3991
|
-
* e.g. "2026.4.10" → '"2026"* AND "4"* AND "10"*'
|
|
3992
|
-
* The AND ensures all parts must appear, matching how the porter tokenizer
|
|
3993
|
-
* indexes dotted strings.
|
|
4019
|
+
* Split one query term the way the tokenizer split the document text, and
|
|
4020
|
+
* sanitize each part.
|
|
4021
|
+
*
|
|
4022
|
+
* `PIO-1384` becomes ["pio", "1384"], `src/lib/i18n.ts` becomes
|
|
4023
|
+
* ["src", "lib", "i18n", "ts"], and a term with no separator in it comes back
|
|
4024
|
+
* as a single part. Callers join the parts into an FTS5 phrase, which is what
|
|
4025
|
+
* makes the parts have to be adjacent in the document rather than merely all
|
|
4026
|
+
* present. Parts that sanitize to nothing are dropped, so a term that is all
|
|
4027
|
+
* punctuation yields an empty list and the caller skips it.
|
|
3994
4028
|
*/
|
|
3995
|
-
function
|
|
3996
|
-
return term.split(
|
|
4029
|
+
function splitFTS5CompoundTerm(term) {
|
|
4030
|
+
return term.split(FTS5_SEPARATOR_RUN).map(p => sanitizeFTS5Term(p)).filter(p => p);
|
|
3997
4031
|
}
|
|
3998
4032
|
/**
|
|
3999
4033
|
* Parse lex query syntax into FTS5 query.
|
|
@@ -4001,7 +4035,8 @@ function sanitizeDottedTerm(term) {
|
|
|
4001
4035
|
* Supports:
|
|
4002
4036
|
* - Quoted phrases: "exact phrase" → "exact phrase" (exact match)
|
|
4003
4037
|
* - Negation: -term or -"phrase" → uses FTS5 NOT operator
|
|
4004
|
-
* -
|
|
4038
|
+
* - Terms holding a separator: multi-agent, DEC-0054, gpt-4, 2026.4.10,
|
|
4039
|
+
* src/lib/i18n.ts, @tobilu/qmd → treated as phrases over their parts
|
|
4005
4040
|
* - Plain terms: term → "term"* (prefix match)
|
|
4006
4041
|
*
|
|
4007
4042
|
* FTS5 NOT is a binary operator: `term1 NOT term2` means "match term1 but not term2".
|
|
@@ -4018,6 +4053,8 @@ function sanitizeDottedTerm(term) {
|
|
|
4018
4053
|
* multi-agent memory → "multi agent" AND "memory"*
|
|
4019
4054
|
* DEC-0054 → "dec 0054"
|
|
4020
4055
|
* -multi-agent → NOT "multi agent"
|
|
4056
|
+
* "DEC-0054" → "dec 0054"
|
|
4057
|
+
* src/lib/i18n.ts → "src lib i18n ts"
|
|
4021
4058
|
*/
|
|
4022
4059
|
function buildFTS5Query(query) {
|
|
4023
4060
|
const positive = [];
|
|
@@ -4061,41 +4098,7 @@ function buildFTS5Query(query) {
|
|
|
4061
4098
|
while (i < s.length && !/[\s"]/.test(s[i]))
|
|
4062
4099
|
i++;
|
|
4063
4100
|
const term = s.slice(start, i);
|
|
4064
|
-
|
|
4065
|
-
// These get split into phrase queries so FTS5 porter tokenizer matches them.
|
|
4066
|
-
if (isHyphenatedToken(term)) {
|
|
4067
|
-
const sanitized = sanitizeHyphenatedTerm(term);
|
|
4068
|
-
if (sanitized) {
|
|
4069
|
-
const ftsPhrase = `"${sanitized}"`; // Phrase match (no prefix)
|
|
4070
|
-
if (negated) {
|
|
4071
|
-
negative.push(ftsPhrase);
|
|
4072
|
-
}
|
|
4073
|
-
else {
|
|
4074
|
-
positive.push(ftsPhrase);
|
|
4075
|
-
}
|
|
4076
|
-
}
|
|
4077
|
-
}
|
|
4078
|
-
else if (isDottedToken(term)) {
|
|
4079
|
-
// Handle dotted version strings: 2026.4.10, 3.14.0, v1.2.3
|
|
4080
|
-
// The porter tokenizer splits on dots, so the index has individual tokens.
|
|
4081
|
-
// We AND all parts together so the query matches documents containing all parts.
|
|
4082
|
-
const sanitized = sanitizeDottedTerm(term);
|
|
4083
|
-
if (sanitized) {
|
|
4084
|
-
// sanitizeDottedTerm already wraps each part in quotes with prefix match
|
|
4085
|
-
if (negated) {
|
|
4086
|
-
// Wrap multi-token AND expression in parens for NOT negation
|
|
4087
|
-
negative.push(`(${sanitized})`);
|
|
4088
|
-
}
|
|
4089
|
-
else {
|
|
4090
|
-
// Flatten individual AND'd terms into the positive list so they combine
|
|
4091
|
-
// correctly with other terms (avoids double-wrapping in outer AND).
|
|
4092
|
-
for (const part of sanitized.split(' AND ')) {
|
|
4093
|
-
positive.push(part.trim());
|
|
4094
|
-
}
|
|
4095
|
-
}
|
|
4096
|
-
}
|
|
4097
|
-
}
|
|
4098
|
-
else if (containsCjk(term)) {
|
|
4101
|
+
if (containsCjk(term)) {
|
|
4099
4102
|
const sanitized = sanitizeFTS5Phrase(term);
|
|
4100
4103
|
if (sanitized) {
|
|
4101
4104
|
const ftsPhrase = `"${sanitized}"`; // CJK phrase over character tokens
|
|
@@ -4108,9 +4111,16 @@ function buildFTS5Query(query) {
|
|
|
4108
4111
|
}
|
|
4109
4112
|
}
|
|
4110
4113
|
else {
|
|
4111
|
-
|
|
4112
|
-
|
|
4113
|
-
|
|
4114
|
+
// Any separator inside the term (multi-agent, DEC-0054, 2026.4.10,
|
|
4115
|
+
// src/lib/i18n.ts, @tobilu/qmd) split it at index time too, so the term
|
|
4116
|
+
// has to be matched as the phrase those parts form. A term with no
|
|
4117
|
+
// separator is one part and keeps its prefix match, which is what makes
|
|
4118
|
+
// a plain word still match longer words that start with it.
|
|
4119
|
+
const parts = splitFTS5CompoundTerm(term);
|
|
4120
|
+
if (parts.length > 0) {
|
|
4121
|
+
const ftsTerm = parts.length > 1
|
|
4122
|
+
? `"${parts.join(' ')}"` // Phrase match (no prefix)
|
|
4123
|
+
: `"${parts[0]}"*`; // Prefix match
|
|
4114
4124
|
if (negated) {
|
|
4115
4125
|
negative.push(ftsTerm);
|
|
4116
4126
|
}
|
|
@@ -4170,7 +4180,28 @@ function normalizeCollectionFilter(filter) {
|
|
|
4170
4180
|
const names = typeof filter === "string" ? [filter] : filter;
|
|
4171
4181
|
return [...new Set(names.filter(name => name.length > 0))];
|
|
4172
4182
|
}
|
|
4173
|
-
function
|
|
4183
|
+
function scopedCollectionNames(scope) {
|
|
4184
|
+
if (scope == null)
|
|
4185
|
+
return undefined;
|
|
4186
|
+
const names = (typeof scope === "string" ? [scope] : Array.from(scope))
|
|
4187
|
+
.map(n => n.trim())
|
|
4188
|
+
.filter(n => n.length > 0);
|
|
4189
|
+
return names.length > 0 ? names : undefined;
|
|
4190
|
+
}
|
|
4191
|
+
function mergeSearchResultsByScore(lists, limit) {
|
|
4192
|
+
const best = new Map();
|
|
4193
|
+
for (const list of lists) {
|
|
4194
|
+
for (const r of list) {
|
|
4195
|
+
const prev = best.get(r.filepath);
|
|
4196
|
+
if (!prev || r.score > prev.score)
|
|
4197
|
+
best.set(r.filepath, r);
|
|
4198
|
+
}
|
|
4199
|
+
}
|
|
4200
|
+
return Array.from(best.values())
|
|
4201
|
+
.sort((a, b) => b.score - a.score)
|
|
4202
|
+
.slice(0, limit);
|
|
4203
|
+
}
|
|
4204
|
+
function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter, filter) {
|
|
4174
4205
|
// Keep collection membership inside the ranked FTS candidate set. Filtering
|
|
4175
4206
|
// after LIMIT can lose every matching row from a smaller collection.
|
|
4176
4207
|
const params = [ftsQuery];
|
|
@@ -4182,7 +4213,10 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4182
4213
|
? `AND filtered_d.active = 1 AND filtered_d.collection IN (${collections.map(() => "?").join(", ")})`
|
|
4183
4214
|
: "";
|
|
4184
4215
|
params.push(...collections);
|
|
4185
|
-
|
|
4216
|
+
// When filtering by metadata, fetch extra candidates from the FTS index
|
|
4217
|
+
// since some will be filtered out. Without a filter we can fetch exactly the requested limit.
|
|
4218
|
+
const ftsLimit = filter ? limit * 10 : limit;
|
|
4219
|
+
params.push(ftsLimit);
|
|
4186
4220
|
let sql = `
|
|
4187
4221
|
WITH fts_matches AS (
|
|
4188
4222
|
SELECT ${table}.rowid AS rowid, bm25(${table}, 1.5, 4.0, 1.0) as bm25_score
|
|
@@ -4199,12 +4233,21 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4199
4233
|
d.title,
|
|
4200
4234
|
content.doc as body,
|
|
4201
4235
|
d.hash,
|
|
4202
|
-
fm.bm25_score
|
|
4236
|
+
fm.bm25_score,
|
|
4237
|
+
dm.metadata_json
|
|
4203
4238
|
FROM fts_matches fm
|
|
4204
4239
|
JOIN documents d ON d.id = fm.rowid
|
|
4205
4240
|
JOIN content ON content.hash = d.hash
|
|
4241
|
+
LEFT JOIN document_metadata dm ON dm.document_id = d.id
|
|
4206
4242
|
WHERE d.active = 1
|
|
4207
4243
|
`;
|
|
4244
|
+
if (filter) {
|
|
4245
|
+
// Only documents with current, error-free extraction can match — an
|
|
4246
|
+
// unprocessed document must not accidentally satisfy `exists: false`.
|
|
4247
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4248
|
+
sql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`;
|
|
4249
|
+
params.push(...compiledFilter.params);
|
|
4250
|
+
}
|
|
4208
4251
|
// bm25 lower is better; sort ascending.
|
|
4209
4252
|
sql += ` ORDER BY fm.bm25_score ASC LIMIT ?`;
|
|
4210
4253
|
params.push(limit);
|
|
@@ -4227,6 +4270,7 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4227
4270
|
bodyLength: row.body.length,
|
|
4228
4271
|
body: row.body,
|
|
4229
4272
|
context: getContextForFile(db, row.filepath),
|
|
4273
|
+
metadata: parseMetadataJson(row.metadata_json),
|
|
4230
4274
|
score,
|
|
4231
4275
|
source: "fts",
|
|
4232
4276
|
};
|
|
@@ -4238,17 +4282,21 @@ function buildCjkSignalQuery(signal) {
|
|
|
4238
4282
|
return null;
|
|
4239
4283
|
return terms.map(term => `"${term.replace(/"/g, '""')}"`).join(" AND ");
|
|
4240
4284
|
}
|
|
4241
|
-
export function searchCharFTS(db, query, limit = 20, collectionFilter) {
|
|
4285
|
+
export function searchCharFTS(db, query, limit = 20, collectionFilter, filter) {
|
|
4242
4286
|
if (!readOnlyDatabases.has(db))
|
|
4243
4287
|
repairDirtyCjkCharFallback(db);
|
|
4244
4288
|
const charQuery = buildFTS5Query(query);
|
|
4245
4289
|
if (!charQuery)
|
|
4246
4290
|
return [];
|
|
4247
|
-
return searchFtsChannel(db, "documents_fts", charQuery, limit, collectionFilter);
|
|
4291
|
+
return searchFtsChannel(db, "documents_fts", charQuery, limit, collectionFilter, filter);
|
|
4248
4292
|
}
|
|
4249
|
-
export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
4293
|
+
export function searchFTS(db, query, limit = 20, collectionFilter, filter) {
|
|
4294
|
+
const names = scopedCollectionNames(collectionFilter);
|
|
4295
|
+
if (names && names.length > 1) {
|
|
4296
|
+
return mergeSearchResultsByScore(names.map(name => searchFTS(db, query, limit, name, filter)), limit);
|
|
4297
|
+
}
|
|
4250
4298
|
if (!containsCjk(query))
|
|
4251
|
-
return searchCharFTS(db, query, limit, collectionFilter);
|
|
4299
|
+
return searchCharFTS(db, query, limit, collectionFilter, filter);
|
|
4252
4300
|
const charQuery = buildFTS5Query(query);
|
|
4253
4301
|
if (!charQuery)
|
|
4254
4302
|
return [];
|
|
@@ -4257,7 +4305,7 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4257
4305
|
const candidateDepth = Math.max(limit, CJK_LEXICAL_CANDIDATE_DEPTH);
|
|
4258
4306
|
const rankedChannels = [{
|
|
4259
4307
|
channel: "char",
|
|
4260
|
-
results: searchCharFTS(db, query, candidateDepth, collectionFilter),
|
|
4308
|
+
results: searchCharFTS(db, query, candidateDepth, collectionFilter, filter),
|
|
4261
4309
|
}];
|
|
4262
4310
|
let secondaryReason = null;
|
|
4263
4311
|
try {
|
|
@@ -4293,7 +4341,7 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4293
4341
|
}
|
|
4294
4342
|
rankedChannels.push({
|
|
4295
4343
|
channel: entry.channel,
|
|
4296
|
-
results: searchFtsChannel(db, entry.table, ftsQuery, candidateDepth, collectionFilter),
|
|
4344
|
+
results: searchFtsChannel(db, entry.table, ftsQuery, candidateDepth, collectionFilter, filter),
|
|
4297
4345
|
});
|
|
4298
4346
|
channels.push({ channel: entry.channel, status: "used" });
|
|
4299
4347
|
}
|
|
@@ -4333,6 +4381,49 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4333
4381
|
// =============================================================================
|
|
4334
4382
|
// Vector Search
|
|
4335
4383
|
// =============================================================================
|
|
4384
|
+
/** sqlite-vec rejects k above this in MATCH queries (v0.1.9). */
|
|
4385
|
+
const SQLITE_VEC_MAX_K = 4096;
|
|
4386
|
+
/**
|
|
4387
|
+
* Max filter-eligible vectors for an exact cosine scan. Above this we fall
|
|
4388
|
+
* back to global ANN with a capped over-fetch. Exact scan avoids the
|
|
4389
|
+
* post-filter starvation of small eligible sets — originally small
|
|
4390
|
+
* collections (#791, #803), now also selective metadata filters; ANN remains
|
|
4391
|
+
* for very large eligible sets where a full scan would be expensive.
|
|
4392
|
+
*/
|
|
4393
|
+
const FILTERED_VEC_EXACT_SCAN_MAX = 20_000;
|
|
4394
|
+
const VEC_HASH_SEQ_IN_CHUNK = 400;
|
|
4395
|
+
/**
|
|
4396
|
+
* Exact cosine-distance scan over a known set of hash_seq keys.
|
|
4397
|
+
* Uses vec_distance_cosine with chunked IN lists (no JOIN with vectors_vec).
|
|
4398
|
+
*/
|
|
4399
|
+
function exactVecScanByHashSeq(db, embedding, hashSeqs, limit) {
|
|
4400
|
+
if (hashSeqs.length === 0 || limit <= 0)
|
|
4401
|
+
return [];
|
|
4402
|
+
const queryVec = new Float32Array(embedding);
|
|
4403
|
+
// Over-fetch a bit so multi-chunk docs can still yield `limit` unique files.
|
|
4404
|
+
const fetchLimit = Math.max(limit * 3, limit);
|
|
4405
|
+
const scored = [];
|
|
4406
|
+
for (let i = 0; i < hashSeqs.length; i += VEC_HASH_SEQ_IN_CHUNK) {
|
|
4407
|
+
const chunk = hashSeqs.slice(i, i + VEC_HASH_SEQ_IN_CHUNK);
|
|
4408
|
+
const placeholders = chunk.map(() => "?").join(",");
|
|
4409
|
+
const rows = db.prepare(`
|
|
4410
|
+
SELECT hash_seq, vec_distance_cosine(embedding, ?) AS distance
|
|
4411
|
+
FROM vectors_vec
|
|
4412
|
+
WHERE hash_seq IN (${placeholders})
|
|
4413
|
+
`).all(queryVec, ...chunk);
|
|
4414
|
+
scored.push(...rows);
|
|
4415
|
+
}
|
|
4416
|
+
scored.sort((a, b) => a.distance - b.distance);
|
|
4417
|
+
return scored.slice(0, fetchLimit);
|
|
4418
|
+
}
|
|
4419
|
+
function annVecScan(db, embedding, k) {
|
|
4420
|
+
const vecK = Math.max(1, Math.min(SQLITE_VEC_MAX_K, k));
|
|
4421
|
+
return db.prepare(`
|
|
4422
|
+
SELECT hash_seq, distance
|
|
4423
|
+
FROM vectors_vec
|
|
4424
|
+
WHERE embedding MATCH ? AND k = ?
|
|
4425
|
+
`).all(new Float32Array(embedding), vecK);
|
|
4426
|
+
}
|
|
4336
4427
|
function resolveReadyProviderEmbeddingIdentity(db, provider) {
|
|
4337
4428
|
const storedIdentity = readStoredEmbeddingIdentity(db);
|
|
4338
4429
|
if (!storedIdentity || storedIdentity.providerId !== provider.providerId
|
|
@@ -4373,10 +4464,29 @@ function hasSearchableVectorIndex(store) {
|
|
|
4373
4464
|
return storedIdentity !== undefined
|
|
4374
4465
|
&& inspectEmbeddingIndexState(store.db, storedIdentity).status === "ready";
|
|
4375
4466
|
}
|
|
4376
|
-
export async function searchVec(db, query, model, limit = 20, collectionFilter, session, precomputedEmbedding,
|
|
4467
|
+
export async function searchVec(db, query, model, limit = 20, collectionFilter, session, precomputedEmbedding, providerOrLlm, authorizeRemoteRequestOrFilter, llmOverride, metadataFilter) {
|
|
4377
4468
|
const tableExists = db.prepare(`SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get();
|
|
4378
4469
|
if (!tableExists)
|
|
4379
4470
|
return [];
|
|
4471
|
+
// Disambiguate overloaded arguments
|
|
4472
|
+
let provider;
|
|
4473
|
+
let authorizeRemoteRequest;
|
|
4474
|
+
let filter = metadataFilter;
|
|
4475
|
+
let effectiveLlmOverride = llmOverride;
|
|
4476
|
+
if (providerOrLlm && typeof providerOrLlm === "object") {
|
|
4477
|
+
if ("providerId" in providerOrLlm) {
|
|
4478
|
+
provider = providerOrLlm;
|
|
4479
|
+
}
|
|
4480
|
+
else {
|
|
4481
|
+
effectiveLlmOverride = providerOrLlm;
|
|
4482
|
+
}
|
|
4483
|
+
}
|
|
4484
|
+
if (typeof authorizeRemoteRequestOrFilter === "function") {
|
|
4485
|
+
authorizeRemoteRequest = authorizeRemoteRequestOrFilter;
|
|
4486
|
+
}
|
|
4487
|
+
else if (typeof authorizeRemoteRequestOrFilter === "object" && authorizeRemoteRequestOrFilter !== null && !filter) {
|
|
4488
|
+
filter = authorizeRemoteRequestOrFilter;
|
|
4489
|
+
}
|
|
4380
4490
|
if (provider && model !== provider.model) {
|
|
4381
4491
|
throw new Error(`Embedding model ${model} does not match borrowed provider model ${provider.model}.`);
|
|
4382
4492
|
}
|
|
@@ -4405,13 +4515,19 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4405
4515
|
})).vector;
|
|
4406
4516
|
}
|
|
4407
4517
|
else if (!embedding) {
|
|
4408
|
-
embedding = await getEmbedding(query, model, true, session,
|
|
4518
|
+
embedding = await getEmbedding(query, model, true, session, effectiveLlmOverride);
|
|
4409
4519
|
}
|
|
4410
4520
|
if (!embedding)
|
|
4411
4521
|
return [];
|
|
4522
|
+
// Multi-collection union: search each collection separately to avoid starvation
|
|
4523
|
+
const names = scopedCollectionNames(collectionFilter);
|
|
4524
|
+
if (names && names.length > 1) {
|
|
4525
|
+
const lists = await Promise.all(names.map(name => searchVec(db, query, model, limit, name, session, embedding, provider ?? effectiveLlmOverride, authorizeRemoteRequest ?? filter, effectiveLlmOverride, filter)));
|
|
4526
|
+
return mergeSearchResultsByScore(lists, limit);
|
|
4527
|
+
}
|
|
4412
4528
|
const activeFingerprint = providerIdentity
|
|
4413
4529
|
? providerIdentity.fingerprint
|
|
4414
|
-
: storedIdentity
|
|
4530
|
+
: storedIdentity?.fingerprint;
|
|
4415
4531
|
const toSearchResult = (row, body, distance) => {
|
|
4416
4532
|
return {
|
|
4417
4533
|
filepath: row.filepath,
|
|
@@ -4424,56 +4540,87 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4424
4540
|
bodyLength: body.length,
|
|
4425
4541
|
body,
|
|
4426
4542
|
context: getContextForFile(db, row.filepath),
|
|
4543
|
+
metadata: parseMetadataJson(row.metadata_json),
|
|
4427
4544
|
score: 1 - distance, // Cosine similarity = 1 - cosine distance
|
|
4428
4545
|
source: "vec",
|
|
4429
4546
|
chunkPos: row.pos,
|
|
4430
4547
|
};
|
|
4431
4548
|
};
|
|
4432
4549
|
const collections = normalizeCollectionFilter(collectionFilter);
|
|
4433
|
-
const queryVec = embedding instanceof Float32Array ? embedding : new Float32Array(embedding);
|
|
4434
|
-
const SQLITE_VEC_MAX_K = 4096;
|
|
4435
|
-
let collectionFilterSql = "";
|
|
4436
|
-
const queryParams = [queryVec];
|
|
4437
|
-
if (collections.length === 1) {
|
|
4438
|
-
collectionFilterSql = "AND collection = ?";
|
|
4439
|
-
queryParams.push(collections[0]);
|
|
4440
|
-
}
|
|
4441
|
-
else if (collections.length > 1) {
|
|
4442
|
-
const placeholders = collections.map(() => "?").join(", ");
|
|
4443
|
-
collectionFilterSql = `AND collection IN (${placeholders})`;
|
|
4444
|
-
queryParams.push(...collections);
|
|
4445
|
-
}
|
|
4446
|
-
// Request generous k to account for multi-chunk deduplication per document
|
|
4447
|
-
const fetchK = Math.min(SQLITE_VEC_MAX_K, Math.max(limit * 15, 60));
|
|
4448
|
-
queryParams.push(fetchK);
|
|
4449
4550
|
let vecRows;
|
|
4450
|
-
|
|
4451
|
-
|
|
4452
|
-
|
|
4453
|
-
|
|
4454
|
-
|
|
4455
|
-
|
|
4456
|
-
|
|
4457
|
-
|
|
4551
|
+
if (filter) {
|
|
4552
|
+
let eligibleSql = `
|
|
4553
|
+
SELECT DISTINCT cv.hash || '_' || cv.seq AS hash_seq
|
|
4554
|
+
FROM content_vectors cv
|
|
4555
|
+
JOIN documents d ON d.hash = cv.hash AND d.active = 1
|
|
4556
|
+
JOIN document_metadata dm ON dm.document_id = d.id
|
|
4557
|
+
`;
|
|
4558
|
+
const eligibleConditions = [];
|
|
4559
|
+
const eligibleParams = [];
|
|
4560
|
+
if (activeFingerprint) {
|
|
4561
|
+
eligibleConditions.push(`cv.model = ? AND cv.embed_fingerprint = ?`);
|
|
4562
|
+
eligibleParams.push(model, activeFingerprint);
|
|
4563
|
+
}
|
|
4564
|
+
if (collections.length === 1) {
|
|
4565
|
+
eligibleConditions.push(`d.collection = ?`);
|
|
4566
|
+
eligibleParams.push(collections[0]);
|
|
4567
|
+
}
|
|
4568
|
+
else if (collections.length > 1) {
|
|
4569
|
+
eligibleConditions.push(`d.collection IN (${collections.map(() => "?").join(", ")})`);
|
|
4570
|
+
eligibleParams.push(...collections);
|
|
4571
|
+
}
|
|
4572
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4573
|
+
eligibleConditions.push(`dm.extraction_version = ${METADATA_EXTRACTION_VERSION}`);
|
|
4574
|
+
eligibleConditions.push(`dm.extraction_error IS NULL`);
|
|
4575
|
+
eligibleConditions.push(compiledFilter.sql);
|
|
4576
|
+
eligibleParams.push(...compiledFilter.params);
|
|
4577
|
+
eligibleSql += ` WHERE ${eligibleConditions.join(" AND ")}`;
|
|
4578
|
+
const eligibleHashSeqs = withLazyContentVectorMigration(db, () => db.prepare(eligibleSql).all(...eligibleParams)).map((r) => r.hash_seq);
|
|
4579
|
+
if (eligibleHashSeqs.length === 0)
|
|
4580
|
+
return [];
|
|
4581
|
+
if (eligibleHashSeqs.length <= FILTERED_VEC_EXACT_SCAN_MAX) {
|
|
4582
|
+
vecRows = exactVecScanByHashSeq(db, embedding, eligibleHashSeqs, limit);
|
|
4583
|
+
}
|
|
4584
|
+
else {
|
|
4585
|
+
vecRows = annVecScan(db, embedding, Math.max(limit * 30, limit * 3));
|
|
4586
|
+
}
|
|
4458
4587
|
}
|
|
4459
|
-
|
|
4460
|
-
//
|
|
4461
|
-
|
|
4462
|
-
|
|
4463
|
-
|
|
4464
|
-
|
|
4465
|
-
|
|
4466
|
-
|
|
4588
|
+
else {
|
|
4589
|
+
// No metadata filter: use fast ANN scan (with collection pushdown if supported)
|
|
4590
|
+
const queryVec = embedding instanceof Float32Array ? embedding : new Float32Array(embedding);
|
|
4591
|
+
let collectionFilterSql = "";
|
|
4592
|
+
const queryParams = [queryVec];
|
|
4593
|
+
if (collections.length === 1) {
|
|
4594
|
+
collectionFilterSql = "AND collection = ?";
|
|
4595
|
+
queryParams.push(collections[0]);
|
|
4596
|
+
}
|
|
4597
|
+
else if (collections.length > 1) {
|
|
4598
|
+
const placeholders = collections.map(() => "?").join(", ");
|
|
4599
|
+
collectionFilterSql = `AND collection IN (${placeholders})`;
|
|
4600
|
+
queryParams.push(...collections);
|
|
4601
|
+
}
|
|
4602
|
+
const fetchK = Math.min(SQLITE_VEC_MAX_K, Math.max(limit * 15, 60));
|
|
4603
|
+
queryParams.push(fetchK);
|
|
4604
|
+
try {
|
|
4605
|
+
vecRows = withLazyContentVectorMigration(db, () => db.prepare(`
|
|
4606
|
+
SELECT hash_seq, distance
|
|
4607
|
+
FROM vectors_vec
|
|
4608
|
+
WHERE embedding MATCH ?
|
|
4609
|
+
${collectionFilterSql}
|
|
4610
|
+
AND k = ?
|
|
4611
|
+
`).all(...queryParams));
|
|
4612
|
+
}
|
|
4613
|
+
catch {
|
|
4614
|
+
// Fallback if table lacks collection column before migration
|
|
4615
|
+
vecRows = annVecScan(db, embedding, fetchK);
|
|
4616
|
+
}
|
|
4467
4617
|
}
|
|
4468
4618
|
if (vecRows.length === 0)
|
|
4469
4619
|
return [];
|
|
4470
4620
|
const keys = vecRows.map(r => r.hash_seq);
|
|
4471
4621
|
const distMap = new Map(vecRows.map(r => [r.hash_seq, r.distance]));
|
|
4472
4622
|
const placeholders = keys.map(() => "?").join(", ");
|
|
4473
|
-
|
|
4474
|
-
? `AND d.collection IN (${collections.map(() => "?").join(", ")})`
|
|
4475
|
-
: "";
|
|
4476
|
-
const metaRows = withLazyContentVectorMigration(db, () => db.prepare(`
|
|
4623
|
+
let metaSql = `
|
|
4477
4624
|
SELECT
|
|
4478
4625
|
cv.hash || '_' || cv.seq AS hash_seq,
|
|
4479
4626
|
cv.hash,
|
|
@@ -4481,14 +4628,32 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4481
4628
|
'qmd://' || d.collection || '/' || d.path AS filepath,
|
|
4482
4629
|
d.collection || '/' || d.path AS display_path,
|
|
4483
4630
|
d.title,
|
|
4484
|
-
d.collection
|
|
4631
|
+
d.collection,
|
|
4632
|
+
dm.metadata_json
|
|
4485
4633
|
FROM content_vectors cv
|
|
4486
4634
|
JOIN documents d ON d.hash = cv.hash AND d.active = 1
|
|
4487
|
-
|
|
4488
|
-
|
|
4489
|
-
|
|
4490
|
-
|
|
4491
|
-
|
|
4635
|
+
LEFT JOIN document_metadata dm ON dm.document_id = d.id
|
|
4636
|
+
WHERE (cv.hash || '_' || cv.seq) IN (${placeholders})
|
|
4637
|
+
`;
|
|
4638
|
+
const metaParams = [...keys];
|
|
4639
|
+
if (activeFingerprint) {
|
|
4640
|
+
metaSql += ` AND cv.model = ? AND cv.embed_fingerprint = ?`;
|
|
4641
|
+
metaParams.push(model, activeFingerprint);
|
|
4642
|
+
}
|
|
4643
|
+
if (collections.length === 1) {
|
|
4644
|
+
metaSql += ` AND d.collection = ?`;
|
|
4645
|
+
metaParams.push(collections[0]);
|
|
4646
|
+
}
|
|
4647
|
+
else if (collections.length > 1) {
|
|
4648
|
+
metaSql += ` AND d.collection IN (${collections.map(() => "?").join(", ")})`;
|
|
4649
|
+
metaParams.push(...collections);
|
|
4650
|
+
}
|
|
4651
|
+
if (filter) {
|
|
4652
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4653
|
+
metaSql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`;
|
|
4654
|
+
metaParams.push(...compiledFilter.params);
|
|
4655
|
+
}
|
|
4656
|
+
const metaRows = withLazyContentVectorMigration(db, () => db.prepare(metaSql).all(...metaParams));
|
|
4492
4657
|
const byFile = new Map();
|
|
4493
4658
|
for (const m of metaRows) {
|
|
4494
4659
|
const dist = distMap.get(m.hash_seq);
|
|
@@ -4767,7 +4932,7 @@ export async function rerank(query, documents, model = DEFAULT_RERANK_MODEL, db,
|
|
|
4767
4932
|
cachedResults.set(doc.text, parseFloat(cached));
|
|
4768
4933
|
}
|
|
4769
4934
|
else {
|
|
4770
|
-
uncachedDocsByChunk.set(doc.text, { file: doc.file, text: doc.text });
|
|
4935
|
+
uncachedDocsByChunk.set(doc.text, { file: doc.file, text: doc.text, title: doc.title });
|
|
4771
4936
|
}
|
|
4772
4937
|
}
|
|
4773
4938
|
// Rerank uncached documents using LlamaCpp
|
|
@@ -5318,6 +5483,7 @@ export function getStatusReadOnly(db, needsEmbedding) {
|
|
|
5318
5483
|
totalDocuments: totalDocs,
|
|
5319
5484
|
needsEmbedding,
|
|
5320
5485
|
hasVectorIndex: hasVectors,
|
|
5486
|
+
pendingMetadata: countDocumentsPendingMetadata(db),
|
|
5321
5487
|
collections,
|
|
5322
5488
|
};
|
|
5323
5489
|
}
|
|
@@ -5443,6 +5609,15 @@ export function addLineNumbers(text, startLine = 1) {
|
|
|
5443
5609
|
const lines = text.split('\n');
|
|
5444
5610
|
return lines.map((line, i) => `${startLine + i}: ${line}`).join('\n');
|
|
5445
5611
|
}
|
|
5612
|
+
/**
|
|
5613
|
+
* Attach canonical metadata to final search results with one batch query.
|
|
5614
|
+
* Runs after RRF/reranking so metadata is never duplicated through the
|
|
5615
|
+
* intermediate ranked lists.
|
|
5616
|
+
*/
|
|
5617
|
+
function attachResultMetadata(db, results) {
|
|
5618
|
+
const metadataByFilepath = getMetadataByFilepath(db, results.map(r => r.file));
|
|
5619
|
+
return results.map(r => ({ ...r, metadata: metadataByFilepath.get(r.file) ?? {} }));
|
|
5620
|
+
}
|
|
5446
5621
|
/**
|
|
5447
5622
|
* RRF list weights for hybridQuery.
|
|
5448
5623
|
*
|
|
@@ -5477,6 +5652,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5477
5652
|
const collectionFilter = options?.collections && options.collections.length > 0
|
|
5478
5653
|
? options.collections
|
|
5479
5654
|
: options?.collection;
|
|
5655
|
+
const filter = options?.filter;
|
|
5480
5656
|
const explain = options?.explain ?? false;
|
|
5481
5657
|
const expansionContext = options?.expansionContext;
|
|
5482
5658
|
const rerankContext = options?.rerankContext;
|
|
@@ -5491,9 +5667,9 @@ export async function hybridQuery(store, query, options) {
|
|
|
5491
5667
|
// When either context is provided, disable strong-signal bypass — the obvious BM25
|
|
5492
5668
|
// match may not be what the caller wants (e.g. "performance" with context
|
|
5493
5669
|
// "web page load times" should NOT shortcut to a sports-performance doc).
|
|
5494
|
-
// Pass collection directly into FTS query (filter at SQL level, not post-hoc)
|
|
5670
|
+
// Pass collection and metadata filter directly into FTS query (filter at SQL level, not post-hoc)
|
|
5495
5671
|
const parsedDirective = parseExpansionDirective(query);
|
|
5496
|
-
const initialFts = store.searchFTS(parsedDirective.query, 20, collectionFilter);
|
|
5672
|
+
const initialFts = store.searchFTS(parsedDirective.query, 20, collectionFilter, filter);
|
|
5497
5673
|
const strongSignal = getLexicalStrongSignal(initialFts);
|
|
5498
5674
|
const topScore = strongSignal.topScore;
|
|
5499
5675
|
let expansionDecision;
|
|
@@ -5501,7 +5677,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5501
5677
|
expansionDecision = resolveExpansionPolicy({
|
|
5502
5678
|
query,
|
|
5503
5679
|
mode: expansionMode,
|
|
5504
|
-
strongSignal: !expansionContext && !rerankContext && strongSignal.strong,
|
|
5680
|
+
strongSignal: !expansionContext && !rerankContext && !containsRelativeTemporalTerms(query) && strongSignal.strong,
|
|
5505
5681
|
allowCjkExpand: Boolean(store.llm?.supportsExpand),
|
|
5506
5682
|
});
|
|
5507
5683
|
}
|
|
@@ -5561,7 +5737,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5561
5737
|
// 3a: Run FTS for all lex expansions right away (no LLM needed)
|
|
5562
5738
|
for (const q of expanded) {
|
|
5563
5739
|
if (q.type === 'lex') {
|
|
5564
|
-
const ftsResults = store.searchFTS(q.query, 20, collectionFilter);
|
|
5740
|
+
const ftsResults = store.searchFTS(q.query, 20, collectionFilter, filter);
|
|
5565
5741
|
if (ftsResults.length > 0) {
|
|
5566
5742
|
for (const r of ftsResults)
|
|
5567
5743
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5593,7 +5769,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5593
5769
|
const embedding = embeddings[i];
|
|
5594
5770
|
if (!embedding || embedding.length === 0)
|
|
5595
5771
|
continue;
|
|
5596
|
-
const vecResults = await store.searchVec(vecQueries[i].text, embedModel, 20, collectionFilter, undefined, embedding);
|
|
5772
|
+
const vecResults = await store.searchVec(vecQueries[i].text, embedModel, 20, collectionFilter, undefined, embedding, filter);
|
|
5597
5773
|
if (vecResults.length > 0) {
|
|
5598
5774
|
for (const r of vecResults)
|
|
5599
5775
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5660,7 +5836,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5660
5836
|
if (skipRerank) {
|
|
5661
5837
|
// Skip LLM reranking — return candidates scored by RRF only
|
|
5662
5838
|
const seenFiles = new Set();
|
|
5663
|
-
|
|
5839
|
+
const rrfResults = candidates
|
|
5664
5840
|
.map((cand, i) => {
|
|
5665
5841
|
const chunkInfo = docChunkMap.get(cand.file);
|
|
5666
5842
|
const bestIdx = chunkInfo?.bestIdx ?? 0;
|
|
@@ -5705,13 +5881,14 @@ export async function hybridQuery(store, query, options) {
|
|
|
5705
5881
|
})
|
|
5706
5882
|
.filter(r => r.score >= minScore)
|
|
5707
5883
|
.slice(0, limit);
|
|
5884
|
+
return attachResultMetadata(store.db, rrfResults);
|
|
5708
5885
|
}
|
|
5709
5886
|
// Step 6: Rerank chunks (NOT full bodies)
|
|
5710
5887
|
const chunksToRerank = [];
|
|
5711
5888
|
for (const cand of candidates) {
|
|
5712
5889
|
const chunkInfo = docChunkMap.get(cand.file);
|
|
5713
5890
|
if (chunkInfo) {
|
|
5714
|
-
chunksToRerank.push({ file: cand.file, text: chunkInfo.chunks[chunkInfo.bestIdx].text });
|
|
5891
|
+
chunksToRerank.push({ file: cand.file, text: chunkInfo.chunks[chunkInfo.bestIdx].text, title: cand.title });
|
|
5715
5892
|
}
|
|
5716
5893
|
}
|
|
5717
5894
|
hooks?.onRerankStart?.(chunksToRerank.length);
|
|
@@ -5771,7 +5948,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5771
5948
|
}).sort((a, b) => b.score - a.score);
|
|
5772
5949
|
// Step 8: Dedup by file (safety net — prevents duplicate output)
|
|
5773
5950
|
const seenFiles = new Set();
|
|
5774
|
-
|
|
5951
|
+
const finalResults = blended
|
|
5775
5952
|
.filter(r => {
|
|
5776
5953
|
if (seenFiles.has(r.file))
|
|
5777
5954
|
return false;
|
|
@@ -5780,6 +5957,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5780
5957
|
})
|
|
5781
5958
|
.filter(r => r.score >= minScore)
|
|
5782
5959
|
.slice(0, limit);
|
|
5960
|
+
return attachResultMetadata(store.db, finalResults);
|
|
5783
5961
|
}
|
|
5784
5962
|
/**
|
|
5785
5963
|
* Vector-only semantic search with query expansion.
|
|
@@ -5794,6 +5972,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5794
5972
|
const limit = options?.limit ?? 10;
|
|
5795
5973
|
const minScore = options?.minScore ?? 0.3;
|
|
5796
5974
|
const collection = options?.collection;
|
|
5975
|
+
const filter = options?.filter;
|
|
5797
5976
|
const expansionContext = options?.expansionContext;
|
|
5798
5977
|
const includeHyde = options?.includeHyde ?? true;
|
|
5799
5978
|
if (!hasSearchableVectorIndex(store))
|
|
@@ -5809,7 +5988,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5809
5988
|
const queryTexts = [query, ...vecExpanded.map(q => q.query)];
|
|
5810
5989
|
const allResults = new Map();
|
|
5811
5990
|
for (const q of queryTexts) {
|
|
5812
|
-
const vecResults = await store.searchVec(q, embedModel, limit, collection);
|
|
5991
|
+
const vecResults = await store.searchVec(q, embedModel, limit, collection, undefined, undefined, filter);
|
|
5813
5992
|
for (const r of vecResults) {
|
|
5814
5993
|
const existing = allResults.get(r.filepath);
|
|
5815
5994
|
if (!existing || r.score > existing.score) {
|
|
@@ -5821,6 +6000,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5821
6000
|
score: r.score,
|
|
5822
6001
|
context: store.getContextForFile(r.filepath),
|
|
5823
6002
|
docid: r.docid,
|
|
6003
|
+
metadata: r.metadata,
|
|
5824
6004
|
});
|
|
5825
6005
|
}
|
|
5826
6006
|
}
|
|
@@ -5857,6 +6037,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5857
6037
|
const skipRerank = options?.skipRerank ?? false;
|
|
5858
6038
|
const hooks = options?.hooks;
|
|
5859
6039
|
const collections = options?.collections;
|
|
6040
|
+
const filter = options?.filter;
|
|
5860
6041
|
if (searches.length === 0)
|
|
5861
6042
|
return [];
|
|
5862
6043
|
// Validate queries before executing
|
|
@@ -5889,7 +6070,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5889
6070
|
for (const [searchIndex, search] of searches.entries()) {
|
|
5890
6071
|
if (search.type === 'lex') {
|
|
5891
6072
|
for (const coll of collectionList) {
|
|
5892
|
-
const ftsResults = store.searchFTS(search.query, 20, coll);
|
|
6073
|
+
const ftsResults = store.searchFTS(search.query, 20, coll, filter);
|
|
5893
6074
|
if (ftsResults.length > 0) {
|
|
5894
6075
|
for (const r of ftsResults)
|
|
5895
6076
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5922,7 +6103,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5922
6103
|
if (!embedding || embedding.length === 0)
|
|
5923
6104
|
continue;
|
|
5924
6105
|
for (const coll of collectionList) {
|
|
5925
|
-
const vecResults = await store.searchVec(vecSearches[i].search.query, embedModel, 20, coll, undefined, embedding);
|
|
6106
|
+
const vecResults = await store.searchVec(vecSearches[i].search.query, embedModel, 20, coll, undefined, embedding, filter);
|
|
5926
6107
|
if (vecResults.length > 0) {
|
|
5927
6108
|
for (const r of vecResults)
|
|
5928
6109
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5985,7 +6166,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5985
6166
|
if (skipRerank) {
|
|
5986
6167
|
// Skip LLM reranking — return candidates scored by RRF only
|
|
5987
6168
|
const seenFiles = new Set();
|
|
5988
|
-
|
|
6169
|
+
const rrfResults = candidates
|
|
5989
6170
|
.map((cand, i) => {
|
|
5990
6171
|
const chunkInfo = docChunkMap.get(cand.file);
|
|
5991
6172
|
const bestIdx = chunkInfo?.bestIdx ?? 0;
|
|
@@ -6030,13 +6211,14 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6030
6211
|
})
|
|
6031
6212
|
.filter(r => r.score >= minScore)
|
|
6032
6213
|
.slice(0, limit);
|
|
6214
|
+
return attachResultMetadata(store.db, rrfResults);
|
|
6033
6215
|
}
|
|
6034
6216
|
// Step 5: Rerank chunks
|
|
6035
6217
|
const chunksToRerank = [];
|
|
6036
6218
|
for (const cand of candidates) {
|
|
6037
6219
|
const chunkInfo = docChunkMap.get(cand.file);
|
|
6038
6220
|
if (chunkInfo) {
|
|
6039
|
-
chunksToRerank.push({ file: cand.file, text: chunkInfo.chunks[chunkInfo.bestIdx].text });
|
|
6221
|
+
chunksToRerank.push({ file: cand.file, text: chunkInfo.chunks[chunkInfo.bestIdx].text, title: cand.title });
|
|
6040
6222
|
}
|
|
6041
6223
|
}
|
|
6042
6224
|
hooks?.onRerankStart?.(chunksToRerank.length);
|
|
@@ -6095,7 +6277,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6095
6277
|
}).sort((a, b) => b.score - a.score);
|
|
6096
6278
|
// Step 7: Dedup by file
|
|
6097
6279
|
const seenFiles = new Set();
|
|
6098
|
-
|
|
6280
|
+
const finalResults = blended
|
|
6099
6281
|
.filter(r => {
|
|
6100
6282
|
if (seenFiles.has(r.file))
|
|
6101
6283
|
return false;
|
|
@@ -6104,4 +6286,5 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6104
6286
|
})
|
|
6105
6287
|
.filter(r => r.score >= minScore)
|
|
6106
6288
|
.slice(0, limit);
|
|
6289
|
+
return attachResultMetadata(store.db, finalResults);
|
|
6107
6290
|
}
|