@wei840222/qmd 2026.9.6 → 2026.9.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +84 -1
- package/dist/cli/build-info.json +2 -2
- package/dist/cli/qmd.js +85 -9
- package/dist/collections.js +9 -4
- package/dist/index.d.ts +10 -0
- package/dist/index.js +14 -2
- package/dist/llm.d.ts +7 -1
- package/dist/llm.js +23 -4
- package/dist/mcp/server.js +70 -6
- package/dist/metadata-filter.d.ts +74 -0
- package/dist/metadata-filter.js +279 -0
- package/dist/metadata-store.d.ts +45 -0
- package/dist/metadata-store.js +173 -0
- package/dist/metadata.d.ts +61 -0
- package/dist/metadata.js +215 -0
- package/dist/search/zh-dict.txt +3 -0
- package/dist/store.d.ts +23 -16
- package/dist/store.js +341 -161
- package/package.json +2 -2
- package/scripts/sync-zh-dict.mjs +4 -1
- package/skills/qmd/SKILL.md +11 -0
- package/skills/release/SKILL.md +0 -141
- package/skills/release/scripts/install-hooks.sh +0 -38
- package/skills/release/scripts/release-context.sh +0 -129
package/dist/store.js
CHANGED
|
@@ -25,6 +25,9 @@ import { analyzeCjkSync, containsCjk } from "./search/cjk-analyzer.js";
|
|
|
25
25
|
import { ExpansionPolicyError, parseExpansionDirective, resolveExpansionPolicy, } from "./search/query-expansion.js";
|
|
26
26
|
import { LlamaCpp, getDefaultLlamaCpp, formatQueryForEmbedding, formatDocForEmbedding, withLLMSessionForLlm, DEFAULT_EMBED_MODEL_URI, DEFAULT_RERANK_MODEL_URI, DEFAULT_GENERATE_MODEL_URI, } from "./llm.js";
|
|
27
27
|
const readOnlyDatabases = new WeakSet();
|
|
28
|
+
import { METADATA_EXTRACTION_VERSION } from "./metadata.js";
|
|
29
|
+
import { compileMetadataFilter } from "./metadata-filter.js";
|
|
30
|
+
import { initializeMetadataSchema, syncDocumentMetadata, countDocumentsPendingMetadata, getMetadataByFilepath, parseMetadataJson, } from "./metadata-store.js";
|
|
28
31
|
// =============================================================================
|
|
29
32
|
// Configuration
|
|
30
33
|
// =============================================================================
|
|
@@ -925,18 +928,15 @@ export function normalizeCjkForFTS(text) {
|
|
|
925
928
|
return text.replace(CJK_RUN_PATTERN, run => ` ${Array.from(run).join(' ')} `);
|
|
926
929
|
}
|
|
927
930
|
function sanitizeFTS5Phrase(phrase) {
|
|
928
|
-
//
|
|
929
|
-
//
|
|
930
|
-
//
|
|
931
|
+
// A quoted phrase is matched against tokens the porter unicode61 tokenizer
|
|
932
|
+
// produced, and that tokenizer splits document text on every separator.
|
|
933
|
+
// Deleting the separators here instead would collapse "1.0.21" to "1021" and
|
|
934
|
+
// "PIO-1384" to "pio1384", tokens no document holds, so the query returns
|
|
935
|
+
// nothing with no error (#757 for dots, #916 for the rest). Split on the same
|
|
936
|
+
// separators the tokenizer does and emit the parts as adjacent phrase terms.
|
|
931
937
|
return normalizeCjkForFTS(phrase)
|
|
932
938
|
.split(/\s+/)
|
|
933
|
-
.flatMap(t =>
|
|
934
|
-
if (isDottedToken(t)) {
|
|
935
|
-
return t.split('.').map(p => sanitizeFTS5Term(p)).filter(p => p);
|
|
936
|
-
}
|
|
937
|
-
const sanitized = sanitizeFTS5Term(t);
|
|
938
|
-
return sanitized ? [sanitized] : [];
|
|
939
|
-
})
|
|
939
|
+
.flatMap(t => splitFTS5CompoundTerm(t))
|
|
940
940
|
.join(' ');
|
|
941
941
|
}
|
|
942
942
|
function getUserVersion(db) {
|
|
@@ -1402,6 +1402,9 @@ function initializeDatabase(db) {
|
|
|
1402
1402
|
`);
|
|
1403
1403
|
ensureEmbeddingIdentitySchema(db);
|
|
1404
1404
|
ensureContentVectorsStatusIndex(db);
|
|
1405
|
+
// Document metadata — extraction state plus normalized value rows for
|
|
1406
|
+
// metadata filtering. Keyed by document identity, not content hash.
|
|
1407
|
+
initializeMetadataSchema(db);
|
|
1405
1408
|
// Store collections — makes the DB self-contained (no external config needed)
|
|
1406
1409
|
db.exec(`
|
|
1407
1410
|
CREATE TABLE IF NOT EXISTS store_collections (
|
|
@@ -1730,7 +1733,7 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1730
1733
|
return !parts.some(part => part.startsWith("."));
|
|
1731
1734
|
});
|
|
1732
1735
|
const total = files.length;
|
|
1733
|
-
let indexed = 0, updated = 0, unchanged = 0, processed = 0;
|
|
1736
|
+
let indexed = 0, updated = 0, unchanged = 0, processed = 0, metadataErrors = 0;
|
|
1734
1737
|
const skippedFiles = [];
|
|
1735
1738
|
const seenPaths = new Set();
|
|
1736
1739
|
// Literal paths of every file in this scan. Passed to the legacy-path
|
|
@@ -1772,8 +1775,12 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1772
1775
|
const hash = await hashContent(content);
|
|
1773
1776
|
const title = extractTitle(content, relativeFile);
|
|
1774
1777
|
const existing = findOrMigrateLegacyDocument(db, collectionName, path, livePaths);
|
|
1778
|
+
let documentId;
|
|
1779
|
+
let contentChanged = true;
|
|
1775
1780
|
if (existing) {
|
|
1781
|
+
documentId = existing.id;
|
|
1776
1782
|
if (existing.hash === hash) {
|
|
1783
|
+
contentChanged = false;
|
|
1777
1784
|
if (existing.title !== title) {
|
|
1778
1785
|
updateDocumentTitle(db, existing.id, title, now);
|
|
1779
1786
|
updated++;
|
|
@@ -1791,8 +1798,12 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1791
1798
|
else {
|
|
1792
1799
|
indexed++;
|
|
1793
1800
|
const stat = statSync(filepath);
|
|
1794
|
-
insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1801
|
+
documentId = insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1795
1802
|
}
|
|
1803
|
+
// Unchanged content still backfills missing or stale extraction state.
|
|
1804
|
+
const extraction = syncDocumentMetadata(db, documentId, content, path, contentChanged ? undefined : { onlyIfStale: true });
|
|
1805
|
+
if (extraction?.error)
|
|
1806
|
+
metadataErrors++;
|
|
1796
1807
|
processed++;
|
|
1797
1808
|
options?.onProgress?.({ file: relativeFile, current: processed, total });
|
|
1798
1809
|
}
|
|
@@ -1806,7 +1817,7 @@ export async function reindexCollection(store, collectionPath, globPattern, coll
|
|
|
1806
1817
|
}
|
|
1807
1818
|
}
|
|
1808
1819
|
const orphanedCleaned = cleanupOrphanedContent(db);
|
|
1809
|
-
return { indexed, updated, unchanged, removed, orphanedCleaned, skipped: skippedFiles.length, skippedFiles };
|
|
1820
|
+
return { indexed, updated, unchanged, removed, orphanedCleaned, skipped: skippedFiles.length, skippedFiles, metadataErrors };
|
|
1810
1821
|
}
|
|
1811
1822
|
function validatePositiveIntegerOption(name, value, fallback) {
|
|
1812
1823
|
if (value === undefined)
|
|
@@ -2464,7 +2475,7 @@ export async function generateEmbeddings(store, options) {
|
|
|
2464
2475
|
pos: chunk.pos,
|
|
2465
2476
|
tokens: chunk.tokenUpperBound,
|
|
2466
2477
|
}))
|
|
2467
|
-
: await
|
|
2478
|
+
: await chunkDocumentByTokensWithLlm((llm ?? getLlm(store) ?? getDefaultLlamaCpp()), doc.body, undefined, undefined, undefined, doc.path, options?.chunkStrategy, session.signal);
|
|
2468
2479
|
if (embeddingLease) {
|
|
2469
2480
|
pruneEmbeddingRowsOutsideLayout(db, doc.hash, model, fingerprint, chunks.length, embeddingLease);
|
|
2470
2481
|
}
|
|
@@ -2641,11 +2652,11 @@ export function createStore(dbPath, options = {}) {
|
|
|
2641
2652
|
const commitPendingDocumentInsert = (collectionName, path, title, hash, createdAt, modifiedAt) => {
|
|
2642
2653
|
const pending = pendingContent.get(hash);
|
|
2643
2654
|
if (!pending) {
|
|
2644
|
-
insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt);
|
|
2645
|
-
return;
|
|
2655
|
+
return insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt);
|
|
2646
2656
|
}
|
|
2647
|
-
insertDocumentWithContent(db, hash, pending.content, pending.createdAt, collectionName, path, title, createdAt, modifiedAt);
|
|
2657
|
+
const docId = insertDocumentWithContent(db, hash, pending.content, pending.createdAt, collectionName, path, title, createdAt, modifiedAt);
|
|
2648
2658
|
pendingContent.delete(hash);
|
|
2659
|
+
return docId;
|
|
2649
2660
|
};
|
|
2650
2661
|
const commitPendingDocumentUpdate = (documentId, title, hash, modifiedAt) => {
|
|
2651
2662
|
const pending = pendingContent.get(hash);
|
|
@@ -2693,9 +2704,9 @@ export function createStore(dbPath, options = {}) {
|
|
|
2693
2704
|
resolveVirtualPath: (virtualPath) => resolveVirtualPath(db, virtualPath),
|
|
2694
2705
|
toVirtualPath: (absolutePath) => toVirtualPath(db, absolutePath),
|
|
2695
2706
|
// Search
|
|
2696
|
-
searchCharFTS: (query, limit, collectionName) => searchCharFTS(db, query, limit, collectionName),
|
|
2697
|
-
searchFTS: (query, limit, collectionName) => searchFTS(db, query, limit, collectionName),
|
|
2698
|
-
searchVec: (query, model, limit, collectionFilter, session, precomputedEmbedding) => searchVec(db, query, model, limit, collectionFilter, session, precomputedEmbedding, store.embeddingProvider, store.authorizeRemoteRequest, store.llm),
|
|
2707
|
+
searchCharFTS: (query, limit, collectionName, filter) => searchCharFTS(db, query, limit, collectionName, filter),
|
|
2708
|
+
searchFTS: (query, limit, collectionName, filter) => searchFTS(db, query, limit, collectionName, filter),
|
|
2709
|
+
searchVec: (query, model, limit, collectionFilter, session, precomputedEmbedding, filter) => searchVec(db, query, model, limit, collectionFilter, session, precomputedEmbedding, store.embeddingProvider, store.authorizeRemoteRequest, store.llm, filter),
|
|
2699
2710
|
// Query expansion & reranking
|
|
2700
2711
|
expandQuery: (query, model, expansionContext, options) => expandQuery(query, model ?? store.localLlm?.generateModelName ?? store.llm?.generateModelName ?? DEFAULT_QUERY_MODEL, db, expansionContext, store.llm, options),
|
|
2701
2712
|
invalidateExpansionCache: (query, expansionContext, options) => deleteExpansionCacheEntry(db, query, store.localLlm?.generateModelName ?? store.llm?.generateModelName ?? DEFAULT_QUERY_MODEL, expansionContext, options),
|
|
@@ -2724,6 +2735,30 @@ export function createStore(dbPath, options = {}) {
|
|
|
2724
2735
|
getActiveDocumentPaths: (collectionName) => getActiveDocumentPaths(db, collectionName),
|
|
2725
2736
|
// Vector/embedding operations
|
|
2726
2737
|
getHashesForEmbedding: () => getHashesForEmbedding(db),
|
|
2738
|
+
insertEmbedding: (hash, seq, pos, embedding, model, embeddedAt, totalChunks, fingerprint, lease) => {
|
|
2739
|
+
const state = db.prepare(`
|
|
2740
|
+
SELECT status, fingerprint, model, generation, lease_owner, lease_expires_at
|
|
2741
|
+
FROM embedding_index_state
|
|
2742
|
+
WHERE singleton = 1
|
|
2743
|
+
`).get();
|
|
2744
|
+
if (!state && !lease) {
|
|
2745
|
+
withLazyContentVectorMigration(db, () => {
|
|
2746
|
+
db.prepare(`
|
|
2747
|
+
INSERT OR REPLACE INTO content_vectors (hash, seq, pos, model, embed_fingerprint, total_chunks, embedded_at)
|
|
2748
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
2749
|
+
`).run(hash, seq, pos, model, fingerprint ?? "", totalChunks ?? 1, embeddedAt);
|
|
2750
|
+
const hasCollectionCol = db.prepare(`SELECT sql FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get();
|
|
2751
|
+
if (hasCollectionCol?.sql?.includes("collection")) {
|
|
2752
|
+
db.prepare(`INSERT OR REPLACE INTO vectors_vec (hash_seq, collection, embedding) VALUES (?, ?, ?)`).run(`${hash}_${seq}`, "", embedding);
|
|
2753
|
+
}
|
|
2754
|
+
else {
|
|
2755
|
+
db.prepare(`INSERT OR REPLACE INTO vectors_vec (hash_seq, embedding) VALUES (?, ?)`).run(`${hash}_${seq}`, embedding);
|
|
2756
|
+
}
|
|
2757
|
+
});
|
|
2758
|
+
return;
|
|
2759
|
+
}
|
|
2760
|
+
insertEmbedding(db, hash, seq, pos, embedding, model, embeddedAt, totalChunks, fingerprint, lease);
|
|
2761
|
+
},
|
|
2727
2762
|
};
|
|
2728
2763
|
return store;
|
|
2729
2764
|
}
|
|
@@ -2891,7 +2926,7 @@ export async function maybeAdoptLegacyEmbeddingFingerprint(store, model = DEFAUL
|
|
|
2891
2926
|
const title = extractTitle(sample.body, sample.path);
|
|
2892
2927
|
const llm = getLlm(store);
|
|
2893
2928
|
return await withLLMSessionForLlm(llm, async (session) => {
|
|
2894
|
-
const chunks = await
|
|
2929
|
+
const chunks = await chunkDocumentByTokensWithLlm(llm, sample.body, undefined, undefined, undefined, sample.path, undefined, session.signal);
|
|
2895
2930
|
const chunk = chunks[sample.seq];
|
|
2896
2931
|
if (!chunk) {
|
|
2897
2932
|
return { checked: true, adopted: 0, reason: `sample chunk ${expectedHashSeq} no longer exists` };
|
|
@@ -3197,9 +3232,10 @@ function rebuildDocumentFTS(db, documentId) {
|
|
|
3197
3232
|
}
|
|
3198
3233
|
/**
|
|
3199
3234
|
* Insert a new document into the documents table.
|
|
3235
|
+
* Returns the document's id so callers can attach document-scoped state.
|
|
3200
3236
|
*/
|
|
3201
3237
|
export function insertDocument(db, collectionName, path, title, hash, createdAt, modifiedAt) {
|
|
3202
|
-
runCjkSynchronizedMutation(db, () => {
|
|
3238
|
+
return runCjkSynchronizedMutation(db, () => {
|
|
3203
3239
|
db.prepare(`
|
|
3204
3240
|
INSERT INTO documents (collection, path, title, hash, created_at, modified_at, active)
|
|
3205
3241
|
VALUES (?, ?, ?, ?, ?, ?, 1)
|
|
@@ -3210,15 +3246,17 @@ export function insertDocument(db, collectionName, path, title, hash, createdAt,
|
|
|
3210
3246
|
active = 1
|
|
3211
3247
|
`).run(collectionName, path, title, hash, createdAt, modifiedAt);
|
|
3212
3248
|
const row = db.prepare(`SELECT id FROM documents WHERE collection = ? AND path = ?`).get(collectionName, path);
|
|
3213
|
-
if (row)
|
|
3214
|
-
|
|
3249
|
+
if (!row)
|
|
3250
|
+
throw new Error(`Document row missing after insert: ${collectionName}/${path}`);
|
|
3251
|
+
rebuildDocumentFTS(db, row.id);
|
|
3252
|
+
return row.id;
|
|
3215
3253
|
});
|
|
3216
3254
|
}
|
|
3217
3255
|
/** Insert immutable content and its document row in the same lexical transaction. */
|
|
3218
3256
|
export function insertDocumentWithContent(db, hash, content, contentCreatedAt, collectionName, path, title, documentCreatedAt, modifiedAt) {
|
|
3219
|
-
runCjkSynchronizedMutation(db, () => {
|
|
3257
|
+
return runCjkSynchronizedMutation(db, () => {
|
|
3220
3258
|
insertContent(db, hash, content, contentCreatedAt);
|
|
3221
|
-
insertDocument(db, collectionName, path, title, hash, documentCreatedAt, modifiedAt);
|
|
3259
|
+
return insertDocument(db, collectionName, path, title, hash, documentCreatedAt, modifiedAt);
|
|
3222
3260
|
});
|
|
3223
3261
|
}
|
|
3224
3262
|
/**
|
|
@@ -3425,8 +3463,7 @@ function stripUnpairedSurrogates(text) {
|
|
|
3425
3463
|
* When filepath and chunkStrategy are provided, uses AST-aware break points
|
|
3426
3464
|
* for supported code files.
|
|
3427
3465
|
*/
|
|
3428
|
-
|
|
3429
|
-
const llm = getDefaultLlamaCpp();
|
|
3466
|
+
async function chunkDocumentByTokensWithLlm(llm, content, maxTokens = CHUNK_SIZE_TOKENS, overlapTokens = CHUNK_OVERLAP_TOKENS, windowTokens = CHUNK_WINDOW_TOKENS, filepath, chunkStrategy = "regex", signal) {
|
|
3430
3467
|
// Use moderate chars/token estimate (prose ~4, code ~2, mixed ~3)
|
|
3431
3468
|
// If chunks exceed limit, they'll be re-split with actual ratio
|
|
3432
3469
|
const avgCharsPerToken = 3;
|
|
@@ -3500,6 +3537,9 @@ export async function chunkDocumentByTokens(content, maxTokens = CHUNK_SIZE_TOKE
|
|
|
3500
3537
|
}
|
|
3501
3538
|
return results;
|
|
3502
3539
|
}
|
|
3540
|
+
export async function chunkDocumentByTokens(content, maxTokens = CHUNK_SIZE_TOKENS, overlapTokens = CHUNK_OVERLAP_TOKENS, windowTokens = CHUNK_WINDOW_TOKENS, filepath, chunkStrategy = "regex", signal) {
|
|
3541
|
+
return await chunkDocumentByTokensWithLlm(getDefaultLlamaCpp(), content, maxTokens, overlapTokens, windowTokens, filepath, chunkStrategy, signal);
|
|
3542
|
+
}
|
|
3503
3543
|
// =============================================================================
|
|
3504
3544
|
// Fuzzy matching
|
|
3505
3545
|
// =============================================================================
|
|
@@ -3962,38 +4002,29 @@ export function sanitizeFTS5Term(term) {
|
|
|
3962
4002
|
return term.replace(/[^\p{L}\p{N}'_]/gu, '').toLowerCase();
|
|
3963
4003
|
}
|
|
3964
4004
|
/**
|
|
3965
|
-
*
|
|
3966
|
-
*
|
|
3967
|
-
|
|
3968
|
-
|
|
3969
|
-
|
|
3970
|
-
|
|
3971
|
-
|
|
3972
|
-
*
|
|
3973
|
-
* and sanitizing each part. Returns the parts joined by spaces for use
|
|
3974
|
-
* inside FTS5 quotes: "multi agent" matches "multi-agent" in porter tokenizer.
|
|
4005
|
+
* A run of characters the FTS tokenizer treats as a separator.
|
|
4006
|
+
*
|
|
4007
|
+
* `documents_fts` is tokenized with `porter unicode61`, which starts a new
|
|
4008
|
+
* token at every character that is not a letter or a digit. Underscore is one
|
|
4009
|
+
* of those, but it is deliberately kept here rather than split on: FTS5 applies
|
|
4010
|
+
* the same tokenizer to a quoted phrase, so leaving `apply_secrets` intact lets
|
|
4011
|
+
* it split symmetrically into `apply secrets` on both sides, and that is the
|
|
4012
|
+
* behaviour #305 shipped. The apostrophe is kept for the same reason.
|
|
3975
4013
|
*/
|
|
3976
|
-
|
|
3977
|
-
return term.split('-').map(t => sanitizeFTS5Term(t)).filter(t => t).join(' ');
|
|
3978
|
-
}
|
|
4014
|
+
const FTS5_SEPARATOR_RUN = /[^\p{L}\p{N}'_]+/u;
|
|
3979
4015
|
/**
|
|
3980
|
-
*
|
|
3981
|
-
*
|
|
3982
|
-
*
|
|
3983
|
-
*
|
|
3984
|
-
|
|
3985
|
-
|
|
3986
|
-
|
|
3987
|
-
|
|
3988
|
-
|
|
3989
|
-
/**
|
|
3990
|
-
* Sanitize a dotted term into individual FTS5 tokens joined with AND.
|
|
3991
|
-
* e.g. "2026.4.10" → '"2026"* AND "4"* AND "10"*'
|
|
3992
|
-
* The AND ensures all parts must appear, matching how the porter tokenizer
|
|
3993
|
-
* indexes dotted strings.
|
|
4016
|
+
* Split one query term the way the tokenizer split the document text, and
|
|
4017
|
+
* sanitize each part.
|
|
4018
|
+
*
|
|
4019
|
+
* `PIO-1384` becomes ["pio", "1384"], `src/lib/i18n.ts` becomes
|
|
4020
|
+
* ["src", "lib", "i18n", "ts"], and a term with no separator in it comes back
|
|
4021
|
+
* as a single part. Callers join the parts into an FTS5 phrase, which is what
|
|
4022
|
+
* makes the parts have to be adjacent in the document rather than merely all
|
|
4023
|
+
* present. Parts that sanitize to nothing are dropped, so a term that is all
|
|
4024
|
+
* punctuation yields an empty list and the caller skips it.
|
|
3994
4025
|
*/
|
|
3995
|
-
function
|
|
3996
|
-
return term.split(
|
|
4026
|
+
function splitFTS5CompoundTerm(term) {
|
|
4027
|
+
return term.split(FTS5_SEPARATOR_RUN).map(p => sanitizeFTS5Term(p)).filter(p => p);
|
|
3997
4028
|
}
|
|
3998
4029
|
/**
|
|
3999
4030
|
* Parse lex query syntax into FTS5 query.
|
|
@@ -4001,7 +4032,8 @@ function sanitizeDottedTerm(term) {
|
|
|
4001
4032
|
* Supports:
|
|
4002
4033
|
* - Quoted phrases: "exact phrase" → "exact phrase" (exact match)
|
|
4003
4034
|
* - Negation: -term or -"phrase" → uses FTS5 NOT operator
|
|
4004
|
-
* -
|
|
4035
|
+
* - Terms holding a separator: multi-agent, DEC-0054, gpt-4, 2026.4.10,
|
|
4036
|
+
* src/lib/i18n.ts, @tobilu/qmd → treated as phrases over their parts
|
|
4005
4037
|
* - Plain terms: term → "term"* (prefix match)
|
|
4006
4038
|
*
|
|
4007
4039
|
* FTS5 NOT is a binary operator: `term1 NOT term2` means "match term1 but not term2".
|
|
@@ -4018,6 +4050,8 @@ function sanitizeDottedTerm(term) {
|
|
|
4018
4050
|
* multi-agent memory → "multi agent" AND "memory"*
|
|
4019
4051
|
* DEC-0054 → "dec 0054"
|
|
4020
4052
|
* -multi-agent → NOT "multi agent"
|
|
4053
|
+
* "DEC-0054" → "dec 0054"
|
|
4054
|
+
* src/lib/i18n.ts → "src lib i18n ts"
|
|
4021
4055
|
*/
|
|
4022
4056
|
function buildFTS5Query(query) {
|
|
4023
4057
|
const positive = [];
|
|
@@ -4061,41 +4095,7 @@ function buildFTS5Query(query) {
|
|
|
4061
4095
|
while (i < s.length && !/[\s"]/.test(s[i]))
|
|
4062
4096
|
i++;
|
|
4063
4097
|
const term = s.slice(start, i);
|
|
4064
|
-
|
|
4065
|
-
// These get split into phrase queries so FTS5 porter tokenizer matches them.
|
|
4066
|
-
if (isHyphenatedToken(term)) {
|
|
4067
|
-
const sanitized = sanitizeHyphenatedTerm(term);
|
|
4068
|
-
if (sanitized) {
|
|
4069
|
-
const ftsPhrase = `"${sanitized}"`; // Phrase match (no prefix)
|
|
4070
|
-
if (negated) {
|
|
4071
|
-
negative.push(ftsPhrase);
|
|
4072
|
-
}
|
|
4073
|
-
else {
|
|
4074
|
-
positive.push(ftsPhrase);
|
|
4075
|
-
}
|
|
4076
|
-
}
|
|
4077
|
-
}
|
|
4078
|
-
else if (isDottedToken(term)) {
|
|
4079
|
-
// Handle dotted version strings: 2026.4.10, 3.14.0, v1.2.3
|
|
4080
|
-
// The porter tokenizer splits on dots, so the index has individual tokens.
|
|
4081
|
-
// We AND all parts together so the query matches documents containing all parts.
|
|
4082
|
-
const sanitized = sanitizeDottedTerm(term);
|
|
4083
|
-
if (sanitized) {
|
|
4084
|
-
// sanitizeDottedTerm already wraps each part in quotes with prefix match
|
|
4085
|
-
if (negated) {
|
|
4086
|
-
// Wrap multi-token AND expression in parens for NOT negation
|
|
4087
|
-
negative.push(`(${sanitized})`);
|
|
4088
|
-
}
|
|
4089
|
-
else {
|
|
4090
|
-
// Flatten individual AND'd terms into the positive list so they combine
|
|
4091
|
-
// correctly with other terms (avoids double-wrapping in outer AND).
|
|
4092
|
-
for (const part of sanitized.split(' AND ')) {
|
|
4093
|
-
positive.push(part.trim());
|
|
4094
|
-
}
|
|
4095
|
-
}
|
|
4096
|
-
}
|
|
4097
|
-
}
|
|
4098
|
-
else if (containsCjk(term)) {
|
|
4098
|
+
if (containsCjk(term)) {
|
|
4099
4099
|
const sanitized = sanitizeFTS5Phrase(term);
|
|
4100
4100
|
if (sanitized) {
|
|
4101
4101
|
const ftsPhrase = `"${sanitized}"`; // CJK phrase over character tokens
|
|
@@ -4108,9 +4108,16 @@ function buildFTS5Query(query) {
|
|
|
4108
4108
|
}
|
|
4109
4109
|
}
|
|
4110
4110
|
else {
|
|
4111
|
-
|
|
4112
|
-
|
|
4113
|
-
|
|
4111
|
+
// Any separator inside the term (multi-agent, DEC-0054, 2026.4.10,
|
|
4112
|
+
// src/lib/i18n.ts, @tobilu/qmd) split it at index time too, so the term
|
|
4113
|
+
// has to be matched as the phrase those parts form. A term with no
|
|
4114
|
+
// separator is one part and keeps its prefix match, which is what makes
|
|
4115
|
+
// a plain word still match longer words that start with it.
|
|
4116
|
+
const parts = splitFTS5CompoundTerm(term);
|
|
4117
|
+
if (parts.length > 0) {
|
|
4118
|
+
const ftsTerm = parts.length > 1
|
|
4119
|
+
? `"${parts.join(' ')}"` // Phrase match (no prefix)
|
|
4120
|
+
: `"${parts[0]}"*`; // Prefix match
|
|
4114
4121
|
if (negated) {
|
|
4115
4122
|
negative.push(ftsTerm);
|
|
4116
4123
|
}
|
|
@@ -4170,7 +4177,28 @@ function normalizeCollectionFilter(filter) {
|
|
|
4170
4177
|
const names = typeof filter === "string" ? [filter] : filter;
|
|
4171
4178
|
return [...new Set(names.filter(name => name.length > 0))];
|
|
4172
4179
|
}
|
|
4173
|
-
function
|
|
4180
|
+
function scopedCollectionNames(scope) {
|
|
4181
|
+
if (scope == null)
|
|
4182
|
+
return undefined;
|
|
4183
|
+
const names = (typeof scope === "string" ? [scope] : Array.from(scope))
|
|
4184
|
+
.map(n => n.trim())
|
|
4185
|
+
.filter(n => n.length > 0);
|
|
4186
|
+
return names.length > 0 ? names : undefined;
|
|
4187
|
+
}
|
|
4188
|
+
function mergeSearchResultsByScore(lists, limit) {
|
|
4189
|
+
const best = new Map();
|
|
4190
|
+
for (const list of lists) {
|
|
4191
|
+
for (const r of list) {
|
|
4192
|
+
const prev = best.get(r.filepath);
|
|
4193
|
+
if (!prev || r.score > prev.score)
|
|
4194
|
+
best.set(r.filepath, r);
|
|
4195
|
+
}
|
|
4196
|
+
}
|
|
4197
|
+
return Array.from(best.values())
|
|
4198
|
+
.sort((a, b) => b.score - a.score)
|
|
4199
|
+
.slice(0, limit);
|
|
4200
|
+
}
|
|
4201
|
+
function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter, filter) {
|
|
4174
4202
|
// Keep collection membership inside the ranked FTS candidate set. Filtering
|
|
4175
4203
|
// after LIMIT can lose every matching row from a smaller collection.
|
|
4176
4204
|
const params = [ftsQuery];
|
|
@@ -4182,7 +4210,10 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4182
4210
|
? `AND filtered_d.active = 1 AND filtered_d.collection IN (${collections.map(() => "?").join(", ")})`
|
|
4183
4211
|
: "";
|
|
4184
4212
|
params.push(...collections);
|
|
4185
|
-
|
|
4213
|
+
// When filtering by metadata, fetch extra candidates from the FTS index
|
|
4214
|
+
// since some will be filtered out. Without a filter we can fetch exactly the requested limit.
|
|
4215
|
+
const ftsLimit = filter ? limit * 10 : limit;
|
|
4216
|
+
params.push(ftsLimit);
|
|
4186
4217
|
let sql = `
|
|
4187
4218
|
WITH fts_matches AS (
|
|
4188
4219
|
SELECT ${table}.rowid AS rowid, bm25(${table}, 1.5, 4.0, 1.0) as bm25_score
|
|
@@ -4199,12 +4230,21 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4199
4230
|
d.title,
|
|
4200
4231
|
content.doc as body,
|
|
4201
4232
|
d.hash,
|
|
4202
|
-
fm.bm25_score
|
|
4233
|
+
fm.bm25_score,
|
|
4234
|
+
dm.metadata_json
|
|
4203
4235
|
FROM fts_matches fm
|
|
4204
4236
|
JOIN documents d ON d.id = fm.rowid
|
|
4205
4237
|
JOIN content ON content.hash = d.hash
|
|
4238
|
+
LEFT JOIN document_metadata dm ON dm.document_id = d.id
|
|
4206
4239
|
WHERE d.active = 1
|
|
4207
4240
|
`;
|
|
4241
|
+
if (filter) {
|
|
4242
|
+
// Only documents with current, error-free extraction can match — an
|
|
4243
|
+
// unprocessed document must not accidentally satisfy `exists: false`.
|
|
4244
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4245
|
+
sql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`;
|
|
4246
|
+
params.push(...compiledFilter.params);
|
|
4247
|
+
}
|
|
4208
4248
|
// bm25 lower is better; sort ascending.
|
|
4209
4249
|
sql += ` ORDER BY fm.bm25_score ASC LIMIT ?`;
|
|
4210
4250
|
params.push(limit);
|
|
@@ -4227,6 +4267,7 @@ function searchFtsChannel(db, table, ftsQuery, limit, collectionFilter) {
|
|
|
4227
4267
|
bodyLength: row.body.length,
|
|
4228
4268
|
body: row.body,
|
|
4229
4269
|
context: getContextForFile(db, row.filepath),
|
|
4270
|
+
metadata: parseMetadataJson(row.metadata_json),
|
|
4230
4271
|
score,
|
|
4231
4272
|
source: "fts",
|
|
4232
4273
|
};
|
|
@@ -4238,17 +4279,21 @@ function buildCjkSignalQuery(signal) {
|
|
|
4238
4279
|
return null;
|
|
4239
4280
|
return terms.map(term => `"${term.replace(/"/g, '""')}"`).join(" AND ");
|
|
4240
4281
|
}
|
|
4241
|
-
export function searchCharFTS(db, query, limit = 20, collectionFilter) {
|
|
4282
|
+
export function searchCharFTS(db, query, limit = 20, collectionFilter, filter) {
|
|
4242
4283
|
if (!readOnlyDatabases.has(db))
|
|
4243
4284
|
repairDirtyCjkCharFallback(db);
|
|
4244
4285
|
const charQuery = buildFTS5Query(query);
|
|
4245
4286
|
if (!charQuery)
|
|
4246
4287
|
return [];
|
|
4247
|
-
return searchFtsChannel(db, "documents_fts", charQuery, limit, collectionFilter);
|
|
4288
|
+
return searchFtsChannel(db, "documents_fts", charQuery, limit, collectionFilter, filter);
|
|
4248
4289
|
}
|
|
4249
|
-
export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
4290
|
+
export function searchFTS(db, query, limit = 20, collectionFilter, filter) {
|
|
4291
|
+
const names = scopedCollectionNames(collectionFilter);
|
|
4292
|
+
if (names && names.length > 1) {
|
|
4293
|
+
return mergeSearchResultsByScore(names.map(name => searchFTS(db, query, limit, name, filter)), limit);
|
|
4294
|
+
}
|
|
4250
4295
|
if (!containsCjk(query))
|
|
4251
|
-
return searchCharFTS(db, query, limit, collectionFilter);
|
|
4296
|
+
return searchCharFTS(db, query, limit, collectionFilter, filter);
|
|
4252
4297
|
const charQuery = buildFTS5Query(query);
|
|
4253
4298
|
if (!charQuery)
|
|
4254
4299
|
return [];
|
|
@@ -4257,7 +4302,7 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4257
4302
|
const candidateDepth = Math.max(limit, CJK_LEXICAL_CANDIDATE_DEPTH);
|
|
4258
4303
|
const rankedChannels = [{
|
|
4259
4304
|
channel: "char",
|
|
4260
|
-
results: searchCharFTS(db, query, candidateDepth, collectionFilter),
|
|
4305
|
+
results: searchCharFTS(db, query, candidateDepth, collectionFilter, filter),
|
|
4261
4306
|
}];
|
|
4262
4307
|
let secondaryReason = null;
|
|
4263
4308
|
try {
|
|
@@ -4293,7 +4338,7 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4293
4338
|
}
|
|
4294
4339
|
rankedChannels.push({
|
|
4295
4340
|
channel: entry.channel,
|
|
4296
|
-
results: searchFtsChannel(db, entry.table, ftsQuery, candidateDepth, collectionFilter),
|
|
4341
|
+
results: searchFtsChannel(db, entry.table, ftsQuery, candidateDepth, collectionFilter, filter),
|
|
4297
4342
|
});
|
|
4298
4343
|
channels.push({ channel: entry.channel, status: "used" });
|
|
4299
4344
|
}
|
|
@@ -4333,6 +4378,49 @@ export function searchFTS(db, query, limit = 20, collectionFilter) {
|
|
|
4333
4378
|
// =============================================================================
|
|
4334
4379
|
// Vector Search
|
|
4335
4380
|
// =============================================================================
|
|
4381
|
+
/** sqlite-vec rejects k above this in MATCH queries (v0.1.9). */
|
|
4382
|
+
const SQLITE_VEC_MAX_K = 4096;
|
|
4383
|
+
/**
|
|
4384
|
+
* Max filter-eligible vectors for an exact cosine scan. Above this we fall
|
|
4385
|
+
* back to global ANN with a capped over-fetch. Exact scan avoids the
|
|
4386
|
+
* post-filter starvation of small eligible sets — originally small
|
|
4387
|
+
* collections (#791, #803), now also selective metadata filters; ANN remains
|
|
4388
|
+
* for very large eligible sets where a full scan would be expensive.
|
|
4389
|
+
*/
|
|
4390
|
+
const FILTERED_VEC_EXACT_SCAN_MAX = 20_000;
|
|
4391
|
+
const VEC_HASH_SEQ_IN_CHUNK = 400;
|
|
4392
|
+
/**
|
|
4393
|
+
* Exact cosine-distance scan over a known set of hash_seq keys.
|
|
4394
|
+
* Uses vec_distance_cosine with chunked IN lists (no JOIN with vectors_vec).
|
|
4395
|
+
*/
|
|
4396
|
+
function exactVecScanByHashSeq(db, embedding, hashSeqs, limit) {
|
|
4397
|
+
if (hashSeqs.length === 0 || limit <= 0)
|
|
4398
|
+
return [];
|
|
4399
|
+
const queryVec = new Float32Array(embedding);
|
|
4400
|
+
// Over-fetch a bit so multi-chunk docs can still yield `limit` unique files.
|
|
4401
|
+
const fetchLimit = Math.max(limit * 3, limit);
|
|
4402
|
+
const scored = [];
|
|
4403
|
+
for (let i = 0; i < hashSeqs.length; i += VEC_HASH_SEQ_IN_CHUNK) {
|
|
4404
|
+
const chunk = hashSeqs.slice(i, i + VEC_HASH_SEQ_IN_CHUNK);
|
|
4405
|
+
const placeholders = chunk.map(() => "?").join(",");
|
|
4406
|
+
const rows = db.prepare(`
|
|
4407
|
+
SELECT hash_seq, vec_distance_cosine(embedding, ?) AS distance
|
|
4408
|
+
FROM vectors_vec
|
|
4409
|
+
WHERE hash_seq IN (${placeholders})
|
|
4410
|
+
`).all(queryVec, ...chunk);
|
|
4411
|
+
scored.push(...rows);
|
|
4412
|
+
}
|
|
4413
|
+
scored.sort((a, b) => a.distance - b.distance);
|
|
4414
|
+
return scored.slice(0, fetchLimit);
|
|
4415
|
+
}
|
|
4416
|
+
function annVecScan(db, embedding, k) {
|
|
4417
|
+
const vecK = Math.max(1, Math.min(SQLITE_VEC_MAX_K, k));
|
|
4418
|
+
return db.prepare(`
|
|
4419
|
+
SELECT hash_seq, distance
|
|
4420
|
+
FROM vectors_vec
|
|
4421
|
+
WHERE embedding MATCH ? AND k = ?
|
|
4422
|
+
`).all(new Float32Array(embedding), vecK);
|
|
4423
|
+
}
|
|
4336
4424
|
function resolveReadyProviderEmbeddingIdentity(db, provider) {
|
|
4337
4425
|
const storedIdentity = readStoredEmbeddingIdentity(db);
|
|
4338
4426
|
if (!storedIdentity || storedIdentity.providerId !== provider.providerId
|
|
@@ -4373,10 +4461,29 @@ function hasSearchableVectorIndex(store) {
|
|
|
4373
4461
|
return storedIdentity !== undefined
|
|
4374
4462
|
&& inspectEmbeddingIndexState(store.db, storedIdentity).status === "ready";
|
|
4375
4463
|
}
|
|
4376
|
-
export async function searchVec(db, query, model, limit = 20, collectionFilter, session, precomputedEmbedding,
|
|
4464
|
+
export async function searchVec(db, query, model, limit = 20, collectionFilter, session, precomputedEmbedding, providerOrLlm, authorizeRemoteRequestOrFilter, llmOverride, metadataFilter) {
|
|
4377
4465
|
const tableExists = db.prepare(`SELECT name FROM sqlite_master WHERE type='table' AND name='vectors_vec'`).get();
|
|
4378
4466
|
if (!tableExists)
|
|
4379
4467
|
return [];
|
|
4468
|
+
// Disambiguate overloaded arguments
|
|
4469
|
+
let provider;
|
|
4470
|
+
let authorizeRemoteRequest;
|
|
4471
|
+
let filter = metadataFilter;
|
|
4472
|
+
let effectiveLlmOverride = llmOverride;
|
|
4473
|
+
if (providerOrLlm && typeof providerOrLlm === "object") {
|
|
4474
|
+
if ("providerId" in providerOrLlm) {
|
|
4475
|
+
provider = providerOrLlm;
|
|
4476
|
+
}
|
|
4477
|
+
else {
|
|
4478
|
+
effectiveLlmOverride = providerOrLlm;
|
|
4479
|
+
}
|
|
4480
|
+
}
|
|
4481
|
+
if (typeof authorizeRemoteRequestOrFilter === "function") {
|
|
4482
|
+
authorizeRemoteRequest = authorizeRemoteRequestOrFilter;
|
|
4483
|
+
}
|
|
4484
|
+
else if (typeof authorizeRemoteRequestOrFilter === "object" && authorizeRemoteRequestOrFilter !== null && !filter) {
|
|
4485
|
+
filter = authorizeRemoteRequestOrFilter;
|
|
4486
|
+
}
|
|
4380
4487
|
if (provider && model !== provider.model) {
|
|
4381
4488
|
throw new Error(`Embedding model ${model} does not match borrowed provider model ${provider.model}.`);
|
|
4382
4489
|
}
|
|
@@ -4405,13 +4512,19 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4405
4512
|
})).vector;
|
|
4406
4513
|
}
|
|
4407
4514
|
else if (!embedding) {
|
|
4408
|
-
embedding = await getEmbedding(query, model, true, session,
|
|
4515
|
+
embedding = await getEmbedding(query, model, true, session, effectiveLlmOverride);
|
|
4409
4516
|
}
|
|
4410
4517
|
if (!embedding)
|
|
4411
4518
|
return [];
|
|
4519
|
+
// Multi-collection union: search each collection separately to avoid starvation
|
|
4520
|
+
const names = scopedCollectionNames(collectionFilter);
|
|
4521
|
+
if (names && names.length > 1) {
|
|
4522
|
+
const lists = await Promise.all(names.map(name => searchVec(db, query, model, limit, name, session, embedding, provider ?? effectiveLlmOverride, authorizeRemoteRequest ?? filter, effectiveLlmOverride, filter)));
|
|
4523
|
+
return mergeSearchResultsByScore(lists, limit);
|
|
4524
|
+
}
|
|
4412
4525
|
const activeFingerprint = providerIdentity
|
|
4413
4526
|
? providerIdentity.fingerprint
|
|
4414
|
-
: storedIdentity
|
|
4527
|
+
: storedIdentity?.fingerprint;
|
|
4415
4528
|
const toSearchResult = (row, body, distance) => {
|
|
4416
4529
|
return {
|
|
4417
4530
|
filepath: row.filepath,
|
|
@@ -4424,56 +4537,87 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4424
4537
|
bodyLength: body.length,
|
|
4425
4538
|
body,
|
|
4426
4539
|
context: getContextForFile(db, row.filepath),
|
|
4540
|
+
metadata: parseMetadataJson(row.metadata_json),
|
|
4427
4541
|
score: 1 - distance, // Cosine similarity = 1 - cosine distance
|
|
4428
4542
|
source: "vec",
|
|
4429
4543
|
chunkPos: row.pos,
|
|
4430
4544
|
};
|
|
4431
4545
|
};
|
|
4432
4546
|
const collections = normalizeCollectionFilter(collectionFilter);
|
|
4433
|
-
const queryVec = embedding instanceof Float32Array ? embedding : new Float32Array(embedding);
|
|
4434
|
-
const SQLITE_VEC_MAX_K = 4096;
|
|
4435
|
-
let collectionFilterSql = "";
|
|
4436
|
-
const queryParams = [queryVec];
|
|
4437
|
-
if (collections.length === 1) {
|
|
4438
|
-
collectionFilterSql = "AND collection = ?";
|
|
4439
|
-
queryParams.push(collections[0]);
|
|
4440
|
-
}
|
|
4441
|
-
else if (collections.length > 1) {
|
|
4442
|
-
const placeholders = collections.map(() => "?").join(", ");
|
|
4443
|
-
collectionFilterSql = `AND collection IN (${placeholders})`;
|
|
4444
|
-
queryParams.push(...collections);
|
|
4445
|
-
}
|
|
4446
|
-
// Request generous k to account for multi-chunk deduplication per document
|
|
4447
|
-
const fetchK = Math.min(SQLITE_VEC_MAX_K, Math.max(limit * 15, 60));
|
|
4448
|
-
queryParams.push(fetchK);
|
|
4449
4547
|
let vecRows;
|
|
4450
|
-
|
|
4451
|
-
|
|
4452
|
-
|
|
4453
|
-
|
|
4454
|
-
|
|
4455
|
-
|
|
4456
|
-
|
|
4457
|
-
|
|
4548
|
+
if (filter) {
|
|
4549
|
+
let eligibleSql = `
|
|
4550
|
+
SELECT DISTINCT cv.hash || '_' || cv.seq AS hash_seq
|
|
4551
|
+
FROM content_vectors cv
|
|
4552
|
+
JOIN documents d ON d.hash = cv.hash AND d.active = 1
|
|
4553
|
+
JOIN document_metadata dm ON dm.document_id = d.id
|
|
4554
|
+
`;
|
|
4555
|
+
const eligibleConditions = [];
|
|
4556
|
+
const eligibleParams = [];
|
|
4557
|
+
if (activeFingerprint) {
|
|
4558
|
+
eligibleConditions.push(`cv.model = ? AND cv.embed_fingerprint = ?`);
|
|
4559
|
+
eligibleParams.push(model, activeFingerprint);
|
|
4560
|
+
}
|
|
4561
|
+
if (collections.length === 1) {
|
|
4562
|
+
eligibleConditions.push(`d.collection = ?`);
|
|
4563
|
+
eligibleParams.push(collections[0]);
|
|
4564
|
+
}
|
|
4565
|
+
else if (collections.length > 1) {
|
|
4566
|
+
eligibleConditions.push(`d.collection IN (${collections.map(() => "?").join(", ")})`);
|
|
4567
|
+
eligibleParams.push(...collections);
|
|
4568
|
+
}
|
|
4569
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4570
|
+
eligibleConditions.push(`dm.extraction_version = ${METADATA_EXTRACTION_VERSION}`);
|
|
4571
|
+
eligibleConditions.push(`dm.extraction_error IS NULL`);
|
|
4572
|
+
eligibleConditions.push(compiledFilter.sql);
|
|
4573
|
+
eligibleParams.push(...compiledFilter.params);
|
|
4574
|
+
eligibleSql += ` WHERE ${eligibleConditions.join(" AND ")}`;
|
|
4575
|
+
const eligibleHashSeqs = withLazyContentVectorMigration(db, () => db.prepare(eligibleSql).all(...eligibleParams)).map((r) => r.hash_seq);
|
|
4576
|
+
if (eligibleHashSeqs.length === 0)
|
|
4577
|
+
return [];
|
|
4578
|
+
if (eligibleHashSeqs.length <= FILTERED_VEC_EXACT_SCAN_MAX) {
|
|
4579
|
+
vecRows = exactVecScanByHashSeq(db, embedding, eligibleHashSeqs, limit);
|
|
4580
|
+
}
|
|
4581
|
+
else {
|
|
4582
|
+
vecRows = annVecScan(db, embedding, Math.max(limit * 30, limit * 3));
|
|
4583
|
+
}
|
|
4458
4584
|
}
|
|
4459
|
-
|
|
4460
|
-
//
|
|
4461
|
-
|
|
4462
|
-
|
|
4463
|
-
|
|
4464
|
-
|
|
4465
|
-
|
|
4466
|
-
|
|
4585
|
+
else {
|
|
4586
|
+
// No metadata filter: use fast ANN scan (with collection pushdown if supported)
|
|
4587
|
+
const queryVec = embedding instanceof Float32Array ? embedding : new Float32Array(embedding);
|
|
4588
|
+
let collectionFilterSql = "";
|
|
4589
|
+
const queryParams = [queryVec];
|
|
4590
|
+
if (collections.length === 1) {
|
|
4591
|
+
collectionFilterSql = "AND collection = ?";
|
|
4592
|
+
queryParams.push(collections[0]);
|
|
4593
|
+
}
|
|
4594
|
+
else if (collections.length > 1) {
|
|
4595
|
+
const placeholders = collections.map(() => "?").join(", ");
|
|
4596
|
+
collectionFilterSql = `AND collection IN (${placeholders})`;
|
|
4597
|
+
queryParams.push(...collections);
|
|
4598
|
+
}
|
|
4599
|
+
const fetchK = Math.min(SQLITE_VEC_MAX_K, Math.max(limit * 15, 60));
|
|
4600
|
+
queryParams.push(fetchK);
|
|
4601
|
+
try {
|
|
4602
|
+
vecRows = withLazyContentVectorMigration(db, () => db.prepare(`
|
|
4603
|
+
SELECT hash_seq, distance
|
|
4604
|
+
FROM vectors_vec
|
|
4605
|
+
WHERE embedding MATCH ?
|
|
4606
|
+
${collectionFilterSql}
|
|
4607
|
+
AND k = ?
|
|
4608
|
+
`).all(...queryParams));
|
|
4609
|
+
}
|
|
4610
|
+
catch {
|
|
4611
|
+
// Fallback if table lacks collection column before migration
|
|
4612
|
+
vecRows = annVecScan(db, embedding, fetchK);
|
|
4613
|
+
}
|
|
4467
4614
|
}
|
|
4468
4615
|
if (vecRows.length === 0)
|
|
4469
4616
|
return [];
|
|
4470
4617
|
const keys = vecRows.map(r => r.hash_seq);
|
|
4471
4618
|
const distMap = new Map(vecRows.map(r => [r.hash_seq, r.distance]));
|
|
4472
4619
|
const placeholders = keys.map(() => "?").join(", ");
|
|
4473
|
-
|
|
4474
|
-
? `AND d.collection IN (${collections.map(() => "?").join(", ")})`
|
|
4475
|
-
: "";
|
|
4476
|
-
const metaRows = withLazyContentVectorMigration(db, () => db.prepare(`
|
|
4620
|
+
let metaSql = `
|
|
4477
4621
|
SELECT
|
|
4478
4622
|
cv.hash || '_' || cv.seq AS hash_seq,
|
|
4479
4623
|
cv.hash,
|
|
@@ -4481,14 +4625,32 @@ export async function searchVec(db, query, model, limit = 20, collectionFilter,
|
|
|
4481
4625
|
'qmd://' || d.collection || '/' || d.path AS filepath,
|
|
4482
4626
|
d.collection || '/' || d.path AS display_path,
|
|
4483
4627
|
d.title,
|
|
4484
|
-
d.collection
|
|
4628
|
+
d.collection,
|
|
4629
|
+
dm.metadata_json
|
|
4485
4630
|
FROM content_vectors cv
|
|
4486
4631
|
JOIN documents d ON d.hash = cv.hash AND d.active = 1
|
|
4487
|
-
|
|
4488
|
-
|
|
4489
|
-
|
|
4490
|
-
|
|
4491
|
-
|
|
4632
|
+
LEFT JOIN document_metadata dm ON dm.document_id = d.id
|
|
4633
|
+
WHERE (cv.hash || '_' || cv.seq) IN (${placeholders})
|
|
4634
|
+
`;
|
|
4635
|
+
const metaParams = [...keys];
|
|
4636
|
+
if (activeFingerprint) {
|
|
4637
|
+
metaSql += ` AND cv.model = ? AND cv.embed_fingerprint = ?`;
|
|
4638
|
+
metaParams.push(model, activeFingerprint);
|
|
4639
|
+
}
|
|
4640
|
+
if (collections.length === 1) {
|
|
4641
|
+
metaSql += ` AND d.collection = ?`;
|
|
4642
|
+
metaParams.push(collections[0]);
|
|
4643
|
+
}
|
|
4644
|
+
else if (collections.length > 1) {
|
|
4645
|
+
metaSql += ` AND d.collection IN (${collections.map(() => "?").join(", ")})`;
|
|
4646
|
+
metaParams.push(...collections);
|
|
4647
|
+
}
|
|
4648
|
+
if (filter) {
|
|
4649
|
+
const compiledFilter = compileMetadataFilter(filter, "d");
|
|
4650
|
+
metaSql += ` AND dm.extraction_version = ${METADATA_EXTRACTION_VERSION} AND dm.extraction_error IS NULL AND ${compiledFilter.sql}`;
|
|
4651
|
+
metaParams.push(...compiledFilter.params);
|
|
4652
|
+
}
|
|
4653
|
+
const metaRows = withLazyContentVectorMigration(db, () => db.prepare(metaSql).all(...metaParams));
|
|
4492
4654
|
const byFile = new Map();
|
|
4493
4655
|
for (const m of metaRows) {
|
|
4494
4656
|
const dist = distMap.get(m.hash_seq);
|
|
@@ -5318,6 +5480,7 @@ export function getStatusReadOnly(db, needsEmbedding) {
|
|
|
5318
5480
|
totalDocuments: totalDocs,
|
|
5319
5481
|
needsEmbedding,
|
|
5320
5482
|
hasVectorIndex: hasVectors,
|
|
5483
|
+
pendingMetadata: countDocumentsPendingMetadata(db),
|
|
5321
5484
|
collections,
|
|
5322
5485
|
};
|
|
5323
5486
|
}
|
|
@@ -5443,6 +5606,15 @@ export function addLineNumbers(text, startLine = 1) {
|
|
|
5443
5606
|
const lines = text.split('\n');
|
|
5444
5607
|
return lines.map((line, i) => `${startLine + i}: ${line}`).join('\n');
|
|
5445
5608
|
}
|
|
5609
|
+
/**
|
|
5610
|
+
* Attach canonical metadata to final search results with one batch query.
|
|
5611
|
+
* Runs after RRF/reranking so metadata is never duplicated through the
|
|
5612
|
+
* intermediate ranked lists.
|
|
5613
|
+
*/
|
|
5614
|
+
function attachResultMetadata(db, results) {
|
|
5615
|
+
const metadataByFilepath = getMetadataByFilepath(db, results.map(r => r.file));
|
|
5616
|
+
return results.map(r => ({ ...r, metadata: metadataByFilepath.get(r.file) ?? {} }));
|
|
5617
|
+
}
|
|
5446
5618
|
/**
|
|
5447
5619
|
* RRF list weights for hybridQuery.
|
|
5448
5620
|
*
|
|
@@ -5477,6 +5649,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5477
5649
|
const collectionFilter = options?.collections && options.collections.length > 0
|
|
5478
5650
|
? options.collections
|
|
5479
5651
|
: options?.collection;
|
|
5652
|
+
const filter = options?.filter;
|
|
5480
5653
|
const explain = options?.explain ?? false;
|
|
5481
5654
|
const expansionContext = options?.expansionContext;
|
|
5482
5655
|
const rerankContext = options?.rerankContext;
|
|
@@ -5491,9 +5664,9 @@ export async function hybridQuery(store, query, options) {
|
|
|
5491
5664
|
// When either context is provided, disable strong-signal bypass — the obvious BM25
|
|
5492
5665
|
// match may not be what the caller wants (e.g. "performance" with context
|
|
5493
5666
|
// "web page load times" should NOT shortcut to a sports-performance doc).
|
|
5494
|
-
// Pass collection directly into FTS query (filter at SQL level, not post-hoc)
|
|
5667
|
+
// Pass collection and metadata filter directly into FTS query (filter at SQL level, not post-hoc)
|
|
5495
5668
|
const parsedDirective = parseExpansionDirective(query);
|
|
5496
|
-
const initialFts = store.searchFTS(parsedDirective.query, 20, collectionFilter);
|
|
5669
|
+
const initialFts = store.searchFTS(parsedDirective.query, 20, collectionFilter, filter);
|
|
5497
5670
|
const strongSignal = getLexicalStrongSignal(initialFts);
|
|
5498
5671
|
const topScore = strongSignal.topScore;
|
|
5499
5672
|
let expansionDecision;
|
|
@@ -5561,7 +5734,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5561
5734
|
// 3a: Run FTS for all lex expansions right away (no LLM needed)
|
|
5562
5735
|
for (const q of expanded) {
|
|
5563
5736
|
if (q.type === 'lex') {
|
|
5564
|
-
const ftsResults = store.searchFTS(q.query, 20, collectionFilter);
|
|
5737
|
+
const ftsResults = store.searchFTS(q.query, 20, collectionFilter, filter);
|
|
5565
5738
|
if (ftsResults.length > 0) {
|
|
5566
5739
|
for (const r of ftsResults)
|
|
5567
5740
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5593,7 +5766,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5593
5766
|
const embedding = embeddings[i];
|
|
5594
5767
|
if (!embedding || embedding.length === 0)
|
|
5595
5768
|
continue;
|
|
5596
|
-
const vecResults = await store.searchVec(vecQueries[i].text, embedModel, 20, collectionFilter, undefined, embedding);
|
|
5769
|
+
const vecResults = await store.searchVec(vecQueries[i].text, embedModel, 20, collectionFilter, undefined, embedding, filter);
|
|
5597
5770
|
if (vecResults.length > 0) {
|
|
5598
5771
|
for (const r of vecResults)
|
|
5599
5772
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5660,7 +5833,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5660
5833
|
if (skipRerank) {
|
|
5661
5834
|
// Skip LLM reranking — return candidates scored by RRF only
|
|
5662
5835
|
const seenFiles = new Set();
|
|
5663
|
-
|
|
5836
|
+
const rrfResults = candidates
|
|
5664
5837
|
.map((cand, i) => {
|
|
5665
5838
|
const chunkInfo = docChunkMap.get(cand.file);
|
|
5666
5839
|
const bestIdx = chunkInfo?.bestIdx ?? 0;
|
|
@@ -5705,6 +5878,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5705
5878
|
})
|
|
5706
5879
|
.filter(r => r.score >= minScore)
|
|
5707
5880
|
.slice(0, limit);
|
|
5881
|
+
return attachResultMetadata(store.db, rrfResults);
|
|
5708
5882
|
}
|
|
5709
5883
|
// Step 6: Rerank chunks (NOT full bodies)
|
|
5710
5884
|
const chunksToRerank = [];
|
|
@@ -5771,7 +5945,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5771
5945
|
}).sort((a, b) => b.score - a.score);
|
|
5772
5946
|
// Step 8: Dedup by file (safety net — prevents duplicate output)
|
|
5773
5947
|
const seenFiles = new Set();
|
|
5774
|
-
|
|
5948
|
+
const finalResults = blended
|
|
5775
5949
|
.filter(r => {
|
|
5776
5950
|
if (seenFiles.has(r.file))
|
|
5777
5951
|
return false;
|
|
@@ -5780,6 +5954,7 @@ export async function hybridQuery(store, query, options) {
|
|
|
5780
5954
|
})
|
|
5781
5955
|
.filter(r => r.score >= minScore)
|
|
5782
5956
|
.slice(0, limit);
|
|
5957
|
+
return attachResultMetadata(store.db, finalResults);
|
|
5783
5958
|
}
|
|
5784
5959
|
/**
|
|
5785
5960
|
* Vector-only semantic search with query expansion.
|
|
@@ -5794,6 +5969,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5794
5969
|
const limit = options?.limit ?? 10;
|
|
5795
5970
|
const minScore = options?.minScore ?? 0.3;
|
|
5796
5971
|
const collection = options?.collection;
|
|
5972
|
+
const filter = options?.filter;
|
|
5797
5973
|
const expansionContext = options?.expansionContext;
|
|
5798
5974
|
const includeHyde = options?.includeHyde ?? true;
|
|
5799
5975
|
if (!hasSearchableVectorIndex(store))
|
|
@@ -5809,7 +5985,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5809
5985
|
const queryTexts = [query, ...vecExpanded.map(q => q.query)];
|
|
5810
5986
|
const allResults = new Map();
|
|
5811
5987
|
for (const q of queryTexts) {
|
|
5812
|
-
const vecResults = await store.searchVec(q, embedModel, limit, collection);
|
|
5988
|
+
const vecResults = await store.searchVec(q, embedModel, limit, collection, undefined, undefined, filter);
|
|
5813
5989
|
for (const r of vecResults) {
|
|
5814
5990
|
const existing = allResults.get(r.filepath);
|
|
5815
5991
|
if (!existing || r.score > existing.score) {
|
|
@@ -5821,6 +5997,7 @@ export async function vectorSearchQuery(store, query, options) {
|
|
|
5821
5997
|
score: r.score,
|
|
5822
5998
|
context: store.getContextForFile(r.filepath),
|
|
5823
5999
|
docid: r.docid,
|
|
6000
|
+
metadata: r.metadata,
|
|
5824
6001
|
});
|
|
5825
6002
|
}
|
|
5826
6003
|
}
|
|
@@ -5857,6 +6034,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5857
6034
|
const skipRerank = options?.skipRerank ?? false;
|
|
5858
6035
|
const hooks = options?.hooks;
|
|
5859
6036
|
const collections = options?.collections;
|
|
6037
|
+
const filter = options?.filter;
|
|
5860
6038
|
if (searches.length === 0)
|
|
5861
6039
|
return [];
|
|
5862
6040
|
// Validate queries before executing
|
|
@@ -5889,7 +6067,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5889
6067
|
for (const [searchIndex, search] of searches.entries()) {
|
|
5890
6068
|
if (search.type === 'lex') {
|
|
5891
6069
|
for (const coll of collectionList) {
|
|
5892
|
-
const ftsResults = store.searchFTS(search.query, 20, coll);
|
|
6070
|
+
const ftsResults = store.searchFTS(search.query, 20, coll, filter);
|
|
5893
6071
|
if (ftsResults.length > 0) {
|
|
5894
6072
|
for (const r of ftsResults)
|
|
5895
6073
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5922,7 +6100,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5922
6100
|
if (!embedding || embedding.length === 0)
|
|
5923
6101
|
continue;
|
|
5924
6102
|
for (const coll of collectionList) {
|
|
5925
|
-
const vecResults = await store.searchVec(vecSearches[i].search.query, embedModel, 20, coll, undefined, embedding);
|
|
6103
|
+
const vecResults = await store.searchVec(vecSearches[i].search.query, embedModel, 20, coll, undefined, embedding, filter);
|
|
5926
6104
|
if (vecResults.length > 0) {
|
|
5927
6105
|
for (const r of vecResults)
|
|
5928
6106
|
docidMap.set(r.filepath, r.docid);
|
|
@@ -5985,7 +6163,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
5985
6163
|
if (skipRerank) {
|
|
5986
6164
|
// Skip LLM reranking — return candidates scored by RRF only
|
|
5987
6165
|
const seenFiles = new Set();
|
|
5988
|
-
|
|
6166
|
+
const rrfResults = candidates
|
|
5989
6167
|
.map((cand, i) => {
|
|
5990
6168
|
const chunkInfo = docChunkMap.get(cand.file);
|
|
5991
6169
|
const bestIdx = chunkInfo?.bestIdx ?? 0;
|
|
@@ -6030,6 +6208,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6030
6208
|
})
|
|
6031
6209
|
.filter(r => r.score >= minScore)
|
|
6032
6210
|
.slice(0, limit);
|
|
6211
|
+
return attachResultMetadata(store.db, rrfResults);
|
|
6033
6212
|
}
|
|
6034
6213
|
// Step 5: Rerank chunks
|
|
6035
6214
|
const chunksToRerank = [];
|
|
@@ -6095,7 +6274,7 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6095
6274
|
}).sort((a, b) => b.score - a.score);
|
|
6096
6275
|
// Step 7: Dedup by file
|
|
6097
6276
|
const seenFiles = new Set();
|
|
6098
|
-
|
|
6277
|
+
const finalResults = blended
|
|
6099
6278
|
.filter(r => {
|
|
6100
6279
|
if (seenFiles.has(r.file))
|
|
6101
6280
|
return false;
|
|
@@ -6104,4 +6283,5 @@ export async function structuredSearch(store, searches, options) {
|
|
|
6104
6283
|
})
|
|
6105
6284
|
.filter(r => r.score >= minScore)
|
|
6106
6285
|
.slice(0, limit);
|
|
6286
|
+
return attachResultMetadata(store.db, finalResults);
|
|
6107
6287
|
}
|