@wei840222/qmd 2026.9.28 → 2026.9.30
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +6 -1
- package/dist/cli/build-info.json +2 -2
- package/dist/remote-jev.js +15 -1
- package/dist/store.d.ts +25 -0
- package/dist/store.js +219 -65
- package/package.json +45 -44
package/CHANGELOG.md
CHANGED
|
@@ -2,6 +2,10 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
### Changed
|
|
6
|
+
|
|
7
|
+
- Dependency upgrades and runtime engine modernization: Updated Node.js engine target to `>=24.21.0` (LTS Krypton) and packageManager to `pnpm@12.6.0`. Upgraded dependencies and toolchain including `node-llama-cpp` to `3.22.1`, `zod` to `4.6.5`, `@modelcontextprotocol/server` to `2.1.0`, `picomatch` to `4.0.7`, `yaml` to `2.9.1`, `tsx` to `4.23.15`, `oxlint` to `1.85.0`, `vitest` to `4.1.11` (patching security advisory GHSA-82fw-gwwq-j7x9), and refreshed patch/minor dependency overrides.
|
|
8
|
+
|
|
5
9
|
### Fixed
|
|
6
10
|
|
|
7
11
|
- Metadata extraction error retry: `isDocumentMetadataCurrent` now requires an error-free extraction, allowing `qmd update` to automatically re-attempt extraction on documents that previously failed without requiring manual edits.
|
|
@@ -10,6 +14,7 @@
|
|
|
10
14
|
|
|
11
15
|
### Added
|
|
12
16
|
|
|
17
|
+
- In-flight Singleflight Deduplication & In-Memory LRU Cache: Added granular in-memory `Map<string, Promise<T>>` query-level singleflight deduplication (`inflightExpansions` and `inflightEmbeddings`) and `LRUCache` fallbacks (`memoryLlmCache` and `memoryEmbeddingCache`, powered by `lru-cache`) in `expandQuery` and `embedQueriesForStore`. Features configurable TTL (`2 hours`), item bounds (1,000 for LLM responses, 2,000 for embeddings), and memory size bounds (50 MB for LLM text, 64 MB for float64 embedding vectors) with automatic byte size calculation. In `readOnly: true` environments (such as MCP stdio/HTTP servers) where SQLite write is disabled, query expansions and embeddings gracefully cache in memory. Concurrent requests seamlessly share in-flight promises across identical or partially overlapping query batches, and subsequent queries hit the memory cache, eliminating redundant remote LLM/Jev/embedding API calls and network latency.
|
|
13
18
|
- TypeSafe Jev provider (`src/remote-jev.ts`) supporting System One-based candidate reranking via Noul judgments and query expansion intent classification via Choice and Noul gating. Passes query expansion context to Jev state for contextual intent classification.
|
|
14
19
|
- Refactored `HybridLLM` into `Hybrid` (`src/hybrid.ts`) supporting 3-way provider fallback: `RemoteJev` → `RemoteLLM` → `LlamaCpp`.
|
|
15
20
|
- Added `models.jev_api_key`, `models.jev_api_model`, and `models.jev_base_url` configuration options with `TYPESAFE_API_KEY`, `TYPESAFE_DEFAULT_MODEL`, and `TYPESAFE_BASE_URL` environment variable fallbacks.
|
|
@@ -18,7 +23,7 @@
|
|
|
18
23
|
- Document file path and title metadata in candidate reranking: candidate documents for chat-based and remote rerankers now retain document file paths and titles, enabling rerankers to evaluate provenance and temporal references.
|
|
19
24
|
- Relative temporal resolution in query expansion: `expandQuery` prompts now instruct models to calculate exact ISO dates from relative time expressions (e.g. "yesterday", "昨天", "today", "last week") against current local time for lexical and vector search.
|
|
20
25
|
- Added file, title, and current local time to TypeSafe Jev candidate reranking and intent classification states, with temporal constraint guidance.
|
|
21
|
-
- Search intent playbook and guidance for query expansion: Added `JEV_STRATEGY_PLAYBOOK` mapping Jev intent classifications (`code_search`, `concept_search`, `factual_lookup`, `broad_exploration`) to concrete lexical and vector search guidance passed to downstream expansion LLMs in dedicated `<search_intent>` XML blocks, keeping untrusted context separate.
|
|
26
|
+
- Search intent playbook and guidance for query expansion: Added `JEV_STRATEGY_PLAYBOOK` mapping Jev intent classifications (`troubleshooting`, `how_to_guide`, `code_search`, `concept_search`, `factual_lookup`, `broad_exploration`) to concrete lexical and vector search guidance passed to downstream expansion LLMs in dedicated `<search_intent>` XML blocks, keeping untrusted context separate.
|
|
22
27
|
|
|
23
28
|
## [2026.9.25] - 2026-09-26
|
|
24
29
|
|
package/dist/cli/build-info.json
CHANGED
package/dist/remote-jev.js
CHANGED
|
@@ -3,6 +3,18 @@ import { getFormattedLocalTime } from "./remote-llm.js";
|
|
|
3
3
|
export const DEFAULT_JEV_TIMEOUT_MS = 30000;
|
|
4
4
|
export const DEFAULT_JEV_RERANK_BATCH_SIZE = 40;
|
|
5
5
|
export const JEV_STRATEGY_PLAYBOOK = {
|
|
6
|
+
troubleshooting: {
|
|
7
|
+
label: "Troubleshooting & Bug Fix",
|
|
8
|
+
objective: "Diagnosing errors, exceptions, stack traces, failure modes, or broken states.",
|
|
9
|
+
lexGuidance: "Prioritize exact error codes, exception class names, HTTP statuses, and failed syscall/function identifiers. Exclude generic noise like 'why', 'error', 'issue'.",
|
|
10
|
+
vecGuidance: "Formulate concrete diagnostic or remediation questions (e.g., 'how to resolve <error> caused by <reason>').",
|
|
11
|
+
},
|
|
12
|
+
how_to_guide: {
|
|
13
|
+
label: "How-To & Procedural Guide",
|
|
14
|
+
objective: "Looking for step-by-step setup guides, workflows, recipes, installation, or migration steps.",
|
|
15
|
+
lexGuidance: "Include tool/command names, CLI flags, configuration filenames, and workflow action verbs without extra filler words.",
|
|
16
|
+
vecGuidance: "Formulate procedural task questions (e.g., 'step-by-step guide to configure or implement <task>').",
|
|
17
|
+
},
|
|
6
18
|
code_search: {
|
|
7
19
|
label: "Code Search",
|
|
8
20
|
objective: "Looking for specific code, functions, APIs, syntax, or implementations.",
|
|
@@ -74,8 +86,10 @@ export class RemoteJev {
|
|
|
74
86
|
model: this.model,
|
|
75
87
|
questions: {
|
|
76
88
|
strategy: choice("What type of search is the user performing given the query and optional context?", {
|
|
89
|
+
troubleshooting: "Diagnosing errors, exceptions, stack traces, or broken behavior",
|
|
90
|
+
how_to_guide: "Step-by-step instructions, installation, configuration, or migration workflows",
|
|
77
91
|
code_search: "Looking for specific code, functions, APIs, or implementations",
|
|
78
|
-
concept_search: "Looking for explanations, concepts, or documentation",
|
|
92
|
+
concept_search: "Looking for explanations, concepts, architecture, or documentation",
|
|
79
93
|
factual_lookup: "Looking for specific facts, configurations, or settings",
|
|
80
94
|
broad_exploration: "Exploring a topic broadly without a specific target",
|
|
81
95
|
}),
|
package/dist/store.d.ts
CHANGED
|
@@ -18,6 +18,16 @@ import type { LLM } from "./llm.js";
|
|
|
18
18
|
import { LlamaCpp, formatQueryForEmbedding, formatDocForEmbedding, type ILLMSession } from "./llm.js";
|
|
19
19
|
import type { NamedCollection, Collection, CollectionConfig } from "./collections.js";
|
|
20
20
|
import type { IndexDiagnostics } from "./diagnostics.js";
|
|
21
|
+
import { LRUCache } from "lru-cache";
|
|
22
|
+
export { LRUCache, LRUCache as LruCache } from "lru-cache";
|
|
23
|
+
export declare const DEFAULT_MEMORY_CACHE_TTL_MS: number;
|
|
24
|
+
export declare const DEFAULT_MEMORY_LLM_CACHE_MAX_ITEMS = 1000;
|
|
25
|
+
export declare const DEFAULT_MEMORY_LLM_CACHE_MAX_BYTES: number;
|
|
26
|
+
export declare const DEFAULT_MEMORY_EMBED_CACHE_MAX_ITEMS = 2000;
|
|
27
|
+
export declare const DEFAULT_MEMORY_EMBED_CACHE_MAX_BYTES: number;
|
|
28
|
+
export declare const memoryLlmCache: LRUCache<string, string, unknown>;
|
|
29
|
+
export declare const memoryEmbeddingCache: LRUCache<string, number[], unknown>;
|
|
30
|
+
export declare function resetInflightState(): void;
|
|
21
31
|
import { type DocumentMetadata } from "./metadata.js";
|
|
22
32
|
import { type MetadataFilter } from "./metadata-filter.js";
|
|
23
33
|
export declare const DEFAULT_EMBED_MODEL = "hf:ggml-org/embeddinggemma-300M-GGUF/embeddinggemma-300M-Q8_0.gguf";
|
|
@@ -48,6 +58,10 @@ export declare const CHUNK_WINDOW_TOKENS = 200;
|
|
|
48
58
|
export declare const CHUNK_WINDOW_CHARS: number;
|
|
49
59
|
export declare function canonicalEmbeddingBuildMaterial(providerIdentity: string, strategy: EmbedOptions["chunkStrategy"]): string;
|
|
50
60
|
export declare function getEmbeddingFingerprint(model?: string): string;
|
|
61
|
+
export declare function embedQueriesForStore(store: Store, queries: string[]): Promise<{
|
|
62
|
+
model: string;
|
|
63
|
+
embeddings: number[][];
|
|
64
|
+
}>;
|
|
51
65
|
/**
|
|
52
66
|
* A potential break point in the document with a base score indicating quality.
|
|
53
67
|
*/
|
|
@@ -655,10 +669,21 @@ export type CacheKeyBody = {
|
|
|
655
669
|
export declare function getCacheKey(url: string, body: CacheKeyBody): string;
|
|
656
670
|
export declare function getCachedResult(db: Database, cacheKey: string): string | null;
|
|
657
671
|
export declare function setCachedResult(db: Database, cacheKey: string, result: string): void;
|
|
672
|
+
/**
|
|
673
|
+
* Clear cached LLM API responses and query embeddings.
|
|
674
|
+
*
|
|
675
|
+
* Note: memoryLlmCache, memoryEmbeddingCache, and in-flight deduplication maps
|
|
676
|
+
* are process-level singletons shared across collections/stores to maximize
|
|
677
|
+
* deduplication when different collections are queried concurrently with the
|
|
678
|
+
* same input. Calling clearCache() here flushes this process-wide in-memory
|
|
679
|
+
* cache as well.
|
|
680
|
+
*/
|
|
658
681
|
export declare function clearCache(db: Database): void;
|
|
659
682
|
/**
|
|
660
683
|
* Delete cached LLM API responses.
|
|
661
684
|
* Returns the number of cached responses deleted.
|
|
685
|
+
*
|
|
686
|
+
* Note: Also flushes the process-wide memoryLlmCache.
|
|
662
687
|
*/
|
|
663
688
|
export declare function deleteLLMCache(db: Database): number;
|
|
664
689
|
/**
|
package/dist/store.js
CHANGED
|
@@ -17,6 +17,7 @@ import { abandonEmbeddingBuild, beginEmbeddingBuild, completeEmbeddingBuild, cre
|
|
|
17
17
|
import picomatch from "picomatch";
|
|
18
18
|
import { createHash, randomUUID } from "crypto";
|
|
19
19
|
import { readFileSync, realpathSync, statSync, mkdirSync } from "node:fs";
|
|
20
|
+
import { Buffer } from "node:buffer";
|
|
20
21
|
// Note: node:path resolve is not imported — we export our own cross-platform resolve()
|
|
21
22
|
import fastGlob from "fast-glob";
|
|
22
23
|
import { qmdHomedir } from "./paths.js";
|
|
@@ -24,7 +25,38 @@ import { cleanupExpiredCjkIndexBuilds, getCjkAnalyzerFingerprint, getCjkLexicalI
|
|
|
24
25
|
import { analyzeCjkSync, containsCjk } from "./search/cjk-analyzer.js";
|
|
25
26
|
import { ExpansionPolicyError, parseExpansionDirective, resolveExpansionPolicy, containsRelativeTemporalTerms, } from "./search/query-expansion.js";
|
|
26
27
|
import { LlamaCpp, getDefaultLlamaCpp, formatQueryForEmbedding, formatDocForEmbedding, withLLMSessionForLlm, DEFAULT_EMBED_MODEL_URI, DEFAULT_RERANK_MODEL_URI, DEFAULT_GENERATE_MODEL_URI, } from "./llm.js";
|
|
28
|
+
import { LRUCache } from "lru-cache";
|
|
29
|
+
export { LRUCache, LRUCache as LruCache } from "lru-cache";
|
|
30
|
+
export const DEFAULT_MEMORY_CACHE_TTL_MS = 2 * 60 * 60 * 1000; // 2 hours
|
|
31
|
+
export const DEFAULT_MEMORY_LLM_CACHE_MAX_ITEMS = 1000;
|
|
32
|
+
export const DEFAULT_MEMORY_LLM_CACHE_MAX_BYTES = 50 * 1024 * 1024; // 50 MB
|
|
33
|
+
export const DEFAULT_MEMORY_EMBED_CACHE_MAX_ITEMS = 2000;
|
|
34
|
+
export const DEFAULT_MEMORY_EMBED_CACHE_MAX_BYTES = 64 * 1024 * 1024; // 64 MB
|
|
35
|
+
let cacheGeneration = 0;
|
|
27
36
|
const readOnlyDatabases = new WeakSet();
|
|
37
|
+
const inflightEmbeddings = new Map();
|
|
38
|
+
const inflightExpansions = new Map();
|
|
39
|
+
export const memoryLlmCache = new LRUCache({
|
|
40
|
+
max: DEFAULT_MEMORY_LLM_CACHE_MAX_ITEMS,
|
|
41
|
+
maxSize: DEFAULT_MEMORY_LLM_CACHE_MAX_BYTES,
|
|
42
|
+
sizeCalculation: (value) => Math.max(1, Buffer.byteLength(value, "utf8")),
|
|
43
|
+
ttl: DEFAULT_MEMORY_CACHE_TTL_MS,
|
|
44
|
+
updateAgeOnGet: true,
|
|
45
|
+
});
|
|
46
|
+
export const memoryEmbeddingCache = new LRUCache({
|
|
47
|
+
max: DEFAULT_MEMORY_EMBED_CACHE_MAX_ITEMS,
|
|
48
|
+
maxSize: DEFAULT_MEMORY_EMBED_CACHE_MAX_BYTES,
|
|
49
|
+
sizeCalculation: (vector) => Math.max(1, vector.length * 8),
|
|
50
|
+
ttl: DEFAULT_MEMORY_CACHE_TTL_MS,
|
|
51
|
+
updateAgeOnGet: true,
|
|
52
|
+
});
|
|
53
|
+
export function resetInflightState() {
|
|
54
|
+
cacheGeneration++;
|
|
55
|
+
inflightEmbeddings.clear();
|
|
56
|
+
inflightExpansions.clear();
|
|
57
|
+
memoryLlmCache.clear();
|
|
58
|
+
memoryEmbeddingCache.clear();
|
|
59
|
+
}
|
|
28
60
|
import { METADATA_EXTRACTION_VERSION } from "./metadata.js";
|
|
29
61
|
import { compileMetadataFilter } from "./metadata-filter.js";
|
|
30
62
|
import { initializeMetadataSchema, syncDocumentMetadata, countDocumentsPendingMetadata, getMetadataByFilepath, parseMetadataJson, } from "./metadata-store.js";
|
|
@@ -167,7 +199,70 @@ function authorizeEmbeddingProviderRequest(provider, authorize, purpose, context
|
|
|
167
199
|
}
|
|
168
200
|
authorize(purpose, context);
|
|
169
201
|
}
|
|
170
|
-
async function
|
|
202
|
+
async function embedQueriesWithSingleflight(modelId, queries, fetchBatch) {
|
|
203
|
+
const currentGeneration = cacheGeneration;
|
|
204
|
+
const missingQueries = Array.from(new Set(queries.filter(q => memoryEmbeddingCache.get(`${modelId}:${q}`) === undefined &&
|
|
205
|
+
!inflightEmbeddings.has(`${modelId}:${q}`))));
|
|
206
|
+
if (missingQueries.length > 0) {
|
|
207
|
+
const resolvers = new Map();
|
|
208
|
+
const promises = new Map();
|
|
209
|
+
for (const q of missingQueries) {
|
|
210
|
+
const inflightKey = `${modelId}:${q}`;
|
|
211
|
+
const promise = new Promise((resolve, reject) => {
|
|
212
|
+
resolvers.set(inflightKey, { resolve, reject });
|
|
213
|
+
});
|
|
214
|
+
promises.set(inflightKey, promise);
|
|
215
|
+
inflightEmbeddings.set(inflightKey, promise);
|
|
216
|
+
}
|
|
217
|
+
(async () => {
|
|
218
|
+
try {
|
|
219
|
+
const resultEmbeddings = await fetchBatch(missingQueries);
|
|
220
|
+
const isCurrentGen = currentGeneration === cacheGeneration;
|
|
221
|
+
for (let i = 0; i < missingQueries.length; i++) {
|
|
222
|
+
const q = missingQueries[i];
|
|
223
|
+
const key = `${modelId}:${q}`;
|
|
224
|
+
const vec = resultEmbeddings[i];
|
|
225
|
+
if (!vec) {
|
|
226
|
+
throw new Error(`Embedding missing for query "${q}"`);
|
|
227
|
+
}
|
|
228
|
+
if (isCurrentGen && inflightEmbeddings.get(key) === promises.get(key)) {
|
|
229
|
+
memoryEmbeddingCache.set(key, vec);
|
|
230
|
+
}
|
|
231
|
+
resolvers.get(key)?.resolve(vec);
|
|
232
|
+
}
|
|
233
|
+
}
|
|
234
|
+
catch (err) {
|
|
235
|
+
for (const q of missingQueries) {
|
|
236
|
+
resolvers.get(`${modelId}:${q}`)?.reject(err);
|
|
237
|
+
}
|
|
238
|
+
}
|
|
239
|
+
finally {
|
|
240
|
+
for (const q of missingQueries) {
|
|
241
|
+
const key = `${modelId}:${q}`;
|
|
242
|
+
if (inflightEmbeddings.get(key) === promises.get(key)) {
|
|
243
|
+
inflightEmbeddings.delete(key);
|
|
244
|
+
}
|
|
245
|
+
}
|
|
246
|
+
}
|
|
247
|
+
})();
|
|
248
|
+
}
|
|
249
|
+
return Promise.all(queries.map(async (q) => {
|
|
250
|
+
const cached = memoryEmbeddingCache.get(`${modelId}:${q}`);
|
|
251
|
+
if (cached)
|
|
252
|
+
return [...cached];
|
|
253
|
+
const inflight = inflightEmbeddings.get(`${modelId}:${q}`);
|
|
254
|
+
if (inflight) {
|
|
255
|
+
const vec = await inflight;
|
|
256
|
+
return [...vec];
|
|
257
|
+
}
|
|
258
|
+
throw new Error(`Embedding missing for query "${q}"`);
|
|
259
|
+
}));
|
|
260
|
+
}
|
|
261
|
+
export async function embedQueriesForStore(store, queries) {
|
|
262
|
+
if (queries.length === 0) {
|
|
263
|
+
const model = store.embeddingProvider?.model ?? getLlm(store).embedModelName;
|
|
264
|
+
return { model, embeddings: [] };
|
|
265
|
+
}
|
|
171
266
|
const provider = store.embeddingProvider;
|
|
172
267
|
if (provider) {
|
|
173
268
|
const identity = resolveReadyProviderEmbeddingIdentity(store.db, provider);
|
|
@@ -175,59 +270,61 @@ async function embedQueriesForStore(store, queries) {
|
|
|
175
270
|
throw new EmbeddingIdentityStateError("IDENTITY_MISMATCH", "Query embedding requires a matching published embedding identity.");
|
|
176
271
|
}
|
|
177
272
|
authorizeEmbeddingProviderRequest(provider, store.authorizeRemoteRequest, "query-embedding", { identity });
|
|
178
|
-
const
|
|
179
|
-
|
|
180
|
-
const
|
|
181
|
-
|
|
182
|
-
|
|
183
|
-
identityFingerprint: identity.fingerprint,
|
|
184
|
-
});
|
|
185
|
-
return {
|
|
186
|
-
model: provider.model,
|
|
187
|
-
embeddings: vectors.map(vector => vector.vector),
|
|
188
|
-
};
|
|
189
|
-
}
|
|
190
|
-
catch (batchError) {
|
|
191
|
-
if (formatted.length <= 1)
|
|
192
|
-
throw batchError;
|
|
193
|
-
const embeddings = [];
|
|
194
|
-
for (const query of formatted) {
|
|
195
|
-
const vector = await provider.embed(query, {
|
|
273
|
+
const modelId = `provider:${identity.fingerprint}`;
|
|
274
|
+
const embeddings = await embedQueriesWithSingleflight(modelId, queries, async (missing) => {
|
|
275
|
+
const formatted = missing.map(query => provider.formatQuery(query));
|
|
276
|
+
try {
|
|
277
|
+
const vectors = await provider.embedBatch(formatted, {
|
|
196
278
|
purpose: "query-embedding",
|
|
197
279
|
kind: "query",
|
|
198
280
|
identityFingerprint: identity.fingerprint,
|
|
199
281
|
});
|
|
200
|
-
|
|
282
|
+
return vectors.map(vector => vector.vector);
|
|
201
283
|
}
|
|
202
|
-
|
|
203
|
-
|
|
204
|
-
|
|
205
|
-
|
|
206
|
-
|
|
207
|
-
|
|
208
|
-
|
|
209
|
-
|
|
210
|
-
|
|
211
|
-
|
|
212
|
-
|
|
213
|
-
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
};
|
|
217
|
-
}
|
|
218
|
-
catch (batchError) {
|
|
219
|
-
if (formatted.length <= 1)
|
|
220
|
-
throw batchError;
|
|
221
|
-
const embeddings = [];
|
|
222
|
-
for (const query of formatted) {
|
|
223
|
-
const result = await llm.embed(query);
|
|
224
|
-
embeddings.push(result?.embedding ?? []);
|
|
225
|
-
}
|
|
284
|
+
catch (batchError) {
|
|
285
|
+
if (formatted.length <= 1)
|
|
286
|
+
throw batchError;
|
|
287
|
+
const fallbackVectors = [];
|
|
288
|
+
for (let i = 0; i < formatted.length; i++) {
|
|
289
|
+
const vector = await provider.embed(formatted[i], {
|
|
290
|
+
purpose: "query-embedding",
|
|
291
|
+
kind: "query",
|
|
292
|
+
identityFingerprint: identity.fingerprint,
|
|
293
|
+
});
|
|
294
|
+
fallbackVectors.push(vector.vector);
|
|
295
|
+
}
|
|
296
|
+
return fallbackVectors;
|
|
297
|
+
}
|
|
298
|
+
});
|
|
226
299
|
return {
|
|
227
|
-
model,
|
|
300
|
+
model: provider.model,
|
|
228
301
|
embeddings,
|
|
229
302
|
};
|
|
230
303
|
}
|
|
304
|
+
const llm = getLlm(store);
|
|
305
|
+
const model = llm.embedModelName;
|
|
306
|
+
const modelId = `llm:${model}`;
|
|
307
|
+
const embeddings = await embedQueriesWithSingleflight(modelId, queries, async (missing) => {
|
|
308
|
+
const formatted = missing.map(query => formatQueryForEmbedding(query, model));
|
|
309
|
+
try {
|
|
310
|
+
const results = await llm.embedBatch(formatted);
|
|
311
|
+
return results.map(result => result?.embedding ?? []);
|
|
312
|
+
}
|
|
313
|
+
catch (batchError) {
|
|
314
|
+
if (formatted.length <= 1)
|
|
315
|
+
throw batchError;
|
|
316
|
+
const fallbackVectors = [];
|
|
317
|
+
for (let i = 0; i < formatted.length; i++) {
|
|
318
|
+
const result = await llm.embed(formatted[i]);
|
|
319
|
+
fallbackVectors.push(result?.embedding ?? []);
|
|
320
|
+
}
|
|
321
|
+
return fallbackVectors;
|
|
322
|
+
}
|
|
323
|
+
});
|
|
324
|
+
return {
|
|
325
|
+
model,
|
|
326
|
+
embeddings,
|
|
327
|
+
};
|
|
231
328
|
}
|
|
232
329
|
function createBorrowedEmbeddingSession(provider, maxDurationMs, getBuildLease, authorize, getIdentity) {
|
|
233
330
|
const signal = AbortSignal.timeout(maxDurationMs);
|
|
@@ -2982,10 +3079,23 @@ export function getCacheKey(url, body) {
|
|
|
2982
3079
|
return hash.digest("hex");
|
|
2983
3080
|
}
|
|
2984
3081
|
export function getCachedResult(db, cacheKey) {
|
|
2985
|
-
const
|
|
2986
|
-
|
|
3082
|
+
const inMemory = memoryLlmCache.get(cacheKey);
|
|
3083
|
+
if (inMemory !== undefined)
|
|
3084
|
+
return inMemory;
|
|
3085
|
+
try {
|
|
3086
|
+
const row = db.prepare(`SELECT result FROM llm_cache WHERE hash = ?`).get(cacheKey);
|
|
3087
|
+
if (row?.result) {
|
|
3088
|
+
memoryLlmCache.set(cacheKey, row.result);
|
|
3089
|
+
return row.result;
|
|
3090
|
+
}
|
|
3091
|
+
}
|
|
3092
|
+
catch {
|
|
3093
|
+
// llm_cache table may not exist yet in uninitialized/read-only in-memory databases
|
|
3094
|
+
}
|
|
3095
|
+
return null;
|
|
2987
3096
|
}
|
|
2988
3097
|
export function setCachedResult(db, cacheKey, result) {
|
|
3098
|
+
memoryLlmCache.set(cacheKey, result);
|
|
2989
3099
|
if (readOnlyDatabases.has(db))
|
|
2990
3100
|
return;
|
|
2991
3101
|
const now = new Date().toISOString();
|
|
@@ -2994,8 +3104,24 @@ export function setCachedResult(db, cacheKey, result) {
|
|
|
2994
3104
|
db.exec(`DELETE FROM llm_cache WHERE hash NOT IN (SELECT hash FROM llm_cache ORDER BY created_at DESC LIMIT 1000)`);
|
|
2995
3105
|
}
|
|
2996
3106
|
}
|
|
3107
|
+
/**
|
|
3108
|
+
* Clear cached LLM API responses and query embeddings.
|
|
3109
|
+
*
|
|
3110
|
+
* Note: memoryLlmCache, memoryEmbeddingCache, and in-flight deduplication maps
|
|
3111
|
+
* are process-level singletons shared across collections/stores to maximize
|
|
3112
|
+
* deduplication when different collections are queried concurrently with the
|
|
3113
|
+
* same input. Calling clearCache() here flushes this process-wide in-memory
|
|
3114
|
+
* cache as well.
|
|
3115
|
+
*/
|
|
2997
3116
|
export function clearCache(db) {
|
|
2998
|
-
|
|
3117
|
+
if (!readOnlyDatabases.has(db)) {
|
|
3118
|
+
db.exec(`DELETE FROM llm_cache`);
|
|
3119
|
+
}
|
|
3120
|
+
cacheGeneration++;
|
|
3121
|
+
inflightExpansions.clear();
|
|
3122
|
+
inflightEmbeddings.clear();
|
|
3123
|
+
memoryLlmCache.clear();
|
|
3124
|
+
memoryEmbeddingCache.clear();
|
|
2999
3125
|
}
|
|
3000
3126
|
// =============================================================================
|
|
3001
3127
|
// Cleanup and maintenance operations
|
|
@@ -3003,8 +3129,14 @@ export function clearCache(db) {
|
|
|
3003
3129
|
/**
|
|
3004
3130
|
* Delete cached LLM API responses.
|
|
3005
3131
|
* Returns the number of cached responses deleted.
|
|
3132
|
+
*
|
|
3133
|
+
* Note: Also flushes the process-wide memoryLlmCache.
|
|
3006
3134
|
*/
|
|
3007
3135
|
export function deleteLLMCache(db) {
|
|
3136
|
+
cacheGeneration++;
|
|
3137
|
+
memoryLlmCache.clear();
|
|
3138
|
+
if (readOnlyDatabases.has(db))
|
|
3139
|
+
return 0;
|
|
3008
3140
|
const result = db.prepare(`DELETE FROM llm_cache`).run();
|
|
3009
3141
|
return Number(result.changes);
|
|
3010
3142
|
}
|
|
@@ -4880,25 +5012,43 @@ export async function expandQuery(query, model = DEFAULT_QUERY_MODEL, db, expans
|
|
|
4880
5012
|
// Old cache format (pre-typed, newline-separated text) — re-expand
|
|
4881
5013
|
}
|
|
4882
5014
|
}
|
|
4883
|
-
const
|
|
4884
|
-
|
|
4885
|
-
|
|
4886
|
-
|
|
4887
|
-
|
|
4888
|
-
|
|
4889
|
-
|
|
4890
|
-
|
|
4891
|
-
|
|
4892
|
-
|
|
4893
|
-
|
|
4894
|
-
|
|
5015
|
+
const currentGeneration = cacheGeneration;
|
|
5016
|
+
let inflight = inflightExpansions.get(cacheKey);
|
|
5017
|
+
if (!inflight) {
|
|
5018
|
+
inflight = (async () => {
|
|
5019
|
+
try {
|
|
5020
|
+
const llm = llmOverride ?? getDefaultLlamaCpp();
|
|
5021
|
+
// Note: LlamaCpp uses hardcoded model, model parameter is ignored
|
|
5022
|
+
const results = await llm.expandQuery(query, {
|
|
5023
|
+
context: expansionContext,
|
|
5024
|
+
includeLexical,
|
|
5025
|
+
includeHyde,
|
|
5026
|
+
});
|
|
5027
|
+
// Map Queryable[] → ExpandedQuery[] (same shape, decoupled from llm.ts internals).
|
|
5028
|
+
// Filter out entries that duplicate the original query text.
|
|
5029
|
+
const expanded = results
|
|
5030
|
+
.filter(r => r.text !== query)
|
|
5031
|
+
.map(r => ({ type: r.type, query: r.text }));
|
|
5032
|
+
if (expanded.length > 0 &&
|
|
5033
|
+
currentGeneration === cacheGeneration &&
|
|
5034
|
+
inflightExpansions.get(cacheKey) === inflight) {
|
|
5035
|
+
setCachedResult(db, cacheKey, JSON.stringify(expanded));
|
|
5036
|
+
}
|
|
5037
|
+
return expanded;
|
|
5038
|
+
}
|
|
5039
|
+
finally {
|
|
5040
|
+
if (inflightExpansions.get(cacheKey) === inflight) {
|
|
5041
|
+
inflightExpansions.delete(cacheKey);
|
|
5042
|
+
}
|
|
5043
|
+
}
|
|
5044
|
+
})();
|
|
5045
|
+
inflightExpansions.set(cacheKey, inflight);
|
|
5046
|
+
}
|
|
5047
|
+
const expanded = await inflight;
|
|
4895
5048
|
if (options?.requireResult && expanded.length === 0) {
|
|
4896
5049
|
throw new QueryExpansionNoResultError();
|
|
4897
5050
|
}
|
|
4898
|
-
|
|
4899
|
-
setCachedResult(db, cacheKey, JSON.stringify(expanded));
|
|
4900
|
-
}
|
|
4901
|
-
return expanded;
|
|
5051
|
+
return expanded.map(e => ({ ...e }));
|
|
4902
5052
|
}
|
|
4903
5053
|
/**
|
|
4904
5054
|
* Delete the cached expansion for a query. hybridQuery() calls this when an
|
|
@@ -4913,7 +5063,11 @@ export function deleteExpansionCacheEntry(db, query, model = DEFAULT_QUERY_MODEL
|
|
|
4913
5063
|
...(options?.includeHyde === false && { noHyde: true }),
|
|
4914
5064
|
...(options?.includeLexical === false && { noLex: true }),
|
|
4915
5065
|
});
|
|
4916
|
-
|
|
5066
|
+
inflightExpansions.delete(cacheKey);
|
|
5067
|
+
memoryLlmCache.delete(cacheKey);
|
|
5068
|
+
if (!readOnlyDatabases.has(db)) {
|
|
5069
|
+
db.prepare(`DELETE FROM llm_cache WHERE hash = ?`).run(cacheKey);
|
|
5070
|
+
}
|
|
4917
5071
|
}
|
|
4918
5072
|
// =============================================================================
|
|
4919
5073
|
// Reranking
|
package/package.json
CHANGED
|
@@ -1,8 +1,30 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@wei840222/qmd",
|
|
3
|
-
"version": "2026.9.
|
|
4
|
-
"packageManager": "pnpm@11.15.1",
|
|
3
|
+
"version": "2026.9.30",
|
|
5
4
|
"description": "Query Markup Documents - On-device hybrid search for markdown files with BM25, vector search, and LLM reranking",
|
|
5
|
+
"author": "Wan, Jiun Wei <wei840222@gmail.com>",
|
|
6
|
+
"license": "MIT",
|
|
7
|
+
"keywords": [
|
|
8
|
+
"markdown",
|
|
9
|
+
"search",
|
|
10
|
+
"fts",
|
|
11
|
+
"full-text-search",
|
|
12
|
+
"vector",
|
|
13
|
+
"semantic-search",
|
|
14
|
+
"sqlite",
|
|
15
|
+
"bm25",
|
|
16
|
+
"embeddings",
|
|
17
|
+
"rag",
|
|
18
|
+
"mcp",
|
|
19
|
+
"reranking",
|
|
20
|
+
"knowledge-base",
|
|
21
|
+
"local-ai",
|
|
22
|
+
"llm"
|
|
23
|
+
],
|
|
24
|
+
"engines": {
|
|
25
|
+
"node": ">=24.21.0"
|
|
26
|
+
},
|
|
27
|
+
"packageManager": "pnpm@12.6.0",
|
|
6
28
|
"type": "module",
|
|
7
29
|
"main": "dist/index.js",
|
|
8
30
|
"types": "dist/index.d.ts",
|
|
@@ -64,20 +86,21 @@
|
|
|
64
86
|
"url": "https://github.com/wei840222/qmd/issues"
|
|
65
87
|
},
|
|
66
88
|
"dependencies": {
|
|
67
|
-
"@modelcontextprotocol/server": "2.
|
|
89
|
+
"@modelcontextprotocol/server": "2.1.0",
|
|
68
90
|
"@node-rs/jieba": "2.0.3",
|
|
69
91
|
"@typesafe-ai/sdk": "^0.6.0",
|
|
70
92
|
"fast-glob": "3.3.3",
|
|
71
|
-
"
|
|
72
|
-
"
|
|
93
|
+
"lru-cache": "^11.5.3",
|
|
94
|
+
"node-llama-cpp": "3.22.1",
|
|
95
|
+
"picomatch": "4.0.7",
|
|
73
96
|
"sqlite-vec": "0.1.9",
|
|
74
97
|
"tree-sitter-go": "0.25.0",
|
|
75
98
|
"tree-sitter-python": "0.25.0",
|
|
76
99
|
"tree-sitter-rust": "0.24.0",
|
|
77
100
|
"tree-sitter-typescript": "0.23.2",
|
|
78
101
|
"web-tree-sitter": "0.26.12",
|
|
79
|
-
"yaml": "2.9.
|
|
80
|
-
"zod": "4.
|
|
102
|
+
"yaml": "2.9.1",
|
|
103
|
+
"zod": "4.6.5"
|
|
81
104
|
},
|
|
82
105
|
"optionalDependencies": {
|
|
83
106
|
"sqlite-vec-darwin-arm64": "0.1.9",
|
|
@@ -87,50 +110,28 @@
|
|
|
87
110
|
"sqlite-vec-windows-x64": "0.1.9"
|
|
88
111
|
},
|
|
89
112
|
"devDependencies": {
|
|
90
|
-
"@oxlint/plugins": "1.
|
|
91
|
-
"oxlint": "1.
|
|
92
|
-
"tsx": "4.23.
|
|
93
|
-
"vitest": "
|
|
113
|
+
"@oxlint/plugins": "1.85.0",
|
|
114
|
+
"oxlint": "1.85.0",
|
|
115
|
+
"tsx": "4.23.15",
|
|
116
|
+
"vitest": "4.1.11"
|
|
94
117
|
},
|
|
95
118
|
"overrides": {
|
|
96
|
-
"@hono/node-server": "2.
|
|
97
|
-
"ajv": "8.
|
|
98
|
-
"esbuild": "0.28.
|
|
119
|
+
"@hono/node-server": "2.1.2",
|
|
120
|
+
"ajv": "8.20.0",
|
|
121
|
+
"esbuild": "0.28.2",
|
|
99
122
|
"fast-uri": "3.1.5",
|
|
100
|
-
"hono": "4.
|
|
101
|
-
"ip-address": "10.
|
|
123
|
+
"hono": "4.13.10",
|
|
124
|
+
"ip-address": "10.7.2",
|
|
102
125
|
"nanoid": "3.3.18",
|
|
103
|
-
"path-to-regexp": "8.4.
|
|
104
|
-
"postcss": "8.5.
|
|
105
|
-
"qs": "6.
|
|
106
|
-
"rollup": "4.
|
|
126
|
+
"path-to-regexp": "8.4.2",
|
|
127
|
+
"postcss": "8.5.28",
|
|
128
|
+
"qs": "6.16.0",
|
|
129
|
+
"rollup": "4.63.5",
|
|
107
130
|
"simple-git": "3.36.0",
|
|
108
|
-
"tar": "7.5.
|
|
131
|
+
"tar": "7.5.22",
|
|
109
132
|
"vite": "7.3.5"
|
|
110
133
|
},
|
|
111
134
|
"peerDependencies": {
|
|
112
135
|
"typescript": "^5.9.3 || ^6.0.0-0"
|
|
113
|
-
}
|
|
114
|
-
"engines": {
|
|
115
|
-
"node": ">=22.16.0"
|
|
116
|
-
},
|
|
117
|
-
"keywords": [
|
|
118
|
-
"markdown",
|
|
119
|
-
"search",
|
|
120
|
-
"fts",
|
|
121
|
-
"full-text-search",
|
|
122
|
-
"vector",
|
|
123
|
-
"semantic-search",
|
|
124
|
-
"sqlite",
|
|
125
|
-
"bm25",
|
|
126
|
-
"embeddings",
|
|
127
|
-
"rag",
|
|
128
|
-
"mcp",
|
|
129
|
-
"reranking",
|
|
130
|
-
"knowledge-base",
|
|
131
|
-
"local-ai",
|
|
132
|
-
"llm"
|
|
133
|
-
],
|
|
134
|
-
"author": "Wan, Jiun Wei <wei840222@gmail.com>",
|
|
135
|
-
"license": "MIT"
|
|
136
|
+
}
|
|
136
137
|
}
|