@wei840222/qmd 2026.8.28 → 2026.9.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +18 -0
- package/README.md +84 -2
- package/dist/cli/build-info.json +2 -2
- package/dist/cli/qmd.js +96 -18
- package/dist/collections.js +9 -4
- package/dist/db.d.ts +16 -29
- package/dist/db.js +63 -40
- package/dist/index.d.ts +10 -0
- package/dist/index.js +14 -2
- package/dist/llm.d.ts +7 -1
- package/dist/llm.js +23 -4
- package/dist/mcp/server.js +70 -6
- package/dist/metadata-filter.d.ts +74 -0
- package/dist/metadata-filter.js +279 -0
- package/dist/metadata-store.d.ts +45 -0
- package/dist/metadata-store.js +173 -0
- package/dist/metadata.d.ts +61 -0
- package/dist/metadata.js +215 -0
- package/dist/search/zh-dict.txt +3 -0
- package/dist/store.d.ts +23 -16
- package/dist/store.js +354 -166
- package/package.json +3 -5
- package/scripts/sync-zh-dict.mjs +4 -1
- package/skills/qmd/SKILL.md +11 -2
- package/skills/qmd/references/query-syntax.md +1 -8
- package/skills/release/SKILL.md +0 -141
- package/skills/release/scripts/install-hooks.sh +0 -38
- package/skills/release/scripts/release-context.sh +0 -129
package/CHANGELOG.md
CHANGED
|
@@ -2,9 +2,27 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
## [2026.9.25] - 2026-09-26
|
|
6
|
+
|
|
7
|
+
### Changed
|
|
8
|
+
|
|
9
|
+
- Replace `better-sqlite3` with Node.js built-in `node:sqlite`, preventing SQLite runtime symbol collisions when QMD is embedded alongside other `node:sqlite` consumers. The minimum supported Node.js version is now 22.16.0.
|
|
10
|
+
- Upgraded `@node-rs/jieba` to `2.0.3` and synchronized bundled Traditional Chinese dictionary assets (`zh-dict.txt`) with upstream `sysprog21/zhtw-mcp`.
|
|
11
|
+
- Relocated repository release management skill to `.agents/skills/release`.
|
|
12
|
+
|
|
5
13
|
### Added
|
|
6
14
|
|
|
15
|
+
- Added Oxlint lint fence.
|
|
16
|
+
- Document metadata and metadata filtering. Markdown documents can opt into typed metadata through a namespaced frontmatter block (`qmd.metadata` with strings, numbers, booleans, or flat homogeneous arrays), and every search surface — CLI `search`/`vsearch`/`query` via `--filter <json>`, the SDK's `filter` option on `search()`/`searchLex()`/`searchVector()`, the MCP `query` tool, and HTTP `POST /query` and `/search` — accepts one shared recursive filter AST discriminated by `operator`: `and`/`or`/`not` logical groups, `eq`/`ne`/`gt`/`gte`/`lt`/`lte` comparisons, `in`/`nin`/`all` membership, and `exists` presence. Every returned result satisfies the filter (applied before RRF fusion and reranking); like collection filtering, highly selective filters remain best-effort for top-K completeness. Frontmatter stays ordinary searchable content — no chunking, embedding, snippet, or line-number changes — and documents without `qmd.metadata` behave exactly as before. JSON/SDK/MCP/HTTP results now include each document's indexed metadata, and `qmd status` reports how many documents still need metadata extraction (a normal `qmd update` backfills existing indexes).
|
|
7
17
|
- **Disable HyDE Expansion Control**: Added `--no-hyde` CLI option for `qmd query` and `qmd vsearch`, `includeHyde` parameter to SDK (`store.search`, `store.expandQuery`) and MCP `query` tool, allowing users to disable generating hypothetical document embeddings during query expansion.
|
|
18
|
+
- Added `typesafe-ai` skill to `.agents/skills/typesafe-ai`.
|
|
19
|
+
|
|
20
|
+
### Fixed
|
|
21
|
+
|
|
22
|
+
- Embedding generation and legacy fingerprint adoption now tokenize documents
|
|
23
|
+
with the store-selected embedding model instead of the global default. This
|
|
24
|
+
keeps chunk boundaries aligned with the model that creates and verifies the
|
|
25
|
+
stored vectors without initializing an unrelated provider.
|
|
8
26
|
|
|
9
27
|
## [2026.8.23-1] - 2026-08-23
|
|
10
28
|
|
package/README.md
CHANGED
|
@@ -132,7 +132,7 @@ runs in a container and a liveness probe connects from a non-loopback address.
|
|
|
132
132
|
|
|
133
133
|
The HTTP server exposes two endpoints:
|
|
134
134
|
- `POST /mcp` — MCP Streamable HTTP (JSON responses, stateless)
|
|
135
|
-
- `POST /query` (alias `/search`) — structured search without the MCP protocol
|
|
135
|
+
- `POST /query` (alias `/search`) — structured search without the MCP protocol. Accepts the same optional `filter` object as the `query` tool (invalid filters return `400`); see [Metadata Filtering](#metadata-filtering)
|
|
136
136
|
- `GET /health` — liveness check with uptime
|
|
137
137
|
|
|
138
138
|
|
|
@@ -170,8 +170,10 @@ Point any MCP client at `http://localhost:8181/mcp` to connect.
|
|
|
170
170
|
| `query` | `searches` | array | Typed sub-queries (`lex`/`vec`/`hyde`), 1–10. Mutually exclusive with `query`; exactly one is required. First gets 2x weight. |
|
|
171
171
|
| `query` | `expansion` | string | Plain-query policy: `auto` (default), `force`, or `skip`. Ignored when `searches` is used. |
|
|
172
172
|
| `query` | `collections` | string[] | Filter by collection names (OR). **Array only** — singular `collection` is silently ignored. |
|
|
173
|
+
| `query` | `filter` | object | Metadata filter (recursive `operator`-discriminated JSON AST; see [Metadata Filtering](#metadata-filtering)) |
|
|
173
174
|
| `query` | `expansionContext` | string | Additional context used only to generate `lex` / `vec` / `hyde` query expansions. |
|
|
174
175
|
| `query` | `rerankContext` | string | Additional context used only for reranking and snippet/chunk selection. |
|
|
176
|
+
| `query` | `intent` | string | Disambiguation context (alias for expansionContext & rerankContext; does not search on its own) |
|
|
175
177
|
| `query` | `limit` | number | Max results (default 10) |
|
|
176
178
|
| `query` | `minScore` | number | Minimum relevance 0–1 (default 0) |
|
|
177
179
|
| `query` | `candidateLimit` | number | Max candidates to rerank (default 40) |
|
|
@@ -279,6 +281,20 @@ const results3 = await store.search({
|
|
|
279
281
|
|
|
280
282
|
// Skip reranking for faster results
|
|
281
283
|
const fast = await store.search({ query: "auth", rerank: false })
|
|
284
|
+
|
|
285
|
+
// Metadata filter — every returned result satisfies it (also available on
|
|
286
|
+
// searchLex() and searchVector()); results expose indexed metadata via
|
|
287
|
+
// r.metadata. See "Metadata Filtering" for the full grammar.
|
|
288
|
+
const published = await store.search({
|
|
289
|
+
query: "authentication flow",
|
|
290
|
+
filter: {
|
|
291
|
+
operator: "and",
|
|
292
|
+
operands: [
|
|
293
|
+
{ key: "topics", operator: "all", value: ["typescript"] },
|
|
294
|
+
{ key: "status", operator: "ne", value: "draft" },
|
|
295
|
+
],
|
|
296
|
+
},
|
|
297
|
+
})
|
|
282
298
|
```
|
|
283
299
|
|
|
284
300
|
For simple queries, explicit `force` or `skip` overrides `auto`. Under `auto`, CJK
|
|
@@ -979,10 +995,10 @@ and `deep-search` (→ `query`).
|
|
|
979
995
|
--full # Show full document content
|
|
980
996
|
--line-numbers # Add line numbers to output
|
|
981
997
|
--explain # Include retrieval score traces (query, JSON/CLI output)
|
|
998
|
+
--filter <json> # Metadata filter (recursive JSON AST; see Metadata Filtering)
|
|
982
999
|
--index <name> # Use named index
|
|
983
1000
|
--intent "<text>" # Legacy CLI alias for rerank context (e.g. "web page load times")
|
|
984
1001
|
--no-rerank # Skip LLM reranking (RRF scores only; faster on CPU)
|
|
985
|
-
--no-hyde # Disable HyDE in query expansion (only lex and vec expansions)
|
|
986
1002
|
-C, --candidate-limit <n> # Max candidates to rerank (default: 40)
|
|
987
1003
|
--full-path # Emit on-disk filesystem paths instead of qmd:// URIs
|
|
988
1004
|
# (a result whose file has moved or been deleted since
|
|
@@ -1023,6 +1039,72 @@ explicitly with `-c`.
|
|
|
1023
1039
|
> lexical and vector candidate cutoff. Matching candidates from the selected
|
|
1024
1040
|
> collections are then ranked together.
|
|
1025
1041
|
|
|
1042
|
+
### Metadata Filtering
|
|
1043
|
+
|
|
1044
|
+
Documents can opt into typed metadata through a namespaced frontmatter block. A document without `qmd.metadata` behaves exactly as before, and the frontmatter stays ordinary searchable content (no chunking, embedding, or line-number changes):
|
|
1045
|
+
|
|
1046
|
+
```markdown
|
|
1047
|
+
---
|
|
1048
|
+
qmd:
|
|
1049
|
+
metadata:
|
|
1050
|
+
topics:
|
|
1051
|
+
- typescript
|
|
1052
|
+
- programming
|
|
1053
|
+
status: published
|
|
1054
|
+
priority: 3
|
|
1055
|
+
reviewed: true
|
|
1056
|
+
---
|
|
1057
|
+
|
|
1058
|
+
# Document body starts here
|
|
1059
|
+
```
|
|
1060
|
+
|
|
1061
|
+
Supported values are strings, numbers, booleans, and flat homogeneous arrays of one of those. Nested objects, nulls, empty arrays, and mixed-type arrays are rejected (the document still indexes; it is excluded from filtered search until corrected). Metadata keys are user-defined data — `tags`, `topics`, and `labels` are all ordinary keys with no special semantics.
|
|
1062
|
+
|
|
1063
|
+
Every search surface (CLI, SDK, MCP, HTTP) accepts the same recursive filter, a JSON AST discriminated by `operator`:
|
|
1064
|
+
|
|
1065
|
+
```sh
|
|
1066
|
+
# One condition
|
|
1067
|
+
qmd search "authentication" \
|
|
1068
|
+
--filter '{"key":"status","operator":"eq","value":"published"}'
|
|
1069
|
+
|
|
1070
|
+
# Composed conditions — works with search, vsearch, and query
|
|
1071
|
+
qmd query "dependency injection" --filter '{
|
|
1072
|
+
"operator": "and",
|
|
1073
|
+
"operands": [
|
|
1074
|
+
{ "key": "topics", "operator": "all", "value": ["typescript", "programming"] },
|
|
1075
|
+
{ "key": "status", "operator": "nin", "value": ["draft", "archived"] },
|
|
1076
|
+
{ "operator": "or", "operands": [
|
|
1077
|
+
{ "key": "priority", "operator": "gte", "value": 3 },
|
|
1078
|
+
{ "key": "reviewed", "operator": "eq", "value": true }
|
|
1079
|
+
] },
|
|
1080
|
+
{ "operator": "not", "operand": { "key": "audience", "operator": "eq", "value": "internal" } }
|
|
1081
|
+
]
|
|
1082
|
+
}'
|
|
1083
|
+
```
|
|
1084
|
+
|
|
1085
|
+
| Node | Shape |
|
|
1086
|
+
|------|-------|
|
|
1087
|
+
| Logical group | `{ "operator": "and" \| "or", "operands": […] }` |
|
|
1088
|
+
| Negation | `{ "operator": "not", "operand": {…} }` |
|
|
1089
|
+
| Comparison | `{ "key", "operator": "eq" \| "ne" \| "gt" \| "gte" \| "lt" \| "lte", "value" }` |
|
|
1090
|
+
| Membership | `{ "key", "operator": "in" \| "nin" \| "all", "value": […] }` |
|
|
1091
|
+
| Presence | `{ "key", "operator": "exists", "value": true \| false }` |
|
|
1092
|
+
|
|
1093
|
+
Semantics:
|
|
1094
|
+
|
|
1095
|
+
- Matching is typed and exact — no string/number/boolean coercion, and a type mismatch never matches (including `ne` and `nin`).
|
|
1096
|
+
- Array-valued metadata is a set: a condition matches when any element satisfies it, `all` requires every filter value to be present.
|
|
1097
|
+
- Missing keys do not match `ne`/`nin`; combine with `{ "operator": "exists", "value": false }` in an `or` group to include them.
|
|
1098
|
+
- Multiple conditions require an explicit `and` group — there is no implicit AND, and no `$`-prefixed shorthand.
|
|
1099
|
+
|
|
1100
|
+
Guarantees and limits:
|
|
1101
|
+
|
|
1102
|
+
- Every returned result satisfies the filter, before RRF fusion and reranking.
|
|
1103
|
+
- Like collection filtering, highly selective filters are best-effort for top-K completeness: backends over-fetch and post-filter, so a very selective filter can return fewer than `limit` results.
|
|
1104
|
+
- Filtered search only considers documents whose metadata has been extracted (run `qmd update` after upgrading; `qmd status` shows the pending count).
|
|
1105
|
+
|
|
1106
|
+
JSON output (`--format json`), the SDK, MCP structured results, and the HTTP endpoints include each result's indexed metadata.
|
|
1107
|
+
|
|
1026
1108
|
### Output Format
|
|
1027
1109
|
|
|
1028
1110
|
Default output is colorized CLI format (respects `NO_COLOR` env).
|
package/dist/cli/build-info.json
CHANGED
package/dist/cli/qmd.js
CHANGED
|
@@ -10,6 +10,8 @@ import { parseArgs } from "util";
|
|
|
10
10
|
import { readFileSync, readdirSync, realpathSync, statSync, existsSync, unlinkSync, writeFileSync, openSync, closeSync, mkdirSync, lstatSync, rmSync, symlinkSync, readlinkSync, copyFileSync } from "fs";
|
|
11
11
|
import { createInterface } from "readline/promises";
|
|
12
12
|
import { getPwd, getRealPath, isPathInsideDir, homedir, resolve, enableProductionMode, searchFTS, extractSnippet, getContextForFile, getContextForPath, listCollections, findSimilarFiles, findDocument, resolveCommaListName, matchFilesByGlob, getHashesNeedingEmbedding, clearAllEmbeddings, insertEmbedding, getStatus, hashContent, extractTitle, formatDocForEmbedding, getEmbeddingFingerprint, chunkDocumentByTokens, clearCache, getCacheKey, getCachedResult, setCachedResult, getIndexHealth, parseVirtualPath, buildVirtualPath, isVirtualPath, isDocid, resolveVirtualPath, toVirtualPath, insertContent, insertDocument, insertDocumentWithContent, findActiveDocument, findOrMigrateLegacyDocument, updateDocumentTitle, updateDocument, updateDocumentWithContent, deactivateDocument, getActiveDocumentPaths, cleanupOrphanedContent, countOrphanedVectors, previewCleanup, runCleanup, getCollectionsWithoutContext, getTopLevelPathsWithoutContext, handelize, escapeLikePattern, hybridQuery, vectorSearchQuery, structuredSearch, addLineNumbers, DEFAULT_EMBED_MODEL, DEFAULT_EMBED_MAX_BATCH_BYTES, DEFAULT_EMBED_MAX_DOCS_PER_BATCH, DEFAULT_RERANK_MODEL, DEFAULT_QUERY_MODEL, DEFAULT_GLOB, splitGlobMask, DEFAULT_MULTI_GET_MAX_BYTES, createStore, getDefaultDbPath, reindexCollection, generateEmbeddings, getPendingEmbeddingDocsReadOnly, syncConfigToDb, } from "../store.js";
|
|
13
|
+
import { syncDocumentMetadata, countDocumentsPendingMetadata } from "../metadata-store.js";
|
|
14
|
+
import { parseMetadataFilter } from "../metadata-filter.js";
|
|
13
15
|
import { disposeDefaultLlamaCpp, getDefaultLlamaCpp, setDefaultLlamaCpp, LlamaCpp, withLLMSession, pullModels, DEFAULT_MODEL_CACHE_DIR, resolveEmbedModel, resolveGenerateModel, resolveRerankModel, resolveModels, inspectGgufFile, isDarwinMetalMitigationActive } from "../llm.js";
|
|
14
16
|
import { rebuildCjkLexicalIndex } from "../search/cjk-index.js";
|
|
15
17
|
import { RemoteLLM } from "../remote-llm.js";
|
|
@@ -137,12 +139,16 @@ function getStore() {
|
|
|
137
139
|
}
|
|
138
140
|
return store;
|
|
139
141
|
}
|
|
140
|
-
function getDoctorStore() {
|
|
142
|
+
function getDoctorStore(options = {}) {
|
|
141
143
|
if (!store) {
|
|
142
144
|
const dbPath = getDbPath();
|
|
145
|
+
const shouldReconcile = existsSync(dbPath) && options.reconcileConfig !== undefined;
|
|
143
146
|
store = existsSync(dbPath)
|
|
144
|
-
? createStore(dbPath, { readOnly:
|
|
147
|
+
? createStore(dbPath, { readOnly: !shouldReconcile })
|
|
145
148
|
: createStore(":memory:");
|
|
149
|
+
if (shouldReconcile) {
|
|
150
|
+
syncConfigToDb(store.db, options.reconcileConfig);
|
|
151
|
+
}
|
|
146
152
|
}
|
|
147
153
|
return store;
|
|
148
154
|
}
|
|
@@ -215,11 +221,16 @@ function mcpDaemonPaths() {
|
|
|
215
221
|
}
|
|
216
222
|
function setIndexName(name) {
|
|
217
223
|
let normalizedName = name;
|
|
218
|
-
// Normalize relative paths to prevent malformed database paths
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
224
|
+
// Normalize relative paths to prevent malformed database paths. Windows
|
|
225
|
+
// absolute paths (C:\..., \\server\share) are already absolute -- skip
|
|
226
|
+
// pathResolve() for those and sanitize `:` and `\` alongside `/`.
|
|
227
|
+
if (name) {
|
|
228
|
+
const isWindowsAbsolute = /^[a-zA-Z]:[\\/]/.test(name) || /^\\/.test(name);
|
|
229
|
+
if (isWindowsAbsolute || /[\\/]/.test(name)) {
|
|
230
|
+
const absolutePath = isWindowsAbsolute ? name : pathResolve(process.cwd(), name);
|
|
231
|
+
// Replace path separators with underscores to create a valid filename
|
|
232
|
+
normalizedName = absolutePath.replace(/[:\\/]+/g, '_').replace(/^_+/, '');
|
|
233
|
+
}
|
|
223
234
|
}
|
|
224
235
|
currentIndexName = normalizedName || "index";
|
|
225
236
|
storeDbPathOverride = normalizedName ? getDefaultDbPath(normalizedName) : undefined;
|
|
@@ -542,6 +553,10 @@ async function showStatus() {
|
|
|
542
553
|
if (needsEmbedding > 0) {
|
|
543
554
|
console.log(` ${c.yellow}Pending: ${needsEmbedding} need embedding${c.reset} (run 'qmd embed')`);
|
|
544
555
|
}
|
|
556
|
+
const pendingMetadata = countDocumentsPendingMetadata(db);
|
|
557
|
+
if (pendingMetadata > 0) {
|
|
558
|
+
console.log(` ${c.yellow}Metadata: ${pendingMetadata} need extraction${c.reset} (run 'qmd update'; excluded from --filter searches)`);
|
|
559
|
+
}
|
|
545
560
|
if (mostRecent.latest) {
|
|
546
561
|
const lastUpdate = new Date(mostRecent.latest);
|
|
547
562
|
console.log(` Updated: ${formatTimeAgo(lastUpdate)}`);
|
|
@@ -950,6 +965,7 @@ async function updateCollections() {
|
|
|
950
965
|
progress.clear();
|
|
951
966
|
console.log(`\nIndexed: ${result.indexed} new, ${result.updated} updated, ${result.unchanged} unchanged, ${result.removed} removed`);
|
|
952
967
|
reportSkippedReads(result.skippedFiles);
|
|
968
|
+
reportMetadataErrors(result.metadataErrors);
|
|
953
969
|
if (result.orphanedCleaned > 0) {
|
|
954
970
|
console.log(`Cleaned up ${result.orphanedCleaned} orphaned content hash(es)`);
|
|
955
971
|
}
|
|
@@ -1854,7 +1870,7 @@ async function indexFiles(pwd, globPattern = DEFAULT_GLOB, collectionName, suppr
|
|
|
1854
1870
|
console.log("No files found matching pattern.");
|
|
1855
1871
|
// Continue so the deactivation pass can mark previously indexed docs as inactive.
|
|
1856
1872
|
}
|
|
1857
|
-
let indexed = 0, updated = 0, unchanged = 0, processed = 0;
|
|
1873
|
+
let indexed = 0, updated = 0, unchanged = 0, processed = 0, metadataErrors = 0;
|
|
1858
1874
|
const skippedFiles = [];
|
|
1859
1875
|
const seenPaths = new Set();
|
|
1860
1876
|
// Literal paths of every file in this scan. Passed to the legacy-path
|
|
@@ -1892,8 +1908,12 @@ async function indexFiles(pwd, globPattern = DEFAULT_GLOB, collectionName, suppr
|
|
|
1892
1908
|
const title = extractTitle(content, relativeFile);
|
|
1893
1909
|
// Check if document exists (also migrates legacy lowercase paths)
|
|
1894
1910
|
const existing = findOrMigrateLegacyDocument(db, collectionName, path, livePaths);
|
|
1911
|
+
let documentId;
|
|
1912
|
+
let contentChanged = true;
|
|
1895
1913
|
if (existing) {
|
|
1914
|
+
documentId = existing.id;
|
|
1896
1915
|
if (existing.hash === hash) {
|
|
1916
|
+
contentChanged = false;
|
|
1897
1917
|
// Hash unchanged, but check if title needs updating
|
|
1898
1918
|
if (existing.title !== title) {
|
|
1899
1919
|
updateDocumentTitle(db, existing.id, title, now);
|
|
@@ -1914,8 +1934,12 @@ async function indexFiles(pwd, globPattern = DEFAULT_GLOB, collectionName, suppr
|
|
|
1914
1934
|
// New document - insert content and document
|
|
1915
1935
|
indexed++;
|
|
1916
1936
|
const stat = statSync(filepath);
|
|
1917
|
-
insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1937
|
+
documentId = insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1918
1938
|
}
|
|
1939
|
+
// Unchanged content still backfills missing or stale extraction state.
|
|
1940
|
+
const extraction = syncDocumentMetadata(db, documentId, content, path, contentChanged ? undefined : { onlyIfStale: true });
|
|
1941
|
+
if (extraction?.error)
|
|
1942
|
+
metadataErrors++;
|
|
1919
1943
|
processed++;
|
|
1920
1944
|
progress.set((processed / total) * 100);
|
|
1921
1945
|
const elapsed = (Date.now() - startTime) / 1000;
|
|
@@ -1941,6 +1965,7 @@ async function indexFiles(pwd, globPattern = DEFAULT_GLOB, collectionName, suppr
|
|
|
1941
1965
|
progress.clear();
|
|
1942
1966
|
console.log(`\nIndexed: ${indexed} new, ${updated} updated, ${unchanged} unchanged, ${removed} removed`);
|
|
1943
1967
|
reportSkippedReads(skippedFiles);
|
|
1968
|
+
reportMetadataErrors(metadataErrors);
|
|
1944
1969
|
if (orphanedContent > 0) {
|
|
1945
1970
|
console.log(`Cleaned up ${orphanedContent} orphaned content hash(es)`);
|
|
1946
1971
|
}
|
|
@@ -1958,6 +1983,11 @@ function fsErrorCode(err) {
|
|
|
1958
1983
|
}
|
|
1959
1984
|
return "ERROR";
|
|
1960
1985
|
}
|
|
1986
|
+
function reportMetadataErrors(metadataErrors) {
|
|
1987
|
+
if (metadataErrors === 0)
|
|
1988
|
+
return;
|
|
1989
|
+
console.warn(`⚠ ${metadataErrors} file(s) have invalid qmd.metadata frontmatter and are excluded from filtered search`);
|
|
1990
|
+
}
|
|
1961
1991
|
function reportSkippedReads(skippedFiles) {
|
|
1962
1992
|
if (skippedFiles.length === 0)
|
|
1963
1993
|
return;
|
|
@@ -2371,6 +2401,7 @@ function outputResults(results, query, opts) {
|
|
|
2371
2401
|
line: snippetInfo.line,
|
|
2372
2402
|
title: row.title,
|
|
2373
2403
|
...(row.context && { context: row.context }),
|
|
2404
|
+
...(row.metadata && Object.keys(row.metadata).length > 0 && { metadata: row.metadata }),
|
|
2374
2405
|
...(body && { body }),
|
|
2375
2406
|
...(snippet && { snippet }),
|
|
2376
2407
|
...(opts.explain && row.explain && { explain: row.explain }),
|
|
@@ -2620,14 +2651,46 @@ export function parseStructuredQuery(query) {
|
|
|
2620
2651
|
}
|
|
2621
2652
|
return typed.length > 0 ? { searches: typed, intent } : null;
|
|
2622
2653
|
}
|
|
2654
|
+
// Parse and validate a --filter JSON string; exits with an actionable
|
|
2655
|
+
// message on malformed JSON or an invalid filter AST.
|
|
2656
|
+
function parseCliMetadataFilter(rawFilter) {
|
|
2657
|
+
if (rawFilter === undefined)
|
|
2658
|
+
return undefined;
|
|
2659
|
+
let filterJson;
|
|
2660
|
+
try {
|
|
2661
|
+
filterJson = JSON.parse(String(rawFilter));
|
|
2662
|
+
}
|
|
2663
|
+
catch (err) {
|
|
2664
|
+
console.error(`Invalid --filter JSON: ${err instanceof Error ? err.message : String(err)}`);
|
|
2665
|
+
console.error(`Example: --filter '{"key":"status","operator":"eq","value":"published"}'`);
|
|
2666
|
+
process.exit(1);
|
|
2667
|
+
}
|
|
2668
|
+
try {
|
|
2669
|
+
return parseMetadataFilter(filterJson);
|
|
2670
|
+
}
|
|
2671
|
+
catch (err) {
|
|
2672
|
+
console.error(err instanceof Error ? err.message : String(err));
|
|
2673
|
+
process.exit(1);
|
|
2674
|
+
}
|
|
2675
|
+
}
|
|
2676
|
+
// Filtered search excludes documents without current metadata extraction;
|
|
2677
|
+
// tell the user when that makes results incomplete.
|
|
2678
|
+
function warnPendingMetadata(db) {
|
|
2679
|
+
const pendingMetadata = countDocumentsPendingMetadata(db);
|
|
2680
|
+
if (pendingMetadata === 0)
|
|
2681
|
+
return;
|
|
2682
|
+
process.stderr.write(`${c.yellow}Warning: ${pendingMetadata} document(s) lack current metadata extraction and are excluded from filtered results. Run 'qmd update'.${c.reset}\n`);
|
|
2683
|
+
}
|
|
2623
2684
|
function search(query, opts) {
|
|
2624
2685
|
const db = getDb();
|
|
2625
2686
|
// Validate collection filter (supports multiple -c flags)
|
|
2626
2687
|
// Use default collections if none specified
|
|
2627
2688
|
const collectionNames = resolveCollectionFilter(opts.collection, true);
|
|
2689
|
+
if (opts.filter)
|
|
2690
|
+
warnPendingMetadata(db);
|
|
2628
2691
|
// Use large limit for --all, otherwise fetch more than needed and let outputResults filter
|
|
2629
2692
|
const fetchLimit = opts.all ? 100000 : Math.max(50, opts.limit * 2);
|
|
2630
|
-
const results = searchFTS(db, query, fetchLimit, collectionNames);
|
|
2693
|
+
const results = searchFTS(db, query, fetchLimit, collectionSearchFilter(collectionNames), opts.filter);
|
|
2631
2694
|
// Add context to results
|
|
2632
2695
|
const resultsWithContext = results.map(r => ({
|
|
2633
2696
|
file: r.filepath,
|
|
@@ -2638,6 +2701,7 @@ function search(query, opts) {
|
|
|
2638
2701
|
context: getContextForFile(db, r.filepath),
|
|
2639
2702
|
hash: r.hash,
|
|
2640
2703
|
docid: r.docid,
|
|
2704
|
+
metadata: r.metadata,
|
|
2641
2705
|
}));
|
|
2642
2706
|
closeDb();
|
|
2643
2707
|
if (resultsWithContext.length === 0) {
|
|
@@ -2668,9 +2732,12 @@ async function vectorSearch(query, opts, _model = DEFAULT_EMBED_MODEL) {
|
|
|
2668
2732
|
// Use default collections if none specified
|
|
2669
2733
|
const collectionNames = resolveCollectionFilter(opts.collection, true);
|
|
2670
2734
|
checkIndexHealth(store.db);
|
|
2735
|
+
if (opts.filter)
|
|
2736
|
+
warnPendingMetadata(store.db);
|
|
2671
2737
|
await withLLMSession(async () => {
|
|
2672
2738
|
const results = await vectorSearchQuery(store, query, {
|
|
2673
|
-
collection: collectionNames,
|
|
2739
|
+
collection: collectionSearchFilter(collectionNames),
|
|
2740
|
+
filter: opts.filter,
|
|
2674
2741
|
limit: opts.all ? 500 : (opts.limit || 10),
|
|
2675
2742
|
minScore: opts.minScore || 0.3,
|
|
2676
2743
|
expansionContext: opts.intent,
|
|
@@ -2695,6 +2762,7 @@ async function vectorSearch(query, opts, _model = DEFAULT_EMBED_MODEL) {
|
|
|
2695
2762
|
score: r.score,
|
|
2696
2763
|
context: r.context,
|
|
2697
2764
|
docid: r.docid,
|
|
2765
|
+
metadata: r.metadata,
|
|
2698
2766
|
})), query, { ...opts, limit: results.length });
|
|
2699
2767
|
}, { maxDuration: 10 * 60 * 1000, name: 'vectorSearch' });
|
|
2700
2768
|
}
|
|
@@ -2704,6 +2772,8 @@ async function querySearch(query, opts, _embedModel = DEFAULT_EMBED_MODEL, _rera
|
|
|
2704
2772
|
// Use default collections if none specified
|
|
2705
2773
|
const collectionNames = resolveCollectionFilter(opts.collection, true);
|
|
2706
2774
|
checkIndexHealth(store.db);
|
|
2775
|
+
if (opts.filter)
|
|
2776
|
+
warnPendingMetadata(store.db);
|
|
2707
2777
|
// Check for structured query syntax (lex:/vec:/hyde:/intent: prefixes)
|
|
2708
2778
|
const parsed = parseStructuredQuery(query);
|
|
2709
2779
|
// Intent can come from --intent flag or from intent: line in query document
|
|
@@ -2731,6 +2801,7 @@ async function querySearch(query, opts, _embedModel = DEFAULT_EMBED_MODEL, _rera
|
|
|
2731
2801
|
process.stderr.write(`${c.dim}└─ Searching...${c.reset}\n`);
|
|
2732
2802
|
results = await structuredSearch(store, structuredQueries, {
|
|
2733
2803
|
collections: collectionNames.length > 0 ? collectionNames : undefined,
|
|
2804
|
+
filter: opts.filter,
|
|
2734
2805
|
limit: opts.all ? 500 : (opts.limit || 10),
|
|
2735
2806
|
minScore: opts.minScore || 0,
|
|
2736
2807
|
candidateLimit: opts.candidateLimit,
|
|
@@ -2760,6 +2831,8 @@ async function querySearch(query, opts, _embedModel = DEFAULT_EMBED_MODEL, _rera
|
|
|
2760
2831
|
// Standard hybrid query with automatic expansion
|
|
2761
2832
|
results = await hybridQuery(store, query, {
|
|
2762
2833
|
collections: collectionNames.length > 0 ? collectionNames : undefined,
|
|
2834
|
+
collection: collectionSearchFilter(collectionNames),
|
|
2835
|
+
filter: opts.filter,
|
|
2763
2836
|
limit: opts.all ? 500 : (opts.limit || 10),
|
|
2764
2837
|
minScore: opts.minScore || 0,
|
|
2765
2838
|
candidateLimit: opts.candidateLimit,
|
|
@@ -2826,6 +2899,7 @@ async function querySearch(query, opts, _embedModel = DEFAULT_EMBED_MODEL, _rera
|
|
|
2826
2899
|
score: r.score,
|
|
2827
2900
|
context: r.context,
|
|
2828
2901
|
docid: r.docid,
|
|
2902
|
+
metadata: r.metadata,
|
|
2829
2903
|
explain: r.explain,
|
|
2830
2904
|
})), displayQuery, { ...opts, limit: results.length });
|
|
2831
2905
|
}, { maxDuration: 10 * 60 * 1000, name: 'querySearch' });
|
|
@@ -2862,6 +2936,7 @@ function parseCLI() {
|
|
|
2862
2936
|
json: { type: "boolean" },
|
|
2863
2937
|
explain: { type: "boolean" },
|
|
2864
2938
|
collection: { type: "string", short: "c", multiple: true }, // Filter by collection(s)
|
|
2939
|
+
filter: { type: "string" }, // Metadata filter (JSON AST) for search/vsearch/query
|
|
2865
2940
|
// Collection options
|
|
2866
2941
|
name: { type: "string" }, // collection name
|
|
2867
2942
|
mask: { type: "string" }, // glob pattern
|
|
@@ -3458,6 +3533,8 @@ function showHelp() {
|
|
|
3458
3533
|
console.log(" --explain - Include retrieval score traces (query, CLI/--format json)");
|
|
3459
3534
|
console.log(" --format <kind> - Output format: cli (default) | json | csv | md | xml | files");
|
|
3460
3535
|
console.log(" -c, --collection <name> - Filter by one or more collections");
|
|
3536
|
+
console.log(" --filter <json> - Metadata filter (recursive JSON AST; search/vsearch/query)");
|
|
3537
|
+
console.log(" e.g. '{\"key\":\"status\",\"operator\":\"eq\",\"value\":\"published\"}'");
|
|
3461
3538
|
console.log("");
|
|
3462
3539
|
console.log("Embed/query options:");
|
|
3463
3540
|
console.log(" --chunk-strategy <auto|regex> - Chunking mode (default: regex; auto uses AST for code files)");
|
|
@@ -3916,10 +3993,6 @@ async function runDoctorDeviceChecks(nextSteps) {
|
|
|
3916
3993
|
}
|
|
3917
3994
|
}
|
|
3918
3995
|
async function showDoctor() {
|
|
3919
|
-
const storeInstance = getDoctorStore();
|
|
3920
|
-
const db = storeInstance.db;
|
|
3921
|
-
const pkg = readPackageJson();
|
|
3922
|
-
const activeModels = resolveModelsForCli();
|
|
3923
3996
|
let doctorConfig;
|
|
3924
3997
|
try {
|
|
3925
3998
|
doctorConfig = loadConfig();
|
|
@@ -3928,6 +4001,9 @@ async function showDoctor() {
|
|
|
3928
4001
|
// The dedicated index-config check below reports parse errors. Keep the
|
|
3929
4002
|
// remaining diagnostics available by falling back to DB/default config.
|
|
3930
4003
|
}
|
|
4004
|
+
const storeInstance = getDoctorStore({ reconcileConfig: doctorConfig });
|
|
4005
|
+
const db = storeInstance.db;
|
|
4006
|
+
const activeModels = resolveModelsForCli();
|
|
3931
4007
|
const doctorEmbedding = resolveEmbeddingConfig({
|
|
3932
4008
|
config: doctorConfig,
|
|
3933
4009
|
dbConfig: readCanonicalEmbeddingConfig(db),
|
|
@@ -3938,7 +4014,7 @@ async function showDoctor() {
|
|
|
3938
4014
|
const nextSteps = [];
|
|
3939
4015
|
console.log(`${c.bold}QMD Doctor${c.reset}\n`);
|
|
3940
4016
|
console.log(`Index: ${getDbPath()}`);
|
|
3941
|
-
console.log(`Runtime:
|
|
4017
|
+
console.log(`Runtime: node:sqlite`);
|
|
3942
4018
|
try {
|
|
3943
4019
|
const row = db.prepare(`SELECT sqlite_version() AS version`).get();
|
|
3944
4020
|
doctorCheck("SQLite runtime", true, row.version);
|
|
@@ -3946,8 +4022,7 @@ async function showDoctor() {
|
|
|
3946
4022
|
catch (error) {
|
|
3947
4023
|
doctorCheck("SQLite runtime", false, error instanceof Error ? error.message : String(error));
|
|
3948
4024
|
}
|
|
3949
|
-
|
|
3950
|
-
doctorCheck("better-sqlite3 package", true, String(betterSqliteVersion));
|
|
4025
|
+
doctorCheck("node:sqlite", true, process.versions.node);
|
|
3951
4026
|
try {
|
|
3952
4027
|
loadSqliteVec(db);
|
|
3953
4028
|
const row = db.prepare(`SELECT vec_version() AS version`).get();
|
|
@@ -4530,6 +4605,7 @@ if (isMain) {
|
|
|
4530
4605
|
console.error("Usage: qmd search [options] <query>");
|
|
4531
4606
|
process.exit(1);
|
|
4532
4607
|
}
|
|
4608
|
+
cli.opts.filter = parseCliMetadataFilter(cli.values.filter);
|
|
4533
4609
|
search(cli.query, cli.opts);
|
|
4534
4610
|
break;
|
|
4535
4611
|
case "vsearch":
|
|
@@ -4542,6 +4618,7 @@ if (isMain) {
|
|
|
4542
4618
|
if (!cli.values["min-score"]) {
|
|
4543
4619
|
cli.opts.minScore = 0.3;
|
|
4544
4620
|
}
|
|
4621
|
+
cli.opts.filter = parseCliMetadataFilter(cli.values.filter);
|
|
4545
4622
|
await resolveLocalConfigTrust();
|
|
4546
4623
|
await vectorSearch(cli.query, cli.opts);
|
|
4547
4624
|
break;
|
|
@@ -4551,6 +4628,7 @@ if (isMain) {
|
|
|
4551
4628
|
console.error("Usage: qmd query [options] <query>");
|
|
4552
4629
|
process.exit(1);
|
|
4553
4630
|
}
|
|
4631
|
+
cli.opts.filter = parseCliMetadataFilter(cli.values.filter);
|
|
4554
4632
|
await resolveLocalConfigTrust();
|
|
4555
4633
|
await querySearch(cli.query, cli.opts);
|
|
4556
4634
|
break;
|
package/dist/collections.js
CHANGED
|
@@ -41,11 +41,16 @@ export function setConfigSource(source) {
|
|
|
41
41
|
* Config file will be ~/.config/qmd/{indexName}.yml
|
|
42
42
|
*/
|
|
43
43
|
export function setConfigIndexName(name) {
|
|
44
|
-
// Resolve relative paths to absolute paths and sanitize for use as filename
|
|
45
|
-
|
|
46
|
-
|
|
44
|
+
// Resolve relative paths to absolute paths and sanitize for use as filename.
|
|
45
|
+
// Windows absolute paths (C:\..., \\server\share) are already absolute --
|
|
46
|
+
// skip resolve() for those (POSIX path.resolve doesn't recognize a drive
|
|
47
|
+
// letter as absolute and would wrongly prefix it with cwd) and sanitize
|
|
48
|
+
// `:` and `\` alongside `/` so the derived filename is valid on Windows.
|
|
49
|
+
const isWindowsAbsolute = /^[a-zA-Z]:[\\/]/.test(name) || /^\\/.test(name);
|
|
50
|
+
if (isWindowsAbsolute || /[\\/]/.test(name)) {
|
|
51
|
+
const absolutePath = isWindowsAbsolute ? name : resolve(process.cwd(), name);
|
|
47
52
|
// Replace path separators with underscores to create a valid filename
|
|
48
|
-
currentIndexName = absolutePath.replace(
|
|
53
|
+
currentIndexName = absolutePath.replace(/[:\\/]+/g, '_').replace(/^_+/, '');
|
|
49
54
|
}
|
|
50
55
|
else {
|
|
51
56
|
currentIndexName = name;
|
package/dist/db.d.ts
CHANGED
|
@@ -1,39 +1,26 @@
|
|
|
1
1
|
/**
|
|
2
2
|
* db.ts - SQLite database connection and extension management
|
|
3
3
|
*
|
|
4
|
-
* Provides
|
|
5
|
-
* and sqlite-vec.
|
|
4
|
+
* Provides a synchronous node:sqlite connection with QMD's transaction helper
|
|
5
|
+
* and sqlite-vec extension loading.
|
|
6
6
|
*/
|
|
7
|
-
import
|
|
7
|
+
import { DatabaseSync, type StatementSync } from "node:sqlite";
|
|
8
8
|
export type SQLiteValue = string | number | bigint | Buffer | Uint8Array | Float32Array | null;
|
|
9
9
|
export type SQLiteParams = readonly SQLiteValue[];
|
|
10
|
-
|
|
11
|
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
17
|
-
|
|
18
|
-
|
|
19
|
-
|
|
20
|
-
|
|
21
|
-
|
|
22
|
-
* Default 120_000 ms outlasts the worst-case batch commit on a multi-GB
|
|
23
|
-
* index. Override with `QMD_SQLITE_BUSY_TIMEOUT` (value in milliseconds; `0`
|
|
24
|
-
* restores the upstream fail-fast behaviour).
|
|
25
|
-
*/
|
|
10
|
+
export type Transaction<TArgs extends unknown[], TResult> = ((...args: TArgs) => TResult) & {
|
|
11
|
+
deferred: (...args: TArgs) => TResult;
|
|
12
|
+
immediate: (...args: TArgs) => TResult;
|
|
13
|
+
exclusive: (...args: TArgs) => TResult;
|
|
14
|
+
};
|
|
15
|
+
/** Synchronous SQLite connection with QMD-compatible transactions. */
|
|
16
|
+
export declare class Database extends DatabaseSync {
|
|
17
|
+
transaction<TArgs extends unknown[], TResult>(operation: (...args: TArgs) => TResult): Transaction<TArgs, TResult>;
|
|
18
|
+
}
|
|
19
|
+
/** Statement type used throughout QMD. */
|
|
20
|
+
export type Statement<T extends SQLiteParams = SQLiteParams> = StatementSync;
|
|
21
|
+
/** Open a writable QMD database using Node's built-in SQLite runtime. */
|
|
26
22
|
export declare function openDatabase(path: string): Database;
|
|
27
23
|
/** Open an existing database without changing journal mode, schema, or user data. */
|
|
28
24
|
export declare function openReadOnlyDatabase(path: string): Database;
|
|
29
|
-
/**
|
|
30
|
-
* Database and Statement types used throughout QMD.
|
|
31
|
-
*/
|
|
32
|
-
export type Database = BetterSqlite3.Database;
|
|
33
|
-
export type Statement<T extends SQLiteParams = SQLiteParams> = BetterSqlite3.Statement<T>;
|
|
34
|
-
/**
|
|
35
|
-
* Load the sqlite-vec extension into a database.
|
|
36
|
-
*
|
|
37
|
-
* Throws with fix instructions when the extension is unavailable.
|
|
38
|
-
*/
|
|
25
|
+
/** Load the sqlite-vec extension into a database. */
|
|
39
26
|
export declare function loadSqliteVec(db: Database): void;
|