@wei840222/qmd 2026.9.6 → 2026.9.27
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +32 -0
- package/README.md +84 -1
- package/dist/cli/build-info.json +2 -2
- package/dist/cli/qmd.js +141 -11
- package/dist/collections.d.ts +3 -0
- package/dist/collections.js +9 -4
- package/dist/{hybrid-llm.d.ts → hybrid.d.ts} +9 -3
- package/dist/hybrid.js +97 -0
- package/dist/index.d.ts +10 -0
- package/dist/index.js +28 -4
- package/dist/llm.d.ts +19 -1
- package/dist/llm.js +28 -5
- package/dist/mcp/server.js +70 -6
- package/dist/metadata-filter.d.ts +74 -0
- package/dist/metadata-filter.js +279 -0
- package/dist/metadata-store.d.ts +54 -0
- package/dist/metadata-store.js +194 -0
- package/dist/metadata.d.ts +61 -0
- package/dist/metadata.js +221 -0
- package/dist/remote-jev.d.ts +38 -0
- package/dist/remote-jev.js +167 -0
- package/dist/remote-llm.d.ts +3 -1
- package/dist/remote-llm.js +21 -4
- package/dist/search/query-expansion.d.ts +1 -0
- package/dist/search/query-expansion.js +4 -1
- package/dist/search/zh-dict.txt +3 -0
- package/dist/store.d.ts +29 -16
- package/dist/store.js +349 -166
- package/package.json +3 -2
- package/scripts/sync-zh-dict.mjs +4 -1
- package/skills/qmd/SKILL.md +11 -0
- package/dist/hybrid-llm.js +0 -53
- package/skills/release/SKILL.md +0 -141
- package/skills/release/scripts/install-hooks.sh +0 -38
- package/skills/release/scripts/release-context.sh +0 -129
package/CHANGELOG.md
CHANGED
|
@@ -2,13 +2,45 @@
|
|
|
2
2
|
|
|
3
3
|
## [Unreleased]
|
|
4
4
|
|
|
5
|
+
### Fixed
|
|
6
|
+
|
|
7
|
+
- Metadata extraction error retry: `isDocumentMetadataCurrent` now requires an error-free extraction, allowing `qmd update` to automatically re-attempt extraction on documents that previously failed without requiring manual edits.
|
|
8
|
+
- Non-QMD frontmatter tolerance: Markdown documents whose leading frontmatter has formatting quirks (such as unquoted colons in titles) but does not declare `qmd:` are no longer treated as extraction failures or excluded from filtered search.
|
|
9
|
+
- Relative temporal query expansion bypass and conjunctive lexical dilution: Relative temporal queries (e.g. "昨天", "前天", "yesterday") now bypass the BM25 strong-signal expansion skip, ensuring that archival documents containing relative words cannot preempt target date resolution. In addition, lexical expansion prompts now instruct models to keep search terms minimal without appending generic synonyms (such as "日誌", "行程", "活動") that inadvertently eliminate valid documents under QMD's conjunctive (AND) FTS5 matching.
|
|
10
|
+
|
|
11
|
+
### Added
|
|
12
|
+
|
|
13
|
+
- TypeSafe Jev provider (`src/remote-jev.ts`) supporting System One-based candidate reranking via Noul judgments and query expansion intent classification via Choice and Noul gating. Passes query expansion context to Jev state for contextual intent classification.
|
|
14
|
+
- Refactored `HybridLLM` into `Hybrid` (`src/hybrid.ts`) supporting 3-way provider fallback: `RemoteJev` → `RemoteLLM` → `LlamaCpp`.
|
|
15
|
+
- Added `models.jev_api_key`, `models.jev_api_model`, and `models.jev_base_url` configuration options with `TYPESAFE_API_KEY`, `TYPESAFE_DEFAULT_MODEL`, and `TYPESAFE_BASE_URL` environment variable fallbacks.
|
|
16
|
+
- Added TypeSafe Jev provider diagnostic check to `qmd doctor`.
|
|
17
|
+
- Detailed metadata extraction error reporting: `qmd status` and `qmd doctor` now list pending/errored document paths with error hints, and `qmd update` outputs specific files and causes when frontmatter extraction errors occur.
|
|
18
|
+
- Document file path and title metadata in candidate reranking: candidate documents for chat-based and remote rerankers now retain document file paths and titles, enabling rerankers to evaluate provenance and temporal references.
|
|
19
|
+
- Relative temporal resolution in query expansion: `expandQuery` prompts now instruct models to calculate exact ISO dates from relative time expressions (e.g. "yesterday", "昨天", "today", "last week") against current local time for lexical and vector search.
|
|
20
|
+
- Added file, title, and current local time to TypeSafe Jev candidate reranking and intent classification states, with temporal constraint guidance.
|
|
21
|
+
- Search intent playbook and guidance for query expansion: Added `JEV_STRATEGY_PLAYBOOK` mapping Jev intent classifications (`code_search`, `concept_search`, `factual_lookup`, `broad_exploration`) to concrete lexical and vector search guidance passed to downstream expansion LLMs in dedicated `<search_intent>` XML blocks, keeping untrusted context separate.
|
|
22
|
+
|
|
23
|
+
## [2026.9.25] - 2026-09-26
|
|
24
|
+
|
|
5
25
|
### Changed
|
|
6
26
|
|
|
7
27
|
- Replace `better-sqlite3` with Node.js built-in `node:sqlite`, preventing SQLite runtime symbol collisions when QMD is embedded alongside other `node:sqlite` consumers. The minimum supported Node.js version is now 22.16.0.
|
|
28
|
+
- Upgraded `@node-rs/jieba` to `2.0.3` and synchronized bundled Traditional Chinese dictionary assets (`zh-dict.txt`) with upstream `sysprog21/zhtw-mcp`.
|
|
29
|
+
- Relocated repository release management skill to `.agents/skills/release`.
|
|
8
30
|
|
|
9
31
|
### Added
|
|
10
32
|
|
|
33
|
+
- Added Oxlint lint fence.
|
|
34
|
+
- Document metadata and metadata filtering. Markdown documents can opt into typed metadata through a namespaced frontmatter block (`qmd.metadata` with strings, numbers, booleans, or flat homogeneous arrays), and every search surface — CLI `search`/`vsearch`/`query` via `--filter <json>`, the SDK's `filter` option on `search()`/`searchLex()`/`searchVector()`, the MCP `query` tool, and HTTP `POST /query` and `/search` — accepts one shared recursive filter AST discriminated by `operator`: `and`/`or`/`not` logical groups, `eq`/`ne`/`gt`/`gte`/`lt`/`lte` comparisons, `in`/`nin`/`all` membership, and `exists` presence. Every returned result satisfies the filter (applied before RRF fusion and reranking); like collection filtering, highly selective filters remain best-effort for top-K completeness. Frontmatter stays ordinary searchable content — no chunking, embedding, snippet, or line-number changes — and documents without `qmd.metadata` behave exactly as before. JSON/SDK/MCP/HTTP results now include each document's indexed metadata, and `qmd status` reports how many documents still need metadata extraction (a normal `qmd update` backfills existing indexes).
|
|
11
35
|
- **Disable HyDE Expansion Control**: Added `--no-hyde` CLI option for `qmd query` and `qmd vsearch`, `includeHyde` parameter to SDK (`store.search`, `store.expandQuery`) and MCP `query` tool, allowing users to disable generating hypothetical document embeddings during query expansion.
|
|
36
|
+
- Added `typesafe-ai` skill to `.agents/skills/typesafe-ai`.
|
|
37
|
+
|
|
38
|
+
### Fixed
|
|
39
|
+
|
|
40
|
+
- Embedding generation and legacy fingerprint adoption now tokenize documents
|
|
41
|
+
with the store-selected embedding model instead of the global default. This
|
|
42
|
+
keeps chunk boundaries aligned with the model that creates and verifies the
|
|
43
|
+
stored vectors without initializing an unrelated provider.
|
|
12
44
|
|
|
13
45
|
## [2026.8.23-1] - 2026-08-23
|
|
14
46
|
|
package/README.md
CHANGED
|
@@ -132,7 +132,7 @@ runs in a container and a liveness probe connects from a non-loopback address.
|
|
|
132
132
|
|
|
133
133
|
The HTTP server exposes two endpoints:
|
|
134
134
|
- `POST /mcp` — MCP Streamable HTTP (JSON responses, stateless)
|
|
135
|
-
- `POST /query` (alias `/search`) — structured search without the MCP protocol
|
|
135
|
+
- `POST /query` (alias `/search`) — structured search without the MCP protocol. Accepts the same optional `filter` object as the `query` tool (invalid filters return `400`); see [Metadata Filtering](#metadata-filtering)
|
|
136
136
|
- `GET /health` — liveness check with uptime
|
|
137
137
|
|
|
138
138
|
|
|
@@ -170,8 +170,10 @@ Point any MCP client at `http://localhost:8181/mcp` to connect.
|
|
|
170
170
|
| `query` | `searches` | array | Typed sub-queries (`lex`/`vec`/`hyde`), 1–10. Mutually exclusive with `query`; exactly one is required. First gets 2x weight. |
|
|
171
171
|
| `query` | `expansion` | string | Plain-query policy: `auto` (default), `force`, or `skip`. Ignored when `searches` is used. |
|
|
172
172
|
| `query` | `collections` | string[] | Filter by collection names (OR). **Array only** — singular `collection` is silently ignored. |
|
|
173
|
+
| `query` | `filter` | object | Metadata filter (recursive `operator`-discriminated JSON AST; see [Metadata Filtering](#metadata-filtering)) |
|
|
173
174
|
| `query` | `expansionContext` | string | Additional context used only to generate `lex` / `vec` / `hyde` query expansions. |
|
|
174
175
|
| `query` | `rerankContext` | string | Additional context used only for reranking and snippet/chunk selection. |
|
|
176
|
+
| `query` | `intent` | string | Disambiguation context (alias for expansionContext & rerankContext; does not search on its own) |
|
|
175
177
|
| `query` | `limit` | number | Max results (default 10) |
|
|
176
178
|
| `query` | `minScore` | number | Minimum relevance 0–1 (default 0) |
|
|
177
179
|
| `query` | `candidateLimit` | number | Max candidates to rerank (default 40) |
|
|
@@ -279,6 +281,20 @@ const results3 = await store.search({
|
|
|
279
281
|
|
|
280
282
|
// Skip reranking for faster results
|
|
281
283
|
const fast = await store.search({ query: "auth", rerank: false })
|
|
284
|
+
|
|
285
|
+
// Metadata filter — every returned result satisfies it (also available on
|
|
286
|
+
// searchLex() and searchVector()); results expose indexed metadata via
|
|
287
|
+
// r.metadata. See "Metadata Filtering" for the full grammar.
|
|
288
|
+
const published = await store.search({
|
|
289
|
+
query: "authentication flow",
|
|
290
|
+
filter: {
|
|
291
|
+
operator: "and",
|
|
292
|
+
operands: [
|
|
293
|
+
{ key: "topics", operator: "all", value: ["typescript"] },
|
|
294
|
+
{ key: "status", operator: "ne", value: "draft" },
|
|
295
|
+
],
|
|
296
|
+
},
|
|
297
|
+
})
|
|
282
298
|
```
|
|
283
299
|
|
|
284
300
|
For simple queries, explicit `force` or `skip` overrides `auto`. Under `auto`, CJK
|
|
@@ -979,6 +995,7 @@ and `deep-search` (→ `query`).
|
|
|
979
995
|
--full # Show full document content
|
|
980
996
|
--line-numbers # Add line numbers to output
|
|
981
997
|
--explain # Include retrieval score traces (query, JSON/CLI output)
|
|
998
|
+
--filter <json> # Metadata filter (recursive JSON AST; see Metadata Filtering)
|
|
982
999
|
--index <name> # Use named index
|
|
983
1000
|
--intent "<text>" # Legacy CLI alias for rerank context (e.g. "web page load times")
|
|
984
1001
|
--no-rerank # Skip LLM reranking (RRF scores only; faster on CPU)
|
|
@@ -1022,6 +1039,72 @@ explicitly with `-c`.
|
|
|
1022
1039
|
> lexical and vector candidate cutoff. Matching candidates from the selected
|
|
1023
1040
|
> collections are then ranked together.
|
|
1024
1041
|
|
|
1042
|
+
### Metadata Filtering
|
|
1043
|
+
|
|
1044
|
+
Documents can opt into typed metadata through a namespaced frontmatter block. A document without `qmd.metadata` behaves exactly as before, and the frontmatter stays ordinary searchable content (no chunking, embedding, or line-number changes):
|
|
1045
|
+
|
|
1046
|
+
```markdown
|
|
1047
|
+
---
|
|
1048
|
+
qmd:
|
|
1049
|
+
metadata:
|
|
1050
|
+
topics:
|
|
1051
|
+
- typescript
|
|
1052
|
+
- programming
|
|
1053
|
+
status: published
|
|
1054
|
+
priority: 3
|
|
1055
|
+
reviewed: true
|
|
1056
|
+
---
|
|
1057
|
+
|
|
1058
|
+
# Document body starts here
|
|
1059
|
+
```
|
|
1060
|
+
|
|
1061
|
+
Supported values are strings, numbers, booleans, and flat homogeneous arrays of one of those. Nested objects, nulls, empty arrays, and mixed-type arrays are rejected (the document still indexes; it is excluded from filtered search until corrected). Metadata keys are user-defined data — `tags`, `topics`, and `labels` are all ordinary keys with no special semantics.
|
|
1062
|
+
|
|
1063
|
+
Every search surface (CLI, SDK, MCP, HTTP) accepts the same recursive filter, a JSON AST discriminated by `operator`:
|
|
1064
|
+
|
|
1065
|
+
```sh
|
|
1066
|
+
# One condition
|
|
1067
|
+
qmd search "authentication" \
|
|
1068
|
+
--filter '{"key":"status","operator":"eq","value":"published"}'
|
|
1069
|
+
|
|
1070
|
+
# Composed conditions — works with search, vsearch, and query
|
|
1071
|
+
qmd query "dependency injection" --filter '{
|
|
1072
|
+
"operator": "and",
|
|
1073
|
+
"operands": [
|
|
1074
|
+
{ "key": "topics", "operator": "all", "value": ["typescript", "programming"] },
|
|
1075
|
+
{ "key": "status", "operator": "nin", "value": ["draft", "archived"] },
|
|
1076
|
+
{ "operator": "or", "operands": [
|
|
1077
|
+
{ "key": "priority", "operator": "gte", "value": 3 },
|
|
1078
|
+
{ "key": "reviewed", "operator": "eq", "value": true }
|
|
1079
|
+
] },
|
|
1080
|
+
{ "operator": "not", "operand": { "key": "audience", "operator": "eq", "value": "internal" } }
|
|
1081
|
+
]
|
|
1082
|
+
}'
|
|
1083
|
+
```
|
|
1084
|
+
|
|
1085
|
+
| Node | Shape |
|
|
1086
|
+
|------|-------|
|
|
1087
|
+
| Logical group | `{ "operator": "and" \| "or", "operands": […] }` |
|
|
1088
|
+
| Negation | `{ "operator": "not", "operand": {…} }` |
|
|
1089
|
+
| Comparison | `{ "key", "operator": "eq" \| "ne" \| "gt" \| "gte" \| "lt" \| "lte", "value" }` |
|
|
1090
|
+
| Membership | `{ "key", "operator": "in" \| "nin" \| "all", "value": […] }` |
|
|
1091
|
+
| Presence | `{ "key", "operator": "exists", "value": true \| false }` |
|
|
1092
|
+
|
|
1093
|
+
Semantics:
|
|
1094
|
+
|
|
1095
|
+
- Matching is typed and exact — no string/number/boolean coercion, and a type mismatch never matches (including `ne` and `nin`).
|
|
1096
|
+
- Array-valued metadata is a set: a condition matches when any element satisfies it, `all` requires every filter value to be present.
|
|
1097
|
+
- Missing keys do not match `ne`/`nin`; combine with `{ "operator": "exists", "value": false }` in an `or` group to include them.
|
|
1098
|
+
- Multiple conditions require an explicit `and` group — there is no implicit AND, and no `$`-prefixed shorthand.
|
|
1099
|
+
|
|
1100
|
+
Guarantees and limits:
|
|
1101
|
+
|
|
1102
|
+
- Every returned result satisfies the filter, before RRF fusion and reranking.
|
|
1103
|
+
- Like collection filtering, highly selective filters are best-effort for top-K completeness: backends over-fetch and post-filter, so a very selective filter can return fewer than `limit` results.
|
|
1104
|
+
- Filtered search only considers documents whose metadata has been extracted (run `qmd update` after upgrading; `qmd status` shows the pending count).
|
|
1105
|
+
|
|
1106
|
+
JSON output (`--format json`), the SDK, MCP structured results, and the HTTP endpoints include each result's indexed metadata.
|
|
1107
|
+
|
|
1025
1108
|
### Output Format
|
|
1026
1109
|
|
|
1027
1110
|
Default output is colorized CLI format (respects `NO_COLOR` env).
|
package/dist/cli/build-info.json
CHANGED
package/dist/cli/qmd.js
CHANGED
|
@@ -10,10 +10,13 @@ import { parseArgs } from "util";
|
|
|
10
10
|
import { readFileSync, readdirSync, realpathSync, statSync, existsSync, unlinkSync, writeFileSync, openSync, closeSync, mkdirSync, lstatSync, rmSync, symlinkSync, readlinkSync, copyFileSync } from "fs";
|
|
11
11
|
import { createInterface } from "readline/promises";
|
|
12
12
|
import { getPwd, getRealPath, isPathInsideDir, homedir, resolve, enableProductionMode, searchFTS, extractSnippet, getContextForFile, getContextForPath, listCollections, findSimilarFiles, findDocument, resolveCommaListName, matchFilesByGlob, getHashesNeedingEmbedding, clearAllEmbeddings, insertEmbedding, getStatus, hashContent, extractTitle, formatDocForEmbedding, getEmbeddingFingerprint, chunkDocumentByTokens, clearCache, getCacheKey, getCachedResult, setCachedResult, getIndexHealth, parseVirtualPath, buildVirtualPath, isVirtualPath, isDocid, resolveVirtualPath, toVirtualPath, insertContent, insertDocument, insertDocumentWithContent, findActiveDocument, findOrMigrateLegacyDocument, updateDocumentTitle, updateDocument, updateDocumentWithContent, deactivateDocument, getActiveDocumentPaths, cleanupOrphanedContent, countOrphanedVectors, previewCleanup, runCleanup, getCollectionsWithoutContext, getTopLevelPathsWithoutContext, handelize, escapeLikePattern, hybridQuery, vectorSearchQuery, structuredSearch, addLineNumbers, DEFAULT_EMBED_MODEL, DEFAULT_EMBED_MAX_BATCH_BYTES, DEFAULT_EMBED_MAX_DOCS_PER_BATCH, DEFAULT_RERANK_MODEL, DEFAULT_QUERY_MODEL, DEFAULT_GLOB, splitGlobMask, DEFAULT_MULTI_GET_MAX_BYTES, createStore, getDefaultDbPath, reindexCollection, generateEmbeddings, getPendingEmbeddingDocsReadOnly, syncConfigToDb, } from "../store.js";
|
|
13
|
+
import { syncDocumentMetadata, countDocumentsPendingMetadata, getDocumentsPendingMetadata } from "../metadata-store.js";
|
|
14
|
+
import { parseMetadataFilter } from "../metadata-filter.js";
|
|
13
15
|
import { disposeDefaultLlamaCpp, getDefaultLlamaCpp, setDefaultLlamaCpp, LlamaCpp, withLLMSession, pullModels, DEFAULT_MODEL_CACHE_DIR, resolveEmbedModel, resolveGenerateModel, resolveRerankModel, resolveModels, inspectGgufFile, isDarwinMetalMitigationActive } from "../llm.js";
|
|
14
16
|
import { rebuildCjkLexicalIndex } from "../search/cjk-index.js";
|
|
15
17
|
import { RemoteLLM } from "../remote-llm.js";
|
|
16
|
-
import {
|
|
18
|
+
import { Hybrid } from "../hybrid.js";
|
|
19
|
+
import { RemoteJev } from "../remote-jev.js";
|
|
17
20
|
import { EmbeddingConfigError, OPENAI_EMBEDDING_MODEL, readCanonicalEmbeddingConfig, resolveEmbeddingConfig, writeCanonicalEmbeddingConfig, } from "../embedding/config.js";
|
|
18
21
|
import { OpenAIEmbeddingProvider, UnavailableOpenAIEmbeddingProvider, } from "../embedding/openai.js";
|
|
19
22
|
import { createCliEmbeddingProviderOwner } from "./embedding-owner.js";
|
|
@@ -131,8 +134,19 @@ function getStore() {
|
|
|
131
134
|
rerankApiKey: config?.models?.rerank_api_key,
|
|
132
135
|
})
|
|
133
136
|
: undefined;
|
|
137
|
+
const jevApiKey = config?.models?.jev_api_key?.trim() || process.env.TYPESAFE_API_KEY?.trim();
|
|
138
|
+
const jevBaseUrl = config?.models?.jev_base_url?.trim() || process.env.TYPESAFE_BASE_URL?.trim();
|
|
139
|
+
const jevModel = config?.models?.jev_api_model?.trim() || process.env.TYPESAFE_DEFAULT_MODEL?.trim() || "jev-1.13";
|
|
140
|
+
const remoteJev = jevApiKey
|
|
141
|
+
? new RemoteJev({
|
|
142
|
+
apiKey: jevApiKey,
|
|
143
|
+
baseUrl: jevBaseUrl,
|
|
144
|
+
model: jevModel,
|
|
145
|
+
timeoutMs: 30000,
|
|
146
|
+
})
|
|
147
|
+
: undefined;
|
|
134
148
|
if (cliLlama) {
|
|
135
|
-
store.llm = remoteLlm ? new
|
|
149
|
+
store.llm = (remoteLlm || remoteJev) ? new Hybrid(cliLlama, remoteLlm, remoteJev) : cliLlama;
|
|
136
150
|
}
|
|
137
151
|
}
|
|
138
152
|
return store;
|
|
@@ -219,11 +233,16 @@ function mcpDaemonPaths() {
|
|
|
219
233
|
}
|
|
220
234
|
function setIndexName(name) {
|
|
221
235
|
let normalizedName = name;
|
|
222
|
-
// Normalize relative paths to prevent malformed database paths
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
236
|
+
// Normalize relative paths to prevent malformed database paths. Windows
|
|
237
|
+
// absolute paths (C:\..., \\server\share) are already absolute -- skip
|
|
238
|
+
// pathResolve() for those and sanitize `:` and `\` alongside `/`.
|
|
239
|
+
if (name) {
|
|
240
|
+
const isWindowsAbsolute = /^[a-zA-Z]:[\\/]/.test(name) || /^\\/.test(name);
|
|
241
|
+
if (isWindowsAbsolute || /[\\/]/.test(name)) {
|
|
242
|
+
const absolutePath = isWindowsAbsolute ? name : pathResolve(process.cwd(), name);
|
|
243
|
+
// Replace path separators with underscores to create a valid filename
|
|
244
|
+
normalizedName = absolutePath.replace(/[:\\/]+/g, '_').replace(/^_+/, '');
|
|
245
|
+
}
|
|
227
246
|
}
|
|
228
247
|
currentIndexName = normalizedName || "index";
|
|
229
248
|
storeDbPathOverride = normalizedName ? getDefaultDbPath(normalizedName) : undefined;
|
|
@@ -546,6 +565,18 @@ async function showStatus() {
|
|
|
546
565
|
if (needsEmbedding > 0) {
|
|
547
566
|
console.log(` ${c.yellow}Pending: ${needsEmbedding} need embedding${c.reset} (run 'qmd embed')`);
|
|
548
567
|
}
|
|
568
|
+
const pendingMetadata = countDocumentsPendingMetadata(db);
|
|
569
|
+
if (pendingMetadata > 0) {
|
|
570
|
+
console.log(` ${c.yellow}Metadata: ${pendingMetadata} need extraction${c.reset} (run 'qmd update'; excluded from --filter searches)`);
|
|
571
|
+
const pendingDocs = getDocumentsPendingMetadata(db, 3);
|
|
572
|
+
for (const doc of pendingDocs) {
|
|
573
|
+
const errHint = doc.error ? `: ${doc.error.split("\n")[0]}` : "";
|
|
574
|
+
console.log(` ${c.dim}• qmd://${doc.collection}/${doc.path}${errHint}${c.reset}`);
|
|
575
|
+
}
|
|
576
|
+
if (pendingMetadata > pendingDocs.length) {
|
|
577
|
+
console.log(` ${c.dim}... and ${pendingMetadata - pendingDocs.length} more${c.reset}`);
|
|
578
|
+
}
|
|
579
|
+
}
|
|
549
580
|
if (mostRecent.latest) {
|
|
550
581
|
const lastUpdate = new Date(mostRecent.latest);
|
|
551
582
|
console.log(` Updated: ${formatTimeAgo(lastUpdate)}`);
|
|
@@ -954,6 +985,7 @@ async function updateCollections() {
|
|
|
954
985
|
progress.clear();
|
|
955
986
|
console.log(`\nIndexed: ${result.indexed} new, ${result.updated} updated, ${result.unchanged} unchanged, ${result.removed} removed`);
|
|
956
987
|
reportSkippedReads(result.skippedFiles);
|
|
988
|
+
reportMetadataErrors(result.metadataErrors, result.metadataErrorFiles);
|
|
957
989
|
if (result.orphanedCleaned > 0) {
|
|
958
990
|
console.log(`Cleaned up ${result.orphanedCleaned} orphaned content hash(es)`);
|
|
959
991
|
}
|
|
@@ -1858,8 +1890,9 @@ async function indexFiles(pwd, globPattern = DEFAULT_GLOB, collectionName, suppr
|
|
|
1858
1890
|
console.log("No files found matching pattern.");
|
|
1859
1891
|
// Continue so the deactivation pass can mark previously indexed docs as inactive.
|
|
1860
1892
|
}
|
|
1861
|
-
let indexed = 0, updated = 0, unchanged = 0, processed = 0;
|
|
1893
|
+
let indexed = 0, updated = 0, unchanged = 0, processed = 0, metadataErrors = 0;
|
|
1862
1894
|
const skippedFiles = [];
|
|
1895
|
+
const metadataErrorFiles = [];
|
|
1863
1896
|
const seenPaths = new Set();
|
|
1864
1897
|
// Literal paths of every file in this scan. Passed to the legacy-path
|
|
1865
1898
|
// migration so it never adopts a row that still belongs to a live file.
|
|
@@ -1896,8 +1929,12 @@ async function indexFiles(pwd, globPattern = DEFAULT_GLOB, collectionName, suppr
|
|
|
1896
1929
|
const title = extractTitle(content, relativeFile);
|
|
1897
1930
|
// Check if document exists (also migrates legacy lowercase paths)
|
|
1898
1931
|
const existing = findOrMigrateLegacyDocument(db, collectionName, path, livePaths);
|
|
1932
|
+
let documentId;
|
|
1933
|
+
let contentChanged = true;
|
|
1899
1934
|
if (existing) {
|
|
1935
|
+
documentId = existing.id;
|
|
1900
1936
|
if (existing.hash === hash) {
|
|
1937
|
+
contentChanged = false;
|
|
1901
1938
|
// Hash unchanged, but check if title needs updating
|
|
1902
1939
|
if (existing.title !== title) {
|
|
1903
1940
|
updateDocumentTitle(db, existing.id, title, now);
|
|
@@ -1918,7 +1955,13 @@ async function indexFiles(pwd, globPattern = DEFAULT_GLOB, collectionName, suppr
|
|
|
1918
1955
|
// New document - insert content and document
|
|
1919
1956
|
indexed++;
|
|
1920
1957
|
const stat = statSync(filepath);
|
|
1921
|
-
insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1958
|
+
documentId = insertDocumentWithContent(db, hash, content, now, collectionName, path, title, stat ? new Date(stat.birthtime).toISOString() : now, stat ? new Date(stat.mtime).toISOString() : now);
|
|
1959
|
+
}
|
|
1960
|
+
// Unchanged content still backfills missing or stale extraction state.
|
|
1961
|
+
const extraction = syncDocumentMetadata(db, documentId, content, path, contentChanged ? undefined : { onlyIfStale: true });
|
|
1962
|
+
if (extraction?.error) {
|
|
1963
|
+
metadataErrors++;
|
|
1964
|
+
metadataErrorFiles.push({ file: relativeFile, error: extraction.error });
|
|
1922
1965
|
}
|
|
1923
1966
|
processed++;
|
|
1924
1967
|
progress.set((processed / total) * 100);
|
|
@@ -1945,6 +1988,7 @@ async function indexFiles(pwd, globPattern = DEFAULT_GLOB, collectionName, suppr
|
|
|
1945
1988
|
progress.clear();
|
|
1946
1989
|
console.log(`\nIndexed: ${indexed} new, ${updated} updated, ${unchanged} unchanged, ${removed} removed`);
|
|
1947
1990
|
reportSkippedReads(skippedFiles);
|
|
1991
|
+
reportMetadataErrors(metadataErrors, metadataErrorFiles);
|
|
1948
1992
|
if (orphanedContent > 0) {
|
|
1949
1993
|
console.log(`Cleaned up ${orphanedContent} orphaned content hash(es)`);
|
|
1950
1994
|
}
|
|
@@ -1962,6 +2006,20 @@ function fsErrorCode(err) {
|
|
|
1962
2006
|
}
|
|
1963
2007
|
return "ERROR";
|
|
1964
2008
|
}
|
|
2009
|
+
function reportMetadataErrors(metadataErrors, errorFiles) {
|
|
2010
|
+
if (metadataErrors === 0)
|
|
2011
|
+
return;
|
|
2012
|
+
console.warn(`⚠ ${metadataErrors} file(s) have invalid qmd.metadata frontmatter and are excluded from filtered search`);
|
|
2013
|
+
if (errorFiles && errorFiles.length > 0) {
|
|
2014
|
+
const displayFiles = errorFiles.slice(0, 5);
|
|
2015
|
+
for (const errFile of displayFiles) {
|
|
2016
|
+
console.warn(` • ${errFile.file}: ${errFile.error}`);
|
|
2017
|
+
}
|
|
2018
|
+
if (errorFiles.length > displayFiles.length) {
|
|
2019
|
+
console.warn(` ...and ${errorFiles.length - displayFiles.length} more`);
|
|
2020
|
+
}
|
|
2021
|
+
}
|
|
2022
|
+
}
|
|
1965
2023
|
function reportSkippedReads(skippedFiles) {
|
|
1966
2024
|
if (skippedFiles.length === 0)
|
|
1967
2025
|
return;
|
|
@@ -2375,6 +2433,7 @@ function outputResults(results, query, opts) {
|
|
|
2375
2433
|
line: snippetInfo.line,
|
|
2376
2434
|
title: row.title,
|
|
2377
2435
|
...(row.context && { context: row.context }),
|
|
2436
|
+
...(row.metadata && Object.keys(row.metadata).length > 0 && { metadata: row.metadata }),
|
|
2378
2437
|
...(body && { body }),
|
|
2379
2438
|
...(snippet && { snippet }),
|
|
2380
2439
|
...(opts.explain && row.explain && { explain: row.explain }),
|
|
@@ -2624,14 +2683,46 @@ export function parseStructuredQuery(query) {
|
|
|
2624
2683
|
}
|
|
2625
2684
|
return typed.length > 0 ? { searches: typed, intent } : null;
|
|
2626
2685
|
}
|
|
2686
|
+
// Parse and validate a --filter JSON string; exits with an actionable
|
|
2687
|
+
// message on malformed JSON or an invalid filter AST.
|
|
2688
|
+
function parseCliMetadataFilter(rawFilter) {
|
|
2689
|
+
if (rawFilter === undefined)
|
|
2690
|
+
return undefined;
|
|
2691
|
+
let filterJson;
|
|
2692
|
+
try {
|
|
2693
|
+
filterJson = JSON.parse(String(rawFilter));
|
|
2694
|
+
}
|
|
2695
|
+
catch (err) {
|
|
2696
|
+
console.error(`Invalid --filter JSON: ${err instanceof Error ? err.message : String(err)}`);
|
|
2697
|
+
console.error(`Example: --filter '{"key":"status","operator":"eq","value":"published"}'`);
|
|
2698
|
+
process.exit(1);
|
|
2699
|
+
}
|
|
2700
|
+
try {
|
|
2701
|
+
return parseMetadataFilter(filterJson);
|
|
2702
|
+
}
|
|
2703
|
+
catch (err) {
|
|
2704
|
+
console.error(err instanceof Error ? err.message : String(err));
|
|
2705
|
+
process.exit(1);
|
|
2706
|
+
}
|
|
2707
|
+
}
|
|
2708
|
+
// Filtered search excludes documents without current metadata extraction;
|
|
2709
|
+
// tell the user when that makes results incomplete.
|
|
2710
|
+
function warnPendingMetadata(db) {
|
|
2711
|
+
const pendingMetadata = countDocumentsPendingMetadata(db);
|
|
2712
|
+
if (pendingMetadata === 0)
|
|
2713
|
+
return;
|
|
2714
|
+
process.stderr.write(`${c.yellow}Warning: ${pendingMetadata} document(s) lack current metadata extraction and are excluded from filtered results. Run 'qmd update'.${c.reset}\n`);
|
|
2715
|
+
}
|
|
2627
2716
|
function search(query, opts) {
|
|
2628
2717
|
const db = getDb();
|
|
2629
2718
|
// Validate collection filter (supports multiple -c flags)
|
|
2630
2719
|
// Use default collections if none specified
|
|
2631
2720
|
const collectionNames = resolveCollectionFilter(opts.collection, true);
|
|
2721
|
+
if (opts.filter)
|
|
2722
|
+
warnPendingMetadata(db);
|
|
2632
2723
|
// Use large limit for --all, otherwise fetch more than needed and let outputResults filter
|
|
2633
2724
|
const fetchLimit = opts.all ? 100000 : Math.max(50, opts.limit * 2);
|
|
2634
|
-
const results = searchFTS(db, query, fetchLimit, collectionNames);
|
|
2725
|
+
const results = searchFTS(db, query, fetchLimit, collectionSearchFilter(collectionNames), opts.filter);
|
|
2635
2726
|
// Add context to results
|
|
2636
2727
|
const resultsWithContext = results.map(r => ({
|
|
2637
2728
|
file: r.filepath,
|
|
@@ -2642,6 +2733,7 @@ function search(query, opts) {
|
|
|
2642
2733
|
context: getContextForFile(db, r.filepath),
|
|
2643
2734
|
hash: r.hash,
|
|
2644
2735
|
docid: r.docid,
|
|
2736
|
+
metadata: r.metadata,
|
|
2645
2737
|
}));
|
|
2646
2738
|
closeDb();
|
|
2647
2739
|
if (resultsWithContext.length === 0) {
|
|
@@ -2672,9 +2764,12 @@ async function vectorSearch(query, opts, _model = DEFAULT_EMBED_MODEL) {
|
|
|
2672
2764
|
// Use default collections if none specified
|
|
2673
2765
|
const collectionNames = resolveCollectionFilter(opts.collection, true);
|
|
2674
2766
|
checkIndexHealth(store.db);
|
|
2767
|
+
if (opts.filter)
|
|
2768
|
+
warnPendingMetadata(store.db);
|
|
2675
2769
|
await withLLMSession(async () => {
|
|
2676
2770
|
const results = await vectorSearchQuery(store, query, {
|
|
2677
|
-
collection: collectionNames,
|
|
2771
|
+
collection: collectionSearchFilter(collectionNames),
|
|
2772
|
+
filter: opts.filter,
|
|
2678
2773
|
limit: opts.all ? 500 : (opts.limit || 10),
|
|
2679
2774
|
minScore: opts.minScore || 0.3,
|
|
2680
2775
|
expansionContext: opts.intent,
|
|
@@ -2699,6 +2794,7 @@ async function vectorSearch(query, opts, _model = DEFAULT_EMBED_MODEL) {
|
|
|
2699
2794
|
score: r.score,
|
|
2700
2795
|
context: r.context,
|
|
2701
2796
|
docid: r.docid,
|
|
2797
|
+
metadata: r.metadata,
|
|
2702
2798
|
})), query, { ...opts, limit: results.length });
|
|
2703
2799
|
}, { maxDuration: 10 * 60 * 1000, name: 'vectorSearch' });
|
|
2704
2800
|
}
|
|
@@ -2708,6 +2804,8 @@ async function querySearch(query, opts, _embedModel = DEFAULT_EMBED_MODEL, _rera
|
|
|
2708
2804
|
// Use default collections if none specified
|
|
2709
2805
|
const collectionNames = resolveCollectionFilter(opts.collection, true);
|
|
2710
2806
|
checkIndexHealth(store.db);
|
|
2807
|
+
if (opts.filter)
|
|
2808
|
+
warnPendingMetadata(store.db);
|
|
2711
2809
|
// Check for structured query syntax (lex:/vec:/hyde:/intent: prefixes)
|
|
2712
2810
|
const parsed = parseStructuredQuery(query);
|
|
2713
2811
|
// Intent can come from --intent flag or from intent: line in query document
|
|
@@ -2735,6 +2833,7 @@ async function querySearch(query, opts, _embedModel = DEFAULT_EMBED_MODEL, _rera
|
|
|
2735
2833
|
process.stderr.write(`${c.dim}└─ Searching...${c.reset}\n`);
|
|
2736
2834
|
results = await structuredSearch(store, structuredQueries, {
|
|
2737
2835
|
collections: collectionNames.length > 0 ? collectionNames : undefined,
|
|
2836
|
+
filter: opts.filter,
|
|
2738
2837
|
limit: opts.all ? 500 : (opts.limit || 10),
|
|
2739
2838
|
minScore: opts.minScore || 0,
|
|
2740
2839
|
candidateLimit: opts.candidateLimit,
|
|
@@ -2764,6 +2863,8 @@ async function querySearch(query, opts, _embedModel = DEFAULT_EMBED_MODEL, _rera
|
|
|
2764
2863
|
// Standard hybrid query with automatic expansion
|
|
2765
2864
|
results = await hybridQuery(store, query, {
|
|
2766
2865
|
collections: collectionNames.length > 0 ? collectionNames : undefined,
|
|
2866
|
+
collection: collectionSearchFilter(collectionNames),
|
|
2867
|
+
filter: opts.filter,
|
|
2767
2868
|
limit: opts.all ? 500 : (opts.limit || 10),
|
|
2768
2869
|
minScore: opts.minScore || 0,
|
|
2769
2870
|
candidateLimit: opts.candidateLimit,
|
|
@@ -2830,6 +2931,7 @@ async function querySearch(query, opts, _embedModel = DEFAULT_EMBED_MODEL, _rera
|
|
|
2830
2931
|
score: r.score,
|
|
2831
2932
|
context: r.context,
|
|
2832
2933
|
docid: r.docid,
|
|
2934
|
+
metadata: r.metadata,
|
|
2833
2935
|
explain: r.explain,
|
|
2834
2936
|
})), displayQuery, { ...opts, limit: results.length });
|
|
2835
2937
|
}, { maxDuration: 10 * 60 * 1000, name: 'querySearch' });
|
|
@@ -2866,6 +2968,7 @@ function parseCLI() {
|
|
|
2866
2968
|
json: { type: "boolean" },
|
|
2867
2969
|
explain: { type: "boolean" },
|
|
2868
2970
|
collection: { type: "string", short: "c", multiple: true }, // Filter by collection(s)
|
|
2971
|
+
filter: { type: "string" }, // Metadata filter (JSON AST) for search/vsearch/query
|
|
2869
2972
|
// Collection options
|
|
2870
2973
|
name: { type: "string" }, // collection name
|
|
2871
2974
|
mask: { type: "string" }, // glob pattern
|
|
@@ -3462,6 +3565,8 @@ function showHelp() {
|
|
|
3462
3565
|
console.log(" --explain - Include retrieval score traces (query, CLI/--format json)");
|
|
3463
3566
|
console.log(" --format <kind> - Output format: cli (default) | json | csv | md | xml | files");
|
|
3464
3567
|
console.log(" -c, --collection <name> - Filter by one or more collections");
|
|
3568
|
+
console.log(" --filter <json> - Metadata filter (recursive JSON AST; search/vsearch/query)");
|
|
3569
|
+
console.log(" e.g. '{\"key\":\"status\",\"operator\":\"eq\",\"value\":\"published\"}'");
|
|
3465
3570
|
console.log("");
|
|
3466
3571
|
console.log("Embed/query options:");
|
|
3467
3572
|
console.log(" --chunk-strategy <auto|regex> - Chunking mode (default: regex; auto uses AST for code files)");
|
|
@@ -4030,6 +4135,12 @@ async function showDoctor() {
|
|
|
4030
4135
|
const rerankModel = configModels.rerank_api_model ?? activeModels.rerank;
|
|
4031
4136
|
doctorCheck("reranking model", true, `${rerankModel} (endpoint: ${rerankEndpoint})`);
|
|
4032
4137
|
}
|
|
4138
|
+
const isJevConfigured = Boolean(configModels?.jev_api_key || process.env.TYPESAFE_API_KEY);
|
|
4139
|
+
if (isJevConfigured) {
|
|
4140
|
+
const jevModel = configModels?.jev_api_model ?? process.env.TYPESAFE_DEFAULT_MODEL ?? "jev-1.13";
|
|
4141
|
+
const jevEndpoint = configModels?.jev_base_url ?? process.env.TYPESAFE_BASE_URL ?? "https://api.typesafe.ai";
|
|
4142
|
+
doctorCheck("typesafe jev", true, `${jevModel} (endpoint: ${jevEndpoint})`);
|
|
4143
|
+
}
|
|
4033
4144
|
await runDoctorDeviceChecks(nextSteps);
|
|
4034
4145
|
const diagnostics = inspectIndexDiagnostics(db, {
|
|
4035
4146
|
fallbackModel: embedModel,
|
|
@@ -4070,6 +4181,22 @@ async function showDoctor() {
|
|
|
4070
4181
|
catch (error) {
|
|
4071
4182
|
doctorCheck("embedding freshness", false, error instanceof Error ? error.message : String(error));
|
|
4072
4183
|
}
|
|
4184
|
+
try {
|
|
4185
|
+
const pendingMetadata = countDocumentsPendingMetadata(db);
|
|
4186
|
+
if (pendingMetadata === 0) {
|
|
4187
|
+
doctorCheck("metadata extraction", true, "all active documents have current metadata");
|
|
4188
|
+
}
|
|
4189
|
+
else {
|
|
4190
|
+
const pendingDocs = getDocumentsPendingMetadata(db, 3);
|
|
4191
|
+
const fileHints = pendingDocs.map(d => `${d.collection}/${d.path}`).join(", ");
|
|
4192
|
+
const extraHint = pendingMetadata > pendingDocs.length ? ` and ${pendingMetadata - pendingDocs.length} more` : "";
|
|
4193
|
+
doctorCheck("metadata extraction", false, `${formatCount(pendingMetadata)} active ${pendingMetadata === 1 ? "document lacks" : "documents lack"} valid metadata (${fileHints}${extraHint}). Next: \`qmd update\``);
|
|
4194
|
+
nextSteps.push(`Inspect and fix frontmatter in ${formatCount(pendingMetadata)} documents, then run \`qmd update\`.`);
|
|
4195
|
+
}
|
|
4196
|
+
}
|
|
4197
|
+
catch (error) {
|
|
4198
|
+
doctorCheck("metadata extraction", false, error instanceof Error ? error.message : String(error));
|
|
4199
|
+
}
|
|
4073
4200
|
try {
|
|
4074
4201
|
const rows = db.prepare(`
|
|
4075
4202
|
SELECT model, embed_fingerprint AS fingerprint, COUNT(DISTINCT hash) AS docs, COUNT(*) AS chunks
|
|
@@ -4532,6 +4659,7 @@ if (isMain) {
|
|
|
4532
4659
|
console.error("Usage: qmd search [options] <query>");
|
|
4533
4660
|
process.exit(1);
|
|
4534
4661
|
}
|
|
4662
|
+
cli.opts.filter = parseCliMetadataFilter(cli.values.filter);
|
|
4535
4663
|
search(cli.query, cli.opts);
|
|
4536
4664
|
break;
|
|
4537
4665
|
case "vsearch":
|
|
@@ -4544,6 +4672,7 @@ if (isMain) {
|
|
|
4544
4672
|
if (!cli.values["min-score"]) {
|
|
4545
4673
|
cli.opts.minScore = 0.3;
|
|
4546
4674
|
}
|
|
4675
|
+
cli.opts.filter = parseCliMetadataFilter(cli.values.filter);
|
|
4547
4676
|
await resolveLocalConfigTrust();
|
|
4548
4677
|
await vectorSearch(cli.query, cli.opts);
|
|
4549
4678
|
break;
|
|
@@ -4553,6 +4682,7 @@ if (isMain) {
|
|
|
4553
4682
|
console.error("Usage: qmd query [options] <query>");
|
|
4554
4683
|
process.exit(1);
|
|
4555
4684
|
}
|
|
4685
|
+
cli.opts.filter = parseCliMetadataFilter(cli.values.filter);
|
|
4556
4686
|
await resolveLocalConfigTrust();
|
|
4557
4687
|
await querySearch(cli.query, cli.opts);
|
|
4558
4688
|
break;
|
package/dist/collections.d.ts
CHANGED
package/dist/collections.js
CHANGED
|
@@ -41,11 +41,16 @@ export function setConfigSource(source) {
|
|
|
41
41
|
* Config file will be ~/.config/qmd/{indexName}.yml
|
|
42
42
|
*/
|
|
43
43
|
export function setConfigIndexName(name) {
|
|
44
|
-
// Resolve relative paths to absolute paths and sanitize for use as filename
|
|
45
|
-
|
|
46
|
-
|
|
44
|
+
// Resolve relative paths to absolute paths and sanitize for use as filename.
|
|
45
|
+
// Windows absolute paths (C:\..., \\server\share) are already absolute --
|
|
46
|
+
// skip resolve() for those (POSIX path.resolve doesn't recognize a drive
|
|
47
|
+
// letter as absolute and would wrongly prefix it with cwd) and sanitize
|
|
48
|
+
// `:` and `\` alongside `/` so the derived filename is valid on Windows.
|
|
49
|
+
const isWindowsAbsolute = /^[a-zA-Z]:[\\/]/.test(name) || /^\\/.test(name);
|
|
50
|
+
if (isWindowsAbsolute || /[\\/]/.test(name)) {
|
|
51
|
+
const absolutePath = isWindowsAbsolute ? name : resolve(process.cwd(), name);
|
|
47
52
|
// Replace path separators with underscores to create a valid filename
|
|
48
|
-
currentIndexName = absolutePath.replace(
|
|
53
|
+
currentIndexName = absolutePath.replace(/[:\\/]+/g, '_').replace(/^_+/, '');
|
|
49
54
|
}
|
|
50
55
|
else {
|
|
51
56
|
currentIndexName = name;
|
|
@@ -1,9 +1,14 @@
|
|
|
1
|
-
import type { LLM, EmbedOptions, EmbeddingResult, GenerateOptions, GenerateResult, ModelInfo, Queryable, RerankDocument, RerankOptions, RerankResult } from "./llm.js";
|
|
1
|
+
import type { LLM, EmbedOptions, EmbeddingResult, GenerateOptions, GenerateResult, ModelInfo, Queryable, RerankDocument, RerankOptions, RerankResult, SearchIntentGuidance } from "./llm.js";
|
|
2
2
|
import type { RemoteLLM } from "./remote-llm.js";
|
|
3
|
-
|
|
3
|
+
import type { RemoteJev } from "./remote-jev.js";
|
|
4
|
+
export declare class Hybrid implements LLM {
|
|
4
5
|
private readonly localLLM;
|
|
5
6
|
private readonly remoteLLM?;
|
|
6
|
-
|
|
7
|
+
private readonly remoteJev?;
|
|
8
|
+
constructor(localLLM: LLM, remoteLLM?: RemoteLLM | undefined, remoteJev?: RemoteJev | undefined);
|
|
9
|
+
get jev(): RemoteJev | undefined;
|
|
10
|
+
get remote(): RemoteLLM | undefined;
|
|
11
|
+
get local(): LLM;
|
|
7
12
|
get supportsExpand(): boolean;
|
|
8
13
|
get supportsRerank(): boolean;
|
|
9
14
|
embed(text: string, options?: EmbedOptions): Promise<EmbeddingResult | null>;
|
|
@@ -13,6 +18,7 @@ export declare class HybridLLM implements LLM {
|
|
|
13
18
|
context?: string;
|
|
14
19
|
includeLexical?: boolean;
|
|
15
20
|
includeHyde?: boolean;
|
|
21
|
+
searchIntent?: SearchIntentGuidance;
|
|
16
22
|
}): Promise<Queryable[]>;
|
|
17
23
|
rerank(query: string, documents: RerankDocument[], options?: RerankOptions): Promise<RerankResult>;
|
|
18
24
|
dispose(): Promise<void>;
|