@wei840222/qmd 2026.9.6 → 2026.9.25

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -15,10 +15,24 @@ import { createMcpHandler, McpServer, ResourceTemplate } from "@modelcontextprot
15
15
  import { serveStdio } from "@modelcontextprotocol/server/stdio";
16
16
  import { z } from "zod";
17
17
  import { existsSync } from "fs";
18
- import { createStore, extractSnippet, addLineNumbers, getDefaultDbPath, DEFAULT_MULTI_GET_MAX_BYTES, } from "../index.js";
18
+ import { createStore, extractSnippet, addLineNumbers, getDefaultDbPath, DEFAULT_MULTI_GET_MAX_BYTES, parseMetadataFilter, } from "../index.js";
19
19
  import { getConfigPath } from "../collections.js";
20
20
  import { enableProductionMode } from "../store.js";
21
21
  import { checkRequestOrigin, resolveOriginGuard } from "./origin-guard.js";
22
+ /**
23
+ * Validate an untrusted `filter` argument through the shared runtime
24
+ * validator. Returns the parse error message when invalid.
25
+ */
26
+ function validateFilterArgument(filter) {
27
+ if (filter === undefined)
28
+ return {};
29
+ try {
30
+ return { filter: parseMetadataFilter(filter) };
31
+ }
32
+ catch (err) {
33
+ return { error: err instanceof Error ? err.message : String(err) };
34
+ }
35
+ }
22
36
  // =============================================================================
23
37
  // Helper functions
24
38
  // =============================================================================
@@ -251,14 +265,21 @@ Context-aware lex (C++ performance, not sports):
251
265
  minScore: z.number().optional().default(0).describe("Min relevance 0-1 (default: 0)"),
252
266
  candidateLimit: z.number().optional().describe("Maximum candidates to rerank (default: 40, lower = faster but may miss results)"),
253
267
  collections: z.array(z.string()).optional().describe("Filter to collections (OR match)"),
268
+ filter: z.record(z.string(), z.unknown()).optional().describe("Metadata filter (recursive JSON AST). Every returned result satisfies it. " +
269
+ "Nodes are operator-discriminated: logical groups {operator:'and'|'or', operands:[...]}, " +
270
+ "negation {operator:'not', operand:{...}}, and conditions {key, operator, value} with " +
271
+ "operators eq/ne/gt/gte/lt/lte (comparison), in/nin/all (membership), exists (presence). " +
272
+ "Values are typed exactly (no coercion); missing keys do not match ne/nin. " +
273
+ "Example: {\"operator\":\"and\",\"operands\":[{\"key\":\"topics\",\"operator\":\"all\",\"value\":[\"typescript\"]}," +
274
+ "{\"key\":\"status\",\"operator\":\"ne\",\"value\":\"draft\"}]}"),
254
275
  expansionContext: z.string().optional().describe("Additional context used only to generate lex, vec, and hyde query expansions."),
255
276
  rerankContext: z.string().optional().describe("Additional context used only to rerank results and select snippets/chunks."),
277
+ intent: z.string().optional().describe("Background context to disambiguate the query. Example: query='performance', intent='web page load times and Core Web Vitals'. Does not search on its own."),
256
278
  rerank: z.boolean().optional().default(true).describe("Rerank results using LLM (default: true). Set to false for faster results on CPU-only machines."),
257
279
  explain: z.boolean().optional().default(false).describe("Include retrieval traces and the shared query-expansion decision or typed expansion error"),
258
280
  includeHyde: z.boolean().optional().default(true).describe("Whether to include HyDE (hypothetical document) in query expansion (default: true)"),
259
281
  }),
260
- }, track(async ({ query, searches, expansion, includeHyde, limit, minScore, candidateLimit, collections, expansionContext, rerankContext, rerank, explain }) => {
261
- // Require exactly one of `query` (plain text with an expansion policy) or `searches` (typed sub-queries).
282
+ }, track(async ({ query, searches, expansion, includeHyde, limit, minScore, candidateLimit, collections, filter, expansionContext, rerankContext, intent, rerank, explain }) => {
262
283
  if (!query && (!searches || searches.length === 0)) {
263
284
  return {
264
285
  content: [{ type: "text", text: "Error: provide either 'query' (plain text) or 'searches' (typed sub-queries)" }],
@@ -271,6 +292,13 @@ Context-aware lex (C++ performance, not sports):
271
292
  isError: true,
272
293
  };
273
294
  }
295
+ const filterValidation = validateFilterArgument(filter);
296
+ if (filterValidation.error) {
297
+ return {
298
+ content: [{ type: "text", text: `Error: ${filterValidation.error}` }],
299
+ isError: true,
300
+ };
301
+ }
274
302
  // Use default collections if none specified
275
303
  const effectiveCollections = collections ?? defaultCollectionNames;
276
304
  // Plain `query` follows the requested SDK expansion policy before fusion and reranking;
@@ -278,6 +306,8 @@ Context-aware lex (C++ performance, not sports):
278
306
  const searchOptions = query
279
307
  ? { query }
280
308
  : { queries: (searches ?? []).map(s => ({ type: s.type, query: s.query })) };
309
+ const effectiveExpansionContext = expansionContext ?? intent;
310
+ const effectiveRerankContext = rerankContext ?? intent;
281
311
  let expansionDecision;
282
312
  let expansionError;
283
313
  let results;
@@ -285,12 +315,13 @@ Context-aware lex (C++ performance, not sports):
285
315
  results = await store.search({
286
316
  ...searchOptions,
287
317
  collections: effectiveCollections.length > 0 ? effectiveCollections : undefined,
318
+ filter: filterValidation.filter,
288
319
  limit,
289
320
  minScore,
290
321
  candidateLimit,
291
322
  rerank,
292
- expansionContext,
293
- rerankContext,
323
+ expansionContext: effectiveExpansionContext,
324
+ rerankContext: effectiveRerankContext,
294
325
  explain,
295
326
  expansion: query ? expansion : undefined,
296
327
  includeHyde,
@@ -323,6 +354,7 @@ Context-aware lex (C++ performance, not sports):
323
354
  title: r.title,
324
355
  score: Math.round(r.score * 100) / 100,
325
356
  context: r.context,
357
+ ...(Object.keys(r.metadata).length > 0 ? { metadata: r.metadata } : {}),
326
358
  line,
327
359
  snippet: addLineNumbers(snippet, line),
328
360
  ...(explain && r.explain ? { explain: r.explain } : {}),
@@ -798,7 +830,21 @@ export async function startMcpHttpServer(port, options = {}) {
798
830
  // REST endpoint: POST /query (alias: /search) — structured search without MCP protocol
799
831
  if ((pathname === "/query" || pathname === "/search") && nodeReq.method === "POST") {
800
832
  const rawBody = await collectBody(nodeReq);
801
- const params = JSON.parse(rawBody);
833
+ let parsedParams;
834
+ try {
835
+ parsedParams = JSON.parse(rawBody);
836
+ }
837
+ catch {
838
+ nodeRes.writeHead(400, { "Content-Type": "application/json" });
839
+ nodeRes.end(JSON.stringify({ error: "Invalid JSON body" }));
840
+ return;
841
+ }
842
+ if (typeof parsedParams !== "object" || parsedParams === null || Array.isArray(parsedParams)) {
843
+ nodeRes.writeHead(400, { "Content-Type": "application/json" });
844
+ nodeRes.end(JSON.stringify({ error: "JSON body must be an object" }));
845
+ return;
846
+ }
847
+ const params = parsedParams;
802
848
  // Validate required fields
803
849
  if (!params.searches || !Array.isArray(params.searches)) {
804
850
  nodeRes.writeHead(400, { "Content-Type": "application/json" });
@@ -811,11 +857,28 @@ export async function startMcpHttpServer(port, options = {}) {
811
857
  type: s.type,
812
858
  query: String(s.query || ""),
813
859
  }));
860
+ // Optional metadata filter — must be an object and a valid filter AST
861
+ let restFilter;
862
+ if (params.filter !== undefined) {
863
+ if (typeof params.filter !== "object" || params.filter === null || Array.isArray(params.filter)) {
864
+ nodeRes.writeHead(400, { "Content-Type": "application/json" });
865
+ nodeRes.end(JSON.stringify({ error: "Invalid field: filter (must be an object)" }));
866
+ return;
867
+ }
868
+ const filterValidation = validateFilterArgument(params.filter);
869
+ if (filterValidation.error) {
870
+ nodeRes.writeHead(400, { "Content-Type": "application/json" });
871
+ nodeRes.end(JSON.stringify({ error: filterValidation.error }));
872
+ return;
873
+ }
874
+ restFilter = filterValidation.filter;
875
+ }
814
876
  // Use default collections if none specified
815
877
  const effectiveCollections = Array.isArray(params.collections) ? params.collections.map(String) : defaultCollectionNames;
816
878
  const results = await store.search({
817
879
  queries,
818
880
  collections: effectiveCollections.length > 0 ? effectiveCollections : undefined,
881
+ filter: restFilter,
819
882
  limit: typeof params.limit === "number" ? params.limit : 10,
820
883
  minScore: typeof params.minScore === "number" ? params.minScore : 0,
821
884
  candidateLimit: typeof params.candidateLimit === "number" ? params.candidateLimit : undefined,
@@ -835,6 +898,7 @@ export async function startMcpHttpServer(port, options = {}) {
835
898
  title: r.title,
836
899
  score: Math.round(r.score * 100) / 100,
837
900
  context: r.context,
901
+ ...(Object.keys(r.metadata).length > 0 ? { metadata: r.metadata } : {}),
838
902
  line,
839
903
  snippet: addLineNumbers(snippet, line),
840
904
  };
@@ -0,0 +1,74 @@
1
+ /**
2
+ * QMD Metadata Filter - Recursive filter AST, strict runtime validation, and
3
+ * parameterized SQL compilation.
4
+ *
5
+ * The filter has one canonical, `operator`-discriminated recursive shape shared
6
+ * by every public search surface (CLI, SDK, MCP, HTTP):
7
+ *
8
+ * { "operator": "and", "operands": [ ... ] }
9
+ * { "operator": "not", "operand": { ... } }
10
+ * { "key": "status", "operator": "eq", "value": "published" }
11
+ *
12
+ * Compilation emits correlated EXISTS/NOT EXISTS subqueries over
13
+ * `document_metadata_values` with every user value bound as a parameter —
14
+ * metadata keys and values are data, never SQL.
15
+ */
16
+ import type { MetadataScalar, MetadataScalarArray } from "./metadata.js";
17
+ export type MetadataFilter = MetadataFilterGroup | MetadataFilterNegation | MetadataCondition;
18
+ export interface MetadataFilterGroup {
19
+ operator: "and" | "or";
20
+ operands: readonly MetadataFilter[];
21
+ }
22
+ export interface MetadataFilterNegation {
23
+ operator: "not";
24
+ operand: MetadataFilter;
25
+ }
26
+ export type MetadataCondition = {
27
+ key: string;
28
+ operator: "eq" | "ne";
29
+ value: MetadataScalar;
30
+ } | {
31
+ key: string;
32
+ operator: "gt" | "gte" | "lt" | "lte";
33
+ value: string | number;
34
+ } | {
35
+ key: string;
36
+ operator: "in" | "nin" | "all";
37
+ value: MetadataScalarArray;
38
+ } | {
39
+ key: string;
40
+ operator: "exists";
41
+ value: boolean;
42
+ };
43
+ export interface CompiledMetadataFilter {
44
+ sql: string;
45
+ params: (string | number)[];
46
+ }
47
+ /** Raised by parseMetadataFilter with the JSON path of the failing node. */
48
+ export declare class MetadataFilterError extends Error {
49
+ readonly path: string;
50
+ constructor(path: string, message: string);
51
+ }
52
+ /** Defensive limits for recursive filters from untrusted callers. */
53
+ export declare const METADATA_FILTER_LIMITS: {
54
+ readonly maxDepth: 16;
55
+ readonly maxNodes: 256;
56
+ readonly maxGroupOperands: 32;
57
+ readonly maxMembershipValues: 64;
58
+ readonly maxKeyBytes: 128;
59
+ readonly maxStringLength: 1024;
60
+ };
61
+ /**
62
+ * Strictly validate an untrusted value as a MetadataFilter.
63
+ * Rejects unknown operators, unknown properties, operator-incompatible values,
64
+ * and inputs exceeding METADATA_FILTER_LIMITS. Canonicalizes membership value
65
+ * arrays by de-duplicating while preserving order.
66
+ */
67
+ export declare function parseMetadataFilter(input: unknown): MetadataFilter;
68
+ /**
69
+ * Compile a validated filter into one parameterized SQL predicate correlated
70
+ * against a documents-table alias (e.g. `d`). All keys and values are bound
71
+ * parameters. The caller is responsible for restricting the surrounding query
72
+ * to active documents with current, error-free metadata extraction.
73
+ */
74
+ export declare function compileMetadataFilter(filter: MetadataFilter, documentsAlias: string): CompiledMetadataFilter;
@@ -0,0 +1,279 @@
1
+ /**
2
+ * QMD Metadata Filter - Recursive filter AST, strict runtime validation, and
3
+ * parameterized SQL compilation.
4
+ *
5
+ * The filter has one canonical, `operator`-discriminated recursive shape shared
6
+ * by every public search surface (CLI, SDK, MCP, HTTP):
7
+ *
8
+ * { "operator": "and", "operands": [ ... ] }
9
+ * { "operator": "not", "operand": { ... } }
10
+ * { "key": "status", "operator": "eq", "value": "published" }
11
+ *
12
+ * Compilation emits correlated EXISTS/NOT EXISTS subqueries over
13
+ * `document_metadata_values` with every user value bound as a parameter —
14
+ * metadata keys and values are data, never SQL.
15
+ */
16
+ import { METADATA_LIMITS } from "./metadata.js";
17
+ /** Raised by parseMetadataFilter with the JSON path of the failing node. */
18
+ export class MetadataFilterError extends Error {
19
+ path;
20
+ constructor(path, message) {
21
+ super(`Invalid metadata filter at ${path}: ${message}`);
22
+ this.name = "MetadataFilterError";
23
+ this.path = path;
24
+ }
25
+ }
26
+ // =============================================================================
27
+ // Limits
28
+ // =============================================================================
29
+ /** Defensive limits for recursive filters from untrusted callers. */
30
+ export const METADATA_FILTER_LIMITS = {
31
+ maxDepth: 16,
32
+ maxNodes: 256,
33
+ maxGroupOperands: 32,
34
+ maxMembershipValues: 64,
35
+ maxKeyBytes: METADATA_LIMITS.maxKeyBytes,
36
+ maxStringLength: METADATA_LIMITS.maxStringLength,
37
+ };
38
+ const GROUP_OPERATORS = new Set(["and", "or"]);
39
+ const COMPARISON_OPERATORS = new Set(["eq", "ne", "gt", "gte", "lt", "lte"]);
40
+ const ORDERED_OPERATORS = new Set(["gt", "gte", "lt", "lte"]);
41
+ const MEMBERSHIP_OPERATORS = new Set(["in", "nin", "all"]);
42
+ const CONDITION_OPERATORS = new Set([...COMPARISON_OPERATORS, ...MEMBERSHIP_OPERATORS, "exists"]);
43
+ const ALL_OPERATORS = [...GROUP_OPERATORS, "not", ...CONDITION_OPERATORS];
44
+ /**
45
+ * Strictly validate an untrusted value as a MetadataFilter.
46
+ * Rejects unknown operators, unknown properties, operator-incompatible values,
47
+ * and inputs exceeding METADATA_FILTER_LIMITS. Canonicalizes membership value
48
+ * arrays by de-duplicating while preserving order.
49
+ */
50
+ export function parseMetadataFilter(input) {
51
+ const state = { nodes: 0 };
52
+ return parseFilterNode(input, "$", 1, state);
53
+ }
54
+ function parseFilterNode(input, path, depth, state) {
55
+ if (depth > METADATA_FILTER_LIMITS.maxDepth) {
56
+ throw new MetadataFilterError(path, `exceeds maximum nesting depth of ${METADATA_FILTER_LIMITS.maxDepth}`);
57
+ }
58
+ state.nodes += 1;
59
+ if (state.nodes > METADATA_FILTER_LIMITS.maxNodes) {
60
+ throw new MetadataFilterError(path, `exceeds maximum of ${METADATA_FILTER_LIMITS.maxNodes} nodes`);
61
+ }
62
+ if (typeof input !== "object" || input === null || Array.isArray(input)) {
63
+ throw new MetadataFilterError(path, "each filter node must be an object");
64
+ }
65
+ const node = input;
66
+ const operator = node["operator"];
67
+ if (typeof operator !== "string") {
68
+ throw new MetadataFilterError(path, "missing 'operator' property");
69
+ }
70
+ if (GROUP_OPERATORS.has(operator)) {
71
+ return parseFilterGroup(node, operator, path, depth, state);
72
+ }
73
+ if (operator === "not") {
74
+ return parseFilterNegation(node, path, depth, state);
75
+ }
76
+ if (CONDITION_OPERATORS.has(operator)) {
77
+ return parseFilterCondition(node, operator, path);
78
+ }
79
+ throw new MetadataFilterError(path, `unknown operator '${operator}' — expected one of: ${ALL_OPERATORS.join(", ")}`);
80
+ }
81
+ function parseFilterGroup(node, operator, path, depth, state) {
82
+ rejectUnknownProperties(node, ["operator", "operands"], path);
83
+ const operands = node["operands"];
84
+ if (!Array.isArray(operands)) {
85
+ throw new MetadataFilterError(path, `'${operator}' requires an 'operands' array`);
86
+ }
87
+ if (operands.length === 0) {
88
+ throw new MetadataFilterError(path, `'${operator}' requires a non-empty 'operands' array`);
89
+ }
90
+ if (operands.length > METADATA_FILTER_LIMITS.maxGroupOperands) {
91
+ throw new MetadataFilterError(path, `'${operator}' exceeds maximum of ${METADATA_FILTER_LIMITS.maxGroupOperands} operands`);
92
+ }
93
+ return {
94
+ operator,
95
+ operands: operands.map((operand, index) => parseFilterNode(operand, `${path}.operands[${index}]`, depth + 1, state)),
96
+ };
97
+ }
98
+ function parseFilterNegation(node, path, depth, state) {
99
+ rejectUnknownProperties(node, ["operator", "operand"], path);
100
+ if (!("operand" in node)) {
101
+ throw new MetadataFilterError(path, "'not' requires exactly one 'operand'");
102
+ }
103
+ return {
104
+ operator: "not",
105
+ operand: parseFilterNode(node["operand"], `${path}.operand`, depth + 1, state),
106
+ };
107
+ }
108
+ function parseFilterCondition(node, operator, path) {
109
+ rejectUnknownProperties(node, ["key", "operator", "value"], path);
110
+ const key = node["key"];
111
+ if (typeof key !== "string" || key.length === 0) {
112
+ throw new MetadataFilterError(path, `'${operator}' requires a non-empty string 'key'`);
113
+ }
114
+ if (Buffer.byteLength(key, "utf-8") > METADATA_FILTER_LIMITS.maxKeyBytes) {
115
+ throw new MetadataFilterError(path, `'key' exceeds ${METADATA_FILTER_LIMITS.maxKeyBytes} bytes`);
116
+ }
117
+ if (!("value" in node)) {
118
+ throw new MetadataFilterError(path, `'${operator}' requires a 'value'`);
119
+ }
120
+ const value = node["value"];
121
+ if (operator === "exists") {
122
+ if (typeof value !== "boolean") {
123
+ throw new MetadataFilterError(`${path}.value`, "'exists' requires a boolean value");
124
+ }
125
+ return { key, operator, value };
126
+ }
127
+ if (MEMBERSHIP_OPERATORS.has(operator)) {
128
+ return {
129
+ key,
130
+ operator: operator,
131
+ value: parseMembershipValues(value, operator, path),
132
+ };
133
+ }
134
+ // Comparison operators: eq, ne, gt, gte, lt, lte.
135
+ const scalar = parseScalarValue(value, `${path}.value`);
136
+ if (ORDERED_OPERATORS.has(operator) && typeof scalar === "boolean") {
137
+ throw new MetadataFilterError(`${path}.value`, `'${operator}' requires a string or number value`);
138
+ }
139
+ return { key, operator, value: scalar };
140
+ }
141
+ function parseMembershipValues(value, operator, path) {
142
+ if (!Array.isArray(value)) {
143
+ throw new MetadataFilterError(`${path}.value`, `'${operator}' requires an array value`);
144
+ }
145
+ if (value.length === 0) {
146
+ throw new MetadataFilterError(`${path}.value`, `'${operator}' requires a non-empty array value`);
147
+ }
148
+ if (value.length > METADATA_FILTER_LIMITS.maxMembershipValues) {
149
+ throw new MetadataFilterError(`${path}.value`, `'${operator}' exceeds maximum of ${METADATA_FILTER_LIMITS.maxMembershipValues} values`);
150
+ }
151
+ const scalars = value.map((element, index) => parseScalarValue(element, `${path}.value[${index}]`));
152
+ // Narrow each homogeneous case explicitly so the public array union remains
153
+ // precise without discarding type evidence through chained assertions.
154
+ if (scalars.every((scalar) => typeof scalar === "string")) {
155
+ return Array.from(new Set(scalars));
156
+ }
157
+ if (scalars.every((scalar) => typeof scalar === "number")) {
158
+ return Array.from(new Set(scalars));
159
+ }
160
+ if (scalars.every((scalar) => typeof scalar === "boolean")) {
161
+ return Array.from(new Set(scalars));
162
+ }
163
+ throw new MetadataFilterError(`${path}.value`, `'${operator}' requires a homogeneous array of one scalar type`);
164
+ }
165
+ function parseScalarValue(value, path) {
166
+ if (typeof value === "string") {
167
+ if (value.length > METADATA_FILTER_LIMITS.maxStringLength) {
168
+ throw new MetadataFilterError(path, `string exceeds ${METADATA_FILTER_LIMITS.maxStringLength} characters`);
169
+ }
170
+ return value;
171
+ }
172
+ if (typeof value === "number") {
173
+ if (!Number.isFinite(value)) {
174
+ throw new MetadataFilterError(path, "numbers must be finite");
175
+ }
176
+ return value;
177
+ }
178
+ if (typeof value === "boolean")
179
+ return value;
180
+ throw new MetadataFilterError(path, "expected a string, number, or boolean");
181
+ }
182
+ function rejectUnknownProperties(node, allowed, path) {
183
+ for (const property of Object.keys(node)) {
184
+ if (!allowed.includes(property)) {
185
+ throw new MetadataFilterError(path, `unknown property '${property}' — allowed: ${allowed.join(", ")}`);
186
+ }
187
+ }
188
+ }
189
+ // =============================================================================
190
+ // SQL compilation
191
+ // =============================================================================
192
+ /**
193
+ * Compile a validated filter into one parameterized SQL predicate correlated
194
+ * against a documents-table alias (e.g. `d`). All keys and values are bound
195
+ * parameters. The caller is responsible for restricting the surrounding query
196
+ * to active documents with current, error-free metadata extraction.
197
+ */
198
+ export function compileMetadataFilter(filter, documentsAlias) {
199
+ const params = [];
200
+ const sql = compileFilterNode(filter, documentsAlias, params);
201
+ return { sql, params };
202
+ }
203
+ function compileFilterNode(filter, alias, params) {
204
+ switch (filter.operator) {
205
+ case "and":
206
+ case "or": {
207
+ const joiner = filter.operator === "and" ? " AND " : " OR ";
208
+ return `(${filter.operands.map(operand => compileFilterNode(operand, alias, params)).join(joiner)})`;
209
+ }
210
+ case "not":
211
+ return `NOT ${compileFilterNode(filter.operand, alias, params)}`;
212
+ case "exists":
213
+ params.push(filter.key);
214
+ return filter.value
215
+ ? buildValueExistsSql(alias, "mv.key = ?")
216
+ : `NOT ${buildValueExistsSql(alias, "mv.key = ?")}`;
217
+ case "eq":
218
+ case "gt":
219
+ case "gte":
220
+ case "lt":
221
+ case "lte": {
222
+ const sqlOperator = { eq: "=", gt: ">", gte: ">=", lt: "<", lte: "<=" }[filter.operator];
223
+ params.push(filter.key, bindScalar(filter.value));
224
+ return buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueTypeOf(filter.value)}' AND mv.${valueColumnOf(filter.value)} ${sqlOperator} ?`);
225
+ }
226
+ case "ne": {
227
+ // Key must have at least one same-type value, and no same-type value
228
+ // may equal the operand. Missing keys and type mismatches do not match.
229
+ const valueType = valueTypeOf(filter.value);
230
+ params.push(filter.key);
231
+ const presentSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}'`);
232
+ params.push(filter.key, bindScalar(filter.value));
233
+ const equalSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${valueColumnOf(filter.value)} = ?`);
234
+ return `(${presentSql} AND NOT ${equalSql})`;
235
+ }
236
+ case "in":
237
+ case "nin": {
238
+ const valueType = valueTypeOf(filter.value[0]);
239
+ const column = valueColumnOf(filter.value[0]);
240
+ const placeholders = filter.value.map(() => "?").join(", ");
241
+ if (filter.operator === "in") {
242
+ params.push(filter.key, ...filter.value.map(bindScalar));
243
+ return buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} IN (${placeholders})`);
244
+ }
245
+ params.push(filter.key);
246
+ const presentSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}'`);
247
+ params.push(filter.key, ...filter.value.map(bindScalar));
248
+ const memberSql = buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} IN (${placeholders})`);
249
+ return `(${presentSql} AND NOT ${memberSql})`;
250
+ }
251
+ case "all": {
252
+ const valueType = valueTypeOf(filter.value[0]);
253
+ const column = valueColumnOf(filter.value[0]);
254
+ const memberSqls = filter.value.map(element => {
255
+ params.push(filter.key, bindScalar(element));
256
+ return buildValueExistsSql(alias, `mv.key = ? AND mv.value_type = '${valueType}' AND mv.${column} = ?`);
257
+ });
258
+ return `(${memberSqls.join(" AND ")})`;
259
+ }
260
+ }
261
+ }
262
+ function buildValueExistsSql(alias, conditionSql) {
263
+ return `EXISTS (SELECT 1 FROM document_metadata_values mv WHERE mv.document_id = ${alias}.id AND ${conditionSql})`;
264
+ }
265
+ function valueTypeOf(scalar) {
266
+ return typeof scalar;
267
+ }
268
+ function valueColumnOf(scalar) {
269
+ if (typeof scalar === "string")
270
+ return "text_value";
271
+ if (typeof scalar === "number")
272
+ return "number_value";
273
+ return "boolean_value";
274
+ }
275
+ function bindScalar(scalar) {
276
+ if (typeof scalar === "boolean")
277
+ return scalar ? 1 : 0;
278
+ return scalar;
279
+ }
@@ -0,0 +1,45 @@
1
+ /**
2
+ * QMD Metadata Store - Schema, persistence, and batch loading for document
3
+ * metadata.
4
+ *
5
+ * Metadata attaches to document identity (`documents.id`), not content
6
+ * identity: two paths can share one content hash while carrying different
7
+ * metadata. SQLite stays a derived index — metadata is rebuilt from source
8
+ * documents on `qmd update`, never mutated in place.
9
+ *
10
+ * `document_metadata` records extraction state per document (including
11
+ * successful-but-empty extraction), so filtered search can distinguish
12
+ * "extracted with no metadata" from "not yet extracted" and "extraction
13
+ * failed". `document_metadata_values` holds one indexed row per scalar value
14
+ * for filtering.
15
+ */
16
+ import type { Database } from "./db.js";
17
+ import { type DocumentMetadata, type MetadataExtractionResult } from "./metadata.js";
18
+ export declare function initializeMetadataSchema(db: Database): void;
19
+ /**
20
+ * Extract and persist metadata for one document, replacing any prior rows.
21
+ *
22
+ * With `onlyIfStale`, extraction is skipped when the document already has a
23
+ * current-version extraction row — the cheap path for unchanged documents
24
+ * during re-index. Returns the extraction result, or null when skipped.
25
+ */
26
+ export declare function syncDocumentMetadata(db: Database, documentId: number, content: string, path: string, options?: {
27
+ onlyIfStale?: boolean;
28
+ }): MetadataExtractionResult | null;
29
+ /**
30
+ * Replace a document's metadata rows atomically. A failed extraction persists
31
+ * empty metadata plus the error, so stale metadata never survives a bad edit.
32
+ */
33
+ export declare function replaceDocumentMetadata(db: Database, documentId: number, extraction: MetadataExtractionResult): void;
34
+ /**
35
+ * Count active documents without a current, error-free metadata extraction.
36
+ * These documents are excluded from filtered search until `qmd update` runs.
37
+ */
38
+ export declare function countDocumentsPendingMetadata(db: Database): number;
39
+ /**
40
+ * Batch-load canonical metadata for a set of result filepaths
41
+ * (`qmd://collection/path`). One query — never per-result lookups.
42
+ */
43
+ export declare function getMetadataByFilepath(db: Database, filepaths: readonly string[]): Map<string, DocumentMetadata>;
44
+ /** Parse a stored `metadata_json` column value, tolerating absent rows. */
45
+ export declare function parseMetadataJson(metadataJson: string | null | undefined): DocumentMetadata;