@wei840222/qmd 2026.9.6 → 2026.9.25
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/CHANGELOG.md +14 -0
- package/README.md +84 -1
- package/dist/cli/build-info.json +2 -2
- package/dist/cli/qmd.js +85 -9
- package/dist/collections.js +9 -4
- package/dist/index.d.ts +10 -0
- package/dist/index.js +14 -2
- package/dist/llm.d.ts +7 -1
- package/dist/llm.js +23 -4
- package/dist/mcp/server.js +70 -6
- package/dist/metadata-filter.d.ts +74 -0
- package/dist/metadata-filter.js +279 -0
- package/dist/metadata-store.d.ts +45 -0
- package/dist/metadata-store.js +173 -0
- package/dist/metadata.d.ts +61 -0
- package/dist/metadata.js +215 -0
- package/dist/search/zh-dict.txt +3 -0
- package/dist/store.d.ts +23 -16
- package/dist/store.js +341 -161
- package/package.json +2 -2
- package/scripts/sync-zh-dict.mjs +4 -1
- package/skills/qmd/SKILL.md +11 -0
- package/skills/release/SKILL.md +0 -141
- package/skills/release/scripts/install-hooks.sh +0 -38
- package/skills/release/scripts/release-context.sh +0 -129
|
@@ -0,0 +1,173 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* QMD Metadata Store - Schema, persistence, and batch loading for document
|
|
3
|
+
* metadata.
|
|
4
|
+
*
|
|
5
|
+
* Metadata attaches to document identity (`documents.id`), not content
|
|
6
|
+
* identity: two paths can share one content hash while carrying different
|
|
7
|
+
* metadata. SQLite stays a derived index — metadata is rebuilt from source
|
|
8
|
+
* documents on `qmd update`, never mutated in place.
|
|
9
|
+
*
|
|
10
|
+
* `document_metadata` records extraction state per document (including
|
|
11
|
+
* successful-but-empty extraction), so filtered search can distinguish
|
|
12
|
+
* "extracted with no metadata" from "not yet extracted" and "extraction
|
|
13
|
+
* failed". `document_metadata_values` holds one indexed row per scalar value
|
|
14
|
+
* for filtering.
|
|
15
|
+
*/
|
|
16
|
+
import { extractDocumentMetadata, METADATA_EXTRACTION_VERSION, } from "./metadata.js";
|
|
17
|
+
// =============================================================================
|
|
18
|
+
// Schema
|
|
19
|
+
// =============================================================================
|
|
20
|
+
export function initializeMetadataSchema(db) {
|
|
21
|
+
db.exec(`
|
|
22
|
+
CREATE TABLE IF NOT EXISTS document_metadata (
|
|
23
|
+
document_id INTEGER PRIMARY KEY,
|
|
24
|
+
metadata_json TEXT NOT NULL DEFAULT '{}',
|
|
25
|
+
extraction_version INTEGER NOT NULL,
|
|
26
|
+
extraction_error TEXT,
|
|
27
|
+
extracted_at TEXT NOT NULL,
|
|
28
|
+
FOREIGN KEY (document_id) REFERENCES documents(id) ON DELETE CASCADE
|
|
29
|
+
)
|
|
30
|
+
`);
|
|
31
|
+
db.exec(`
|
|
32
|
+
CREATE TABLE IF NOT EXISTS document_metadata_values (
|
|
33
|
+
document_id INTEGER NOT NULL,
|
|
34
|
+
key TEXT NOT NULL,
|
|
35
|
+
ordinal INTEGER NOT NULL,
|
|
36
|
+
value_type TEXT NOT NULL,
|
|
37
|
+
text_value TEXT,
|
|
38
|
+
number_value REAL,
|
|
39
|
+
boolean_value INTEGER,
|
|
40
|
+
PRIMARY KEY (document_id, key, ordinal),
|
|
41
|
+
FOREIGN KEY (document_id)
|
|
42
|
+
REFERENCES document_metadata(document_id)
|
|
43
|
+
ON DELETE CASCADE,
|
|
44
|
+
CHECK (value_type IN ('string', 'number', 'boolean')),
|
|
45
|
+
CHECK (
|
|
46
|
+
(value_type = 'string' AND text_value IS NOT NULL AND number_value IS NULL AND boolean_value IS NULL)
|
|
47
|
+
OR (value_type = 'number' AND number_value IS NOT NULL AND text_value IS NULL AND boolean_value IS NULL)
|
|
48
|
+
OR (value_type = 'boolean' AND boolean_value IN (0, 1) AND text_value IS NULL AND number_value IS NULL)
|
|
49
|
+
)
|
|
50
|
+
)
|
|
51
|
+
`);
|
|
52
|
+
db.exec(`
|
|
53
|
+
CREATE INDEX IF NOT EXISTS idx_metadata_text_lookup
|
|
54
|
+
ON document_metadata_values(key, text_value, document_id)
|
|
55
|
+
WHERE value_type = 'string'
|
|
56
|
+
`);
|
|
57
|
+
db.exec(`
|
|
58
|
+
CREATE INDEX IF NOT EXISTS idx_metadata_number_lookup
|
|
59
|
+
ON document_metadata_values(key, number_value, document_id)
|
|
60
|
+
WHERE value_type = 'number'
|
|
61
|
+
`);
|
|
62
|
+
db.exec(`
|
|
63
|
+
CREATE INDEX IF NOT EXISTS idx_metadata_boolean_lookup
|
|
64
|
+
ON document_metadata_values(key, boolean_value, document_id)
|
|
65
|
+
WHERE value_type = 'boolean'
|
|
66
|
+
`);
|
|
67
|
+
}
|
|
68
|
+
// =============================================================================
|
|
69
|
+
// Persistence
|
|
70
|
+
// =============================================================================
|
|
71
|
+
/**
|
|
72
|
+
* Extract and persist metadata for one document, replacing any prior rows.
|
|
73
|
+
*
|
|
74
|
+
* With `onlyIfStale`, extraction is skipped when the document already has a
|
|
75
|
+
* current-version extraction row — the cheap path for unchanged documents
|
|
76
|
+
* during re-index. Returns the extraction result, or null when skipped.
|
|
77
|
+
*/
|
|
78
|
+
export function syncDocumentMetadata(db, documentId, content, path, options) {
|
|
79
|
+
if (options?.onlyIfStale && isDocumentMetadataCurrent(db, documentId))
|
|
80
|
+
return null;
|
|
81
|
+
const extraction = extractDocumentMetadata(content, path);
|
|
82
|
+
replaceDocumentMetadata(db, documentId, extraction);
|
|
83
|
+
return extraction;
|
|
84
|
+
}
|
|
85
|
+
/**
|
|
86
|
+
* Replace a document's metadata rows atomically. A failed extraction persists
|
|
87
|
+
* empty metadata plus the error, so stale metadata never survives a bad edit.
|
|
88
|
+
*/
|
|
89
|
+
export function replaceDocumentMetadata(db, documentId, extraction) {
|
|
90
|
+
const replace = db.transaction(() => {
|
|
91
|
+
db.prepare(`
|
|
92
|
+
INSERT INTO document_metadata (document_id, metadata_json, extraction_version, extraction_error, extracted_at)
|
|
93
|
+
VALUES (?, ?, ?, ?, ?)
|
|
94
|
+
ON CONFLICT(document_id) DO UPDATE SET
|
|
95
|
+
metadata_json = excluded.metadata_json,
|
|
96
|
+
extraction_version = excluded.extraction_version,
|
|
97
|
+
extraction_error = excluded.extraction_error,
|
|
98
|
+
extracted_at = excluded.extracted_at
|
|
99
|
+
`).run(documentId, JSON.stringify(extraction.metadata), extraction.extractionVersion, extraction.error ?? null, new Date().toISOString());
|
|
100
|
+
db.prepare(`DELETE FROM document_metadata_values WHERE document_id = ?`).run(documentId);
|
|
101
|
+
const insertValue = db.prepare(`
|
|
102
|
+
INSERT INTO document_metadata_values (document_id, key, ordinal, value_type, text_value, number_value, boolean_value)
|
|
103
|
+
VALUES (?, ?, ?, ?, ?, ?, ?)
|
|
104
|
+
`);
|
|
105
|
+
for (const [key, value] of Object.entries(extraction.metadata)) {
|
|
106
|
+
const scalars = Array.isArray(value) ? value : [value];
|
|
107
|
+
scalars.forEach((scalar, ordinal) => {
|
|
108
|
+
insertValue.run(documentId, key, ordinal, typeof scalar, typeof scalar === "string" ? scalar : null, typeof scalar === "number" ? scalar : null, typeof scalar === "boolean" ? (scalar ? 1 : 0) : null);
|
|
109
|
+
});
|
|
110
|
+
}
|
|
111
|
+
});
|
|
112
|
+
replace();
|
|
113
|
+
}
|
|
114
|
+
function isDocumentMetadataCurrent(db, documentId) {
|
|
115
|
+
const row = db.prepare(`SELECT extraction_version FROM document_metadata WHERE document_id = ?`)
|
|
116
|
+
.get(documentId);
|
|
117
|
+
return row?.extraction_version === METADATA_EXTRACTION_VERSION;
|
|
118
|
+
}
|
|
119
|
+
// =============================================================================
|
|
120
|
+
// Queries
|
|
121
|
+
// =============================================================================
|
|
122
|
+
/**
|
|
123
|
+
* Count active documents without a current, error-free metadata extraction.
|
|
124
|
+
* These documents are excluded from filtered search until `qmd update` runs.
|
|
125
|
+
*/
|
|
126
|
+
export function countDocumentsPendingMetadata(db) {
|
|
127
|
+
const row = db.prepare(`
|
|
128
|
+
SELECT COUNT(*) as c FROM documents d
|
|
129
|
+
WHERE d.active = 1
|
|
130
|
+
AND NOT EXISTS (
|
|
131
|
+
SELECT 1 FROM document_metadata dm
|
|
132
|
+
WHERE dm.document_id = d.id
|
|
133
|
+
AND dm.extraction_version = ?
|
|
134
|
+
AND dm.extraction_error IS NULL
|
|
135
|
+
)
|
|
136
|
+
`).get(METADATA_EXTRACTION_VERSION);
|
|
137
|
+
return row.c;
|
|
138
|
+
}
|
|
139
|
+
/**
|
|
140
|
+
* Batch-load canonical metadata for a set of result filepaths
|
|
141
|
+
* (`qmd://collection/path`). One query — never per-result lookups.
|
|
142
|
+
*/
|
|
143
|
+
export function getMetadataByFilepath(db, filepaths) {
|
|
144
|
+
const metadataByFilepath = new Map();
|
|
145
|
+
if (filepaths.length === 0)
|
|
146
|
+
return metadataByFilepath;
|
|
147
|
+
const placeholders = filepaths.map(() => "?").join(", ");
|
|
148
|
+
const stmt = db.prepare(`
|
|
149
|
+
SELECT 'qmd://' || d.collection || '/' || d.path AS filepath, dm.metadata_json
|
|
150
|
+
FROM documents d
|
|
151
|
+
JOIN document_metadata dm ON dm.document_id = d.id
|
|
152
|
+
WHERE d.active = 1
|
|
153
|
+
AND 'qmd://' || d.collection || '/' || d.path IN (${placeholders})
|
|
154
|
+
`);
|
|
155
|
+
const rows = typeof stmt?.all === "function"
|
|
156
|
+
? stmt.all(...filepaths)
|
|
157
|
+
: [];
|
|
158
|
+
for (const row of rows) {
|
|
159
|
+
metadataByFilepath.set(row.filepath, parseMetadataJson(row.metadata_json));
|
|
160
|
+
}
|
|
161
|
+
return metadataByFilepath;
|
|
162
|
+
}
|
|
163
|
+
/** Parse a stored `metadata_json` column value, tolerating absent rows. */
|
|
164
|
+
export function parseMetadataJson(metadataJson) {
|
|
165
|
+
if (!metadataJson)
|
|
166
|
+
return {};
|
|
167
|
+
try {
|
|
168
|
+
return JSON.parse(metadataJson);
|
|
169
|
+
}
|
|
170
|
+
catch {
|
|
171
|
+
return {};
|
|
172
|
+
}
|
|
173
|
+
}
|
|
@@ -0,0 +1,61 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* QMD Metadata - Public metadata types and frontmatter extraction.
|
|
3
|
+
*
|
|
4
|
+
* Documents opt into metadata through a namespaced Markdown frontmatter block:
|
|
5
|
+
*
|
|
6
|
+
* ---
|
|
7
|
+
* qmd:
|
|
8
|
+
* metadata:
|
|
9
|
+
* topics:
|
|
10
|
+
* - typescript
|
|
11
|
+
* - programming
|
|
12
|
+
* status: published
|
|
13
|
+
* ---
|
|
14
|
+
*
|
|
15
|
+
* Extraction is source-agnostic at the persistence boundary: this module
|
|
16
|
+
* produces a canonical `MetadataExtractionResult`, and future non-frontmatter
|
|
17
|
+
* sources can produce the same shape without touching storage or filtering.
|
|
18
|
+
*
|
|
19
|
+
* The raw document is never modified — frontmatter stays part of the stored,
|
|
20
|
+
* indexed, chunked, and embedded content.
|
|
21
|
+
*/
|
|
22
|
+
export type MetadataScalar = string | number | boolean;
|
|
23
|
+
export type MetadataScalarArray = readonly string[] | readonly number[] | readonly boolean[];
|
|
24
|
+
export type MetadataValue = MetadataScalar | MetadataScalarArray;
|
|
25
|
+
export type DocumentMetadata = Record<string, MetadataValue>;
|
|
26
|
+
/**
|
|
27
|
+
* Result of extracting metadata from one document.
|
|
28
|
+
*
|
|
29
|
+
* `error` is set when the document opted into `qmd.metadata` but the value was
|
|
30
|
+
* invalid — the document still indexes normally, but it is excluded from
|
|
31
|
+
* filtered search until the metadata is corrected and re-indexed.
|
|
32
|
+
*/
|
|
33
|
+
export interface MetadataExtractionResult {
|
|
34
|
+
metadata: DocumentMetadata;
|
|
35
|
+
error?: string;
|
|
36
|
+
extractionVersion: number;
|
|
37
|
+
}
|
|
38
|
+
/**
|
|
39
|
+
* Bump when extraction or normalization semantics change so existing rows are
|
|
40
|
+
* re-extracted on the next `qmd update`.
|
|
41
|
+
*/
|
|
42
|
+
export declare const METADATA_EXTRACTION_VERSION = 1;
|
|
43
|
+
/** Defensive limits for metadata from untrusted repositories. */
|
|
44
|
+
export declare const METADATA_LIMITS: {
|
|
45
|
+
readonly maxFrontmatterBytes: number;
|
|
46
|
+
readonly maxKeys: 64;
|
|
47
|
+
readonly maxKeyBytes: 128;
|
|
48
|
+
readonly maxStringLength: 1024;
|
|
49
|
+
readonly maxArrayLength: 128;
|
|
50
|
+
readonly maxYamlAliasCount: 100;
|
|
51
|
+
readonly maxErrorLength: 200;
|
|
52
|
+
};
|
|
53
|
+
/**
|
|
54
|
+
* Extract `qmd.metadata` from a document's leading YAML frontmatter.
|
|
55
|
+
*
|
|
56
|
+
* Never throws. A document without frontmatter, without a `qmd` namespace, or
|
|
57
|
+
* with a non-frontmatter extension yields empty metadata with no error.
|
|
58
|
+
* Invalid frontmatter or invalid metadata yields empty metadata plus a bounded
|
|
59
|
+
* extraction error.
|
|
60
|
+
*/
|
|
61
|
+
export declare function extractDocumentMetadata(content: string, path: string): MetadataExtractionResult;
|
package/dist/metadata.js
ADDED
|
@@ -0,0 +1,215 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* QMD Metadata - Public metadata types and frontmatter extraction.
|
|
3
|
+
*
|
|
4
|
+
* Documents opt into metadata through a namespaced Markdown frontmatter block:
|
|
5
|
+
*
|
|
6
|
+
* ---
|
|
7
|
+
* qmd:
|
|
8
|
+
* metadata:
|
|
9
|
+
* topics:
|
|
10
|
+
* - typescript
|
|
11
|
+
* - programming
|
|
12
|
+
* status: published
|
|
13
|
+
* ---
|
|
14
|
+
*
|
|
15
|
+
* Extraction is source-agnostic at the persistence boundary: this module
|
|
16
|
+
* produces a canonical `MetadataExtractionResult`, and future non-frontmatter
|
|
17
|
+
* sources can produce the same shape without touching storage or filtering.
|
|
18
|
+
*
|
|
19
|
+
* The raw document is never modified — frontmatter stays part of the stored,
|
|
20
|
+
* indexed, chunked, and embedded content.
|
|
21
|
+
*/
|
|
22
|
+
import YAML from "yaml";
|
|
23
|
+
// =============================================================================
|
|
24
|
+
// Limits
|
|
25
|
+
// =============================================================================
|
|
26
|
+
/**
|
|
27
|
+
* Bump when extraction or normalization semantics change so existing rows are
|
|
28
|
+
* re-extracted on the next `qmd update`.
|
|
29
|
+
*/
|
|
30
|
+
export const METADATA_EXTRACTION_VERSION = 1;
|
|
31
|
+
/** Defensive limits for metadata from untrusted repositories. */
|
|
32
|
+
export const METADATA_LIMITS = {
|
|
33
|
+
maxFrontmatterBytes: 64 * 1024,
|
|
34
|
+
maxKeys: 64,
|
|
35
|
+
maxKeyBytes: 128,
|
|
36
|
+
maxStringLength: 1024,
|
|
37
|
+
maxArrayLength: 128,
|
|
38
|
+
maxYamlAliasCount: 100,
|
|
39
|
+
maxErrorLength: 200,
|
|
40
|
+
};
|
|
41
|
+
/** File extensions parsed for frontmatter metadata. */
|
|
42
|
+
const FRONTMATTER_FILE_EXTENSIONS = new Set([".md", ".markdown", ".mdx"]);
|
|
43
|
+
// =============================================================================
|
|
44
|
+
// Extraction
|
|
45
|
+
// =============================================================================
|
|
46
|
+
/**
|
|
47
|
+
* Extract `qmd.metadata` from a document's leading YAML frontmatter.
|
|
48
|
+
*
|
|
49
|
+
* Never throws. A document without frontmatter, without a `qmd` namespace, or
|
|
50
|
+
* with a non-frontmatter extension yields empty metadata with no error.
|
|
51
|
+
* Invalid frontmatter or invalid metadata yields empty metadata plus a bounded
|
|
52
|
+
* extraction error.
|
|
53
|
+
*/
|
|
54
|
+
export function extractDocumentMetadata(content, path) {
|
|
55
|
+
const success = (metadata) => ({ metadata, extractionVersion: METADATA_EXTRACTION_VERSION });
|
|
56
|
+
const failure = (message) => ({
|
|
57
|
+
metadata: {},
|
|
58
|
+
error: truncateErrorMessage(message),
|
|
59
|
+
extractionVersion: METADATA_EXTRACTION_VERSION,
|
|
60
|
+
});
|
|
61
|
+
if (!hasFrontmatterFileExtension(path))
|
|
62
|
+
return success({});
|
|
63
|
+
const frontmatterYaml = getFrontmatterYaml(content);
|
|
64
|
+
if (frontmatterYaml === null)
|
|
65
|
+
return success({});
|
|
66
|
+
if (Buffer.byteLength(frontmatterYaml, "utf-8") > METADATA_LIMITS.maxFrontmatterBytes) {
|
|
67
|
+
return failure(`frontmatter exceeds ${METADATA_LIMITS.maxFrontmatterBytes} bytes`);
|
|
68
|
+
}
|
|
69
|
+
let frontmatter;
|
|
70
|
+
try {
|
|
71
|
+
frontmatter = YAML.parse(frontmatterYaml, { maxAliasCount: METADATA_LIMITS.maxYamlAliasCount });
|
|
72
|
+
}
|
|
73
|
+
catch (err) {
|
|
74
|
+
return failure(`invalid frontmatter YAML: ${err instanceof Error ? err.message : String(err)}`);
|
|
75
|
+
}
|
|
76
|
+
if (!isPlainObject(frontmatter))
|
|
77
|
+
return success({});
|
|
78
|
+
const qmdNamespace = frontmatter["qmd"];
|
|
79
|
+
if (qmdNamespace === undefined)
|
|
80
|
+
return success({});
|
|
81
|
+
if (!isPlainObject(qmdNamespace)) {
|
|
82
|
+
return failure("frontmatter 'qmd' must be a mapping");
|
|
83
|
+
}
|
|
84
|
+
const rawMetadata = qmdNamespace["metadata"];
|
|
85
|
+
if (rawMetadata === undefined)
|
|
86
|
+
return success({});
|
|
87
|
+
if (!isPlainObject(rawMetadata)) {
|
|
88
|
+
return failure("frontmatter 'qmd.metadata' must be a mapping");
|
|
89
|
+
}
|
|
90
|
+
try {
|
|
91
|
+
return success(normalizeMetadata(rawMetadata));
|
|
92
|
+
}
|
|
93
|
+
catch (err) {
|
|
94
|
+
return failure(err instanceof Error ? err.message : String(err));
|
|
95
|
+
}
|
|
96
|
+
}
|
|
97
|
+
function hasFrontmatterFileExtension(path) {
|
|
98
|
+
const dotIndex = path.lastIndexOf(".");
|
|
99
|
+
if (dotIndex < 0)
|
|
100
|
+
return false;
|
|
101
|
+
return FRONTMATTER_FILE_EXTENSIONS.has(path.slice(dotIndex).toLowerCase());
|
|
102
|
+
}
|
|
103
|
+
/**
|
|
104
|
+
* Slice the YAML between a leading `---` line and a closing `---` or `...`
|
|
105
|
+
* line. Tolerates a UTF-8 BOM and CRLF line endings. Returns null when the
|
|
106
|
+
* document has no complete leading frontmatter block.
|
|
107
|
+
*/
|
|
108
|
+
function getFrontmatterYaml(content) {
|
|
109
|
+
const body = content.charCodeAt(0) === 0xfeff ? content.slice(1) : content;
|
|
110
|
+
const openMatch = body.match(/^---[ \t]*\r?\n/);
|
|
111
|
+
if (!openMatch)
|
|
112
|
+
return null;
|
|
113
|
+
const yamlStart = openMatch[0].length;
|
|
114
|
+
const closePattern = /^(?:---|\.\.\.)[ \t]*(?:\r?\n|$)/m;
|
|
115
|
+
const closeMatch = body.slice(yamlStart).match(closePattern);
|
|
116
|
+
if (!closeMatch || closeMatch.index === undefined)
|
|
117
|
+
return null;
|
|
118
|
+
return body.slice(yamlStart, yamlStart + closeMatch.index);
|
|
119
|
+
}
|
|
120
|
+
// =============================================================================
|
|
121
|
+
// Normalization
|
|
122
|
+
// =============================================================================
|
|
123
|
+
/**
|
|
124
|
+
* Normalize raw `qmd.metadata` into canonical `DocumentMetadata`.
|
|
125
|
+
* Throws on any unsupported key or value — extraction is all-or-nothing per
|
|
126
|
+
* document so stale partial metadata can never persist.
|
|
127
|
+
*/
|
|
128
|
+
function normalizeMetadata(rawMetadata) {
|
|
129
|
+
const keys = Object.keys(rawMetadata);
|
|
130
|
+
if (keys.length > METADATA_LIMITS.maxKeys) {
|
|
131
|
+
throw new Error(`metadata has ${keys.length} keys (max ${METADATA_LIMITS.maxKeys})`);
|
|
132
|
+
}
|
|
133
|
+
const entries = [];
|
|
134
|
+
for (const key of keys) {
|
|
135
|
+
validateMetadataKey(key);
|
|
136
|
+
entries.push([key, normalizeMetadataValue(key, rawMetadata[key])]);
|
|
137
|
+
}
|
|
138
|
+
// Object.fromEntries defines "__proto__" as ordinary data instead of
|
|
139
|
+
// invoking Object.prototype's legacy setter. Metadata keys are unrestricted
|
|
140
|
+
// user data, so prototype-shaped names must round-trip like every other key.
|
|
141
|
+
return Object.fromEntries(entries);
|
|
142
|
+
}
|
|
143
|
+
function validateMetadataKey(key) {
|
|
144
|
+
if (key.length === 0) {
|
|
145
|
+
throw new Error("metadata keys must be non-empty strings");
|
|
146
|
+
}
|
|
147
|
+
if (Buffer.byteLength(key, "utf-8") > METADATA_LIMITS.maxKeyBytes) {
|
|
148
|
+
throw new Error(`metadata key exceeds ${METADATA_LIMITS.maxKeyBytes} bytes`);
|
|
149
|
+
}
|
|
150
|
+
// eslint-disable-next-line no-control-regex
|
|
151
|
+
if (/[\u0000-\u001f\u007f]/.test(key)) {
|
|
152
|
+
throw new Error("metadata keys must not contain control characters");
|
|
153
|
+
}
|
|
154
|
+
}
|
|
155
|
+
function normalizeMetadataValue(key, rawValue) {
|
|
156
|
+
if (Array.isArray(rawValue)) {
|
|
157
|
+
return normalizeMetadataArray(key, rawValue);
|
|
158
|
+
}
|
|
159
|
+
return normalizeMetadataScalar(key, rawValue);
|
|
160
|
+
}
|
|
161
|
+
function normalizeMetadataScalar(key, rawValue) {
|
|
162
|
+
if (typeof rawValue === "string") {
|
|
163
|
+
if (rawValue.length > METADATA_LIMITS.maxStringLength) {
|
|
164
|
+
throw new Error(`metadata key "${key}": string exceeds ${METADATA_LIMITS.maxStringLength} characters`);
|
|
165
|
+
}
|
|
166
|
+
return rawValue;
|
|
167
|
+
}
|
|
168
|
+
if (typeof rawValue === "number") {
|
|
169
|
+
if (!Number.isFinite(rawValue)) {
|
|
170
|
+
throw new Error(`metadata key "${key}": numbers must be finite`);
|
|
171
|
+
}
|
|
172
|
+
return rawValue;
|
|
173
|
+
}
|
|
174
|
+
if (typeof rawValue === "boolean")
|
|
175
|
+
return rawValue;
|
|
176
|
+
if (rawValue === null) {
|
|
177
|
+
throw new Error(`metadata key "${key}": null is not supported — omit the key instead`);
|
|
178
|
+
}
|
|
179
|
+
throw new Error(`metadata key "${key}": unsupported value type — use strings, numbers, booleans, or flat arrays of one of those`);
|
|
180
|
+
}
|
|
181
|
+
function normalizeMetadataArray(key, rawValues) {
|
|
182
|
+
if (rawValues.length === 0) {
|
|
183
|
+
throw new Error(`metadata key "${key}": empty arrays are not supported — omit the key instead`);
|
|
184
|
+
}
|
|
185
|
+
if (rawValues.length > METADATA_LIMITS.maxArrayLength) {
|
|
186
|
+
throw new Error(`metadata key "${key}": array exceeds ${METADATA_LIMITS.maxArrayLength} values`);
|
|
187
|
+
}
|
|
188
|
+
const scalars = rawValues.map(rawValue => {
|
|
189
|
+
if (Array.isArray(rawValue)) {
|
|
190
|
+
throw new Error(`metadata key "${key}": nested arrays are not supported`);
|
|
191
|
+
}
|
|
192
|
+
return normalizeMetadataScalar(key, rawValue);
|
|
193
|
+
});
|
|
194
|
+
// Narrow each homogeneous case explicitly so the public array union remains
|
|
195
|
+
// precise without discarding type evidence through chained assertions.
|
|
196
|
+
if (scalars.every((scalar) => typeof scalar === "string")) {
|
|
197
|
+
return Array.from(new Set(scalars));
|
|
198
|
+
}
|
|
199
|
+
if (scalars.every((scalar) => typeof scalar === "number")) {
|
|
200
|
+
return Array.from(new Set(scalars));
|
|
201
|
+
}
|
|
202
|
+
if (scalars.every((scalar) => typeof scalar === "boolean")) {
|
|
203
|
+
return Array.from(new Set(scalars));
|
|
204
|
+
}
|
|
205
|
+
throw new Error(`metadata key "${key}": mixed-type arrays are not supported`);
|
|
206
|
+
}
|
|
207
|
+
function isPlainObject(value) {
|
|
208
|
+
return typeof value === "object" && value !== null && !Array.isArray(value);
|
|
209
|
+
}
|
|
210
|
+
function truncateErrorMessage(message) {
|
|
211
|
+
const singleLine = message.replace(/\s+/g, " ").trim();
|
|
212
|
+
if (singleLine.length <= METADATA_LIMITS.maxErrorLength)
|
|
213
|
+
return singleLine;
|
|
214
|
+
return singleLine.slice(0, METADATA_LIMITS.maxErrorLength - 3) + "...";
|
|
215
|
+
}
|
package/dist/search/zh-dict.txt
CHANGED
|
@@ -179849,6 +179849,7 @@ c++ 3 nz
|
|
|
179849
179849
|
實際性 4 N
|
|
179850
179850
|
實際數 3 N
|
|
179851
179851
|
實際節 1 N
|
|
179852
|
+
實際耗時 1000000 nz
|
|
179852
179853
|
實際面 6 N
|
|
179853
179854
|
實領 4 N
|
|
179854
179855
|
實馬 1 N
|
|
@@ -403350,6 +403351,7 @@ c++ 3 nz
|
|
|
403350
403351
|
真實度 3 N
|
|
403351
403352
|
真實性 86 N
|
|
403352
403353
|
真實感 20 N
|
|
403354
|
+
真實時間 1000000 nz
|
|
403353
403355
|
真實模式 1000000 nz
|
|
403354
403356
|
真實版 16 N
|
|
403355
403357
|
真實面 7 N
|
|
@@ -437457,6 +437459,7 @@ c++ 3 nz
|
|
|
437457
437459
|
總編輯 112 N
|
|
437458
437460
|
總署 191 N
|
|
437459
437461
|
總而言之 20 ADV
|
|
437462
|
+
總耗時 1000000 nz
|
|
437460
437463
|
總膽管 3 N
|
|
437461
437464
|
總荷 1 N
|
|
437462
437465
|
總菌 3 N
|
package/dist/store.d.ts
CHANGED
|
@@ -18,6 +18,8 @@ import type { LLM } from "./llm.js";
|
|
|
18
18
|
import { LlamaCpp, formatQueryForEmbedding, formatDocForEmbedding, type ILLMSession } from "./llm.js";
|
|
19
19
|
import type { NamedCollection, Collection, CollectionConfig } from "./collections.js";
|
|
20
20
|
import type { IndexDiagnostics } from "./diagnostics.js";
|
|
21
|
+
import { type DocumentMetadata } from "./metadata.js";
|
|
22
|
+
import { type MetadataFilter } from "./metadata-filter.js";
|
|
21
23
|
export declare const DEFAULT_EMBED_MODEL = "hf:ggml-org/embeddinggemma-300M-GGUF/embeddinggemma-300M-Q8_0.gguf";
|
|
22
24
|
export declare const DEFAULT_RERANK_MODEL = "hf:ggml-org/Qwen3-Reranker-0.6B-Q8_0-GGUF/qwen3-reranker-0.6b-q8_0.gguf";
|
|
23
25
|
export declare const DEFAULT_QUERY_MODEL = "hf:tobil/qmd-query-expansion-1.7B-gguf/qmd-query-expansion-1.7B-q4_k_m.gguf";
|
|
@@ -310,9 +312,9 @@ export type Store = {
|
|
|
310
312
|
isVirtualPath: typeof isVirtualPath;
|
|
311
313
|
resolveVirtualPath: (virtualPath: string) => string | null;
|
|
312
314
|
toVirtualPath: (absolutePath: string) => string | null;
|
|
313
|
-
searchCharFTS: (query: string, limit?: number, collectionFilter?: CollectionFilter) => SearchResult[];
|
|
314
|
-
searchFTS: (query: string, limit?: number, collectionFilter?: CollectionFilter) => SearchResult[];
|
|
315
|
-
searchVec: (query: string, model: string, limit?: number, collectionFilter?: CollectionFilter, session?: ILLMSession, precomputedEmbedding?: number[]) => Promise<SearchResult[]>;
|
|
315
|
+
searchCharFTS: (query: string, limit?: number, collectionFilter?: CollectionFilter, filter?: MetadataFilter) => SearchResult[];
|
|
316
|
+
searchFTS: (query: string, limit?: number, collectionFilter?: CollectionFilter, filter?: MetadataFilter) => SearchResult[];
|
|
317
|
+
searchVec: (query: string, model: string, limit?: number, collectionFilter?: CollectionFilter, session?: ILLMSession, precomputedEmbedding?: number[], filter?: MetadataFilter) => Promise<SearchResult[]>;
|
|
316
318
|
expandQuery: (query: string, model?: string, expansionContext?: string, options?: QueryExpansionOptions) => Promise<ExpandedQuery[]>;
|
|
317
319
|
/** Drop the cached expansion for a query so the next call regenerates. */
|
|
318
320
|
invalidateExpansionCache: (query: string, expansionContext?: string, options?: QueryExpansionOptions) => void;
|
|
@@ -347,7 +349,7 @@ export type Store = {
|
|
|
347
349
|
hash: string;
|
|
348
350
|
} | null;
|
|
349
351
|
insertContent: (hash: string, content: string, createdAt: string) => void;
|
|
350
|
-
insertDocument: (collectionName: string, path: string, title: string, hash: string, createdAt: string, modifiedAt: string) =>
|
|
352
|
+
insertDocument: (collectionName: string, path: string, title: string, hash: string, createdAt: string, modifiedAt: string) => number;
|
|
351
353
|
findActiveDocument: (collectionName: string, path: string) => {
|
|
352
354
|
id: number;
|
|
353
355
|
hash: string;
|
|
@@ -367,6 +369,7 @@ export type Store = {
|
|
|
367
369
|
body: string;
|
|
368
370
|
path: string;
|
|
369
371
|
}[];
|
|
372
|
+
insertEmbedding: (hash: string, seq: number, pos: number, embedding: Float32Array, model: string, embeddedAt: string, totalChunks?: number, fingerprint?: string, lease?: EmbeddingBuildLease) => void;
|
|
370
373
|
};
|
|
371
374
|
export type ReindexProgress = {
|
|
372
375
|
file: string;
|
|
@@ -385,6 +388,8 @@ export type ReindexResult = {
|
|
|
385
388
|
orphanedCleaned: number;
|
|
386
389
|
skipped: number;
|
|
387
390
|
skippedFiles: ReindexSkippedFile[];
|
|
391
|
+
/** Documents whose qmd.metadata frontmatter failed extraction this pass. */
|
|
392
|
+
metadataErrors: number;
|
|
388
393
|
};
|
|
389
394
|
/**
|
|
390
395
|
* Re-index a single collection by scanning the filesystem and updating the database.
|
|
@@ -501,6 +506,7 @@ export declare function handelize(path: string): string;
|
|
|
501
506
|
* Search result extends DocumentResult with score and source info
|
|
502
507
|
*/
|
|
503
508
|
export type SearchResult = DocumentResult & {
|
|
509
|
+
metadata: DocumentMetadata;
|
|
504
510
|
score: number;
|
|
505
511
|
source: "fts" | "vec";
|
|
506
512
|
chunkPos?: number;
|
|
@@ -612,6 +618,8 @@ export type IndexStatus = {
|
|
|
612
618
|
totalDocuments: number;
|
|
613
619
|
needsEmbedding: number;
|
|
614
620
|
hasVectorIndex: boolean;
|
|
621
|
+
/** Active documents without current, error-free metadata extraction. */
|
|
622
|
+
pendingMetadata: number;
|
|
615
623
|
collections: CollectionInfo[];
|
|
616
624
|
/** Additive diagnostics populated by high-level composition roots. */
|
|
617
625
|
diagnostics?: IndexDiagnostics;
|
|
@@ -707,10 +715,11 @@ export declare function extractTitle(content: string, filename: string): string;
|
|
|
707
715
|
export declare function insertContent(db: Database, hash: string, content: string, createdAt: string): void;
|
|
708
716
|
/**
|
|
709
717
|
* Insert a new document into the documents table.
|
|
718
|
+
* Returns the document's id so callers can attach document-scoped state.
|
|
710
719
|
*/
|
|
711
|
-
export declare function insertDocument(db: Database, collectionName: string, path: string, title: string, hash: string, createdAt: string, modifiedAt: string):
|
|
720
|
+
export declare function insertDocument(db: Database, collectionName: string, path: string, title: string, hash: string, createdAt: string, modifiedAt: string): number;
|
|
712
721
|
/** Insert immutable content and its document row in the same lexical transaction. */
|
|
713
|
-
export declare function insertDocumentWithContent(db: Database, hash: string, content: string, contentCreatedAt: string, collectionName: string, path: string, title: string, documentCreatedAt: string, modifiedAt: string):
|
|
722
|
+
export declare function insertDocumentWithContent(db: Database, hash: string, content: string, contentCreatedAt: string, collectionName: string, path: string, title: string, documentCreatedAt: string, modifiedAt: string): number;
|
|
714
723
|
/**
|
|
715
724
|
* Find an active document by collection name and path.
|
|
716
725
|
*/
|
|
@@ -778,13 +787,6 @@ export declare function chunkDocumentAsync(content: string, maxChars?: number, o
|
|
|
778
787
|
text: string;
|
|
779
788
|
pos: number;
|
|
780
789
|
}[]>;
|
|
781
|
-
/**
|
|
782
|
-
* Chunk a document by actual token count using the LLM tokenizer.
|
|
783
|
-
* More accurate than character-based chunking but requires async.
|
|
784
|
-
*
|
|
785
|
-
* When filepath and chunkStrategy are provided, uses AST-aware break points
|
|
786
|
-
* for supported code files.
|
|
787
|
-
*/
|
|
788
790
|
export declare function chunkDocumentByTokens(content: string, maxTokens?: number, overlapTokens?: number, windowTokens?: number, filepath?: string, chunkStrategy?: ChunkStrategy, signal?: AbortSignal): Promise<{
|
|
789
791
|
text: string;
|
|
790
792
|
pos: number;
|
|
@@ -929,9 +931,9 @@ export declare const CJK_LEXICAL_CANDIDATE_DEPTH = 60;
|
|
|
929
931
|
export declare const CJK_LEXICAL_RRF_WEIGHTS: Readonly<Record<CjkLexicalChannel, number>>;
|
|
930
932
|
export type CollectionFilter = string | readonly string[];
|
|
931
933
|
export type CollectionScope = string | readonly string[] | undefined;
|
|
932
|
-
export declare function searchCharFTS(db: Database, query: string, limit?: number, collectionFilter?: CollectionFilter): SearchResult[];
|
|
933
|
-
export declare function searchFTS(db: Database, query: string, limit?: number, collectionFilter?:
|
|
934
|
-
export declare function searchVec(db: Database, query: string, model: string, limit?: number, collectionFilter?: CollectionFilter, session?: ILLMSession, precomputedEmbedding?: number[],
|
|
934
|
+
export declare function searchCharFTS(db: Database, query: string, limit?: number, collectionFilter?: CollectionFilter, filter?: MetadataFilter): SearchResult[];
|
|
935
|
+
export declare function searchFTS(db: Database, query: string, limit?: number, collectionFilter?: CollectionScope, filter?: MetadataFilter): SearchResult[];
|
|
936
|
+
export declare function searchVec(db: Database, query: string, model: string, limit?: number, collectionFilter?: CollectionFilter, session?: ILLMSession, precomputedEmbedding?: number[], providerOrLlm?: EmbeddingProvider | LLM, authorizeRemoteRequestOrFilter?: Store["authorizeRemoteRequest"] | MetadataFilter, llmOverride?: LLM, metadataFilter?: MetadataFilter): Promise<SearchResult[]>;
|
|
935
937
|
/**
|
|
936
938
|
* Get all unique content hashes that need embeddings (from active documents).
|
|
937
939
|
* Returns hash, document body, and a sample path for display purposes.
|
|
@@ -1104,6 +1106,7 @@ export type ExpansionErrorEvent = {
|
|
|
1104
1106
|
export interface HybridQueryOptions {
|
|
1105
1107
|
collection?: string | readonly string[];
|
|
1106
1108
|
collections?: readonly string[];
|
|
1109
|
+
filter?: MetadataFilter;
|
|
1107
1110
|
limit?: number;
|
|
1108
1111
|
minScore?: number;
|
|
1109
1112
|
candidateLimit?: number;
|
|
@@ -1128,6 +1131,7 @@ export interface HybridQueryResult {
|
|
|
1128
1131
|
score: number;
|
|
1129
1132
|
context: string | null;
|
|
1130
1133
|
docid: string;
|
|
1134
|
+
metadata: DocumentMetadata;
|
|
1131
1135
|
explain?: HybridQueryExplain;
|
|
1132
1136
|
}
|
|
1133
1137
|
export type RankedListMeta = {
|
|
@@ -1163,6 +1167,7 @@ export declare function getHybridRrfWeights(rankedListMeta: RankedListMeta[]): n
|
|
|
1163
1167
|
export declare function hybridQuery(store: Store, query: string, options?: HybridQueryOptions): Promise<HybridQueryResult[]>;
|
|
1164
1168
|
export interface VectorSearchOptions {
|
|
1165
1169
|
collection?: CollectionFilter;
|
|
1170
|
+
filter?: MetadataFilter;
|
|
1166
1171
|
limit?: number;
|
|
1167
1172
|
minScore?: number;
|
|
1168
1173
|
/** Additional context used only while generating query expansions. */
|
|
@@ -1179,6 +1184,7 @@ export interface VectorSearchResult {
|
|
|
1179
1184
|
score: number;
|
|
1180
1185
|
context: string | null;
|
|
1181
1186
|
docid: string;
|
|
1187
|
+
metadata: DocumentMetadata;
|
|
1182
1188
|
}
|
|
1183
1189
|
/**
|
|
1184
1190
|
* Vector-only semantic search with query expansion.
|
|
@@ -1196,6 +1202,7 @@ export declare function vectorSearchQuery(store: Store, query: string, options?:
|
|
|
1196
1202
|
*/
|
|
1197
1203
|
export interface StructuredSearchOptions {
|
|
1198
1204
|
collections?: string[];
|
|
1205
|
+
filter?: MetadataFilter;
|
|
1199
1206
|
limit?: number;
|
|
1200
1207
|
minScore?: number;
|
|
1201
1208
|
candidateLimit?: number;
|