@sanqianx/project-knowledge 4.7.0 → 4.8.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.agents/plugins/marketplace.json +17 -17
- package/.claude-plugin/marketplace.json +1 -1
- package/CHANGELOG.md +817 -799
- package/README.md +48 -36
- package/_modules/knowledge-engine/README.md +88 -0
- package/_modules/knowledge-engine/package.json +38 -0
- package/_modules/knowledge-engine/src/analysis-contract.js +25 -0
- package/_modules/knowledge-engine/src/analysis-worker.js +54 -0
- package/_modules/knowledge-engine/src/bin.js +52 -0
- package/_modules/knowledge-engine/src/bridge-client.js +24 -0
- package/_modules/knowledge-engine/src/coordinator.js +255 -0
- package/_modules/knowledge-engine/src/headless-analyzer.js +75 -0
- package/_modules/knowledge-engine/src/memories.js +85 -0
- package/_modules/knowledge-engine/src/migration.js +88 -0
- package/_modules/knowledge-engine/src/retrieval.js +116 -0
- package/_modules/knowledge-engine/src/runtime/README.md +7 -0
- package/_modules/knowledge-engine/src/runtime/atomic-file.js +204 -0
- package/_modules/knowledge-engine/src/runtime/automation-config.js +140 -0
- package/_modules/knowledge-engine/src/runtime/commit-conversation-binder.js +249 -0
- package/_modules/knowledge-engine/src/runtime/commit-processing-ledger.js +26 -0
- package/_modules/knowledge-engine/src/runtime/commit-prompt.js +202 -0
- package/_modules/knowledge-engine/src/runtime/commit-reconciler.js +432 -0
- package/_modules/knowledge-engine/src/runtime/content-hash.js +4 -0
- package/_modules/knowledge-engine/src/runtime/contracts.js +279 -0
- package/_modules/knowledge-engine/src/runtime/conversation-exclusions.js +204 -0
- package/_modules/knowledge-engine/src/runtime/conversation-store.js +387 -0
- package/_modules/knowledge-engine/src/runtime/evidence-bundle.js +228 -0
- package/_modules/knowledge-engine/src/runtime/git-runner.js +62 -0
- package/_modules/knowledge-engine/src/runtime/knowledge-promotion.js +444 -0
- package/_modules/knowledge-engine/src/runtime/knowledge-retrieval-service.js +402 -0
- package/_modules/knowledge-engine/src/runtime/layout.js +44 -0
- package/_modules/knowledge-engine/src/runtime/markdown-knowledge-indexer.js +230 -0
- package/_modules/knowledge-engine/src/runtime/scanner.js +298 -0
- package/_modules/knowledge-engine/src/server.js +155 -0
- package/_modules/knowledge-engine/src/store.js +239 -0
- package/_modules/knowledge-engine/ui/app.css +172 -0
- package/_modules/knowledge-engine/ui/app.js +297 -0
- package/_modules/knowledge-engine/ui/index.html +95 -0
- package/_site/README.md +30 -30
- package/_site/_test/ai-profile-resolver-test.js +137 -137
- package/_site/_test/automation-queue-test.js +14 -14
- package/_site/_test/automation-ui-test.js +40 -40
- package/_site/_test/background-task-registry-test.js +43 -43
- package/_site/_test/baseline-schema-test.js +91 -91
- package/_site/_test/chat-claudecodeui-match-test.js +92 -92
- package/_site/_test/claude-executable-discovery-test.js +62 -62
- package/_site/_test/claude-workbench-test.js +170 -170
- package/_site/_test/codex-conversation-projection-test.js +2 -1
- package/_site/_test/commit-conversation-binding-test.js +95 -95
- package/_site/_test/commit-evidence-test.js +110 -110
- package/_site/_test/conversation-api-test.js +109 -109
- package/_site/_test/conversation-store-test.js +136 -136
- package/_site/_test/data-dir-migration-test.js +57 -57
- package/_site/_test/desktop-browser-compat-test.js +28 -28
- package/_site/_test/desktop-hook-runtime-regression-test.js +204 -204
- package/_site/_test/explicit-commit-processor-test.js +37 -37
- package/_site/_test/fixtures/make-git-repos.js +12 -1
- package/_site/_test/folder-picker-output-test.js +19 -19
- package/_site/_test/full-integration-e2e-test.js +165 -165
- package/_site/_test/git-validation-test.js +2 -2
- package/_site/_test/hook-runtime-endpoint-test.js +88 -88
- package/_site/_test/hook-status-repair-api-test.js +152 -152
- package/_site/_test/hook-trigger-test.js +85 -85
- package/_site/_test/import-preflight-api-test.js +234 -233
- package/_site/_test/index-writer-concurrency-test.js +135 -135
- package/_site/_test/integration-adapters-test.js +171 -171
- package/_site/_test/integration-surface-coverage-test.js +164 -164
- package/_site/_test/knowledge-language-control-test.js +170 -170
- package/_site/_test/knowledge-promotion-recovery-test.js +260 -260
- package/_site/_test/knowledge-query-test.js +54 -54
- package/_site/_test/knowledge-retrieval-service-test.js +77 -77
- package/_site/_test/knowledge-storage-startup-test.js +52 -52
- package/_site/_test/knowledge-store-logs-supervision-test.js +89 -89
- package/_site/_test/legacy-forward-compat-test.js +257 -257
- package/_site/_test/legacy-project-upgrade-e2e-test.js +363 -363
- package/_site/_test/logging-api-test.js +89 -89
- package/_site/_test/logging-sse-no-gap-test.js +88 -88
- package/_site/_test/logging-ui-test.js +106 -104
- package/_site/_test/markdown-delta-overlay-test.js +85 -85
- package/_site/_test/markdown-maintenance-api-test.js +75 -75
- package/_site/_test/mcp-server-test.js +149 -149
- package/_site/_test/module-artifact-integrity-test.js +27 -0
- package/_site/_test/module-boundary-test.js +30 -0
- package/_site/_test/module-bridge-eventbridge-test.js +7 -0
- package/_site/_test/module-model-configuration-test.js +43 -0
- package/_site/_test/module-process-lifecycle-test.js +24 -0
- package/_site/_test/module-stream-proxy-test.js +34 -0
- package/_site/_test/non-release-ci-test.js +34 -34
- package/_site/_test/offline-boundary-isolation-test.js +32 -32
- package/_site/_test/p0-data-migration-characterization-test.js +37 -37
- package/_site/_test/p0-e2e-gate-test.js +347 -347
- package/_site/_test/packaged-ui-smoke-test.js +11 -9
- package/_site/_test/path-consistency-test.js +151 -151
- package/_site/_test/pending-sweep-test.js +9 -9
- package/_site/_test/post-commit-automation-test.js +87 -87
- package/_site/_test/product-diagnostics-ui-test.js +75 -0
- package/_site/_test/product-import-ui-test.js +92 -0
- package/_site/_test/project-delete-recovery-test.js +64 -64
- package/_site/_test/project-goal-editor-test.js +144 -144
- package/_site/_test/project-layout-v2-migration-test.js +127 -127
- package/_site/_test/project-lifecycle-transaction-test.js +106 -106
- package/_site/_test/prompt-settings-test.js +115 -115
- package/_site/_test/protected-architecture-gate-test.js +122 -122
- package/_site/_test/refactor-characterization-test.js +36 -36
- package/_site/_test/release-version-sync-test.js +1 -1
- package/_site/_test/requirement-recorder-test.js +173 -173
- package/_site/_test/run-all-tests.js +159 -156
- package/_site/_test/server-runtime-migration-safety-test.js +27 -27
- package/_site/_test/shared-contracts-test.js +58 -58
- package/_site/_test/startup-analysis-disabled-test.js +10 -10
- package/_site/_test/storage-foundation-test.js +78 -78
- package/_site/_test/structured-logger-test.js +101 -101
- package/_site/_test/tracking-start-test.js +108 -108
- package/_site/_test/v4122-upgrade-data-contract-test.js +56 -56
- package/_site/_test/workbench-permission-test.js +117 -117
- package/_site/_test/workspace-ui-contract-test.js +28 -43
- package/_site/lib/ai-profile-resolver.js +78 -78
- package/_site/lib/ai-workspace.js +101 -101
- package/_site/lib/automation-config.js +140 -140
- package/_site/lib/bridge-adapter.js +1 -0
- package/_site/lib/bridge-consumer-service.js +402 -402
- package/_site/lib/claude-cli-runner.js +1510 -1510
- package/_site/lib/commit-conversation-binder.js +234 -234
- package/_site/lib/commit-processing-ledger.js +27 -27
- package/_site/lib/commit-prompt.js +202 -202
- package/_site/lib/commit-reconciler.js +406 -406
- package/_site/lib/contracts.js +284 -279
- package/_site/lib/conversation-query-service.js +144 -144
- package/_site/lib/conversation-store.js +385 -385
- package/_site/lib/data-dir.js +74 -74
- package/_site/lib/data-state-classifier.js +40 -40
- package/_site/lib/engine-cutover.js +97 -0
- package/_site/lib/evidence-bundle.js +228 -228
- package/_site/lib/folder-picker-output.js +21 -21
- package/_site/lib/git-runner.js +62 -62
- package/_site/lib/github-team-store.js +1172 -1172
- package/_site/lib/hook-manager.js +203 -202
- package/_site/lib/index-service.js +173 -173
- package/_site/lib/integration-installer.js +907 -907
- package/_site/lib/kb-framework.js +176 -176
- package/_site/lib/kb-validator.js +124 -124
- package/_site/lib/knowledge-promotion.js +435 -435
- package/_site/lib/knowledge-retrieval-service.js +401 -401
- package/_site/lib/knowledge-tool-runtime.js +359 -359
- package/_site/lib/legacy-data-manifest.js +26 -26
- package/_site/lib/llm-client.js +161 -161
- package/_site/lib/markdown-knowledge-indexer.js +230 -230
- package/_site/lib/migration-service.js +442 -442
- package/_site/lib/module-bridge.js +198 -33
- package/_site/lib/post-commit-automation.js +87 -87
- package/_site/lib/project-lifecycle-service.js +519 -497
- package/_site/lib/project-store.js +268 -262
- package/_site/lib/requirement-recorder.js +278 -278
- package/_site/lib/runtime-endpoint.js +155 -155
- package/_site/lib/scanner.js +298 -298
- package/_site/lib/server-app.js +1590 -1520
- package/_site/lib/settings-store.js +133 -113
- package/_site/lib/storage-layout.js +162 -162
- package/_site/lib/structured-logger.js +574 -574
- package/_site/scripts/folder-picker.ps1 +156 -156
- package/_site/scripts/hook-trigger.js +157 -156
- package/_site/scripts/install-module-candidates.js +69 -0
- package/_site/scripts/pack-module-candidates.js +72 -0
- package/_site/scripts/sync-release-version.js +2 -2
- package/_site/scripts/vendor-modules.js +47 -0
- package/_site/scripts/verify-module-runtime.js +45 -0
- package/_site/scripts/verify-product-runtime.js +34 -0
- package/bin/project-knowledge-mcp.js +194 -194
- package/docs/README.zh-CN.md +34 -25
- package/docs/project-registry-schema.md +22 -22
- package/docs/testing-strategy.md +33 -33
- package/module-runtime-manifest.json +220 -0
- package/package.json +13 -9
- package/plugins/project-knowledge/.claude-plugin/plugin.json +1 -1
- package/plugins/project-knowledge/.codex-plugin/plugin.json +1 -1
- package/plugins/project-knowledge/.mcp.json +1 -1
- package/plugins/project-knowledge/opencode/project-knowledge.md +3 -3
- package/plugins/project-knowledge/skills/project-knowledge/SKILL.md +28 -28
- package/templates/project-readme.md +34 -34
- package/ui/favicon.svg +38 -38
- package/ui/index.html +51 -149
- package/ui/product.css +9 -0
- package/ui/product.js +386 -0
- package/vendor-manifest.json +46 -0
- package/_site/_test/import-ui-flow-test.js +0 -171
- package/_site/_test/project-control-panel-task14-test.js +0 -63
- package/_site/_test/task15-20-ui-flow-test.js +0 -148
- package/_site/_test/ui-i18n-toggle-test.js +0 -114
- package/_site/_test/ui-smoke-test.js +0 -73
- package/ui/app.css +0 -58
- package/ui/app.js +0 -607
- package/ui/i18n.js +0 -146
|
@@ -1,230 +1,230 @@
|
|
|
1
|
-
const fs = require('fs');
|
|
2
|
-
const path = require('path');
|
|
3
|
-
const { sha256 } = require('./knowledge-schema');
|
|
4
|
-
|
|
5
|
-
const CHUNKER_VERSION = 1;
|
|
6
|
-
const DEFAULT_MAX_CHARS = 1200;
|
|
7
|
-
const DEFAULT_OVERLAP_CHARS = 120;
|
|
8
|
-
const EXCLUDED_DIRS = new Set(['.git', '_ai', '_backup', 'node_modules']);
|
|
9
|
-
const DERIVED_INDEX_FILENAME = '00-index.md';
|
|
10
|
-
|
|
11
|
-
function normalizeRel(value) {
|
|
12
|
-
return String(value || '').replace(/\\/g, '/').replace(/^\.\//, '');
|
|
13
|
-
}
|
|
14
|
-
|
|
15
|
-
function stripFrontmatter(markdown) {
|
|
16
|
-
return String(markdown || '').replace(/^---\s*\r?\n[\s\S]*?\r?\n---\s*\r?\n?/, '');
|
|
17
|
-
}
|
|
18
|
-
|
|
19
|
-
function parseListValue(value) {
|
|
20
|
-
const raw = String(value || '').trim();
|
|
21
|
-
const body = raw.startsWith('[') && raw.endsWith(']') ? raw.slice(1, -1) : raw;
|
|
22
|
-
return body.split(',').map(item => item.trim().replace(/^['"]|['"]$/g, '')).filter(Boolean);
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
function parseKnowledgeMetadata(markdown) {
|
|
26
|
-
const match = /^---\s*\r?\n([\s\S]*?)\r?\n---\s*(?:\r?\n|$)/.exec(String(markdown || ''));
|
|
27
|
-
if (!match) return { tags: [], sourcePaths: [], routes: [], symbols: [], affectedModules: [] };
|
|
28
|
-
const values = {};
|
|
29
|
-
for (const line of match[1].split(/\r?\n/)) {
|
|
30
|
-
const field = /^([A-Za-z][A-Za-z0-9_-]*)\s*:\s*(.*)$/.exec(line);
|
|
31
|
-
if (field) values[field[1].toLowerCase()] = field[2];
|
|
32
|
-
}
|
|
33
|
-
return {
|
|
34
|
-
tags: parseListValue(values.tags),
|
|
35
|
-
sourcePaths: parseListValue(values.sourcepaths || values.source_paths),
|
|
36
|
-
routes: parseListValue(values.routes),
|
|
37
|
-
symbols: parseListValue(values.symbols),
|
|
38
|
-
affectedModules: parseListValue(values.affectedmodules || values.affected_modules),
|
|
39
|
-
};
|
|
40
|
-
}
|
|
41
|
-
|
|
42
|
-
function splitLongText(text, maxChars, overlapChars) {
|
|
43
|
-
const clean = String(text || '').trim();
|
|
44
|
-
if (!clean) return [];
|
|
45
|
-
if (clean.length <= maxChars) return [clean];
|
|
46
|
-
const parts = [];
|
|
47
|
-
let start = 0;
|
|
48
|
-
while (start < clean.length) {
|
|
49
|
-
let end = Math.min(start + maxChars, clean.length);
|
|
50
|
-
if (end < clean.length) {
|
|
51
|
-
const floor = start + Math.floor(maxChars * 0.6);
|
|
52
|
-
const boundary = Math.max(clean.lastIndexOf('\n\n', end), clean.lastIndexOf('。', end), clean.lastIndexOf('. ', end));
|
|
53
|
-
if (boundary >= floor) end = boundary + 1;
|
|
54
|
-
}
|
|
55
|
-
parts.push(clean.slice(start, end).trim());
|
|
56
|
-
if (end >= clean.length) break;
|
|
57
|
-
start = Math.max(start + 1, end - overlapChars);
|
|
58
|
-
}
|
|
59
|
-
return parts.filter(Boolean);
|
|
60
|
-
}
|
|
61
|
-
|
|
62
|
-
function chunkMarkdown(markdown, options = {}) {
|
|
63
|
-
const maxChars = Number(options.maxChars || DEFAULT_MAX_CHARS);
|
|
64
|
-
const overlapChars = Math.min(Number(options.overlapChars || DEFAULT_OVERLAP_CHARS), Math.floor(maxChars / 3));
|
|
65
|
-
const body = stripFrontmatter(markdown).replace(/\r\n/g, '\n');
|
|
66
|
-
const lines = body.split('\n');
|
|
67
|
-
const headings = [];
|
|
68
|
-
const sections = [];
|
|
69
|
-
let buffer = [];
|
|
70
|
-
let sectionHeadings = [];
|
|
71
|
-
|
|
72
|
-
const flush = () => {
|
|
73
|
-
const text = buffer.join('\n').trim();
|
|
74
|
-
if (text) sections.push({ headingPath: [...sectionHeadings], text });
|
|
75
|
-
buffer = [];
|
|
76
|
-
};
|
|
77
|
-
|
|
78
|
-
for (const line of lines) {
|
|
79
|
-
const match = /^(#{1,6})\s+(.+?)\s*$/.exec(line);
|
|
80
|
-
if (match) {
|
|
81
|
-
flush();
|
|
82
|
-
const level = match[1].length;
|
|
83
|
-
headings.length = level - 1;
|
|
84
|
-
headings[level - 1] = match[2].trim();
|
|
85
|
-
sectionHeadings = headings.filter(Boolean);
|
|
86
|
-
buffer.push(line);
|
|
87
|
-
} else {
|
|
88
|
-
buffer.push(line);
|
|
89
|
-
}
|
|
90
|
-
}
|
|
91
|
-
flush();
|
|
92
|
-
|
|
93
|
-
const chunks = [];
|
|
94
|
-
for (const section of sections) {
|
|
95
|
-
for (const text of splitLongText(section.text, maxChars, overlapChars)) {
|
|
96
|
-
chunks.push({
|
|
97
|
-
chunkOrder: chunks.length,
|
|
98
|
-
headingPath: section.headingPath,
|
|
99
|
-
chunkText: text,
|
|
100
|
-
});
|
|
101
|
-
}
|
|
102
|
-
}
|
|
103
|
-
return chunks;
|
|
104
|
-
}
|
|
105
|
-
|
|
106
|
-
function isDerivedIndex(filePath) {
|
|
107
|
-
return path.basename(String(filePath || '')).toLowerCase() === DERIVED_INDEX_FILENAME;
|
|
108
|
-
}
|
|
109
|
-
|
|
110
|
-
function listMarkdownFiles(root, options = {}) {
|
|
111
|
-
const includeDerived = options.includeDerived === true;
|
|
112
|
-
const files = [];
|
|
113
|
-
const walk = (dir) => {
|
|
114
|
-
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
115
|
-
if (entry.name.startsWith('.') || EXCLUDED_DIRS.has(entry.name)) continue;
|
|
116
|
-
const abs = path.join(dir, entry.name);
|
|
117
|
-
if (entry.isDirectory()) walk(abs);
|
|
118
|
-
else if (entry.isFile() && /\.md$/i.test(entry.name) && (includeDerived || !isDerivedIndex(entry.name))) files.push(abs);
|
|
119
|
-
}
|
|
120
|
-
};
|
|
121
|
-
if (fs.existsSync(root)) walk(path.resolve(root));
|
|
122
|
-
return files.sort();
|
|
123
|
-
}
|
|
124
|
-
|
|
125
|
-
function inferEntryType(relativePath) {
|
|
126
|
-
const rel = normalizeRel(relativePath).toLowerCase();
|
|
127
|
-
if (rel === 'goal.md') return 'goal';
|
|
128
|
-
if (rel === 'architecture.md') return 'architecture';
|
|
129
|
-
if (rel.startsWith('modules/')) return 'module';
|
|
130
|
-
if (rel.startsWith('changes/')) return 'change';
|
|
131
|
-
return 'document';
|
|
132
|
-
}
|
|
133
|
-
|
|
134
|
-
function inferTitle(markdown, relativePath) {
|
|
135
|
-
const heading = /^#\s+(.+)$/m.exec(stripFrontmatter(markdown));
|
|
136
|
-
return heading ? heading[1].trim() : path.basename(relativePath, path.extname(relativePath));
|
|
137
|
-
}
|
|
138
|
-
|
|
139
|
-
class MarkdownKnowledgeIndexer {
|
|
140
|
-
constructor(options = {}) {
|
|
141
|
-
if (!options.database) throw new Error('database is required');
|
|
142
|
-
if (!options.embedder) throw new Error('embedder is required');
|
|
143
|
-
this.database = options.database;
|
|
144
|
-
this.embedder = options.embedder;
|
|
145
|
-
this.maxChars = options.maxChars || DEFAULT_MAX_CHARS;
|
|
146
|
-
this.overlapChars = options.overlapChars || DEFAULT_OVERLAP_CHARS;
|
|
147
|
-
}
|
|
148
|
-
|
|
149
|
-
async indexFile(input) {
|
|
150
|
-
const filePath = path.resolve(input.filePath);
|
|
151
|
-
const relativePath = normalizeRel(input.relativePath || path.basename(filePath));
|
|
152
|
-
const markdown = fs.readFileSync(filePath, 'utf8');
|
|
153
|
-
const documentHash = sha256(`chunker:${CHUNKER_VERSION}\n${markdown}`);
|
|
154
|
-
const existing = await this.database.rowsForEntry(input.spaceId, relativePath);
|
|
155
|
-
if (existing.length && existing.every(row => row.document_hash === documentHash)) {
|
|
156
|
-
return { ok: true, action: 'unchanged', entryId: relativePath, chunks: existing.length, documentHash };
|
|
157
|
-
}
|
|
158
|
-
const title = inferTitle(markdown, relativePath);
|
|
159
|
-
const metadata = parseKnowledgeMetadata(markdown);
|
|
160
|
-
const rawChunks = chunkMarkdown(markdown, { maxChars: this.maxChars, overlapChars: this.overlapChars });
|
|
161
|
-
const chunks = [];
|
|
162
|
-
for (const chunk of rawChunks) {
|
|
163
|
-
const embeddingText = [title, chunk.headingPath.join(' > '), chunk.chunkText].filter(Boolean).join('\n');
|
|
164
|
-
chunks.push({
|
|
165
|
-
...chunk,
|
|
166
|
-
title,
|
|
167
|
-
entryType: inferEntryType(relativePath),
|
|
168
|
-
// Keyword search now indexes chunk_text directly. Keep only compact
|
|
169
|
-
// title/heading metadata in the legacy search_text column so new rows
|
|
170
|
-
// do not duplicate the entire document body.
|
|
171
|
-
searchText: [title, chunk.headingPath.join(' > ')].filter(Boolean).join('\n'),
|
|
172
|
-
vector: await this.embedder.embedPassage(embeddingText),
|
|
173
|
-
tags: [...new Set([...metadata.tags, ...metadata.affectedModules])],
|
|
174
|
-
sourcePaths: metadata.sourcePaths.length ? metadata.sourcePaths : input.sourcePaths || [],
|
|
175
|
-
routes: metadata.routes,
|
|
176
|
-
symbols: metadata.symbols,
|
|
177
|
-
sourceProjectId: input.sourceProjectId || '',
|
|
178
|
-
sourceCommit: input.sourceCommit || '',
|
|
179
|
-
documentHash,
|
|
180
|
-
});
|
|
181
|
-
}
|
|
182
|
-
const result = await this.database.replaceEntry(input.spaceId, relativePath, chunks);
|
|
183
|
-
return { ...result, entryId: relativePath, chunks: chunks.length, documentHash };
|
|
184
|
-
}
|
|
185
|
-
|
|
186
|
-
async indexDirectory(input) {
|
|
187
|
-
const root = path.resolve(input.kbPath);
|
|
188
|
-
const files = listMarkdownFiles(root);
|
|
189
|
-
const present = new Set();
|
|
190
|
-
const results = [];
|
|
191
|
-
for (const filePath of files) {
|
|
192
|
-
const relativePath = normalizeRel(path.relative(root, filePath));
|
|
193
|
-
present.add(relativePath);
|
|
194
|
-
results.push(await this.indexFile({ ...input, filePath, relativePath }));
|
|
195
|
-
}
|
|
196
|
-
const stale = (await this.database.entryIds(input.spaceId)).filter(entryId => !present.has(entryId));
|
|
197
|
-
let deletedChunks = 0;
|
|
198
|
-
for (const entryId of stale) deletedChunks += (await this.database.deleteEntry(input.spaceId, entryId)).deleted;
|
|
199
|
-
const changed = results.filter(item => item.action !== 'unchanged');
|
|
200
|
-
const operations = changed.length + stale.length;
|
|
201
|
-
const rows = changed.reduce((sum, item) => sum + Number(item.upserted || 0) + Number(item.deleted || 0), 0) + deletedChunks;
|
|
202
|
-
const mutationState = this.database.noteMutations({ operations, rows });
|
|
203
|
-
let maintenance = { ok: true, optimized: false, deferred: input.deferMaintenance === true, state: mutationState };
|
|
204
|
-
if (await this.database.count([input.spaceId])) {
|
|
205
|
-
await this.database.ensureSearchIndexes();
|
|
206
|
-
if (input.deferMaintenance !== true) maintenance = await this.database.maybeOptimize();
|
|
207
|
-
}
|
|
208
|
-
return {
|
|
209
|
-
ok: true,
|
|
210
|
-
spaceId: input.spaceId,
|
|
211
|
-
files: files.length,
|
|
212
|
-
indexed: changed.length,
|
|
213
|
-
unchanged: results.filter(item => item.action === 'unchanged').length,
|
|
214
|
-
deletedEntries: stale.length,
|
|
215
|
-
deletedChunks,
|
|
216
|
-
maintenance,
|
|
217
|
-
results,
|
|
218
|
-
};
|
|
219
|
-
}
|
|
220
|
-
}
|
|
221
|
-
|
|
222
|
-
module.exports = {
|
|
223
|
-
MarkdownKnowledgeIndexer,
|
|
224
|
-
CHUNKER_VERSION,
|
|
225
|
-
chunkMarkdown,
|
|
226
|
-
listMarkdownFiles,
|
|
227
|
-
isDerivedIndex,
|
|
228
|
-
inferEntryType,
|
|
229
|
-
parseKnowledgeMetadata,
|
|
230
|
-
};
|
|
1
|
+
const fs = require('fs');
|
|
2
|
+
const path = require('path');
|
|
3
|
+
const { sha256 } = require('./knowledge-schema');
|
|
4
|
+
|
|
5
|
+
const CHUNKER_VERSION = 1;
|
|
6
|
+
const DEFAULT_MAX_CHARS = 1200;
|
|
7
|
+
const DEFAULT_OVERLAP_CHARS = 120;
|
|
8
|
+
const EXCLUDED_DIRS = new Set(['.git', '_ai', '_backup', 'node_modules']);
|
|
9
|
+
const DERIVED_INDEX_FILENAME = '00-index.md';
|
|
10
|
+
|
|
11
|
+
function normalizeRel(value) {
|
|
12
|
+
return String(value || '').replace(/\\/g, '/').replace(/^\.\//, '');
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
function stripFrontmatter(markdown) {
|
|
16
|
+
return String(markdown || '').replace(/^---\s*\r?\n[\s\S]*?\r?\n---\s*\r?\n?/, '');
|
|
17
|
+
}
|
|
18
|
+
|
|
19
|
+
function parseListValue(value) {
|
|
20
|
+
const raw = String(value || '').trim();
|
|
21
|
+
const body = raw.startsWith('[') && raw.endsWith(']') ? raw.slice(1, -1) : raw;
|
|
22
|
+
return body.split(',').map(item => item.trim().replace(/^['"]|['"]$/g, '')).filter(Boolean);
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
function parseKnowledgeMetadata(markdown) {
|
|
26
|
+
const match = /^---\s*\r?\n([\s\S]*?)\r?\n---\s*(?:\r?\n|$)/.exec(String(markdown || ''));
|
|
27
|
+
if (!match) return { tags: [], sourcePaths: [], routes: [], symbols: [], affectedModules: [] };
|
|
28
|
+
const values = {};
|
|
29
|
+
for (const line of match[1].split(/\r?\n/)) {
|
|
30
|
+
const field = /^([A-Za-z][A-Za-z0-9_-]*)\s*:\s*(.*)$/.exec(line);
|
|
31
|
+
if (field) values[field[1].toLowerCase()] = field[2];
|
|
32
|
+
}
|
|
33
|
+
return {
|
|
34
|
+
tags: parseListValue(values.tags),
|
|
35
|
+
sourcePaths: parseListValue(values.sourcepaths || values.source_paths),
|
|
36
|
+
routes: parseListValue(values.routes),
|
|
37
|
+
symbols: parseListValue(values.symbols),
|
|
38
|
+
affectedModules: parseListValue(values.affectedmodules || values.affected_modules),
|
|
39
|
+
};
|
|
40
|
+
}
|
|
41
|
+
|
|
42
|
+
function splitLongText(text, maxChars, overlapChars) {
|
|
43
|
+
const clean = String(text || '').trim();
|
|
44
|
+
if (!clean) return [];
|
|
45
|
+
if (clean.length <= maxChars) return [clean];
|
|
46
|
+
const parts = [];
|
|
47
|
+
let start = 0;
|
|
48
|
+
while (start < clean.length) {
|
|
49
|
+
let end = Math.min(start + maxChars, clean.length);
|
|
50
|
+
if (end < clean.length) {
|
|
51
|
+
const floor = start + Math.floor(maxChars * 0.6);
|
|
52
|
+
const boundary = Math.max(clean.lastIndexOf('\n\n', end), clean.lastIndexOf('。', end), clean.lastIndexOf('. ', end));
|
|
53
|
+
if (boundary >= floor) end = boundary + 1;
|
|
54
|
+
}
|
|
55
|
+
parts.push(clean.slice(start, end).trim());
|
|
56
|
+
if (end >= clean.length) break;
|
|
57
|
+
start = Math.max(start + 1, end - overlapChars);
|
|
58
|
+
}
|
|
59
|
+
return parts.filter(Boolean);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function chunkMarkdown(markdown, options = {}) {
|
|
63
|
+
const maxChars = Number(options.maxChars || DEFAULT_MAX_CHARS);
|
|
64
|
+
const overlapChars = Math.min(Number(options.overlapChars || DEFAULT_OVERLAP_CHARS), Math.floor(maxChars / 3));
|
|
65
|
+
const body = stripFrontmatter(markdown).replace(/\r\n/g, '\n');
|
|
66
|
+
const lines = body.split('\n');
|
|
67
|
+
const headings = [];
|
|
68
|
+
const sections = [];
|
|
69
|
+
let buffer = [];
|
|
70
|
+
let sectionHeadings = [];
|
|
71
|
+
|
|
72
|
+
const flush = () => {
|
|
73
|
+
const text = buffer.join('\n').trim();
|
|
74
|
+
if (text) sections.push({ headingPath: [...sectionHeadings], text });
|
|
75
|
+
buffer = [];
|
|
76
|
+
};
|
|
77
|
+
|
|
78
|
+
for (const line of lines) {
|
|
79
|
+
const match = /^(#{1,6})\s+(.+?)\s*$/.exec(line);
|
|
80
|
+
if (match) {
|
|
81
|
+
flush();
|
|
82
|
+
const level = match[1].length;
|
|
83
|
+
headings.length = level - 1;
|
|
84
|
+
headings[level - 1] = match[2].trim();
|
|
85
|
+
sectionHeadings = headings.filter(Boolean);
|
|
86
|
+
buffer.push(line);
|
|
87
|
+
} else {
|
|
88
|
+
buffer.push(line);
|
|
89
|
+
}
|
|
90
|
+
}
|
|
91
|
+
flush();
|
|
92
|
+
|
|
93
|
+
const chunks = [];
|
|
94
|
+
for (const section of sections) {
|
|
95
|
+
for (const text of splitLongText(section.text, maxChars, overlapChars)) {
|
|
96
|
+
chunks.push({
|
|
97
|
+
chunkOrder: chunks.length,
|
|
98
|
+
headingPath: section.headingPath,
|
|
99
|
+
chunkText: text,
|
|
100
|
+
});
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
return chunks;
|
|
104
|
+
}
|
|
105
|
+
|
|
106
|
+
function isDerivedIndex(filePath) {
|
|
107
|
+
return path.basename(String(filePath || '')).toLowerCase() === DERIVED_INDEX_FILENAME;
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
function listMarkdownFiles(root, options = {}) {
|
|
111
|
+
const includeDerived = options.includeDerived === true;
|
|
112
|
+
const files = [];
|
|
113
|
+
const walk = (dir) => {
|
|
114
|
+
for (const entry of fs.readdirSync(dir, { withFileTypes: true })) {
|
|
115
|
+
if (entry.name.startsWith('.') || EXCLUDED_DIRS.has(entry.name)) continue;
|
|
116
|
+
const abs = path.join(dir, entry.name);
|
|
117
|
+
if (entry.isDirectory()) walk(abs);
|
|
118
|
+
else if (entry.isFile() && /\.md$/i.test(entry.name) && (includeDerived || !isDerivedIndex(entry.name))) files.push(abs);
|
|
119
|
+
}
|
|
120
|
+
};
|
|
121
|
+
if (fs.existsSync(root)) walk(path.resolve(root));
|
|
122
|
+
return files.sort();
|
|
123
|
+
}
|
|
124
|
+
|
|
125
|
+
function inferEntryType(relativePath) {
|
|
126
|
+
const rel = normalizeRel(relativePath).toLowerCase();
|
|
127
|
+
if (rel === 'goal.md') return 'goal';
|
|
128
|
+
if (rel === 'architecture.md') return 'architecture';
|
|
129
|
+
if (rel.startsWith('modules/')) return 'module';
|
|
130
|
+
if (rel.startsWith('changes/')) return 'change';
|
|
131
|
+
return 'document';
|
|
132
|
+
}
|
|
133
|
+
|
|
134
|
+
function inferTitle(markdown, relativePath) {
|
|
135
|
+
const heading = /^#\s+(.+)$/m.exec(stripFrontmatter(markdown));
|
|
136
|
+
return heading ? heading[1].trim() : path.basename(relativePath, path.extname(relativePath));
|
|
137
|
+
}
|
|
138
|
+
|
|
139
|
+
class MarkdownKnowledgeIndexer {
|
|
140
|
+
constructor(options = {}) {
|
|
141
|
+
if (!options.database) throw new Error('database is required');
|
|
142
|
+
if (!options.embedder) throw new Error('embedder is required');
|
|
143
|
+
this.database = options.database;
|
|
144
|
+
this.embedder = options.embedder;
|
|
145
|
+
this.maxChars = options.maxChars || DEFAULT_MAX_CHARS;
|
|
146
|
+
this.overlapChars = options.overlapChars || DEFAULT_OVERLAP_CHARS;
|
|
147
|
+
}
|
|
148
|
+
|
|
149
|
+
async indexFile(input) {
|
|
150
|
+
const filePath = path.resolve(input.filePath);
|
|
151
|
+
const relativePath = normalizeRel(input.relativePath || path.basename(filePath));
|
|
152
|
+
const markdown = fs.readFileSync(filePath, 'utf8');
|
|
153
|
+
const documentHash = sha256(`chunker:${CHUNKER_VERSION}\n${markdown}`);
|
|
154
|
+
const existing = await this.database.rowsForEntry(input.spaceId, relativePath);
|
|
155
|
+
if (existing.length && existing.every(row => row.document_hash === documentHash)) {
|
|
156
|
+
return { ok: true, action: 'unchanged', entryId: relativePath, chunks: existing.length, documentHash };
|
|
157
|
+
}
|
|
158
|
+
const title = inferTitle(markdown, relativePath);
|
|
159
|
+
const metadata = parseKnowledgeMetadata(markdown);
|
|
160
|
+
const rawChunks = chunkMarkdown(markdown, { maxChars: this.maxChars, overlapChars: this.overlapChars });
|
|
161
|
+
const chunks = [];
|
|
162
|
+
for (const chunk of rawChunks) {
|
|
163
|
+
const embeddingText = [title, chunk.headingPath.join(' > '), chunk.chunkText].filter(Boolean).join('\n');
|
|
164
|
+
chunks.push({
|
|
165
|
+
...chunk,
|
|
166
|
+
title,
|
|
167
|
+
entryType: inferEntryType(relativePath),
|
|
168
|
+
// Keyword search now indexes chunk_text directly. Keep only compact
|
|
169
|
+
// title/heading metadata in the legacy search_text column so new rows
|
|
170
|
+
// do not duplicate the entire document body.
|
|
171
|
+
searchText: [title, chunk.headingPath.join(' > ')].filter(Boolean).join('\n'),
|
|
172
|
+
vector: await this.embedder.embedPassage(embeddingText),
|
|
173
|
+
tags: [...new Set([...metadata.tags, ...metadata.affectedModules])],
|
|
174
|
+
sourcePaths: metadata.sourcePaths.length ? metadata.sourcePaths : input.sourcePaths || [],
|
|
175
|
+
routes: metadata.routes,
|
|
176
|
+
symbols: metadata.symbols,
|
|
177
|
+
sourceProjectId: input.sourceProjectId || '',
|
|
178
|
+
sourceCommit: input.sourceCommit || '',
|
|
179
|
+
documentHash,
|
|
180
|
+
});
|
|
181
|
+
}
|
|
182
|
+
const result = await this.database.replaceEntry(input.spaceId, relativePath, chunks);
|
|
183
|
+
return { ...result, entryId: relativePath, chunks: chunks.length, documentHash };
|
|
184
|
+
}
|
|
185
|
+
|
|
186
|
+
async indexDirectory(input) {
|
|
187
|
+
const root = path.resolve(input.kbPath);
|
|
188
|
+
const files = listMarkdownFiles(root);
|
|
189
|
+
const present = new Set();
|
|
190
|
+
const results = [];
|
|
191
|
+
for (const filePath of files) {
|
|
192
|
+
const relativePath = normalizeRel(path.relative(root, filePath));
|
|
193
|
+
present.add(relativePath);
|
|
194
|
+
results.push(await this.indexFile({ ...input, filePath, relativePath }));
|
|
195
|
+
}
|
|
196
|
+
const stale = (await this.database.entryIds(input.spaceId)).filter(entryId => !present.has(entryId));
|
|
197
|
+
let deletedChunks = 0;
|
|
198
|
+
for (const entryId of stale) deletedChunks += (await this.database.deleteEntry(input.spaceId, entryId)).deleted;
|
|
199
|
+
const changed = results.filter(item => item.action !== 'unchanged');
|
|
200
|
+
const operations = changed.length + stale.length;
|
|
201
|
+
const rows = changed.reduce((sum, item) => sum + Number(item.upserted || 0) + Number(item.deleted || 0), 0) + deletedChunks;
|
|
202
|
+
const mutationState = this.database.noteMutations({ operations, rows });
|
|
203
|
+
let maintenance = { ok: true, optimized: false, deferred: input.deferMaintenance === true, state: mutationState };
|
|
204
|
+
if (await this.database.count([input.spaceId])) {
|
|
205
|
+
await this.database.ensureSearchIndexes();
|
|
206
|
+
if (input.deferMaintenance !== true) maintenance = await this.database.maybeOptimize();
|
|
207
|
+
}
|
|
208
|
+
return {
|
|
209
|
+
ok: true,
|
|
210
|
+
spaceId: input.spaceId,
|
|
211
|
+
files: files.length,
|
|
212
|
+
indexed: changed.length,
|
|
213
|
+
unchanged: results.filter(item => item.action === 'unchanged').length,
|
|
214
|
+
deletedEntries: stale.length,
|
|
215
|
+
deletedChunks,
|
|
216
|
+
maintenance,
|
|
217
|
+
results,
|
|
218
|
+
};
|
|
219
|
+
}
|
|
220
|
+
}
|
|
221
|
+
|
|
222
|
+
module.exports = {
|
|
223
|
+
MarkdownKnowledgeIndexer,
|
|
224
|
+
CHUNKER_VERSION,
|
|
225
|
+
chunkMarkdown,
|
|
226
|
+
listMarkdownFiles,
|
|
227
|
+
isDerivedIndex,
|
|
228
|
+
inferEntryType,
|
|
229
|
+
parseKnowledgeMetadata,
|
|
230
|
+
};
|