@tangle-network/agent-knowledge 1.1.1 → 1.3.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +36 -0
- package/README.md +182 -0
- package/dist/{chunk-TXNYP4WI.js → chunk-4PNXQ2NT.js} +1 -1
- package/dist/chunk-4PNXQ2NT.js.map +1 -0
- package/dist/{chunk-JWGMOXAT.js → chunk-HKYD765Q.js} +599 -427
- package/dist/chunk-HKYD765Q.js.map +1 -0
- package/dist/chunk-WCYW2GDA.js +551 -0
- package/dist/chunk-WCYW2GDA.js.map +1 -0
- package/dist/chunk-YMKHCTS2.js +19 -0
- package/dist/chunk-YMKHCTS2.js.map +1 -0
- package/dist/cli.js +77 -34
- package/dist/cli.js.map +1 -1
- package/dist/index.d.ts +705 -263
- package/dist/index.js +771 -167
- package/dist/index.js.map +1 -1
- package/dist/sources/index.d.ts +386 -0
- package/dist/sources/index.js +34 -0
- package/dist/sources/index.js.map +1 -0
- package/dist/{types-YkFkfwJG.d.ts → types-CAeh7Lwb.d.ts} +2 -2
- package/dist/viz/index.d.ts +1 -1
- package/dist/viz/index.js +1 -1
- package/package.json +11 -3
- package/dist/chunk-JWGMOXAT.js.map +0 -1
- package/dist/chunk-TXNYP4WI.js.map +0 -1
|
@@ -1,14 +1,158 @@
|
|
|
1
|
-
|
|
2
|
-
|
|
3
|
-
|
|
4
|
-
|
|
1
|
+
import {
|
|
2
|
+
sha256,
|
|
3
|
+
slugify,
|
|
4
|
+
stableId
|
|
5
|
+
} from "./chunk-YMKHCTS2.js";
|
|
6
|
+
|
|
7
|
+
// src/adapters.ts
|
|
8
|
+
var textSourceAdapter = {
|
|
9
|
+
id: "text",
|
|
10
|
+
canLoad: (input) => Boolean(input.text) || /\.(md|txt|json|csv)$/i.test(input.uri),
|
|
11
|
+
load: (input) => ({
|
|
12
|
+
title: input.uri.split("/").pop(),
|
|
13
|
+
mediaType: mediaTypeFor(input.uri),
|
|
14
|
+
text: decodeText(input),
|
|
15
|
+
anchors: anchorsForText(input.uri, decodeText(input)),
|
|
16
|
+
metadata: input.metadata
|
|
17
|
+
})
|
|
18
|
+
};
|
|
19
|
+
function mediaTypeFor(uri) {
|
|
20
|
+
const lower = uri.toLowerCase();
|
|
21
|
+
if (lower.endsWith(".md")) return "text/markdown";
|
|
22
|
+
if (lower.endsWith(".txt")) return "text/plain";
|
|
23
|
+
if (lower.endsWith(".json")) return "application/json";
|
|
24
|
+
if (lower.endsWith(".csv")) return "text/csv";
|
|
25
|
+
if (lower.endsWith(".pdf")) return "application/pdf";
|
|
26
|
+
return "application/octet-stream";
|
|
5
27
|
}
|
|
6
|
-
function
|
|
7
|
-
|
|
8
|
-
return out || "untitled";
|
|
28
|
+
function decodeText(input) {
|
|
29
|
+
return input.text ?? (input.bytes ? new TextDecoder().decode(input.bytes).slice(0, 2e5) : void 0);
|
|
9
30
|
}
|
|
10
|
-
function
|
|
11
|
-
|
|
31
|
+
function anchorsForText(uri, text) {
|
|
32
|
+
if (!text) return [];
|
|
33
|
+
const lines = text.split("\n");
|
|
34
|
+
const anchors = [
|
|
35
|
+
{ id: "all", sourceId: "", label: "Full source", lineStart: 1, lineEnd: lines.length }
|
|
36
|
+
];
|
|
37
|
+
for (let i = 0; i < lines.length; i += 50) {
|
|
38
|
+
anchors.push({
|
|
39
|
+
id: `l${i + 1}`,
|
|
40
|
+
sourceId: "",
|
|
41
|
+
label: `${uri}:${i + 1}`,
|
|
42
|
+
lineStart: i + 1,
|
|
43
|
+
lineEnd: Math.min(lines.length, i + 50)
|
|
44
|
+
});
|
|
45
|
+
}
|
|
46
|
+
return anchors;
|
|
47
|
+
}
|
|
48
|
+
|
|
49
|
+
// src/search.ts
|
|
50
|
+
var RRF_K = 60;
|
|
51
|
+
var STOP_WORDS = /* @__PURE__ */ new Set([
|
|
52
|
+
"the",
|
|
53
|
+
"is",
|
|
54
|
+
"a",
|
|
55
|
+
"an",
|
|
56
|
+
"what",
|
|
57
|
+
"how",
|
|
58
|
+
"are",
|
|
59
|
+
"was",
|
|
60
|
+
"were",
|
|
61
|
+
"to",
|
|
62
|
+
"for",
|
|
63
|
+
"of",
|
|
64
|
+
"with",
|
|
65
|
+
"by",
|
|
66
|
+
"in",
|
|
67
|
+
"on",
|
|
68
|
+
"and"
|
|
69
|
+
]);
|
|
70
|
+
function searchKnowledge(index, query, limit = 10) {
|
|
71
|
+
const trimmed = query.trim();
|
|
72
|
+
if (trimmed === "") return [];
|
|
73
|
+
const tokenRanked = rankByTokens(index.pages, trimmed);
|
|
74
|
+
const graphRanked = rankByGraph(index.pages, tokenRanked);
|
|
75
|
+
const scores = reciprocalRankFusion([tokenRanked.map((p) => p.id), graphRanked.map((p) => p.id)]);
|
|
76
|
+
const byId = new Map(index.pages.map((page) => [page.id, page]));
|
|
77
|
+
const ranked = [...scores.entries()].map(([id, score]) => ({ page: byId.get(id), score })).filter((item) => Boolean(item.page)).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).slice(0, limit);
|
|
78
|
+
const topScore = ranked[0]?.score ?? 0;
|
|
79
|
+
return ranked.map((item, i) => ({
|
|
80
|
+
page: item.page,
|
|
81
|
+
score: item.score,
|
|
82
|
+
rrfScore: item.score,
|
|
83
|
+
normalizedScore: topScore > 0 ? item.score / topScore : 0,
|
|
84
|
+
rank: i + 1,
|
|
85
|
+
snippet: buildSnippet(item.page.text, trimmed),
|
|
86
|
+
reasons: reasonsFor(item.page, trimmed)
|
|
87
|
+
}));
|
|
88
|
+
}
|
|
89
|
+
function tokenizeQuery(query) {
|
|
90
|
+
const raw = query.toLowerCase().split(/[\s,,。!?、;:""''()()\-_/\\·~~…]+/).filter((token) => token.length > 1 && !STOP_WORDS.has(token));
|
|
91
|
+
const tokens = [];
|
|
92
|
+
for (const token of raw) {
|
|
93
|
+
if (/[\u4e00-\u9fff\u3400-\u4dbf]/.test(token) && token.length > 2) {
|
|
94
|
+
const chars = [...token];
|
|
95
|
+
for (let i = 0; i < chars.length - 1; i++) tokens.push(chars[i] + chars[i + 1]);
|
|
96
|
+
tokens.push(...chars);
|
|
97
|
+
}
|
|
98
|
+
tokens.push(token);
|
|
99
|
+
}
|
|
100
|
+
return [...new Set(tokens)];
|
|
101
|
+
}
|
|
102
|
+
function reciprocalRankFusion(rankLists, k = RRF_K) {
|
|
103
|
+
const scores = /* @__PURE__ */ new Map();
|
|
104
|
+
for (const list of rankLists) {
|
|
105
|
+
list.forEach((id, idx) => {
|
|
106
|
+
scores.set(id, (scores.get(id) ?? 0) + 1 / (k + idx + 1));
|
|
107
|
+
});
|
|
108
|
+
}
|
|
109
|
+
return scores;
|
|
110
|
+
}
|
|
111
|
+
function rankByTokens(pages, query) {
|
|
112
|
+
const tokens = tokenizeQuery(query);
|
|
113
|
+
const effective = tokens.length > 0 ? tokens : [query.toLowerCase()];
|
|
114
|
+
return pages.map((page) => ({ page, score: tokenScore(page, query, effective) })).filter((item) => item.score > 0).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).map((item) => item.page);
|
|
115
|
+
}
|
|
116
|
+
function rankByGraph(pages, tokenRanked) {
|
|
117
|
+
if (tokenRanked.length === 0) return [];
|
|
118
|
+
const seeds = new Set(tokenRanked.slice(0, 5).map((page) => page.id));
|
|
119
|
+
return pages.map((page) => ({
|
|
120
|
+
page,
|
|
121
|
+
score: page.outLinks.filter((link) => seeds.has(link)).length + page.sourceIds.filter(
|
|
122
|
+
(source) => tokenRanked.some((seed) => seed.sourceIds.includes(source))
|
|
123
|
+
).length
|
|
124
|
+
})).filter((item) => item.score > 0).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).map((item) => item.page);
|
|
125
|
+
}
|
|
126
|
+
function tokenScore(page, query, tokens) {
|
|
127
|
+
const title = page.title.toLowerCase();
|
|
128
|
+
const path = page.path.toLowerCase();
|
|
129
|
+
const body = page.text.toLowerCase();
|
|
130
|
+
const phrase = query.toLowerCase();
|
|
131
|
+
let score = 0;
|
|
132
|
+
if (path.endsWith(`${phrase}.md`) || title === phrase) score += 200;
|
|
133
|
+
if (title.includes(phrase)) score += 50;
|
|
134
|
+
if (body.includes(phrase)) score += 20;
|
|
135
|
+
for (const token of tokens) {
|
|
136
|
+
if (title.includes(token)) score += 5;
|
|
137
|
+
if (body.includes(token)) score += 1;
|
|
138
|
+
if (path.includes(token)) score += 3;
|
|
139
|
+
}
|
|
140
|
+
return score;
|
|
141
|
+
}
|
|
142
|
+
function buildSnippet(text, query) {
|
|
143
|
+
const compact = text.replace(/\s+/g, " ").trim();
|
|
144
|
+
const idx = compact.toLowerCase().indexOf(query.toLowerCase());
|
|
145
|
+
if (idx < 0) return compact.slice(0, 180);
|
|
146
|
+
return compact.slice(Math.max(0, idx - 80), Math.min(compact.length, idx + query.length + 100));
|
|
147
|
+
}
|
|
148
|
+
function reasonsFor(page, query) {
|
|
149
|
+
const lower = `${page.title}
|
|
150
|
+
${page.text}`.toLowerCase();
|
|
151
|
+
const reasons = [];
|
|
152
|
+
if (lower.includes(query.toLowerCase())) reasons.push("phrase");
|
|
153
|
+
if (page.sourceIds.length > 0) reasons.push("sourced");
|
|
154
|
+
if (page.outLinks.length > 0) reasons.push("linked");
|
|
155
|
+
return reasons;
|
|
12
156
|
}
|
|
13
157
|
|
|
14
158
|
// src/frontmatter.ts
|
|
@@ -63,7 +207,7 @@ function parseSimpleYaml(raw) {
|
|
|
63
207
|
}
|
|
64
208
|
function formatYamlField(key, value) {
|
|
65
209
|
if (Array.isArray(value)) {
|
|
66
|
-
return [key
|
|
210
|
+
return [`${key}:`, ...value.map((item) => ` - ${String(item)}`)];
|
|
67
211
|
}
|
|
68
212
|
if (typeof value === "string") return [`${key}: ${value}`];
|
|
69
213
|
if (typeof value === "number" || typeof value === "boolean") return [`${key}: ${String(value)}`];
|
|
@@ -88,238 +232,6 @@ function normalizeLinkTarget(target) {
|
|
|
88
232
|
return target.trim().replace(/\.md$/i, "").toLowerCase().replace(/\s+/g, "-");
|
|
89
233
|
}
|
|
90
234
|
|
|
91
|
-
// src/write-protocol.ts
|
|
92
|
-
var OPENER_LINE = /^---\s*FILE:\s*(.+?)\s*---\s*$/i;
|
|
93
|
-
var CLOSER_LINE = /^---\s*END\s+FILE\s*---\s*$/i;
|
|
94
|
-
var FENCE_LINE = /^\s{0,3}(```+|~~~+)/;
|
|
95
|
-
function isSafeKnowledgePath(path, allowedPrefixes = ["knowledge/"]) {
|
|
96
|
-
if (typeof path !== "string" || path.trim() === "") return false;
|
|
97
|
-
if (/[\x00-\x1f]/.test(path)) return false;
|
|
98
|
-
if (path.startsWith("/") || path.startsWith("\\")) return false;
|
|
99
|
-
if (/^[a-zA-Z]:/.test(path)) return false;
|
|
100
|
-
const normalized = path.replace(/\\/g, "/");
|
|
101
|
-
if (normalized.split("/").some((part) => part === "..")) return false;
|
|
102
|
-
return allowedPrefixes.some((prefix) => normalized.startsWith(prefix));
|
|
103
|
-
}
|
|
104
|
-
function parseKnowledgeWriteBlocks(text, allowedPrefixes = ["knowledge/"]) {
|
|
105
|
-
const lines = text.replace(/\r\n/g, "\n").split("\n");
|
|
106
|
-
const blocks = [];
|
|
107
|
-
const warnings = [];
|
|
108
|
-
let i = 0;
|
|
109
|
-
while (i < lines.length) {
|
|
110
|
-
const opener = OPENER_LINE.exec(lines[i]);
|
|
111
|
-
if (!opener) {
|
|
112
|
-
i++;
|
|
113
|
-
continue;
|
|
114
|
-
}
|
|
115
|
-
const path = opener[1].trim();
|
|
116
|
-
i++;
|
|
117
|
-
const contentLines = [];
|
|
118
|
-
let fenceMarker = null;
|
|
119
|
-
let fenceLen = 0;
|
|
120
|
-
let closed = false;
|
|
121
|
-
while (i < lines.length) {
|
|
122
|
-
const line = lines[i];
|
|
123
|
-
const fence = FENCE_LINE.exec(line);
|
|
124
|
-
if (fence) {
|
|
125
|
-
const run = fence[1];
|
|
126
|
-
const char = run[0];
|
|
127
|
-
if (fenceMarker === null) {
|
|
128
|
-
fenceMarker = char;
|
|
129
|
-
fenceLen = run.length;
|
|
130
|
-
} else if (char === fenceMarker && run.length >= fenceLen) {
|
|
131
|
-
fenceMarker = null;
|
|
132
|
-
fenceLen = 0;
|
|
133
|
-
}
|
|
134
|
-
contentLines.push(line);
|
|
135
|
-
i++;
|
|
136
|
-
continue;
|
|
137
|
-
}
|
|
138
|
-
if (fenceMarker === null && CLOSER_LINE.test(line)) {
|
|
139
|
-
closed = true;
|
|
140
|
-
i++;
|
|
141
|
-
break;
|
|
142
|
-
}
|
|
143
|
-
contentLines.push(line);
|
|
144
|
-
i++;
|
|
145
|
-
}
|
|
146
|
-
if (!closed) {
|
|
147
|
-
warnings.push(`FILE block "${path || "(empty)"}" was not closed before end of stream.`);
|
|
148
|
-
continue;
|
|
149
|
-
}
|
|
150
|
-
if (!isSafeKnowledgePath(path, allowedPrefixes)) {
|
|
151
|
-
warnings.push(`FILE block with unsafe path "${path}" rejected.`);
|
|
152
|
-
continue;
|
|
153
|
-
}
|
|
154
|
-
blocks.push({ path, content: contentLines.join("\n") });
|
|
155
|
-
}
|
|
156
|
-
return { blocks, warnings };
|
|
157
|
-
}
|
|
158
|
-
|
|
159
|
-
// src/adapters.ts
|
|
160
|
-
var textSourceAdapter = {
|
|
161
|
-
id: "text",
|
|
162
|
-
canLoad: (input) => Boolean(input.text) || /\.(md|txt|json|csv)$/i.test(input.uri),
|
|
163
|
-
load: (input) => ({
|
|
164
|
-
title: input.uri.split("/").pop(),
|
|
165
|
-
mediaType: mediaTypeFor(input.uri),
|
|
166
|
-
text: decodeText(input),
|
|
167
|
-
anchors: anchorsForText(input.uri, decodeText(input)),
|
|
168
|
-
metadata: input.metadata
|
|
169
|
-
})
|
|
170
|
-
};
|
|
171
|
-
function mediaTypeFor(uri) {
|
|
172
|
-
const lower = uri.toLowerCase();
|
|
173
|
-
if (lower.endsWith(".md")) return "text/markdown";
|
|
174
|
-
if (lower.endsWith(".txt")) return "text/plain";
|
|
175
|
-
if (lower.endsWith(".json")) return "application/json";
|
|
176
|
-
if (lower.endsWith(".csv")) return "text/csv";
|
|
177
|
-
if (lower.endsWith(".pdf")) return "application/pdf";
|
|
178
|
-
return "application/octet-stream";
|
|
179
|
-
}
|
|
180
|
-
function decodeText(input) {
|
|
181
|
-
return input.text ?? (input.bytes ? new TextDecoder().decode(input.bytes).slice(0, 2e5) : void 0);
|
|
182
|
-
}
|
|
183
|
-
function anchorsForText(uri, text) {
|
|
184
|
-
if (!text) return [];
|
|
185
|
-
const lines = text.split("\n");
|
|
186
|
-
const anchors = [{ id: "all", sourceId: "", label: "Full source", lineStart: 1, lineEnd: lines.length }];
|
|
187
|
-
for (let i = 0; i < lines.length; i += 50) {
|
|
188
|
-
anchors.push({
|
|
189
|
-
id: `l${i + 1}`,
|
|
190
|
-
sourceId: "",
|
|
191
|
-
label: `${uri}:${i + 1}`,
|
|
192
|
-
lineStart: i + 1,
|
|
193
|
-
lineEnd: Math.min(lines.length, i + 50)
|
|
194
|
-
});
|
|
195
|
-
}
|
|
196
|
-
return anchors;
|
|
197
|
-
}
|
|
198
|
-
|
|
199
|
-
// src/proposals.ts
|
|
200
|
-
import { mkdir, readFile, writeFile } from "fs/promises";
|
|
201
|
-
import { dirname, join } from "path";
|
|
202
|
-
async function applyKnowledgeWriteBlocks(root, proposalText) {
|
|
203
|
-
const parsed = parseKnowledgeWriteBlocks(proposalText);
|
|
204
|
-
const written = [];
|
|
205
|
-
for (const block of parsed.blocks) {
|
|
206
|
-
const path = join(root, block.path);
|
|
207
|
-
await mkdir(dirname(path), { recursive: true });
|
|
208
|
-
await writeFile(path, block.content.endsWith("\n") ? block.content : `${block.content}
|
|
209
|
-
`, "utf8");
|
|
210
|
-
written.push(block.path);
|
|
211
|
-
}
|
|
212
|
-
return { written, warnings: parsed.warnings };
|
|
213
|
-
}
|
|
214
|
-
async function applyKnowledgeWriteBlocksFile(root, proposalPath) {
|
|
215
|
-
return applyKnowledgeWriteBlocks(root, await readFile(proposalPath, "utf8"));
|
|
216
|
-
}
|
|
217
|
-
|
|
218
|
-
// src/schemas.ts
|
|
219
|
-
import { z } from "zod";
|
|
220
|
-
var SourceAnchorSchema = z.object({
|
|
221
|
-
id: z.string().min(1),
|
|
222
|
-
sourceId: z.string().min(1),
|
|
223
|
-
label: z.string().optional(),
|
|
224
|
-
page: z.number().int().positive().optional(),
|
|
225
|
-
lineStart: z.number().int().positive().optional(),
|
|
226
|
-
lineEnd: z.number().int().positive().optional(),
|
|
227
|
-
charStart: z.number().int().nonnegative().optional(),
|
|
228
|
-
charEnd: z.number().int().nonnegative().optional(),
|
|
229
|
-
timestampMs: z.number().nonnegative().optional(),
|
|
230
|
-
metadata: z.record(z.string(), z.unknown()).optional()
|
|
231
|
-
});
|
|
232
|
-
var SourceRecordSchema = z.object({
|
|
233
|
-
id: z.string().min(1),
|
|
234
|
-
uri: z.string().min(1),
|
|
235
|
-
title: z.string().optional(),
|
|
236
|
-
mediaType: z.string().optional(),
|
|
237
|
-
contentHash: z.string().min(16),
|
|
238
|
-
text: z.string().optional(),
|
|
239
|
-
anchors: z.array(SourceAnchorSchema).optional(),
|
|
240
|
-
metadata: z.record(z.string(), z.unknown()).optional(),
|
|
241
|
-
createdAt: z.string().min(1)
|
|
242
|
-
});
|
|
243
|
-
var KnowledgePageSchema = z.object({
|
|
244
|
-
id: z.string().min(1),
|
|
245
|
-
path: z.string().min(1),
|
|
246
|
-
title: z.string().min(1),
|
|
247
|
-
text: z.string(),
|
|
248
|
-
frontmatter: z.record(z.string(), z.unknown()),
|
|
249
|
-
sourceIds: z.array(z.string()),
|
|
250
|
-
tags: z.array(z.string()),
|
|
251
|
-
outLinks: z.array(z.string())
|
|
252
|
-
});
|
|
253
|
-
var KnowledgeGraphNodeSchema = z.object({
|
|
254
|
-
id: z.string(),
|
|
255
|
-
title: z.string(),
|
|
256
|
-
path: z.string(),
|
|
257
|
-
tags: z.array(z.string()),
|
|
258
|
-
sourceIds: z.array(z.string()),
|
|
259
|
-
outDegree: z.number().int().nonnegative(),
|
|
260
|
-
inDegree: z.number().int().nonnegative()
|
|
261
|
-
});
|
|
262
|
-
var KnowledgeGraphEdgeSchema = z.object({
|
|
263
|
-
source: z.string(),
|
|
264
|
-
target: z.string(),
|
|
265
|
-
weight: z.number(),
|
|
266
|
-
reasons: z.array(z.string())
|
|
267
|
-
});
|
|
268
|
-
var KnowledgeIndexSchema = z.object({
|
|
269
|
-
root: z.string(),
|
|
270
|
-
generatedAt: z.string(),
|
|
271
|
-
sources: z.array(SourceRecordSchema),
|
|
272
|
-
pages: z.array(KnowledgePageSchema),
|
|
273
|
-
graph: z.object({
|
|
274
|
-
nodes: z.array(KnowledgeGraphNodeSchema),
|
|
275
|
-
edges: z.array(KnowledgeGraphEdgeSchema)
|
|
276
|
-
})
|
|
277
|
-
});
|
|
278
|
-
var KnowledgeEventSchema = z.object({
|
|
279
|
-
id: z.string().min(1),
|
|
280
|
-
type: z.enum(["source.added", "proposal.applied", "index.built", "lint.run", "optimization.run", "release.promoted", "release.rejected"]),
|
|
281
|
-
createdAt: z.string().min(1),
|
|
282
|
-
actor: z.string().optional(),
|
|
283
|
-
target: z.string().optional(),
|
|
284
|
-
metadata: z.record(z.string(), z.unknown()).optional()
|
|
285
|
-
});
|
|
286
|
-
var KnowledgeBaseCandidateSchema = z.object({
|
|
287
|
-
id: z.string().min(1),
|
|
288
|
-
units: z.array(z.object({
|
|
289
|
-
id: z.string().min(1),
|
|
290
|
-
title: z.string().min(1),
|
|
291
|
-
text: z.string(),
|
|
292
|
-
claims: z.array(z.object({
|
|
293
|
-
id: z.string().min(1),
|
|
294
|
-
text: z.string().min(1),
|
|
295
|
-
refs: z.array(z.object({
|
|
296
|
-
sourceId: z.string().min(1),
|
|
297
|
-
anchorId: z.string().optional(),
|
|
298
|
-
quote: z.string().optional()
|
|
299
|
-
})),
|
|
300
|
-
confidence: z.number().min(0).max(1).optional(),
|
|
301
|
-
status: z.enum(["draft", "active", "superseded", "rejected"]).optional(),
|
|
302
|
-
metadata: z.record(z.string(), z.unknown()).optional()
|
|
303
|
-
})).optional(),
|
|
304
|
-
relations: z.array(z.object({
|
|
305
|
-
sourceId: z.string(),
|
|
306
|
-
targetId: z.string(),
|
|
307
|
-
predicate: z.string(),
|
|
308
|
-
weight: z.number().optional(),
|
|
309
|
-
metadata: z.record(z.string(), z.unknown()).optional()
|
|
310
|
-
})).optional(),
|
|
311
|
-
sourceIds: z.array(z.string()).optional(),
|
|
312
|
-
tags: z.array(z.string()).optional(),
|
|
313
|
-
metadata: z.record(z.string(), z.unknown()).optional(),
|
|
314
|
-
updatedAt: z.string().optional()
|
|
315
|
-
})),
|
|
316
|
-
retrievalPolicy: z.string().optional(),
|
|
317
|
-
synthesisPolicy: z.string().optional(),
|
|
318
|
-
questionPolicy: z.string().optional(),
|
|
319
|
-
updatePolicy: z.string().optional(),
|
|
320
|
-
metadata: z.record(z.string(), z.unknown()).optional()
|
|
321
|
-
});
|
|
322
|
-
|
|
323
235
|
// src/graph.ts
|
|
324
236
|
function buildKnowledgeGraph(pages) {
|
|
325
237
|
const byId = /* @__PURE__ */ new Map();
|
|
@@ -344,7 +256,13 @@ function buildKnowledgeGraph(pages) {
|
|
|
344
256
|
const key = `${page.id}->${target.id}`;
|
|
345
257
|
const edge = edgesByKey.get(key);
|
|
346
258
|
if (edge) edge.weight += 1;
|
|
347
|
-
else
|
|
259
|
+
else
|
|
260
|
+
edgesByKey.set(key, {
|
|
261
|
+
source: page.id,
|
|
262
|
+
target: target.id,
|
|
263
|
+
weight: 1,
|
|
264
|
+
reasons: ["wikilink"]
|
|
265
|
+
});
|
|
348
266
|
outgoing.set(page.id, (outgoing.get(page.id) ?? 0) + 1);
|
|
349
267
|
incoming.set(target.id, (incoming.get(target.id) ?? 0) + 1);
|
|
350
268
|
}
|
|
@@ -386,17 +304,17 @@ function addSourceOverlapEdges(pages, edges) {
|
|
|
386
304
|
}
|
|
387
305
|
|
|
388
306
|
// src/store.ts
|
|
389
|
-
import { mkdir
|
|
390
|
-
import { dirname
|
|
307
|
+
import { mkdir, readdir, readFile, stat, writeFile } from "fs/promises";
|
|
308
|
+
import { dirname, join, relative } from "path";
|
|
391
309
|
function layoutFor(root) {
|
|
392
310
|
return {
|
|
393
311
|
root,
|
|
394
|
-
knowledgeDir:
|
|
395
|
-
rawSourcesDir:
|
|
396
|
-
sourceRegistryPath:
|
|
397
|
-
indexPath:
|
|
398
|
-
logPath:
|
|
399
|
-
cacheDir:
|
|
312
|
+
knowledgeDir: join(root, "knowledge"),
|
|
313
|
+
rawSourcesDir: join(root, "raw", "sources"),
|
|
314
|
+
sourceRegistryPath: join(root, ".agent-knowledge", "sources.json"),
|
|
315
|
+
indexPath: join(root, "knowledge", "index.md"),
|
|
316
|
+
logPath: join(root, "knowledge", "log.md"),
|
|
317
|
+
cacheDir: join(root, ".agent-knowledge")
|
|
400
318
|
};
|
|
401
319
|
}
|
|
402
320
|
var SCAFFOLD_PAGE_BASENAMES = ["index.md", "log.md"];
|
|
@@ -410,12 +328,15 @@ function isScaffoldPath(path) {
|
|
|
410
328
|
}
|
|
411
329
|
async function initKnowledgeBase(root) {
|
|
412
330
|
const layout = layoutFor(root);
|
|
413
|
-
await
|
|
414
|
-
await
|
|
415
|
-
await
|
|
331
|
+
await mkdir(layout.knowledgeDir, { recursive: true });
|
|
332
|
+
await mkdir(layout.rawSourcesDir, { recursive: true });
|
|
333
|
+
await mkdir(layout.cacheDir, { recursive: true });
|
|
416
334
|
await writeIfMissing(layout.indexPath, "# Knowledge Index\n\n");
|
|
417
335
|
await writeIfMissing(layout.logPath, "# Knowledge Log\n\n");
|
|
418
|
-
await writeIfMissing(
|
|
336
|
+
await writeIfMissing(
|
|
337
|
+
layout.sourceRegistryPath,
|
|
338
|
+
'{\n "generatedAt": "1970-01-01T00:00:00.000Z",\n "sources": []\n}\n'
|
|
339
|
+
);
|
|
419
340
|
return layout;
|
|
420
341
|
}
|
|
421
342
|
async function loadKnowledgePages(root) {
|
|
@@ -425,7 +346,7 @@ async function loadKnowledgePages(root) {
|
|
|
425
346
|
for (const file of files) {
|
|
426
347
|
const rel = relative(root, file).replace(/\\/g, "/");
|
|
427
348
|
if (isScaffoldPath(rel)) continue;
|
|
428
|
-
const content = await
|
|
349
|
+
const content = await readFile(file, "utf8");
|
|
429
350
|
const { frontmatter, body } = parseFrontmatter(content);
|
|
430
351
|
const title = stringField(frontmatter.title) ?? firstHeading(body) ?? rel.split("/").pop().replace(/\.md$/, "");
|
|
431
352
|
const sourceIds = arrayField(frontmatter.sources);
|
|
@@ -445,15 +366,16 @@ async function loadKnowledgePages(root) {
|
|
|
445
366
|
return pages;
|
|
446
367
|
}
|
|
447
368
|
async function writeJson(path, value) {
|
|
448
|
-
await
|
|
449
|
-
await
|
|
369
|
+
await mkdir(dirname(path), { recursive: true });
|
|
370
|
+
await writeFile(path, `${JSON.stringify(value, null, 2)}
|
|
371
|
+
`, "utf8");
|
|
450
372
|
}
|
|
451
373
|
async function writeIfMissing(path, content) {
|
|
452
374
|
try {
|
|
453
375
|
await stat(path);
|
|
454
376
|
} catch {
|
|
455
|
-
await
|
|
456
|
-
await
|
|
377
|
+
await mkdir(dirname(path), { recursive: true });
|
|
378
|
+
await writeFile(path, content, "utf8");
|
|
457
379
|
}
|
|
458
380
|
}
|
|
459
381
|
async function listMarkdownFiles(root) {
|
|
@@ -461,7 +383,7 @@ async function listMarkdownFiles(root) {
|
|
|
461
383
|
const entries = await readdir(root, { withFileTypes: true });
|
|
462
384
|
const out = [];
|
|
463
385
|
for (const entry of entries) {
|
|
464
|
-
const full =
|
|
386
|
+
const full = join(root, entry.name);
|
|
465
387
|
if (entry.isDirectory()) out.push(...await listMarkdownFiles(full));
|
|
466
388
|
else if (entry.isFile() && entry.name.endsWith(".md")) out.push(full);
|
|
467
389
|
}
|
|
@@ -481,12 +403,12 @@ function firstHeading(body) {
|
|
|
481
403
|
}
|
|
482
404
|
|
|
483
405
|
// src/sources.ts
|
|
484
|
-
import { copyFile, mkdir as
|
|
485
|
-
import { basename, dirname as
|
|
406
|
+
import { copyFile, mkdir as mkdir2, readdir as readdir2, readFile as readFile2, stat as stat2, writeFile as writeFile2 } from "fs/promises";
|
|
407
|
+
import { basename, dirname as dirname2, join as join2, relative as relative2 } from "path";
|
|
486
408
|
async function loadSourceRegistry(root) {
|
|
487
409
|
const path = sourceRegistryPath(root);
|
|
488
410
|
try {
|
|
489
|
-
const parsed = JSON.parse(await
|
|
411
|
+
const parsed = JSON.parse(await readFile2(path, "utf8"));
|
|
490
412
|
return {
|
|
491
413
|
generatedAt: typeof parsed.generatedAt === "string" ? parsed.generatedAt : (/* @__PURE__ */ new Date(0)).toISOString(),
|
|
492
414
|
sources: Array.isArray(parsed.sources) ? parsed.sources : []
|
|
@@ -497,8 +419,9 @@ async function loadSourceRegistry(root) {
|
|
|
497
419
|
}
|
|
498
420
|
async function writeSourceRegistry(root, registry) {
|
|
499
421
|
const path = sourceRegistryPath(root);
|
|
500
|
-
await
|
|
501
|
-
await
|
|
422
|
+
await mkdir2(dirname2(path), { recursive: true });
|
|
423
|
+
await writeFile2(path, `${JSON.stringify(registry, null, 2)}
|
|
424
|
+
`, "utf8");
|
|
502
425
|
}
|
|
503
426
|
async function addSourcePath(root, sourcePath, options = {}) {
|
|
504
427
|
const s = await stat2(sourcePath);
|
|
@@ -510,18 +433,22 @@ async function addSourcePath(root, sourcePath, options = {}) {
|
|
|
510
433
|
return out;
|
|
511
434
|
}
|
|
512
435
|
const layout = layoutFor(root);
|
|
513
|
-
await
|
|
514
|
-
const bytes = await
|
|
436
|
+
await mkdir2(layout.rawSourcesDir, { recursive: true });
|
|
437
|
+
const bytes = await readFile2(sourcePath);
|
|
515
438
|
const contentHash = sha256(bytes.toString("base64"));
|
|
516
439
|
const fileName = basename(sourcePath);
|
|
517
440
|
const adapters = options.adapters ?? [textSourceAdapter];
|
|
518
441
|
const adapter = adapters.find((candidate) => candidate.canLoad({ uri: sourcePath, bytes }));
|
|
519
442
|
const loaded = adapter ? await adapter.load({ uri: sourcePath, bytes }) : {};
|
|
520
443
|
const id = stableId("src", `${contentHash}:${fileName}`);
|
|
521
|
-
const targetRel =
|
|
522
|
-
|
|
444
|
+
const targetRel = join2(
|
|
445
|
+
"raw",
|
|
446
|
+
"sources",
|
|
447
|
+
`${slugify(fileName.replace(/\.[^.]+$/, ""))}-${contentHash.slice(0, 8)}${ext(fileName)}`
|
|
448
|
+
).replace(/\\/g, "/");
|
|
449
|
+
const targetAbs = join2(root, targetRel);
|
|
523
450
|
if (options.copyIntoRaw ?? true) {
|
|
524
|
-
await
|
|
451
|
+
await mkdir2(dirname2(targetAbs), { recursive: true });
|
|
525
452
|
await copyFile(sourcePath, targetAbs);
|
|
526
453
|
}
|
|
527
454
|
const existing = await loadSourceRegistry(root);
|
|
@@ -548,129 +475,84 @@ async function addSourcePath(root, sourcePath, options = {}) {
|
|
|
548
475
|
await writeSourceRegistry(root, next);
|
|
549
476
|
return [record];
|
|
550
477
|
}
|
|
551
|
-
function
|
|
552
|
-
|
|
553
|
-
|
|
554
|
-
|
|
555
|
-
const
|
|
556
|
-
const
|
|
557
|
-
|
|
558
|
-
|
|
559
|
-
|
|
560
|
-
|
|
561
|
-
|
|
562
|
-
|
|
563
|
-
|
|
564
|
-
|
|
565
|
-
|
|
566
|
-
|
|
567
|
-
}
|
|
568
|
-
|
|
569
|
-
|
|
570
|
-
|
|
571
|
-
|
|
572
|
-
|
|
573
|
-
|
|
574
|
-
|
|
575
|
-
|
|
576
|
-
|
|
577
|
-
|
|
578
|
-
|
|
579
|
-
|
|
580
|
-
|
|
581
|
-
|
|
582
|
-
|
|
583
|
-
|
|
584
|
-
|
|
585
|
-
|
|
586
|
-
|
|
587
|
-
|
|
588
|
-
|
|
589
|
-
|
|
590
|
-
|
|
591
|
-
|
|
592
|
-
|
|
593
|
-
|
|
594
|
-
|
|
595
|
-
|
|
596
|
-
|
|
597
|
-
|
|
598
|
-
|
|
599
|
-
|
|
600
|
-
|
|
601
|
-
|
|
602
|
-
|
|
603
|
-
|
|
604
|
-
|
|
605
|
-
function tokenizeQuery(query) {
|
|
606
|
-
const raw = query.toLowerCase().split(/[\s,,。!?、;:""''()()\-_/\\·~~…]+/).filter((token) => token.length > 1 && !STOP_WORDS.has(token));
|
|
607
|
-
const tokens = [];
|
|
608
|
-
for (const token of raw) {
|
|
609
|
-
if (/[\u4e00-\u9fff\u3400-\u4dbf]/.test(token) && token.length > 2) {
|
|
610
|
-
const chars = [...token];
|
|
611
|
-
for (let i = 0; i < chars.length - 1; i++) tokens.push(chars[i] + chars[i + 1]);
|
|
612
|
-
tokens.push(...chars);
|
|
613
|
-
}
|
|
614
|
-
tokens.push(token);
|
|
615
|
-
}
|
|
616
|
-
return [...new Set(tokens)];
|
|
617
|
-
}
|
|
618
|
-
function reciprocalRankFusion(rankLists, k = RRF_K) {
|
|
619
|
-
const scores = /* @__PURE__ */ new Map();
|
|
620
|
-
for (const list of rankLists) {
|
|
621
|
-
list.forEach((id, idx) => {
|
|
622
|
-
scores.set(id, (scores.get(id) ?? 0) + 1 / (k + idx + 1));
|
|
623
|
-
});
|
|
624
|
-
}
|
|
625
|
-
return scores;
|
|
626
|
-
}
|
|
627
|
-
function rankByTokens(pages, query) {
|
|
628
|
-
const tokens = tokenizeQuery(query);
|
|
629
|
-
const effective = tokens.length > 0 ? tokens : [query.toLowerCase()];
|
|
630
|
-
return pages.map((page) => ({ page, score: tokenScore(page, query, effective) })).filter((item) => item.score > 0).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).map((item) => item.page);
|
|
631
|
-
}
|
|
632
|
-
function rankByGraph(pages, tokenRanked) {
|
|
633
|
-
if (tokenRanked.length === 0) return [];
|
|
634
|
-
const seeds = new Set(tokenRanked.slice(0, 5).map((page) => page.id));
|
|
635
|
-
return pages.map((page) => ({
|
|
636
|
-
page,
|
|
637
|
-
score: page.outLinks.filter((link) => seeds.has(link)).length + page.sourceIds.filter((source) => tokenRanked.some((seed) => seed.sourceIds.includes(source))).length
|
|
638
|
-
})).filter((item) => item.score > 0).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).map((item) => item.page);
|
|
639
|
-
}
|
|
640
|
-
function tokenScore(page, query, tokens) {
|
|
641
|
-
const title = page.title.toLowerCase();
|
|
642
|
-
const path = page.path.toLowerCase();
|
|
643
|
-
const body = page.text.toLowerCase();
|
|
644
|
-
const phrase = query.toLowerCase();
|
|
645
|
-
let score = 0;
|
|
646
|
-
if (path.endsWith(`${phrase}.md`) || title === phrase) score += 200;
|
|
647
|
-
if (title.includes(phrase)) score += 50;
|
|
648
|
-
if (body.includes(phrase)) score += 20;
|
|
649
|
-
for (const token of tokens) {
|
|
650
|
-
if (title.includes(token)) score += 5;
|
|
651
|
-
if (body.includes(token)) score += 1;
|
|
652
|
-
if (path.includes(token)) score += 3;
|
|
478
|
+
async function addSourceText(root, input, options = {}) {
|
|
479
|
+
const text = input.text;
|
|
480
|
+
const contentHash = sha256(text);
|
|
481
|
+
const fileName = basename(input.uri) || `${slugify(input.title ?? input.uri)}.txt`;
|
|
482
|
+
const adapterInput = { uri: input.uri, text, metadata: input.metadata };
|
|
483
|
+
const adapter = (options.adapters ?? [textSourceAdapter]).find(
|
|
484
|
+
(candidate) => candidate.canLoad(adapterInput)
|
|
485
|
+
);
|
|
486
|
+
const loaded = adapter ? await adapter.load(adapterInput) : {};
|
|
487
|
+
const id = stableId("src", `${contentHash}:${input.uri}`);
|
|
488
|
+
const targetRel = join2(
|
|
489
|
+
"raw",
|
|
490
|
+
"sources",
|
|
491
|
+
`${slugify(fileName.replace(/\.[^.]+$/, ""))}-${contentHash.slice(0, 8)}.txt`
|
|
492
|
+
).replace(/\\/g, "/");
|
|
493
|
+
const targetAbs = join2(root, targetRel);
|
|
494
|
+
await mkdir2(dirname2(targetAbs), { recursive: true });
|
|
495
|
+
await writeFile2(targetAbs, text.endsWith("\n") ? text : `${text}
|
|
496
|
+
`, "utf8");
|
|
497
|
+
const existing = await loadSourceRegistry(root);
|
|
498
|
+
const record = {
|
|
499
|
+
id,
|
|
500
|
+
uri: targetRel,
|
|
501
|
+
title: input.title ?? loaded.title ?? fileName,
|
|
502
|
+
mediaType: input.mediaType ?? loaded.mediaType ?? "text/plain",
|
|
503
|
+
contentHash,
|
|
504
|
+
text: loaded.text ?? text.slice(0, 2e5),
|
|
505
|
+
anchors: (loaded.anchors ?? []).map((anchor) => ({ ...anchor, sourceId: id })),
|
|
506
|
+
validUntil: input.validUntil,
|
|
507
|
+
lastVerifiedAt: input.lastVerifiedAt,
|
|
508
|
+
createdAt: (options.now ?? (() => /* @__PURE__ */ new Date()))().toISOString(),
|
|
509
|
+
metadata: {
|
|
510
|
+
...loaded.metadata ?? {},
|
|
511
|
+
...input.metadata ?? {},
|
|
512
|
+
originalUri: input.uri,
|
|
513
|
+
sizeBytes: Buffer.byteLength(text, "utf8")
|
|
514
|
+
}
|
|
515
|
+
};
|
|
516
|
+
await writeSourceRegistry(root, {
|
|
517
|
+
generatedAt: (/* @__PURE__ */ new Date()).toISOString(),
|
|
518
|
+
sources: [record, ...existing.sources.filter((source) => source.id !== id)]
|
|
519
|
+
});
|
|
520
|
+
return record;
|
|
521
|
+
}
|
|
522
|
+
function sourceRegistryPath(root) {
|
|
523
|
+
return join2(layoutFor(root).cacheDir, "sources.json");
|
|
524
|
+
}
|
|
525
|
+
async function listFiles(root) {
|
|
526
|
+
const entries = await readdir2(root, { withFileTypes: true });
|
|
527
|
+
const out = [];
|
|
528
|
+
for (const entry of entries) {
|
|
529
|
+
const full = join2(root, entry.name);
|
|
530
|
+
if (entry.isDirectory()) out.push(...await listFiles(full));
|
|
531
|
+
else if (entry.isFile()) out.push(full);
|
|
653
532
|
}
|
|
654
|
-
return
|
|
533
|
+
return out;
|
|
655
534
|
}
|
|
656
|
-
function
|
|
657
|
-
const
|
|
658
|
-
|
|
659
|
-
if (idx < 0) return compact.slice(0, 180);
|
|
660
|
-
return compact.slice(Math.max(0, idx - 80), Math.min(compact.length, idx + query.length + 100));
|
|
535
|
+
function ext(fileName) {
|
|
536
|
+
const idx = fileName.lastIndexOf(".");
|
|
537
|
+
return idx >= 0 ? fileName.slice(idx) : "";
|
|
661
538
|
}
|
|
662
|
-
function
|
|
663
|
-
const lower =
|
|
664
|
-
|
|
665
|
-
|
|
666
|
-
if (lower.
|
|
667
|
-
if (
|
|
668
|
-
if (
|
|
669
|
-
return
|
|
539
|
+
function mediaTypeFor2(fileName) {
|
|
540
|
+
const lower = fileName.toLowerCase();
|
|
541
|
+
if (lower.endsWith(".md")) return "text/markdown";
|
|
542
|
+
if (lower.endsWith(".txt")) return "text/plain";
|
|
543
|
+
if (lower.endsWith(".json")) return "application/json";
|
|
544
|
+
if (lower.endsWith(".csv")) return "text/csv";
|
|
545
|
+
if (lower.endsWith(".pdf")) return "application/pdf";
|
|
546
|
+
return "application/octet-stream";
|
|
547
|
+
}
|
|
548
|
+
function textPreview(fileName, bytes) {
|
|
549
|
+
const mediaType = mediaTypeFor2(fileName);
|
|
550
|
+
if (!mediaType.startsWith("text/") && mediaType !== "application/json") return void 0;
|
|
551
|
+
return bytes.toString("utf8").slice(0, 2e5);
|
|
670
552
|
}
|
|
671
553
|
|
|
672
554
|
// src/indexer.ts
|
|
673
|
-
import { join as
|
|
555
|
+
import { join as join3 } from "path";
|
|
674
556
|
async function buildKnowledgeIndex(root) {
|
|
675
557
|
const [pages, sourceRegistry] = await Promise.all([
|
|
676
558
|
loadKnowledgePages(root),
|
|
@@ -687,7 +569,7 @@ async function buildKnowledgeIndex(root) {
|
|
|
687
569
|
}
|
|
688
570
|
async function writeKnowledgeIndex(root) {
|
|
689
571
|
const index = await buildKnowledgeIndex(root);
|
|
690
|
-
await writeJson(
|
|
572
|
+
await writeJson(join3(layoutFor(root).cacheDir, "index.json"), index);
|
|
691
573
|
return index;
|
|
692
574
|
}
|
|
693
575
|
|
|
@@ -697,7 +579,12 @@ function lintKnowledgeIndex(index) {
|
|
|
697
579
|
const byTarget = /* @__PURE__ */ new Set();
|
|
698
580
|
const titles = /* @__PURE__ */ new Map();
|
|
699
581
|
const sourceIds = new Set(index.sources.map((source) => source.id));
|
|
700
|
-
const anchorIds = new Map(
|
|
582
|
+
const anchorIds = new Map(
|
|
583
|
+
index.sources.map((source) => [
|
|
584
|
+
source.id,
|
|
585
|
+
new Set((source.anchors ?? []).map((anchor) => anchor.id))
|
|
586
|
+
])
|
|
587
|
+
);
|
|
701
588
|
const pageIds = /* @__PURE__ */ new Map();
|
|
702
589
|
const sourceHashes = /* @__PURE__ */ new Map();
|
|
703
590
|
for (const page of index.pages) {
|
|
@@ -709,54 +596,111 @@ function lintKnowledgeIndex(index) {
|
|
|
709
596
|
titles.set(titleKey, [...titles.get(titleKey) ?? [], page.path]);
|
|
710
597
|
}
|
|
711
598
|
for (const source of index.sources) {
|
|
712
|
-
sourceHashes.set(source.contentHash, [
|
|
599
|
+
sourceHashes.set(source.contentHash, [
|
|
600
|
+
...sourceHashes.get(source.contentHash) ?? [],
|
|
601
|
+
source.id
|
|
602
|
+
]);
|
|
713
603
|
}
|
|
714
604
|
const inbound = /* @__PURE__ */ new Map();
|
|
715
605
|
for (const page of index.pages) inbound.set(page.id, 0);
|
|
716
606
|
for (const page of index.pages) {
|
|
717
607
|
if (page.outLinks.length === 0 && !isStructural(page.path)) {
|
|
718
|
-
findings.push({
|
|
608
|
+
findings.push({
|
|
609
|
+
type: "no-outlinks",
|
|
610
|
+
severity: "info",
|
|
611
|
+
page: page.path,
|
|
612
|
+
message: "Page has no wikilinks to other knowledge pages."
|
|
613
|
+
});
|
|
719
614
|
}
|
|
720
615
|
for (const link of page.outLinks) {
|
|
721
616
|
if (!byTarget.has(normalizeLinkTarget(link))) {
|
|
722
|
-
findings.push({
|
|
617
|
+
findings.push({
|
|
618
|
+
type: "broken-link",
|
|
619
|
+
severity: "warning",
|
|
620
|
+
page: page.path,
|
|
621
|
+
message: `Broken wikilink [[${link}]].`
|
|
622
|
+
});
|
|
723
623
|
}
|
|
724
624
|
}
|
|
725
625
|
}
|
|
726
|
-
for (const edge of index.graph.edges)
|
|
626
|
+
for (const edge of index.graph.edges)
|
|
627
|
+
inbound.set(edge.target, (inbound.get(edge.target) ?? 0) + 1);
|
|
727
628
|
for (const page of index.pages) {
|
|
728
629
|
if (!isStructural(page.path) && (inbound.get(page.id) ?? 0) === 0) {
|
|
729
|
-
findings.push({
|
|
630
|
+
findings.push({
|
|
631
|
+
type: "orphan",
|
|
632
|
+
severity: "info",
|
|
633
|
+
page: page.path,
|
|
634
|
+
message: "No other page links to this page."
|
|
635
|
+
});
|
|
730
636
|
}
|
|
731
637
|
if (/\bclaim\b/i.test(page.text) && page.sourceIds.length === 0) {
|
|
732
|
-
findings.push({
|
|
638
|
+
findings.push({
|
|
639
|
+
type: "uncited-claim",
|
|
640
|
+
severity: "warning",
|
|
641
|
+
page: page.path,
|
|
642
|
+
message: "Page appears to contain claims but has no sources frontmatter."
|
|
643
|
+
});
|
|
733
644
|
}
|
|
734
645
|
for (const sourceId of page.sourceIds) {
|
|
735
646
|
if (!sourceIds.has(sourceId)) {
|
|
736
|
-
findings.push({
|
|
647
|
+
findings.push({
|
|
648
|
+
type: "missing-source",
|
|
649
|
+
severity: "error",
|
|
650
|
+
page: page.path,
|
|
651
|
+
message: `Page cites unknown source "${sourceId}".`,
|
|
652
|
+
metadata: { sourceId }
|
|
653
|
+
});
|
|
737
654
|
}
|
|
738
655
|
}
|
|
739
656
|
for (const ref of extractSourceRefs(page.text)) {
|
|
740
657
|
if (!sourceIds.has(ref.sourceId)) {
|
|
741
|
-
findings.push({
|
|
658
|
+
findings.push({
|
|
659
|
+
type: "missing-source",
|
|
660
|
+
severity: "error",
|
|
661
|
+
page: page.path,
|
|
662
|
+
message: `Page cites unknown source "${ref.sourceId}".`,
|
|
663
|
+
metadata: ref
|
|
664
|
+
});
|
|
742
665
|
} else if (ref.anchorId && !anchorIds.get(ref.sourceId)?.has(ref.anchorId)) {
|
|
743
|
-
findings.push({
|
|
666
|
+
findings.push({
|
|
667
|
+
type: "missing-source",
|
|
668
|
+
severity: "error",
|
|
669
|
+
page: page.path,
|
|
670
|
+
message: `Page cites unknown source anchor "${ref.sourceId}#${ref.anchorId}".`,
|
|
671
|
+
metadata: ref
|
|
672
|
+
});
|
|
744
673
|
}
|
|
745
674
|
}
|
|
746
675
|
}
|
|
747
676
|
for (const [title, paths] of titles) {
|
|
748
677
|
if (title && paths.length > 1) {
|
|
749
|
-
findings.push({
|
|
678
|
+
findings.push({
|
|
679
|
+
type: "duplicate-title",
|
|
680
|
+
severity: "warning",
|
|
681
|
+
message: `Duplicate title "${title}" in ${paths.join(", ")}.`,
|
|
682
|
+
metadata: { paths }
|
|
683
|
+
});
|
|
750
684
|
}
|
|
751
685
|
}
|
|
752
686
|
for (const [id, paths] of pageIds) {
|
|
753
687
|
if (id && paths.length > 1) {
|
|
754
|
-
findings.push({
|
|
688
|
+
findings.push({
|
|
689
|
+
type: "duplicate-page-id",
|
|
690
|
+
severity: "error",
|
|
691
|
+
message: `Duplicate page id "${id}" in ${paths.join(", ")}.`,
|
|
692
|
+
metadata: { paths }
|
|
693
|
+
});
|
|
755
694
|
}
|
|
756
695
|
}
|
|
757
696
|
for (const [hash, ids] of sourceHashes) {
|
|
758
697
|
if (hash && ids.length > 1) {
|
|
759
|
-
findings.push({
|
|
698
|
+
findings.push({
|
|
699
|
+
type: "duplicate-source-hash",
|
|
700
|
+
severity: "warning",
|
|
701
|
+
message: `Duplicate source content hash across ${ids.join(", ")}.`,
|
|
702
|
+
metadata: { sourceIds: ids }
|
|
703
|
+
});
|
|
760
704
|
}
|
|
761
705
|
}
|
|
762
706
|
return findings;
|
|
@@ -788,7 +732,12 @@ function inspectKnowledgeIndex(index, options = {}) {
|
|
|
788
732
|
edgeCount: index.graph.edges.length,
|
|
789
733
|
findingCount: findings.length,
|
|
790
734
|
blockingFindingCount: findings.filter((finding) => finding.severity === "error").length,
|
|
791
|
-
topPages: [...index.pages].sort((a, b) => (degree.get(b.id) ?? 0) - (degree.get(a.id) ?? 0)).slice(0, 10).map((page) => ({
|
|
735
|
+
topPages: [...index.pages].sort((a, b) => (degree.get(b.id) ?? 0) - (degree.get(a.id) ?? 0)).slice(0, 10).map((page) => ({
|
|
736
|
+
path: page.path,
|
|
737
|
+
title: page.title,
|
|
738
|
+
degree: degree.get(page.id) ?? 0,
|
|
739
|
+
sources: page.sourceIds.length
|
|
740
|
+
})),
|
|
792
741
|
sourceFreshness,
|
|
793
742
|
findings
|
|
794
743
|
};
|
|
@@ -804,9 +753,21 @@ function stringMetadata(metadata, key) {
|
|
|
804
753
|
return typeof value === "string" ? value : void 0;
|
|
805
754
|
}
|
|
806
755
|
function explainKnowledgeTarget(index, target) {
|
|
807
|
-
const page = index.pages.find(
|
|
808
|
-
|
|
809
|
-
|
|
756
|
+
const page = index.pages.find(
|
|
757
|
+
(candidate) => candidate.path === target || candidate.id === target || candidate.title.toLowerCase() === target.toLowerCase()
|
|
758
|
+
);
|
|
759
|
+
const inbound = page ? index.graph.edges.filter((edge) => edge.target === page.id).map(
|
|
760
|
+
(edge) => index.pages.find((candidate) => candidate.id === edge.source)?.path ?? edge.source
|
|
761
|
+
) : [];
|
|
762
|
+
const related = page ? searchKnowledge(index, `${page.title} ${page.tags.join(" ")}`, 6).filter((result) => result.page.id !== page.id).map((result) => ({
|
|
763
|
+
path: result.page.path,
|
|
764
|
+
title: result.page.title,
|
|
765
|
+
score: result.score
|
|
766
|
+
})) : searchKnowledge(index, target, 6).map((result) => ({
|
|
767
|
+
path: result.page.path,
|
|
768
|
+
title: result.page.title,
|
|
769
|
+
score: result.score
|
|
770
|
+
}));
|
|
810
771
|
return {
|
|
811
772
|
target,
|
|
812
773
|
page,
|
|
@@ -817,6 +778,219 @@ function explainKnowledgeTarget(index, target) {
|
|
|
817
778
|
};
|
|
818
779
|
}
|
|
819
780
|
|
|
781
|
+
// src/write-protocol.ts
|
|
782
|
+
var OPENER_LINE = /^---\s*FILE:\s*(.+?)\s*---\s*$/i;
|
|
783
|
+
var CLOSER_LINE = /^---\s*END\s+FILE\s*---\s*$/i;
|
|
784
|
+
var FENCE_LINE = /^\s{0,3}(```+|~~~+)/;
|
|
785
|
+
function isSafeKnowledgePath(path, allowedPrefixes = ["knowledge/"]) {
|
|
786
|
+
if (typeof path !== "string" || path.trim() === "") return false;
|
|
787
|
+
const controlRangeRegex = new RegExp(`[${String.fromCharCode(0)}-${String.fromCharCode(31)}]`);
|
|
788
|
+
if (controlRangeRegex.test(path)) return false;
|
|
789
|
+
if (path.startsWith("/") || path.startsWith("\\")) return false;
|
|
790
|
+
if (/^[a-zA-Z]:/.test(path)) return false;
|
|
791
|
+
const normalized = path.replace(/\\/g, "/");
|
|
792
|
+
if (normalized.split("/").some((part) => part === "..")) return false;
|
|
793
|
+
return allowedPrefixes.some((prefix) => normalized.startsWith(prefix));
|
|
794
|
+
}
|
|
795
|
+
function parseKnowledgeWriteBlocks(text, allowedPrefixes = ["knowledge/"]) {
|
|
796
|
+
const lines = text.replace(/\r\n/g, "\n").split("\n");
|
|
797
|
+
const blocks = [];
|
|
798
|
+
const warnings = [];
|
|
799
|
+
let i = 0;
|
|
800
|
+
while (i < lines.length) {
|
|
801
|
+
const opener = OPENER_LINE.exec(lines[i]);
|
|
802
|
+
if (!opener) {
|
|
803
|
+
i++;
|
|
804
|
+
continue;
|
|
805
|
+
}
|
|
806
|
+
const path = opener[1].trim();
|
|
807
|
+
i++;
|
|
808
|
+
const contentLines = [];
|
|
809
|
+
let fenceMarker = null;
|
|
810
|
+
let fenceLen = 0;
|
|
811
|
+
let closed = false;
|
|
812
|
+
while (i < lines.length) {
|
|
813
|
+
const line = lines[i];
|
|
814
|
+
const fence = FENCE_LINE.exec(line);
|
|
815
|
+
if (fence) {
|
|
816
|
+
const run = fence[1];
|
|
817
|
+
const char = run[0];
|
|
818
|
+
if (fenceMarker === null) {
|
|
819
|
+
fenceMarker = char;
|
|
820
|
+
fenceLen = run.length;
|
|
821
|
+
} else if (char === fenceMarker && run.length >= fenceLen) {
|
|
822
|
+
fenceMarker = null;
|
|
823
|
+
fenceLen = 0;
|
|
824
|
+
}
|
|
825
|
+
contentLines.push(line);
|
|
826
|
+
i++;
|
|
827
|
+
continue;
|
|
828
|
+
}
|
|
829
|
+
if (fenceMarker === null && CLOSER_LINE.test(line)) {
|
|
830
|
+
closed = true;
|
|
831
|
+
i++;
|
|
832
|
+
break;
|
|
833
|
+
}
|
|
834
|
+
contentLines.push(line);
|
|
835
|
+
i++;
|
|
836
|
+
}
|
|
837
|
+
if (!closed) {
|
|
838
|
+
warnings.push(`FILE block "${path || "(empty)"}" was not closed before end of stream.`);
|
|
839
|
+
continue;
|
|
840
|
+
}
|
|
841
|
+
if (!isSafeKnowledgePath(path, allowedPrefixes)) {
|
|
842
|
+
warnings.push(`FILE block with unsafe path "${path}" rejected.`);
|
|
843
|
+
continue;
|
|
844
|
+
}
|
|
845
|
+
blocks.push({ path, content: contentLines.join("\n") });
|
|
846
|
+
}
|
|
847
|
+
return { blocks, warnings };
|
|
848
|
+
}
|
|
849
|
+
|
|
850
|
+
// src/proposals.ts
|
|
851
|
+
import { mkdir as mkdir3, readFile as readFile3, writeFile as writeFile3 } from "fs/promises";
|
|
852
|
+
import { dirname as dirname3, join as join4 } from "path";
|
|
853
|
+
async function applyKnowledgeWriteBlocks(root, proposalText) {
|
|
854
|
+
const parsed = parseKnowledgeWriteBlocks(proposalText);
|
|
855
|
+
const written = [];
|
|
856
|
+
for (const block of parsed.blocks) {
|
|
857
|
+
const path = join4(root, block.path);
|
|
858
|
+
await mkdir3(dirname3(path), { recursive: true });
|
|
859
|
+
await writeFile3(
|
|
860
|
+
path,
|
|
861
|
+
block.content.endsWith("\n") ? block.content : `${block.content}
|
|
862
|
+
`,
|
|
863
|
+
"utf8"
|
|
864
|
+
);
|
|
865
|
+
written.push(block.path);
|
|
866
|
+
}
|
|
867
|
+
return { written, warnings: parsed.warnings };
|
|
868
|
+
}
|
|
869
|
+
async function applyKnowledgeWriteBlocksFile(root, proposalPath) {
|
|
870
|
+
return applyKnowledgeWriteBlocks(root, await readFile3(proposalPath, "utf8"));
|
|
871
|
+
}
|
|
872
|
+
|
|
873
|
+
// src/schemas.ts
|
|
874
|
+
import { z } from "zod";
|
|
875
|
+
var SourceAnchorSchema = z.object({
|
|
876
|
+
id: z.string().min(1),
|
|
877
|
+
sourceId: z.string().min(1),
|
|
878
|
+
label: z.string().optional(),
|
|
879
|
+
page: z.number().int().positive().optional(),
|
|
880
|
+
lineStart: z.number().int().positive().optional(),
|
|
881
|
+
lineEnd: z.number().int().positive().optional(),
|
|
882
|
+
charStart: z.number().int().nonnegative().optional(),
|
|
883
|
+
charEnd: z.number().int().nonnegative().optional(),
|
|
884
|
+
timestampMs: z.number().nonnegative().optional(),
|
|
885
|
+
metadata: z.record(z.string(), z.unknown()).optional()
|
|
886
|
+
});
|
|
887
|
+
var SourceRecordSchema = z.object({
|
|
888
|
+
id: z.string().min(1),
|
|
889
|
+
uri: z.string().min(1),
|
|
890
|
+
title: z.string().optional(),
|
|
891
|
+
mediaType: z.string().optional(),
|
|
892
|
+
contentHash: z.string().min(16),
|
|
893
|
+
text: z.string().optional(),
|
|
894
|
+
anchors: z.array(SourceAnchorSchema).optional(),
|
|
895
|
+
metadata: z.record(z.string(), z.unknown()).optional(),
|
|
896
|
+
createdAt: z.string().min(1)
|
|
897
|
+
});
|
|
898
|
+
var KnowledgePageSchema = z.object({
|
|
899
|
+
id: z.string().min(1),
|
|
900
|
+
path: z.string().min(1),
|
|
901
|
+
title: z.string().min(1),
|
|
902
|
+
text: z.string(),
|
|
903
|
+
frontmatter: z.record(z.string(), z.unknown()),
|
|
904
|
+
sourceIds: z.array(z.string()),
|
|
905
|
+
tags: z.array(z.string()),
|
|
906
|
+
outLinks: z.array(z.string())
|
|
907
|
+
});
|
|
908
|
+
var KnowledgeGraphNodeSchema = z.object({
|
|
909
|
+
id: z.string(),
|
|
910
|
+
title: z.string(),
|
|
911
|
+
path: z.string(),
|
|
912
|
+
tags: z.array(z.string()),
|
|
913
|
+
sourceIds: z.array(z.string()),
|
|
914
|
+
outDegree: z.number().int().nonnegative(),
|
|
915
|
+
inDegree: z.number().int().nonnegative()
|
|
916
|
+
});
|
|
917
|
+
var KnowledgeGraphEdgeSchema = z.object({
|
|
918
|
+
source: z.string(),
|
|
919
|
+
target: z.string(),
|
|
920
|
+
weight: z.number(),
|
|
921
|
+
reasons: z.array(z.string())
|
|
922
|
+
});
|
|
923
|
+
var KnowledgeIndexSchema = z.object({
|
|
924
|
+
root: z.string(),
|
|
925
|
+
generatedAt: z.string(),
|
|
926
|
+
sources: z.array(SourceRecordSchema),
|
|
927
|
+
pages: z.array(KnowledgePageSchema),
|
|
928
|
+
graph: z.object({
|
|
929
|
+
nodes: z.array(KnowledgeGraphNodeSchema),
|
|
930
|
+
edges: z.array(KnowledgeGraphEdgeSchema)
|
|
931
|
+
})
|
|
932
|
+
});
|
|
933
|
+
var KnowledgeEventSchema = z.object({
|
|
934
|
+
id: z.string().min(1),
|
|
935
|
+
type: z.enum([
|
|
936
|
+
"source.added",
|
|
937
|
+
"proposal.applied",
|
|
938
|
+
"index.built",
|
|
939
|
+
"lint.run",
|
|
940
|
+
"optimization.run",
|
|
941
|
+
"release.promoted",
|
|
942
|
+
"release.rejected"
|
|
943
|
+
]),
|
|
944
|
+
createdAt: z.string().min(1),
|
|
945
|
+
actor: z.string().optional(),
|
|
946
|
+
target: z.string().optional(),
|
|
947
|
+
metadata: z.record(z.string(), z.unknown()).optional()
|
|
948
|
+
});
|
|
949
|
+
var KnowledgeBaseCandidateSchema = z.object({
|
|
950
|
+
id: z.string().min(1),
|
|
951
|
+
units: z.array(
|
|
952
|
+
z.object({
|
|
953
|
+
id: z.string().min(1),
|
|
954
|
+
title: z.string().min(1),
|
|
955
|
+
text: z.string(),
|
|
956
|
+
claims: z.array(
|
|
957
|
+
z.object({
|
|
958
|
+
id: z.string().min(1),
|
|
959
|
+
text: z.string().min(1),
|
|
960
|
+
refs: z.array(
|
|
961
|
+
z.object({
|
|
962
|
+
sourceId: z.string().min(1),
|
|
963
|
+
anchorId: z.string().optional(),
|
|
964
|
+
quote: z.string().optional()
|
|
965
|
+
})
|
|
966
|
+
),
|
|
967
|
+
confidence: z.number().min(0).max(1).optional(),
|
|
968
|
+
status: z.enum(["draft", "active", "superseded", "rejected"]).optional(),
|
|
969
|
+
metadata: z.record(z.string(), z.unknown()).optional()
|
|
970
|
+
})
|
|
971
|
+
).optional(),
|
|
972
|
+
relations: z.array(
|
|
973
|
+
z.object({
|
|
974
|
+
sourceId: z.string(),
|
|
975
|
+
targetId: z.string(),
|
|
976
|
+
predicate: z.string(),
|
|
977
|
+
weight: z.number().optional(),
|
|
978
|
+
metadata: z.record(z.string(), z.unknown()).optional()
|
|
979
|
+
})
|
|
980
|
+
).optional(),
|
|
981
|
+
sourceIds: z.array(z.string()).optional(),
|
|
982
|
+
tags: z.array(z.string()).optional(),
|
|
983
|
+
metadata: z.record(z.string(), z.unknown()).optional(),
|
|
984
|
+
updatedAt: z.string().optional()
|
|
985
|
+
})
|
|
986
|
+
),
|
|
987
|
+
retrievalPolicy: z.string().optional(),
|
|
988
|
+
synthesisPolicy: z.string().optional(),
|
|
989
|
+
questionPolicy: z.string().optional(),
|
|
990
|
+
updatePolicy: z.string().optional(),
|
|
991
|
+
metadata: z.record(z.string(), z.unknown()).optional()
|
|
992
|
+
});
|
|
993
|
+
|
|
820
994
|
// src/validate.ts
|
|
821
995
|
function validateKnowledgeIndex(index, options = {}) {
|
|
822
996
|
const findings = [...lintKnowledgeIndex(index)];
|
|
@@ -848,28 +1022,16 @@ function isStructuralPage(path) {
|
|
|
848
1022
|
}
|
|
849
1023
|
|
|
850
1024
|
export {
|
|
851
|
-
|
|
852
|
-
|
|
853
|
-
|
|
1025
|
+
textSourceAdapter,
|
|
1026
|
+
mediaTypeFor,
|
|
1027
|
+
searchKnowledge,
|
|
1028
|
+
tokenizeQuery,
|
|
1029
|
+
reciprocalRankFusion,
|
|
854
1030
|
parseFrontmatter,
|
|
855
1031
|
formatFrontmatter,
|
|
856
1032
|
WIKILINK_REGEX,
|
|
857
1033
|
extractWikilinks,
|
|
858
1034
|
normalizeLinkTarget,
|
|
859
|
-
isSafeKnowledgePath,
|
|
860
|
-
parseKnowledgeWriteBlocks,
|
|
861
|
-
textSourceAdapter,
|
|
862
|
-
mediaTypeFor,
|
|
863
|
-
applyKnowledgeWriteBlocks,
|
|
864
|
-
applyKnowledgeWriteBlocksFile,
|
|
865
|
-
SourceAnchorSchema,
|
|
866
|
-
SourceRecordSchema,
|
|
867
|
-
KnowledgePageSchema,
|
|
868
|
-
KnowledgeGraphNodeSchema,
|
|
869
|
-
KnowledgeGraphEdgeSchema,
|
|
870
|
-
KnowledgeIndexSchema,
|
|
871
|
-
KnowledgeEventSchema,
|
|
872
|
-
KnowledgeBaseCandidateSchema,
|
|
873
1035
|
buildKnowledgeGraph,
|
|
874
1036
|
layoutFor,
|
|
875
1037
|
SCAFFOLD_PAGE_BASENAMES,
|
|
@@ -880,15 +1042,25 @@ export {
|
|
|
880
1042
|
loadSourceRegistry,
|
|
881
1043
|
writeSourceRegistry,
|
|
882
1044
|
addSourcePath,
|
|
1045
|
+
addSourceText,
|
|
883
1046
|
sourceRegistryPath,
|
|
884
|
-
searchKnowledge,
|
|
885
|
-
tokenizeQuery,
|
|
886
|
-
reciprocalRankFusion,
|
|
887
1047
|
buildKnowledgeIndex,
|
|
888
1048
|
writeKnowledgeIndex,
|
|
889
1049
|
lintKnowledgeIndex,
|
|
890
1050
|
inspectKnowledgeIndex,
|
|
891
1051
|
explainKnowledgeTarget,
|
|
1052
|
+
isSafeKnowledgePath,
|
|
1053
|
+
parseKnowledgeWriteBlocks,
|
|
1054
|
+
applyKnowledgeWriteBlocks,
|
|
1055
|
+
applyKnowledgeWriteBlocksFile,
|
|
1056
|
+
SourceAnchorSchema,
|
|
1057
|
+
SourceRecordSchema,
|
|
1058
|
+
KnowledgePageSchema,
|
|
1059
|
+
KnowledgeGraphNodeSchema,
|
|
1060
|
+
KnowledgeGraphEdgeSchema,
|
|
1061
|
+
KnowledgeIndexSchema,
|
|
1062
|
+
KnowledgeEventSchema,
|
|
1063
|
+
KnowledgeBaseCandidateSchema,
|
|
892
1064
|
validateKnowledgeIndex
|
|
893
1065
|
};
|
|
894
|
-
//# sourceMappingURL=chunk-
|
|
1066
|
+
//# sourceMappingURL=chunk-HKYD765Q.js.map
|