@tangle-network/agent-knowledge 8.0.10 → 10.7.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +14 -2
- package/CHANGELOG.md +106 -0
- package/README.md +127 -5
- package/dist/benchmarks/index.d.ts +1 -1
- package/dist/benchmarks/index.js +1 -1
- package/dist/{benchmarks-B_07vX0Q.js → benchmarks-B6fCb6AD.js} +2 -2
- package/dist/{benchmarks-B_07vX0Q.js.map → benchmarks-B6fCb6AD.js.map} +1 -1
- package/dist/cli.js +53 -21
- package/dist/cli.js.map +1 -1
- package/dist/{index-DypNCZtP.d.ts → index-DZeFm-BP.d.ts} +2 -2
- package/dist/{index-DypNCZtP.d.ts.map → index-DZeFm-BP.d.ts.map} +1 -1
- package/dist/{index-Ij4giqqj.d.ts → index-eDsIXWyM.d.ts} +3 -3
- package/dist/{index-Ij4giqqj.d.ts.map → index-eDsIXWyM.d.ts.map} +1 -1
- package/dist/index.d.ts +911 -291
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +1020 -392
- package/dist/index.js.map +1 -1
- package/dist/{inspect-0tmv3xV8.js → inspect-DALsvG10.js} +680 -108
- package/dist/inspect-DALsvG10.js.map +1 -0
- package/dist/memory/index.d.ts +2 -2
- package/dist/memory/index.js +2 -2
- package/dist/{memory-BPgPIEhj.js → memory-BRGsy2QN.js} +2 -2
- package/dist/{memory-BPgPIEhj.js.map → memory-BRGsy2QN.js.map} +1 -1
- package/dist/search-Cw6eYfSd.js +258 -0
- package/dist/search-Cw6eYfSd.js.map +1 -0
- package/dist/{types-CTT16XnO.d.ts → types-m2QB86fF.d.ts} +17 -3
- package/dist/types-m2QB86fF.d.ts.map +1 -0
- package/dist/viz/index.d.ts +1 -1
- package/docs/architecture.md +1 -0
- package/docs/knowledge-use-receipts.md +50 -12
- package/package.json +5 -5
- package/dist/inspect-0tmv3xV8.js.map +0 -1
- package/dist/search-CtVJ0PKX.js +0 -132
- package/dist/search-CtVJ0PKX.js.map +0 -1
- package/dist/types-CTT16XnO.d.ts.map +0 -1
|
@@ -0,0 +1,258 @@
|
|
|
1
|
+
//#region src/lexical-index.ts
|
|
2
|
+
const STOP_WORDS = /* @__PURE__ */ new Set([
|
|
3
|
+
"the",
|
|
4
|
+
"is",
|
|
5
|
+
"a",
|
|
6
|
+
"an",
|
|
7
|
+
"what",
|
|
8
|
+
"how",
|
|
9
|
+
"are",
|
|
10
|
+
"was",
|
|
11
|
+
"were",
|
|
12
|
+
"to",
|
|
13
|
+
"for",
|
|
14
|
+
"of",
|
|
15
|
+
"with",
|
|
16
|
+
"by",
|
|
17
|
+
"in",
|
|
18
|
+
"on",
|
|
19
|
+
"and"
|
|
20
|
+
]);
|
|
21
|
+
/**
|
|
22
|
+
* The token stream of one text: lower-cased, split on whitespace and
|
|
23
|
+
* punctuation, single characters and stop words removed, and a CJK run
|
|
24
|
+
* expanded into its bigrams and characters. Repeats are kept so a term
|
|
25
|
+
* frequency can be counted. Indexing and querying share this function, so the
|
|
26
|
+
* two vocabularies cannot drift.
|
|
27
|
+
*/
|
|
28
|
+
function tokenizeText(text) {
|
|
29
|
+
const raw = text.toLowerCase().split(/[\s,,。!?、;:""''()()\-_/\\·~~…]+/).filter((token) => token.length > 1 && !STOP_WORDS.has(token));
|
|
30
|
+
const tokens = [];
|
|
31
|
+
for (const token of raw) {
|
|
32
|
+
if (/[\u4e00-\u9fff\u3400-\u4dbf]/.test(token) && token.length > 2) {
|
|
33
|
+
const chars = [...token];
|
|
34
|
+
for (let i = 0; i < chars.length - 1; i++) tokens.push(chars[i] + chars[i + 1]);
|
|
35
|
+
tokens.push(...chars);
|
|
36
|
+
}
|
|
37
|
+
tokens.push(token);
|
|
38
|
+
}
|
|
39
|
+
return tokens;
|
|
40
|
+
}
|
|
41
|
+
/** The distinct query terms, in first-occurrence order. */
|
|
42
|
+
function tokenizeQuery(query) {
|
|
43
|
+
return [...new Set(tokenizeText(query))];
|
|
44
|
+
}
|
|
45
|
+
const DEFAULT_FIELD_BOOSTS = Object.freeze({
|
|
46
|
+
title: 3,
|
|
47
|
+
path: 2,
|
|
48
|
+
text: 1
|
|
49
|
+
});
|
|
50
|
+
function buildKnowledgeLexicalIndex(pages, options = {}) {
|
|
51
|
+
const tokenize = options.tokenize ?? tokenizeText;
|
|
52
|
+
const fieldBoosts = resolveFieldBoosts(options.fieldBoosts);
|
|
53
|
+
const postings = /* @__PURE__ */ new Map();
|
|
54
|
+
const documentLengths = [];
|
|
55
|
+
let totalLength = 0;
|
|
56
|
+
pages.forEach((page, ordinal) => {
|
|
57
|
+
const frequencies = /* @__PURE__ */ new Map();
|
|
58
|
+
let length = 0;
|
|
59
|
+
for (const [text, boost] of [
|
|
60
|
+
[page.title, fieldBoosts.title],
|
|
61
|
+
[page.path.replace(/\.md$/, ""), fieldBoosts.path],
|
|
62
|
+
[page.text, fieldBoosts.text]
|
|
63
|
+
]) {
|
|
64
|
+
if (boost === 0) continue;
|
|
65
|
+
for (const token of tokenize(text)) {
|
|
66
|
+
frequencies.set(token, (frequencies.get(token) ?? 0) + boost);
|
|
67
|
+
length += boost;
|
|
68
|
+
}
|
|
69
|
+
}
|
|
70
|
+
documentLengths.push(length);
|
|
71
|
+
totalLength += length;
|
|
72
|
+
for (const [term, tf] of frequencies) {
|
|
73
|
+
let list = postings.get(term);
|
|
74
|
+
if (!list) {
|
|
75
|
+
list = [];
|
|
76
|
+
postings.set(term, list);
|
|
77
|
+
}
|
|
78
|
+
list.push({
|
|
79
|
+
ordinal,
|
|
80
|
+
tf
|
|
81
|
+
});
|
|
82
|
+
}
|
|
83
|
+
});
|
|
84
|
+
return {
|
|
85
|
+
pages,
|
|
86
|
+
postings,
|
|
87
|
+
documentLengths,
|
|
88
|
+
averageDocumentLength: pages.length > 0 ? totalLength / pages.length : 0,
|
|
89
|
+
documentCount: pages.length,
|
|
90
|
+
tokenize,
|
|
91
|
+
fieldBoosts
|
|
92
|
+
};
|
|
93
|
+
}
|
|
94
|
+
/**
|
|
95
|
+
* Okapi BM25 over the distinct query terms, with the Lucene inverse document
|
|
96
|
+
* frequency `ln(1 + (N - df + 0.5) / (df + 0.5))`, which is positive for every
|
|
97
|
+
* indexed term. Pages with no matching term are absent. The result is ordered
|
|
98
|
+
* by score, then by path, so it does not depend on page order.
|
|
99
|
+
*/
|
|
100
|
+
function scoreBm25(index, tokens, options = {}) {
|
|
101
|
+
const k1 = options.k1 ?? 1.2;
|
|
102
|
+
const b = options.b ?? .75;
|
|
103
|
+
if (!Number.isFinite(k1) || k1 < 0) throw new Error(`bm25 k1 must be >= 0, got ${String(k1)}`);
|
|
104
|
+
if (!Number.isFinite(b) || b < 0 || b > 1) throw new Error(`bm25 b must lie in [0, 1], got ${String(b)}`);
|
|
105
|
+
const scores = /* @__PURE__ */ new Map();
|
|
106
|
+
const { documentCount, averageDocumentLength, documentLengths } = index;
|
|
107
|
+
for (const term of new Set(tokens)) {
|
|
108
|
+
const list = index.postings.get(term);
|
|
109
|
+
if (!list) continue;
|
|
110
|
+
const df = list.length;
|
|
111
|
+
const idf = Math.log(1 + (documentCount - df + .5) / (df + .5));
|
|
112
|
+
for (const { ordinal, tf } of list) {
|
|
113
|
+
const lengthRatio = averageDocumentLength > 0 ? documentLengths[ordinal] / averageDocumentLength : 1;
|
|
114
|
+
const saturated = tf * (k1 + 1) / (tf + k1 * (1 - b + b * lengthRatio));
|
|
115
|
+
scores.set(ordinal, (scores.get(ordinal) ?? 0) + idf * saturated);
|
|
116
|
+
}
|
|
117
|
+
}
|
|
118
|
+
return [...scores.entries()].map(([ordinal, score]) => ({
|
|
119
|
+
page: index.pages[ordinal],
|
|
120
|
+
score
|
|
121
|
+
})).sort((left, right) => right.score - left.score || left.page.path.localeCompare(right.page.path));
|
|
122
|
+
}
|
|
123
|
+
function resolveFieldBoosts(boosts) {
|
|
124
|
+
const resolved = {
|
|
125
|
+
...DEFAULT_FIELD_BOOSTS,
|
|
126
|
+
...boosts
|
|
127
|
+
};
|
|
128
|
+
for (const [field, boost] of Object.entries(resolved)) if (!Number.isFinite(boost) || boost < 0) throw new Error(`lexical field boost ${field} must be >= 0, got ${String(boost)}`);
|
|
129
|
+
return Object.freeze(resolved);
|
|
130
|
+
}
|
|
131
|
+
//#endregion
|
|
132
|
+
//#region src/search.ts
|
|
133
|
+
const RRF_K = 60;
|
|
134
|
+
/**
|
|
135
|
+
* Identity of the ranking `searchKnowledge` performs: BM25 over title, path,
|
|
136
|
+
* and body, exact-title and phrase matches ahead of bag-of-words matches, and
|
|
137
|
+
* reciprocal rank fusion with link and shared-source structure. Declare it as
|
|
138
|
+
* the retriever id of a retrieval receipt minted from these results.
|
|
139
|
+
*/
|
|
140
|
+
const KNOWLEDGE_SEARCH_RETRIEVER_ID = "bm25-rrf-v1";
|
|
141
|
+
function searchKnowledge(index, query, limitOrOptions = 10) {
|
|
142
|
+
return searchKnowledgePages(index.pages, query, limitOrOptions);
|
|
143
|
+
}
|
|
144
|
+
/**
|
|
145
|
+
* Rank a page set that is not a built index, such as the chain a run can see.
|
|
146
|
+
*
|
|
147
|
+
* `searchKnowledge` is this function over `index.pages`, so both entry points
|
|
148
|
+
* rank identically and neither can drift from the other.
|
|
149
|
+
*/
|
|
150
|
+
function searchKnowledgePages(pages, query, limitOrOptions = 10) {
|
|
151
|
+
const trimmed = query.trim();
|
|
152
|
+
if (trimmed === "") return [];
|
|
153
|
+
const options = typeof limitOrOptions === "number" ? { limit: limitOrOptions } : { ...limitOrOptions };
|
|
154
|
+
const limit = options.limit ?? 10;
|
|
155
|
+
if (!Number.isInteger(limit) || limit < 0) throw new Error(`search limit must be a non-negative integer, got ${String(limit)}`);
|
|
156
|
+
const matched = filterPages(pages, options);
|
|
157
|
+
const lexicalRanked = rankLexical(matched, trimmed, options.lexicalIndex ? assertLexicalIndexMatches(options.lexicalIndex, pages) : buildKnowledgeLexicalIndex(pages));
|
|
158
|
+
const graphRanked = rankByGraph(matched, lexicalRanked);
|
|
159
|
+
const scores = reciprocalRankFusion([lexicalRanked.map((p) => p.id), graphRanked.map((p) => p.id)]);
|
|
160
|
+
const byId = new Map(matched.map((page) => [page.id, page]));
|
|
161
|
+
const ranked = [...scores.entries()].map(([id, score]) => ({
|
|
162
|
+
page: byId.get(id),
|
|
163
|
+
score
|
|
164
|
+
})).filter((item) => Boolean(item.page)).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).slice(0, limit);
|
|
165
|
+
const topScore = ranked[0]?.score ?? 0;
|
|
166
|
+
return ranked.map((item, i) => ({
|
|
167
|
+
citationId: item.page.id,
|
|
168
|
+
page: item.page,
|
|
169
|
+
score: item.score,
|
|
170
|
+
rrfScore: item.score,
|
|
171
|
+
normalizedScore: topScore > 0 ? item.score / topScore : 0,
|
|
172
|
+
rank: i + 1,
|
|
173
|
+
snippet: buildSnippet(item.page.text, trimmed),
|
|
174
|
+
reasons: reasonsFor(item.page, trimmed)
|
|
175
|
+
}));
|
|
176
|
+
}
|
|
177
|
+
function reciprocalRankFusion(rankLists, k = RRF_K) {
|
|
178
|
+
const scores = /* @__PURE__ */ new Map();
|
|
179
|
+
for (const list of rankLists) list.forEach((id, idx) => {
|
|
180
|
+
scores.set(id, (scores.get(id) ?? 0) + 1 / (k + idx + 1));
|
|
181
|
+
});
|
|
182
|
+
return scores;
|
|
183
|
+
}
|
|
184
|
+
function filterPages(pages, options) {
|
|
185
|
+
const pageIds = options.pageIds ? new Set(options.pageIds) : null;
|
|
186
|
+
const tags = options.tags ? new Set(options.tags) : null;
|
|
187
|
+
const kinds = options.kinds ? new Set(options.kinds) : null;
|
|
188
|
+
return pages.filter((page) => {
|
|
189
|
+
if (options.excludeInvalidated && page.invalidation !== void 0) return false;
|
|
190
|
+
if (pageIds && !pageIds.has(page.id)) return false;
|
|
191
|
+
if (tags && !page.tags.some((tag) => tags.has(tag))) return false;
|
|
192
|
+
if (kinds) {
|
|
193
|
+
const kind = page.frontmatter.kind;
|
|
194
|
+
if (typeof kind !== "string" || !kinds.has(kind)) return false;
|
|
195
|
+
}
|
|
196
|
+
return options.predicate?.(page) ?? true;
|
|
197
|
+
});
|
|
198
|
+
}
|
|
199
|
+
function assertLexicalIndexMatches(lexicalIndex, pages) {
|
|
200
|
+
if (lexicalIndex.pages.length !== pages.length || lexicalIndex.pages.some((page, ordinal) => page !== pages[ordinal])) throw new Error("lexical index was not built from the searched pages");
|
|
201
|
+
return lexicalIndex;
|
|
202
|
+
}
|
|
203
|
+
/**
|
|
204
|
+
* The lexical rank list. An exact title or path match outranks a title that
|
|
205
|
+
* contains the query, which outranks a body that contains the query, which
|
|
206
|
+
* outranks a bag-of-words match; BM25 orders pages inside each of those tiers.
|
|
207
|
+
* The tiers are an ordering, not a score, so a strong bag-of-words page cannot
|
|
208
|
+
* overtake an exact match by term repetition alone.
|
|
209
|
+
*/
|
|
210
|
+
function rankLexical(pages, query, lexicalIndex) {
|
|
211
|
+
const tokens = [...new Set(lexicalIndex.tokenize(query))];
|
|
212
|
+
const bm25 = new Map(scoreBm25(lexicalIndex, tokens).map((hit) => [hit.page, hit.score]));
|
|
213
|
+
const phrase = query.toLowerCase();
|
|
214
|
+
return pages.flatMap((page) => {
|
|
215
|
+
const score = bm25.get(page) ?? 0;
|
|
216
|
+
if (score === 0 && tokens.length > 0) return [];
|
|
217
|
+
const tier = phraseTier(page, phrase);
|
|
218
|
+
if (score === 0 && tier === 0) return [];
|
|
219
|
+
return [{
|
|
220
|
+
page,
|
|
221
|
+
tier,
|
|
222
|
+
score
|
|
223
|
+
}];
|
|
224
|
+
}).sort((a, b) => b.tier - a.tier || b.score - a.score || a.page.path.localeCompare(b.page.path)).map((item) => item.page);
|
|
225
|
+
}
|
|
226
|
+
function phraseTier(page, phrase) {
|
|
227
|
+
const title = page.title.toLowerCase();
|
|
228
|
+
if (title === phrase || page.path.toLowerCase().endsWith(`${phrase}.md`)) return 3;
|
|
229
|
+
if (title.includes(phrase)) return 2;
|
|
230
|
+
if (page.text.toLowerCase().includes(phrase)) return 1;
|
|
231
|
+
return 0;
|
|
232
|
+
}
|
|
233
|
+
function rankByGraph(pages, lexicalRanked) {
|
|
234
|
+
if (lexicalRanked.length === 0) return [];
|
|
235
|
+
const seeds = new Set(lexicalRanked.slice(0, 5).map((page) => page.id));
|
|
236
|
+
return pages.map((page) => ({
|
|
237
|
+
page,
|
|
238
|
+
score: page.outLinks.filter((link) => seeds.has(link)).length + page.sourceIds.filter((source) => lexicalRanked.some((seed) => seed.sourceIds.includes(source))).length
|
|
239
|
+
})).filter((item) => item.score > 0).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).map((item) => item.page);
|
|
240
|
+
}
|
|
241
|
+
function buildSnippet(text, query) {
|
|
242
|
+
const compact = text.replace(/\s+/g, " ").trim();
|
|
243
|
+
const idx = compact.toLowerCase().indexOf(query.toLowerCase());
|
|
244
|
+
if (idx < 0) return compact.slice(0, 180);
|
|
245
|
+
return compact.slice(Math.max(0, idx - 80), Math.min(compact.length, idx + query.length + 100));
|
|
246
|
+
}
|
|
247
|
+
function reasonsFor(page, query) {
|
|
248
|
+
const lower = `${page.title}\n${page.text}`.toLowerCase();
|
|
249
|
+
const reasons = [];
|
|
250
|
+
if (lower.includes(query.toLowerCase())) reasons.push("phrase");
|
|
251
|
+
if (page.sourceIds.length > 0) reasons.push("sourced");
|
|
252
|
+
if (page.outLinks.length > 0) reasons.push("linked");
|
|
253
|
+
return reasons;
|
|
254
|
+
}
|
|
255
|
+
//#endregion
|
|
256
|
+
export { buildKnowledgeLexicalIndex as a, tokenizeText as c, searchKnowledgePages as i, reciprocalRankFusion as n, scoreBm25 as o, searchKnowledge as r, tokenizeQuery as s, KNOWLEDGE_SEARCH_RETRIEVER_ID as t };
|
|
257
|
+
|
|
258
|
+
//# sourceMappingURL=search-Cw6eYfSd.js.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"search-Cw6eYfSd.js","names":[],"sources":["../src/lexical-index.ts","../src/search.ts"],"sourcesContent":["import type { KnowledgePage } from './types'\n\nconst STOP_WORDS = new Set([\n 'the',\n 'is',\n 'a',\n 'an',\n 'what',\n 'how',\n 'are',\n 'was',\n 'were',\n 'to',\n 'for',\n 'of',\n 'with',\n 'by',\n 'in',\n 'on',\n 'and',\n])\n\n/**\n * The token stream of one text: lower-cased, split on whitespace and\n * punctuation, single characters and stop words removed, and a CJK run\n * expanded into its bigrams and characters. Repeats are kept so a term\n * frequency can be counted. Indexing and querying share this function, so the\n * two vocabularies cannot drift.\n */\nexport function tokenizeText(text: string): string[] {\n const raw = text\n .toLowerCase()\n .split(/[\\s,,。!?、;:\"\"''()()\\-_/\\\\·~~…]+/)\n .filter((token) => token.length > 1 && !STOP_WORDS.has(token))\n const tokens: string[] = []\n for (const token of raw) {\n if (/[\\u4e00-\\u9fff\\u3400-\\u4dbf]/.test(token) && token.length > 2) {\n const chars = [...token]\n for (let i = 0; i < chars.length - 1; i++) tokens.push(chars[i]! + chars[i + 1]!)\n tokens.push(...chars)\n }\n tokens.push(token)\n }\n return tokens\n}\n\n/** The distinct query terms, in first-occurrence order. */\nexport function tokenizeQuery(query: string): string[] {\n return [...new Set(tokenizeText(query))]\n}\n\nexport interface KnowledgeLexicalFieldBoosts {\n /** Multiplier for a term occurrence in the page title. Defaults to 3. */\n title?: number\n /** Multiplier for a term occurrence in the page path without its extension. Defaults to 2. */\n path?: number\n /** Multiplier for a term occurrence in the page body. Defaults to 1. */\n text?: number\n}\n\nexport interface KnowledgeLexicalIndexOptions {\n /** Token stream of one text. Defaults to `tokenizeText`. */\n tokenize?: (text: string) => string[]\n fieldBoosts?: KnowledgeLexicalFieldBoosts\n}\n\nexport interface KnowledgeLexicalPosting {\n /** Position of the page in `KnowledgeLexicalIndex.pages`. */\n ordinal: number\n /** Field-boosted term frequency in that page. */\n tf: number\n}\n\n/**\n * Inverted index over a fixed page list.\n *\n * Ordinals are positions in `pages`. `documentLengths` are field-boosted token\n * counts, so length normalization and term frequency use one scale.\n */\nexport interface KnowledgeLexicalIndex {\n readonly pages: readonly KnowledgePage[]\n readonly postings: ReadonlyMap<string, readonly KnowledgeLexicalPosting[]>\n readonly documentLengths: readonly number[]\n readonly averageDocumentLength: number\n readonly documentCount: number\n readonly tokenize: (text: string) => string[]\n readonly fieldBoosts: Readonly<Required<KnowledgeLexicalFieldBoosts>>\n}\n\nexport interface Bm25Options {\n /** Term-frequency saturation. Defaults to 1.2. */\n k1?: number\n /** Length-normalization strength in [0, 1]. Defaults to 0.75. */\n b?: number\n}\n\nexport interface Bm25Hit {\n page: KnowledgePage\n score: number\n}\n\nconst DEFAULT_FIELD_BOOSTS: Readonly<Required<KnowledgeLexicalFieldBoosts>> = Object.freeze({\n title: 3,\n path: 2,\n text: 1,\n})\n\nexport function buildKnowledgeLexicalIndex(\n pages: readonly KnowledgePage[],\n options: KnowledgeLexicalIndexOptions = {},\n): KnowledgeLexicalIndex {\n const tokenize = options.tokenize ?? tokenizeText\n const fieldBoosts = resolveFieldBoosts(options.fieldBoosts)\n const postings = new Map<string, KnowledgeLexicalPosting[]>()\n const documentLengths: number[] = []\n let totalLength = 0\n\n pages.forEach((page, ordinal) => {\n const frequencies = new Map<string, number>()\n let length = 0\n for (const [text, boost] of [\n [page.title, fieldBoosts.title],\n [page.path.replace(/\\.md$/, ''), fieldBoosts.path],\n [page.text, fieldBoosts.text],\n ] as const) {\n if (boost === 0) continue\n for (const token of tokenize(text)) {\n frequencies.set(token, (frequencies.get(token) ?? 0) + boost)\n length += boost\n }\n }\n documentLengths.push(length)\n totalLength += length\n for (const [term, tf] of frequencies) {\n let list = postings.get(term)\n if (!list) {\n list = []\n postings.set(term, list)\n }\n list.push({ ordinal, tf })\n }\n })\n\n return {\n pages,\n postings,\n documentLengths,\n averageDocumentLength: pages.length > 0 ? totalLength / pages.length : 0,\n documentCount: pages.length,\n tokenize,\n fieldBoosts,\n }\n}\n\n/**\n * Okapi BM25 over the distinct query terms, with the Lucene inverse document\n * frequency `ln(1 + (N - df + 0.5) / (df + 0.5))`, which is positive for every\n * indexed term. Pages with no matching term are absent. The result is ordered\n * by score, then by path, so it does not depend on page order.\n */\nexport function scoreBm25(\n index: KnowledgeLexicalIndex,\n tokens: readonly string[],\n options: Bm25Options = {},\n): Bm25Hit[] {\n const k1 = options.k1 ?? 1.2\n const b = options.b ?? 0.75\n if (!Number.isFinite(k1) || k1 < 0) throw new Error(`bm25 k1 must be >= 0, got ${String(k1)}`)\n if (!Number.isFinite(b) || b < 0 || b > 1) {\n throw new Error(`bm25 b must lie in [0, 1], got ${String(b)}`)\n }\n const scores = new Map<number, number>()\n const { documentCount, averageDocumentLength, documentLengths } = index\n for (const term of new Set(tokens)) {\n const list = index.postings.get(term)\n if (!list) continue\n const df = list.length\n const idf = Math.log(1 + (documentCount - df + 0.5) / (df + 0.5))\n for (const { ordinal, tf } of list) {\n const lengthRatio =\n averageDocumentLength > 0 ? documentLengths[ordinal]! / averageDocumentLength : 1\n const saturated = (tf * (k1 + 1)) / (tf + k1 * (1 - b + b * lengthRatio))\n scores.set(ordinal, (scores.get(ordinal) ?? 0) + idf * saturated)\n }\n }\n return [...scores.entries()]\n .map(([ordinal, score]) => ({ page: index.pages[ordinal]!, score }))\n .sort(\n (left, right) => right.score - left.score || left.page.path.localeCompare(right.page.path),\n )\n}\n\nfunction resolveFieldBoosts(\n boosts: KnowledgeLexicalFieldBoosts | undefined,\n): Readonly<Required<KnowledgeLexicalFieldBoosts>> {\n const resolved = { ...DEFAULT_FIELD_BOOSTS, ...boosts }\n for (const [field, boost] of Object.entries(resolved)) {\n if (!Number.isFinite(boost) || boost < 0) {\n throw new Error(`lexical field boost ${field} must be >= 0, got ${String(boost)}`)\n }\n }\n return Object.freeze(resolved)\n}\n","import { buildKnowledgeLexicalIndex, type KnowledgeLexicalIndex, scoreBm25 } from './lexical-index'\nimport type { KnowledgeId, KnowledgeIndex, KnowledgePage, KnowledgeSearchResult } from './types'\n\nconst RRF_K = 60\n\n/**\n * Identity of the ranking `searchKnowledge` performs: BM25 over title, path,\n * and body, exact-title and phrase matches ahead of bag-of-words matches, and\n * reciprocal rank fusion with link and shared-source structure. Declare it as\n * the retriever id of a retrieval receipt minted from these results.\n */\nexport const KNOWLEDGE_SEARCH_RETRIEVER_ID = 'bm25-rrf-v1'\n\nexport interface SearchKnowledgeOptions {\n /** Maximum results returned. Defaults to 10. */\n limit?: number\n /** Match pages whose stable id is one of these values. */\n pageIds?: readonly KnowledgeId[]\n /** Match pages carrying at least one of these tags. */\n tags?: readonly string[]\n /** Match the exact string stored in `frontmatter.kind`. */\n kinds?: readonly string[]\n /**\n * Drop pages whose own evidence refuted them. Defaults to false: a caller\n * that reads history needs them, and a silent change of what search returns\n * is worse than an explicit option.\n */\n excludeInvalidated?: boolean\n /** Additional caller-owned filter, applied before either ranking stage. */\n predicate?: (page: KnowledgePage) => boolean\n /**\n * A lexical index built from exactly the searched pages, supplied by a caller\n * that searches one page set repeatedly. When absent, one is built for this\n * call.\n */\n lexicalIndex?: KnowledgeLexicalIndex\n}\n\n/**\n * A retrieval result with an explicit citation handle.\n *\n * `citationId` is exactly `page.id`; later writes should persist this value\n * when they cite the page. Keeping it at the result's top level prevents tool\n * renderers from accidentally hiding the only stable handle a model can copy.\n */\nexport interface KnowledgeSearchHit extends KnowledgeSearchResult {\n citationId: KnowledgeId\n}\n\nexport function searchKnowledge(\n index: KnowledgeIndex,\n query: string,\n limit?: number,\n): KnowledgeSearchHit[]\nexport function searchKnowledge(\n index: KnowledgeIndex,\n query: string,\n options?: SearchKnowledgeOptions,\n): KnowledgeSearchHit[]\nexport function searchKnowledge(\n index: KnowledgeIndex,\n query: string,\n limitOrOptions: number | SearchKnowledgeOptions = 10,\n): KnowledgeSearchHit[] {\n return searchKnowledgePages(index.pages, query, limitOrOptions)\n}\n\n/**\n * Rank a page set that is not a built index, such as the chain a run can see.\n *\n * `searchKnowledge` is this function over `index.pages`, so both entry points\n * rank identically and neither can drift from the other.\n */\nexport function searchKnowledgePages(\n pages: readonly KnowledgePage[],\n query: string,\n limitOrOptions: number | SearchKnowledgeOptions = 10,\n): KnowledgeSearchHit[] {\n const trimmed = query.trim()\n if (trimmed === '') return []\n const options =\n typeof limitOrOptions === 'number' ? { limit: limitOrOptions } : { ...limitOrOptions }\n const limit = options.limit ?? 10\n if (!Number.isInteger(limit) || limit < 0) {\n throw new Error(`search limit must be a non-negative integer, got ${String(limit)}`)\n }\n\n const matched = filterPages(pages, options)\n const lexicalIndex = options.lexicalIndex\n ? assertLexicalIndexMatches(options.lexicalIndex, pages)\n : buildKnowledgeLexicalIndex(pages)\n const lexicalRanked = rankLexical(matched, trimmed, lexicalIndex)\n const graphRanked = rankByGraph(matched, lexicalRanked)\n const scores = reciprocalRankFusion([\n lexicalRanked.map((p) => p.id),\n graphRanked.map((p) => p.id),\n ])\n const byId = new Map(matched.map((page) => [page.id, page]))\n\n const ranked = [...scores.entries()]\n .map(([id, score]) => ({ page: byId.get(id), score }))\n .filter((item): item is { page: KnowledgePage; score: number } => Boolean(item.page))\n .sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path))\n .slice(0, limit)\n\n // Normalize against the top hit so callers can compare against natural\n // [0, 1] thresholds. Raw RRF values are typically ~0.016, which reads as\n // \"no relevance\" to humans even when the result is the best available.\n // The top hit becomes 1 by definition; lower-ranked hits scale linearly.\n const topScore = ranked[0]?.score ?? 0\n\n return ranked.map((item, i) => ({\n citationId: item.page.id,\n page: item.page,\n score: item.score,\n rrfScore: item.score,\n normalizedScore: topScore > 0 ? item.score / topScore : 0,\n rank: i + 1,\n snippet: buildSnippet(item.page.text, trimmed),\n reasons: reasonsFor(item.page, trimmed),\n }))\n}\n\nexport function reciprocalRankFusion(rankLists: string[][], k = RRF_K): Map<string, number> {\n const scores = new Map<string, number>()\n for (const list of rankLists) {\n list.forEach((id, idx) => {\n scores.set(id, (scores.get(id) ?? 0) + 1 / (k + idx + 1))\n })\n }\n return scores\n}\n\nfunction filterPages(\n pages: readonly KnowledgePage[],\n options: SearchKnowledgeOptions,\n): KnowledgePage[] {\n const pageIds = options.pageIds ? new Set(options.pageIds) : null\n const tags = options.tags ? new Set(options.tags) : null\n const kinds = options.kinds ? new Set(options.kinds) : null\n\n return pages.filter((page) => {\n if (options.excludeInvalidated && page.invalidation !== undefined) return false\n if (pageIds && !pageIds.has(page.id)) return false\n if (tags && !page.tags.some((tag) => tags.has(tag))) return false\n if (kinds) {\n const kind = page.frontmatter.kind\n if (typeof kind !== 'string' || !kinds.has(kind)) return false\n }\n return options.predicate?.(page) ?? true\n })\n}\n\nfunction assertLexicalIndexMatches(\n lexicalIndex: KnowledgeLexicalIndex,\n pages: readonly KnowledgePage[],\n): KnowledgeLexicalIndex {\n if (\n lexicalIndex.pages.length !== pages.length ||\n lexicalIndex.pages.some((page, ordinal) => page !== pages[ordinal])\n ) {\n throw new Error('lexical index was not built from the searched pages')\n }\n return lexicalIndex\n}\n\n/**\n * The lexical rank list. An exact title or path match outranks a title that\n * contains the query, which outranks a body that contains the query, which\n * outranks a bag-of-words match; BM25 orders pages inside each of those tiers.\n * The tiers are an ordering, not a score, so a strong bag-of-words page cannot\n * overtake an exact match by term repetition alone.\n */\nfunction rankLexical(\n pages: KnowledgePage[],\n query: string,\n lexicalIndex: KnowledgeLexicalIndex,\n): KnowledgePage[] {\n const tokens = [...new Set(lexicalIndex.tokenize(query))]\n const bm25 = new Map(scoreBm25(lexicalIndex, tokens).map((hit) => [hit.page, hit.score]))\n const phrase = query.toLowerCase()\n return pages\n .flatMap((page) => {\n const score = bm25.get(page) ?? 0\n // A query that tokenizes to nothing (stop words, single characters) can\n // still match as a phrase; otherwise only pages with a scored term are\n // candidates, and every phrase match is one of them.\n if (score === 0 && tokens.length > 0) return []\n const tier = phraseTier(page, phrase)\n if (score === 0 && tier === 0) return []\n return [{ page, tier, score }]\n })\n .sort((a, b) => b.tier - a.tier || b.score - a.score || a.page.path.localeCompare(b.page.path))\n .map((item) => item.page)\n}\n\nfunction phraseTier(page: KnowledgePage, phrase: string): number {\n const title = page.title.toLowerCase()\n if (title === phrase || page.path.toLowerCase().endsWith(`${phrase}.md`)) return 3\n if (title.includes(phrase)) return 2\n if (page.text.toLowerCase().includes(phrase)) return 1\n return 0\n}\n\nfunction rankByGraph(pages: KnowledgePage[], lexicalRanked: KnowledgePage[]): KnowledgePage[] {\n if (lexicalRanked.length === 0) return []\n const seeds = new Set(lexicalRanked.slice(0, 5).map((page) => page.id))\n return pages\n .map((page) => ({\n page,\n score:\n page.outLinks.filter((link) => seeds.has(link)).length +\n page.sourceIds.filter((source) =>\n lexicalRanked.some((seed) => seed.sourceIds.includes(source)),\n ).length,\n }))\n .filter((item) => item.score > 0)\n .sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path))\n .map((item) => item.page)\n}\n\nfunction buildSnippet(text: string, query: string): string {\n const compact = text.replace(/\\s+/g, ' ').trim()\n const idx = compact.toLowerCase().indexOf(query.toLowerCase())\n if (idx < 0) return compact.slice(0, 180)\n return compact.slice(Math.max(0, idx - 80), Math.min(compact.length, idx + query.length + 100))\n}\n\nfunction reasonsFor(page: KnowledgePage, query: string): string[] {\n const lower = `${page.title}\\n${page.text}`.toLowerCase()\n const reasons: string[] = []\n if (lower.includes(query.toLowerCase())) reasons.push('phrase')\n if (page.sourceIds.length > 0) reasons.push('sourced')\n if (page.outLinks.length > 0) reasons.push('linked')\n return reasons\n}\n"],"mappings":";AAEA,MAAM,6BAAa,IAAI,IAAI;CACzB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;;;;;;;;AASD,SAAgB,aAAa,MAAwB;CACnD,MAAM,MAAM,KACT,YAAY,CAAC,CACb,MAAM,iCAAiC,CAAC,CACxC,QAAQ,UAAU,MAAM,SAAS,KAAK,CAAC,WAAW,IAAI,KAAK,CAAC;CAC/D,MAAM,SAAmB,CAAC;CAC1B,KAAK,MAAM,SAAS,KAAK;EACvB,IAAI,+BAA+B,KAAK,KAAK,KAAK,MAAM,SAAS,GAAG;GAClE,MAAM,QAAQ,CAAC,GAAG,KAAK;GACvB,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,SAAS,GAAG,KAAK,OAAO,KAAK,MAAM,KAAM,MAAM,IAAI,EAAG;GAChF,OAAO,KAAK,GAAG,KAAK;EACtB;EACA,OAAO,KAAK,KAAK;CACnB;CACA,OAAO;AACT;;AAGA,SAAgB,cAAc,OAAyB;CACrD,OAAO,CAAC,GAAG,IAAI,IAAI,aAAa,KAAK,CAAC,CAAC;AACzC;AAoDA,MAAM,uBAAwE,OAAO,OAAO;CAC1F,OAAO;CACP,MAAM;CACN,MAAM;AACR,CAAC;AAED,SAAgB,2BACd,OACA,UAAwC,CAAC,GAClB;CACvB,MAAM,WAAW,QAAQ,YAAY;CACrC,MAAM,cAAc,mBAAmB,QAAQ,WAAW;CAC1D,MAAM,2BAAW,IAAI,IAAuC;CAC5D,MAAM,kBAA4B,CAAC;CACnC,IAAI,cAAc;CAElB,MAAM,SAAS,MAAM,YAAY;EAC/B,MAAM,8BAAc,IAAI,IAAoB;EAC5C,IAAI,SAAS;EACb,KAAK,MAAM,CAAC,MAAM,UAAU;GAC1B,CAAC,KAAK,OAAO,YAAY,KAAK;GAC9B,CAAC,KAAK,KAAK,QAAQ,SAAS,EAAE,GAAG,YAAY,IAAI;GACjD,CAAC,KAAK,MAAM,YAAY,IAAI;EAC9B,GAAY;GACV,IAAI,UAAU,GAAG;GACjB,KAAK,MAAM,SAAS,SAAS,IAAI,GAAG;IAClC,YAAY,IAAI,QAAQ,YAAY,IAAI,KAAK,KAAK,KAAK,KAAK;IAC5D,UAAU;GACZ;EACF;EACA,gBAAgB,KAAK,MAAM;EAC3B,eAAe;EACf,KAAK,MAAM,CAAC,MAAM,OAAO,aAAa;GACpC,IAAI,OAAO,SAAS,IAAI,IAAI;GAC5B,IAAI,CAAC,MAAM;IACT,OAAO,CAAC;IACR,SAAS,IAAI,MAAM,IAAI;GACzB;GACA,KAAK,KAAK;IAAE;IAAS;GAAG,CAAC;EAC3B;CACF,CAAC;CAED,OAAO;EACL;EACA;EACA;EACA,uBAAuB,MAAM,SAAS,IAAI,cAAc,MAAM,SAAS;EACvE,eAAe,MAAM;EACrB;EACA;CACF;AACF;;;;;;;AAQA,SAAgB,UACd,OACA,QACA,UAAuB,CAAC,GACb;CACX,MAAM,KAAK,QAAQ,MAAM;CACzB,MAAM,IAAI,QAAQ,KAAK;CACvB,IAAI,CAAC,OAAO,SAAS,EAAE,KAAK,KAAK,GAAG,MAAM,IAAI,MAAM,6BAA6B,OAAO,EAAE,GAAG;CAC7F,IAAI,CAAC,OAAO,SAAS,CAAC,KAAK,IAAI,KAAK,IAAI,GACtC,MAAM,IAAI,MAAM,kCAAkC,OAAO,CAAC,GAAG;CAE/D,MAAM,yBAAS,IAAI,IAAoB;CACvC,MAAM,EAAE,eAAe,uBAAuB,oBAAoB;CAClE,KAAK,MAAM,QAAQ,IAAI,IAAI,MAAM,GAAG;EAClC,MAAM,OAAO,MAAM,SAAS,IAAI,IAAI;EACpC,IAAI,CAAC,MAAM;EACX,MAAM,KAAK,KAAK;EAChB,MAAM,MAAM,KAAK,IAAI,KAAK,gBAAgB,KAAK,OAAQ,KAAK,GAAI;EAChE,KAAK,MAAM,EAAE,SAAS,QAAQ,MAAM;GAClC,MAAM,cACJ,wBAAwB,IAAI,gBAAgB,WAAY,wBAAwB;GAClF,MAAM,YAAa,MAAM,KAAK,MAAO,KAAK,MAAM,IAAI,IAAI,IAAI;GAC5D,OAAO,IAAI,UAAU,OAAO,IAAI,OAAO,KAAK,KAAK,MAAM,SAAS;EAClE;CACF;CACA,OAAO,CAAC,GAAG,OAAO,QAAQ,CAAC,CAAC,CACzB,KAAK,CAAC,SAAS,YAAY;EAAE,MAAM,MAAM,MAAM;EAAW;CAAM,EAAE,CAAC,CACnE,MACE,MAAM,UAAU,MAAM,QAAQ,KAAK,SAAS,KAAK,KAAK,KAAK,cAAc,MAAM,KAAK,IAAI,CAC3F;AACJ;AAEA,SAAS,mBACP,QACiD;CACjD,MAAM,WAAW;EAAE,GAAG;EAAsB,GAAG;CAAO;CACtD,KAAK,MAAM,CAAC,OAAO,UAAU,OAAO,QAAQ,QAAQ,GAClD,IAAI,CAAC,OAAO,SAAS,KAAK,KAAK,QAAQ,GACrC,MAAM,IAAI,MAAM,uBAAuB,MAAM,qBAAqB,OAAO,KAAK,GAAG;CAGrF,OAAO,OAAO,OAAO,QAAQ;AAC/B;;;ACvMA,MAAM,QAAQ;;;;;;;AAQd,MAAa,gCAAgC;AAgD7C,SAAgB,gBACd,OACA,OACA,iBAAkD,IAC5B;CACtB,OAAO,qBAAqB,MAAM,OAAO,OAAO,cAAc;AAChE;;;;;;;AAQA,SAAgB,qBACd,OACA,OACA,iBAAkD,IAC5B;CACtB,MAAM,UAAU,MAAM,KAAK;CAC3B,IAAI,YAAY,IAAI,OAAO,CAAC;CAC5B,MAAM,UACJ,OAAO,mBAAmB,WAAW,EAAE,OAAO,eAAe,IAAI,EAAE,GAAG,eAAe;CACvF,MAAM,QAAQ,QAAQ,SAAS;CAC/B,IAAI,CAAC,OAAO,UAAU,KAAK,KAAK,QAAQ,GACtC,MAAM,IAAI,MAAM,oDAAoD,OAAO,KAAK,GAAG;CAGrF,MAAM,UAAU,YAAY,OAAO,OAAO;CAI1C,MAAM,gBAAgB,YAAY,SAAS,SAHtB,QAAQ,eACzB,0BAA0B,QAAQ,cAAc,KAAK,IACrD,2BAA2B,KAAK,CAC4B;CAChE,MAAM,cAAc,YAAY,SAAS,aAAa;CACtD,MAAM,SAAS,qBAAqB,CAClC,cAAc,KAAK,MAAM,EAAE,EAAE,GAC7B,YAAY,KAAK,MAAM,EAAE,EAAE,CAC7B,CAAC;CACD,MAAM,OAAO,IAAI,IAAI,QAAQ,KAAK,SAAS,CAAC,KAAK,IAAI,IAAI,CAAC,CAAC;CAE3D,MAAM,SAAS,CAAC,GAAG,OAAO,QAAQ,CAAC,CAAC,CACjC,KAAK,CAAC,IAAI,YAAY;EAAE,MAAM,KAAK,IAAI,EAAE;EAAG;CAAM,EAAE,CAAC,CACrD,QAAQ,SAAyD,QAAQ,KAAK,IAAI,CAAC,CAAC,CACpF,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,SAAS,EAAE,KAAK,KAAK,cAAc,EAAE,KAAK,IAAI,CAAC,CAAC,CAC3E,MAAM,GAAG,KAAK;CAMjB,MAAM,WAAW,OAAO,EAAE,EAAE,SAAS;CAErC,OAAO,OAAO,KAAK,MAAM,OAAO;EAC9B,YAAY,KAAK,KAAK;EACtB,MAAM,KAAK;EACX,OAAO,KAAK;EACZ,UAAU,KAAK;EACf,iBAAiB,WAAW,IAAI,KAAK,QAAQ,WAAW;EACxD,MAAM,IAAI;EACV,SAAS,aAAa,KAAK,KAAK,MAAM,OAAO;EAC7C,SAAS,WAAW,KAAK,MAAM,OAAO;CACxC,EAAE;AACJ;AAEA,SAAgB,qBAAqB,WAAuB,IAAI,OAA4B;CAC1F,MAAM,yBAAS,IAAI,IAAoB;CACvC,KAAK,MAAM,QAAQ,WACjB,KAAK,SAAS,IAAI,QAAQ;EACxB,OAAO,IAAI,KAAK,OAAO,IAAI,EAAE,KAAK,KAAK,KAAK,IAAI,MAAM,EAAE;CAC1D,CAAC;CAEH,OAAO;AACT;AAEA,SAAS,YACP,OACA,SACiB;CACjB,MAAM,UAAU,QAAQ,UAAU,IAAI,IAAI,QAAQ,OAAO,IAAI;CAC7D,MAAM,OAAO,QAAQ,OAAO,IAAI,IAAI,QAAQ,IAAI,IAAI;CACpD,MAAM,QAAQ,QAAQ,QAAQ,IAAI,IAAI,QAAQ,KAAK,IAAI;CAEvD,OAAO,MAAM,QAAQ,SAAS;EAC5B,IAAI,QAAQ,sBAAsB,KAAK,iBAAiB,KAAA,GAAW,OAAO;EAC1E,IAAI,WAAW,CAAC,QAAQ,IAAI,KAAK,EAAE,GAAG,OAAO;EAC7C,IAAI,QAAQ,CAAC,KAAK,KAAK,MAAM,QAAQ,KAAK,IAAI,GAAG,CAAC,GAAG,OAAO;EAC5D,IAAI,OAAO;GACT,MAAM,OAAO,KAAK,YAAY;GAC9B,IAAI,OAAO,SAAS,YAAY,CAAC,MAAM,IAAI,IAAI,GAAG,OAAO;EAC3D;EACA,OAAO,QAAQ,YAAY,IAAI,KAAK;CACtC,CAAC;AACH;AAEA,SAAS,0BACP,cACA,OACuB;CACvB,IACE,aAAa,MAAM,WAAW,MAAM,UACpC,aAAa,MAAM,MAAM,MAAM,YAAY,SAAS,MAAM,QAAQ,GAElE,MAAM,IAAI,MAAM,qDAAqD;CAEvE,OAAO;AACT;;;;;;;;AASA,SAAS,YACP,OACA,OACA,cACiB;CACjB,MAAM,SAAS,CAAC,GAAG,IAAI,IAAI,aAAa,SAAS,KAAK,CAAC,CAAC;CACxD,MAAM,OAAO,IAAI,IAAI,UAAU,cAAc,MAAM,CAAC,CAAC,KAAK,QAAQ,CAAC,IAAI,MAAM,IAAI,KAAK,CAAC,CAAC;CACxF,MAAM,SAAS,MAAM,YAAY;CACjC,OAAO,MACJ,SAAS,SAAS;EACjB,MAAM,QAAQ,KAAK,IAAI,IAAI,KAAK;EAIhC,IAAI,UAAU,KAAK,OAAO,SAAS,GAAG,OAAO,CAAC;EAC9C,MAAM,OAAO,WAAW,MAAM,MAAM;EACpC,IAAI,UAAU,KAAK,SAAS,GAAG,OAAO,CAAC;EACvC,OAAO,CAAC;GAAE;GAAM;GAAM;EAAM,CAAC;CAC/B,CAAC,CAAC,CACD,MAAM,GAAG,MAAM,EAAE,OAAO,EAAE,QAAQ,EAAE,QAAQ,EAAE,SAAS,EAAE,KAAK,KAAK,cAAc,EAAE,KAAK,IAAI,CAAC,CAAC,CAC9F,KAAK,SAAS,KAAK,IAAI;AAC5B;AAEA,SAAS,WAAW,MAAqB,QAAwB;CAC/D,MAAM,QAAQ,KAAK,MAAM,YAAY;CACrC,IAAI,UAAU,UAAU,KAAK,KAAK,YAAY,CAAC,CAAC,SAAS,GAAG,OAAO,IAAI,GAAG,OAAO;CACjF,IAAI,MAAM,SAAS,MAAM,GAAG,OAAO;CACnC,IAAI,KAAK,KAAK,YAAY,CAAC,CAAC,SAAS,MAAM,GAAG,OAAO;CACrD,OAAO;AACT;AAEA,SAAS,YAAY,OAAwB,eAAiD;CAC5F,IAAI,cAAc,WAAW,GAAG,OAAO,CAAC;CACxC,MAAM,QAAQ,IAAI,IAAI,cAAc,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,SAAS,KAAK,EAAE,CAAC;CACtE,OAAO,MACJ,KAAK,UAAU;EACd;EACA,OACE,KAAK,SAAS,QAAQ,SAAS,MAAM,IAAI,IAAI,CAAC,CAAC,CAAC,SAChD,KAAK,UAAU,QAAQ,WACrB,cAAc,MAAM,SAAS,KAAK,UAAU,SAAS,MAAM,CAAC,CAC9D,CAAC,CAAC;CACN,EAAE,CAAC,CACF,QAAQ,SAAS,KAAK,QAAQ,CAAC,CAAC,CAChC,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,SAAS,EAAE,KAAK,KAAK,cAAc,EAAE,KAAK,IAAI,CAAC,CAAC,CAC3E,KAAK,SAAS,KAAK,IAAI;AAC5B;AAEA,SAAS,aAAa,MAAc,OAAuB;CACzD,MAAM,UAAU,KAAK,QAAQ,QAAQ,GAAG,CAAC,CAAC,KAAK;CAC/C,MAAM,MAAM,QAAQ,YAAY,CAAC,CAAC,QAAQ,MAAM,YAAY,CAAC;CAC7D,IAAI,MAAM,GAAG,OAAO,QAAQ,MAAM,GAAG,GAAG;CACxC,OAAO,QAAQ,MAAM,KAAK,IAAI,GAAG,MAAM,EAAE,GAAG,KAAK,IAAI,QAAQ,QAAQ,MAAM,MAAM,SAAS,GAAG,CAAC;AAChG;AAEA,SAAS,WAAW,MAAqB,OAAyB;CAChE,MAAM,QAAQ,GAAG,KAAK,MAAM,IAAI,KAAK,OAAO,YAAY;CACxD,MAAM,UAAoB,CAAC;CAC3B,IAAI,MAAM,SAAS,MAAM,YAAY,CAAC,GAAG,QAAQ,KAAK,QAAQ;CAC9D,IAAI,KAAK,UAAU,SAAS,GAAG,QAAQ,KAAK,SAAS;CACrD,IAAI,KAAK,SAAS,SAAS,GAAG,QAAQ,KAAK,QAAQ;CACnD,OAAO;AACT"}
|
|
@@ -51,6 +51,20 @@ interface KnowledgeRelation {
|
|
|
51
51
|
weight?: number;
|
|
52
52
|
metadata?: Record<string, unknown>;
|
|
53
53
|
}
|
|
54
|
+
/** A caller-declared vertex of a labeled relation graph. */
|
|
55
|
+
interface KnowledgeRelationNode {
|
|
56
|
+
id: KnowledgeId;
|
|
57
|
+
/** Caller vocabulary, such as `run`, `claim`, `page`, or `model`. */
|
|
58
|
+
kind: string;
|
|
59
|
+
label?: string;
|
|
60
|
+
metadata?: Record<string, unknown>;
|
|
61
|
+
}
|
|
62
|
+
/** A labeled multi-edge graph with one edge per `(sourceId, targetId, predicate)`. */
|
|
63
|
+
interface KnowledgeRelationGraph {
|
|
64
|
+
/** Declared nodes in declaration order; empty when the graph was built from relations alone. */
|
|
65
|
+
nodes: KnowledgeRelationNode[];
|
|
66
|
+
edges: KnowledgeRelation[];
|
|
67
|
+
}
|
|
54
68
|
interface KnowledgeUnit {
|
|
55
69
|
id: KnowledgeId;
|
|
56
70
|
title: string;
|
|
@@ -144,7 +158,7 @@ interface KnowledgeSearchResult {
|
|
|
144
158
|
reasons: string[];
|
|
145
159
|
}
|
|
146
160
|
interface KnowledgeLintFinding {
|
|
147
|
-
type: 'broken-link' | 'broken-citation' | 'ambiguous-citation' | 'orphan' | 'no-outlinks' | 'uncited-claim' | 'missing-source' | 'duplicate-title' | 'duplicate-page-id' | 'duplicate-source-hash' | 'missing-frontmatter' | 'ungradeable-evidence' | 'missing-evidence-path' | 'nonportable-evidence' | 'broken-contradiction' | 'invalid-invalidation';
|
|
161
|
+
type: 'broken-link' | 'broken-citation' | 'ambiguous-citation' | 'orphan' | 'no-outlinks' | 'uncited-claim' | 'missing-source' | 'duplicate-title' | 'duplicate-page-id' | 'duplicate-source-hash' | 'missing-frontmatter' | 'ungradeable-evidence' | 'missing-evidence-path' | 'nonportable-evidence' | 'broken-contradiction' | 'invalid-invalidation' | 'cites-invalidated';
|
|
148
162
|
severity: 'info' | 'warning' | 'error';
|
|
149
163
|
page?: string;
|
|
150
164
|
message: string;
|
|
@@ -324,5 +338,5 @@ interface KnowledgeRelease {
|
|
|
324
338
|
metadata?: Record<string, unknown>;
|
|
325
339
|
}
|
|
326
340
|
//#endregion
|
|
327
|
-
export {
|
|
328
|
-
//# sourceMappingURL=types-
|
|
341
|
+
export { SourceAnchor as A, KnowledgeUnit as C, ResearchClaimLedger as D, ResearchClaimEvidence as E, SourceRegistry as M, TrackedClaim as N, ResearchClaimRecord as O, KnowledgeSearchResult as S, KnowledgeWriteParseResult as T, KnowledgePolicy as _, KnowledgeBaseCandidate as a, KnowledgeRelationNode as b, KnowledgeEventType as c, KnowledgeGraphNode as d, KnowledgeId as f, KnowledgePageInvalidation as g, KnowledgePage as h, KNOWLEDGE_EVENT_TYPES as i, SourceRecord as j, ResearchSourceVersion as k, KnowledgeGraph as l, KnowledgeLintFinding as m, DeepQuestion as n, KnowledgeClaim as o, KnowledgeIndex as p, DeepQuestionKind as r, KnowledgeEvent as s, ClaimRef as t, KnowledgeGraphEdge as u, KnowledgeRelation as v, KnowledgeWriteBlock as w, KnowledgeRelease as x, KnowledgeRelationGraph as y };
|
|
342
|
+
//# sourceMappingURL=types-m2QB86fF.d.ts.map
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
{"version":3,"file":"types-m2QB86fF.d.ts","names":[],"sources":["../src/types.ts"],"mappings":";KAAY;UAEK;EACf;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA;EACA,WAAW;;UAGI;EACf,IAAI;EACJ;EACA;EACA;EACA;EACA;EACA,UAAU;;EAEV;;EAEA;EACA,WAAW;EACX;;UAGe;EACf;EACA,SAAS;;UAGM;EACf;EACA;EACA;;UAGe;EACf,IAAI;EACJ;EACA,MAAM;EACN;EACA;EACA,WAAW;;UAGI;EACf,UAAU;EACV,UAAU;EACV;EACA;EACA,WAAW;;;UAII;EACf,IAAI;;EAEJ;EACA;EACA,WAAW;;;UAII;;EAEf,OAAO;EACP,OAAO;;UAGQ;EACf,IAAI;EACJ;EACA;EACA,SAAS;EACT,YAAY;EACZ;EACA;EACA,WAAW;EACX;;;;;;;;;;UAWe;EACf;EACA;EACA;EACA;EACA;EACA,WAAW;;UAGI;EACf,IAAI;EACJ;EACA;EACA;EACA,aAAa;EACb;EACA;EACA;;EAEA,QAAQ;;EAER,cAAc;;EAEd,eAAe;;UAGA;EACf,IAAI;EACJ;EACA;EACA;EACA;EACA;EACA;;UAGe;EACf,QAAQ;EACR,QAAQ;EACR;EACA;;UAGe;EACf,OAAO;EACP,OAAO;;UAGQ;EACf;EACA;EACA,SAAS;EACT,OAAO;EACP,OAAO;;UAGQ;EACf,MAAM;;;;;;;EAON;;EAEA;;;;;;;;EAQA;EACA;EACA;EACA;;UAGe;EACf;EAkBA;EACA;EACA;EACA,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA,WAAW;;UAGI;EACf,IAAI;EACJ,OAAO;EACP;EACA;EACA;EACA;EACA,WAAW;;UAGI;EACf;EACA;;UAGe;EACf,QAAQ;EACR;;;;;;;;;;cAWW;KAWD,6BAA6B;UAExB;EACf;EACA,MAAM;EACN;EACA;EACA;EACA,WAAW;;;KAID;;;;;;;;;UAUK;EACf,MAAM;EACN;;EAEA;;EAEA;;EAEA;;EAEA;;;UAIe;EACf;;EAEA;;EAEA,iBAAiB;;EAEjB;;EAEA,aAAa;;;;;;EAMb;EACA;;;;;;;;;UAUe;EACf;EACA;EACA;EACA;EACA;EACA;EACA;;;UAIe;;EAEf,UAAU;;EAEV;;EAEA;;;;;;;;;;;UAYe;;EAEf;EACA;EACA;;EAEA,UAAU;EACV;;EAEA;;EAEA;EACA;;;;;;;;;;UAWe;;EAEf;EACA;;EAEA;;EAEA;;EAEA;;;;;EAKA;;;;;EAKA,eAAe;;EAEf,mBAAmB;EACnB,QAAQ;EACR,WAAW;;UAGI;EACf;EACA;EACA;EACA;EACA;EACA;EACA,WAAW"}
|
package/dist/viz/index.d.ts
CHANGED
|
@@ -1,4 +1,4 @@
|
|
|
1
|
-
import { d as KnowledgeGraphNode, l as KnowledgeGraph, u as KnowledgeGraphEdge } from "../types-
|
|
1
|
+
import { d as KnowledgeGraphNode, l as KnowledgeGraph, u as KnowledgeGraphEdge } from "../types-m2QB86fF.js";
|
|
2
2
|
//#region src/viz/index.d.ts
|
|
3
3
|
interface KnowledgeVizNode extends KnowledgeGraphNode {
|
|
4
4
|
degree: number;
|
package/docs/architecture.md
CHANGED
|
@@ -8,6 +8,7 @@ It owns the small set of primitives every serious agent knowledge system needs:
|
|
|
8
8
|
- generated knowledge pages and units
|
|
9
9
|
- claims with source references
|
|
10
10
|
- deterministic indexing, graph construction, search, and lint
|
|
11
|
+
- labeled relation graphs with one edge per `(source, target, predicate)` and neighbor, walk, and reachability queries
|
|
11
12
|
- retrieval/RAG candidate surfaces, gold-target scoring, and eval-loop adapters
|
|
12
13
|
- safe LLM write proposals
|
|
13
14
|
- eval-gated release confidence through `@tangle-network/agent-eval`
|
|
@@ -7,7 +7,7 @@ This module records two immutable links:
|
|
|
7
7
|
```text
|
|
8
8
|
exact visible knowledge snapshot
|
|
9
9
|
↓
|
|
10
|
-
retrieval receipt
|
|
10
|
+
retrieval receipt (references the snapshot by digest)
|
|
11
11
|
↓
|
|
12
12
|
selected returned result
|
|
13
13
|
↓
|
|
@@ -41,6 +41,27 @@ console.log(visibility.snapshotDigest)
|
|
|
41
41
|
|
|
42
42
|
A repeated path at the same origin is refused. The same stable page id may remain visible at different origins; ambiguity-safe citation resolution is handled separately.
|
|
43
43
|
|
|
44
|
+
Create the snapshot once per knowledge view and reuse it for every retrieval over that view. Snapshot creation hashes every visible page; each retrieval receipt over a reused snapshot then costs only its own results. `verifyKnowledgeVisibilitySnapshot()` proves a snapshot's schema and canonical digest, and runs once per snapshot object.
|
|
45
|
+
|
|
46
|
+
## Durable snapshot storage
|
|
47
|
+
|
|
48
|
+
The snapshot is the evidence a compact receipt points at, so a production adapter persists it once per view:
|
|
49
|
+
|
|
50
|
+
```ts
|
|
51
|
+
import {
|
|
52
|
+
encodeKnowledgeVisibilitySnapshot,
|
|
53
|
+
knowledgeVisibilityArtifactRef,
|
|
54
|
+
} from '@tangle-network/agent-knowledge'
|
|
55
|
+
|
|
56
|
+
const bytes = encodeKnowledgeVisibilitySnapshot(visibility)
|
|
57
|
+
await store.put('artifact://run/visibility.json', bytes)
|
|
58
|
+
const artifact = knowledgeVisibilityArtifactRef({ uri: 'artifact://run/visibility.json', bytes })
|
|
59
|
+
```
|
|
60
|
+
|
|
61
|
+
`encodeKnowledgeVisibilitySnapshot()` returns the snapshot's canonical bytes; `decodeKnowledgeVisibilitySnapshot()` parses stored bytes and verifies their digest. `knowledgeVisibilityArtifactRef()` binds the storage uri to the exact stored bytes: their sha256 digest and byte length. Pass the reference as `visibilityArtifact` when creating receipts so a later verifier can locate the snapshot.
|
|
62
|
+
|
|
63
|
+
The artifact reference is optional at this contract layer: a caller that retains the snapshot in another durable record may omit it. A production adapter should require a durable locator.
|
|
64
|
+
|
|
44
65
|
## Retrieval receipt
|
|
45
66
|
|
|
46
67
|
`createKnowledgeRetrievalReceipt()` binds:
|
|
@@ -49,17 +70,20 @@ A repeated path at the same origin is refused. The same stable page id may remai
|
|
|
49
70
|
- optional actor, profile, and execution identities;
|
|
50
71
|
- exact query;
|
|
51
72
|
- retriever id, version, and configuration digest;
|
|
52
|
-
-
|
|
73
|
+
- a compact visibility reference: the snapshot digest, the visible page count, and the optional artifact locator;
|
|
53
74
|
- every ranked returned page, origin, path, digest, score, snippet, and reason;
|
|
54
75
|
- trace or artifact evidence references;
|
|
55
76
|
- bounded scalar attributes;
|
|
56
77
|
- creation timestamp.
|
|
57
78
|
|
|
58
|
-
A result is accepted only when its exact page bytes, path, id, and origin occur in
|
|
79
|
+
The input takes the precomputed `visibility` snapshot. A result is accepted only when its exact page bytes, path, id, and origin occur in that snapshot. Ranks must be unique and contiguous from one. Scores must be finite, and normalized scores must lie in `[0, 1]`. The receipt stores the snapshot's reference, not its entries, so a receipt's size is bounded by its results.
|
|
59
80
|
|
|
60
81
|
```ts
|
|
61
82
|
import { canonicalCandidateDigest } from '@tangle-network/agent-interface'
|
|
62
|
-
import {
|
|
83
|
+
import {
|
|
84
|
+
createKnowledgeRetrievalReceipt,
|
|
85
|
+
KNOWLEDGE_SEARCH_RETRIEVER_ID,
|
|
86
|
+
} from '@tangle-network/agent-knowledge'
|
|
63
87
|
|
|
64
88
|
const receipt = createKnowledgeRetrievalReceipt({
|
|
65
89
|
runId,
|
|
@@ -68,17 +92,26 @@ const receipt = createKnowledgeRetrievalReceipt({
|
|
|
68
92
|
executionRef,
|
|
69
93
|
query: 'prior obstruction calibrated verifier',
|
|
70
94
|
retriever: {
|
|
71
|
-
id:
|
|
95
|
+
id: KNOWLEDGE_SEARCH_RETRIEVER_ID,
|
|
72
96
|
version: '1.0.0',
|
|
73
|
-
configDigest: canonicalCandidateDigest({
|
|
97
|
+
configDigest: canonicalCandidateDigest({ k1: 1.2, b: 0.75, limit: 5 }),
|
|
74
98
|
},
|
|
75
|
-
|
|
99
|
+
visibility,
|
|
100
|
+
visibilityArtifact: artifact,
|
|
76
101
|
results,
|
|
77
102
|
evidenceRefs: [{ kind: 'event', uri: `event://${runId}/retrieval-1` }],
|
|
78
103
|
})
|
|
79
104
|
```
|
|
80
105
|
|
|
81
|
-
|
|
106
|
+
## Verification split
|
|
107
|
+
|
|
108
|
+
Three checks make three different claims. The names say which:
|
|
109
|
+
|
|
110
|
+
- `verifyKnowledgeRetrievalReceipt(receipt)` recomputes the receipt digest and checks rank continuity, finite scores, and a well-formed visibility reference. It proves the receipt is internally consistent and unmutated. It does **not** prove that the results occur in the referenced snapshot, because the receipt does not carry the snapshot.
|
|
111
|
+
- `assertKnowledgeRetrievalMatchesVisibility(receipt, snapshot | visiblePages)` recomputes the snapshot digest and page count and proves every returned result occurs in that exact snapshot with the same page id and bytes.
|
|
112
|
+
- `assertKnowledgeRetrievalMatchesVisibilityArtifact(receipt, loadArtifact)` loads the referenced artifact, proves the stored bytes match the reference (digest and byte length), decodes and verifies the snapshot, and then proves the same join. A snapshot that cannot be obtained — no artifact reference, nothing stored at the uri, or a failing loader — raises `KnowledgeVisibilityUnavailableError` with the reason; it is never treated as an empty snapshot.
|
|
113
|
+
|
|
114
|
+
Schema version `1.0.0` receipts embedded the snapshot and are refused by every verifier.
|
|
82
115
|
|
|
83
116
|
## Use receipt
|
|
84
117
|
|
|
@@ -119,17 +152,21 @@ The use receipt copies the selected result's rank, page id, origin, path, and di
|
|
|
119
152
|
|
|
120
153
|
## Canonical serialization
|
|
121
154
|
|
|
122
|
-
Optional actor, profile, execution, consumer-digest, and evidence-excerpt fields are omitted when absent. They are never emitted with a JavaScript `undefined` value. Empty evidence and attribute collections remain explicit empty arrays or objects because they are part of the receipt contract.
|
|
155
|
+
Optional actor, profile, execution, consumer-digest, artifact, and evidence-excerpt fields are omitted when absent. They are never emitted with a JavaScript `undefined` value. Empty evidence and attribute collections remain explicit empty arrays or objects because they are part of the receipt contract.
|
|
123
156
|
|
|
124
157
|
Attribute values are deliberately limited to strings, finite numbers, booleans, and `null`. Nested arbitrary objects are refused rather than passed through a language-specific serializer. More structured evidence belongs in an artifact or a versioned contract referenced by digest.
|
|
125
158
|
|
|
126
|
-
The receipt digest covers the complete canonical material except the digest field itself
|
|
159
|
+
The receipt digest covers the complete canonical material except the digest field itself, including the visibility reference's snapshot digest and page count. Copying a digest onto modified content does not verify, and a truncated snapshot cannot satisfy a reference whose page count it no longer matches.
|
|
127
160
|
|
|
128
161
|
## What a receipt proves
|
|
129
162
|
|
|
130
|
-
A
|
|
163
|
+
A verified retrieval receipt alone proves:
|
|
164
|
+
|
|
165
|
+
> Under this exact run/actor/profile/execution identity, this exact retriever configuration searched the knowledge snapshot with this exact digest and page count, using this exact query, and returned these exact ranked page versions.
|
|
166
|
+
|
|
167
|
+
A retrieval receipt joined against its snapshot — through `assertKnowledgeRetrievalMatchesVisibility` or the artifact verifier — additionally proves:
|
|
131
168
|
|
|
132
|
-
>
|
|
169
|
+
> Every returned page version occurred, byte for byte, in that exact ordered snapshot.
|
|
133
170
|
|
|
134
171
|
A valid use receipt additionally proves:
|
|
135
172
|
|
|
@@ -156,6 +193,7 @@ Recommended event attributes:
|
|
|
156
193
|
knowledge.receipt.kind
|
|
157
194
|
knowledge.receipt.digest
|
|
158
195
|
knowledge.visibility.digest
|
|
196
|
+
knowledge.visibility.artifact_uri
|
|
159
197
|
knowledge.retriever.id
|
|
160
198
|
knowledge.retriever.version
|
|
161
199
|
knowledge.result.count
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@tangle-network/agent-knowledge",
|
|
3
|
-
"version": "
|
|
3
|
+
"version": "10.7.0",
|
|
4
4
|
"description": "Build, search, evaluate, and improve source-backed knowledge bases.",
|
|
5
5
|
"homepage": "https://github.com/tangle-network/agent-knowledge#readme",
|
|
6
6
|
"repository": {
|
|
@@ -83,15 +83,15 @@
|
|
|
83
83
|
"zod": "^4.4.3"
|
|
84
84
|
},
|
|
85
85
|
"peerDependencies": {
|
|
86
|
-
"@tangle-network/agent-eval": ">=0.
|
|
87
|
-
"@tangle-network/agent-interface": "^1.
|
|
86
|
+
"@tangle-network/agent-eval": ">=0.163.2 <0.164.0",
|
|
87
|
+
"@tangle-network/agent-interface": "^1.4.0"
|
|
88
88
|
},
|
|
89
89
|
"devDependencies": {
|
|
90
90
|
"@arethetypeswrong/cli": "^0.18.5",
|
|
91
91
|
"@biomejs/biome": "^2.5.5",
|
|
92
92
|
"@neo4j-labs/agent-memory": "0.4.1",
|
|
93
|
-
"@tangle-network/agent-eval": "0.
|
|
94
|
-
"@tangle-network/agent-interface": "1.
|
|
93
|
+
"@tangle-network/agent-eval": "0.163.2",
|
|
94
|
+
"@tangle-network/agent-interface": "1.4.0",
|
|
95
95
|
"@types/node": "^26.1.1",
|
|
96
96
|
"mem0ai": "3.1.2",
|
|
97
97
|
"oxc-parser": "0.144.0",
|