@tangle-network/agent-knowledge 6.0.0 → 6.1.1

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (57) hide show
  1. package/CHANGELOG.md +17 -0
  2. package/README.md +1 -1
  3. package/dist/benchmarks/index.d.ts +2 -53
  4. package/dist/benchmarks/index.js +2 -49
  5. package/dist/benchmarks-CmW6iORW.js +2718 -0
  6. package/dist/benchmarks-CmW6iORW.js.map +1 -0
  7. package/dist/cli.d.ts +1 -1
  8. package/dist/cli.js +180 -274
  9. package/dist/cli.js.map +1 -1
  10. package/dist/ids-DRqPZ42_.js +15 -0
  11. package/dist/ids-DRqPZ42_.js.map +1 -0
  12. package/dist/index-CGBctbit.d.ts +857 -0
  13. package/dist/index-CGBctbit.d.ts.map +1 -0
  14. package/dist/index-CIW3G4s_.d.ts +680 -0
  15. package/dist/index-CIW3G4s_.d.ts.map +1 -0
  16. package/dist/index.d.ts +1671 -1868
  17. package/dist/index.d.ts.map +1 -0
  18. package/dist/index.js +5836 -6528
  19. package/dist/index.js.map +1 -1
  20. package/dist/inspect-D5iarJc2.js +1864 -0
  21. package/dist/inspect-D5iarJc2.js.map +1 -0
  22. package/dist/memory/index.d.ts +3 -8
  23. package/dist/memory/index.js +3 -81
  24. package/dist/memory-C6KPRhoU.js +4494 -0
  25. package/dist/memory-C6KPRhoU.js.map +1 -0
  26. package/dist/search-CP0QtBJZ.js +113 -0
  27. package/dist/search-CP0QtBJZ.js.map +1 -0
  28. package/dist/sources/index.d.ts +212 -205
  29. package/dist/sources/index.d.ts.map +1 -0
  30. package/dist/sources/index.js +614 -33
  31. package/dist/sources/index.js.map +1 -1
  32. package/dist/types-DcCCzreS.d.ts +175 -0
  33. package/dist/types-DcCCzreS.d.ts.map +1 -0
  34. package/dist/viz/index.d.ts +23 -22
  35. package/dist/viz/index.d.ts.map +1 -0
  36. package/dist/viz/index.js +134 -10
  37. package/dist/viz/index.js.map +1 -1
  38. package/package.json +22 -11
  39. package/dist/benchmarks/index.js.map +0 -1
  40. package/dist/chunk-46YPZHAX.js +0 -5443
  41. package/dist/chunk-46YPZHAX.js.map +0 -1
  42. package/dist/chunk-4PNXQ2NT.js +0 -147
  43. package/dist/chunk-4PNXQ2NT.js.map +0 -1
  44. package/dist/chunk-AKYJG2MR.js +0 -2183
  45. package/dist/chunk-AKYJG2MR.js.map +0 -1
  46. package/dist/chunk-DQ3PDMDP.js +0 -115
  47. package/dist/chunk-DQ3PDMDP.js.map +0 -1
  48. package/dist/chunk-MYFM6LKH.js +0 -551
  49. package/dist/chunk-MYFM6LKH.js.map +0 -1
  50. package/dist/chunk-PVCSESAF.js +0 -3153
  51. package/dist/chunk-PVCSESAF.js.map +0 -1
  52. package/dist/chunk-YMKHCTS2.js +0 -19
  53. package/dist/chunk-YMKHCTS2.js.map +0 -1
  54. package/dist/index-C--N5wQV.d.ts +0 -796
  55. package/dist/memory/index.js.map +0 -1
  56. package/dist/types-6x0OpfW6.d.ts +0 -173
  57. package/dist/types-BY-xLVw-.d.ts +0 -622
@@ -0,0 +1,113 @@
1
+ //#region src/search.ts
2
+ const RRF_K = 60;
3
+ const STOP_WORDS = /* @__PURE__ */ new Set([
4
+ "the",
5
+ "is",
6
+ "a",
7
+ "an",
8
+ "what",
9
+ "how",
10
+ "are",
11
+ "was",
12
+ "were",
13
+ "to",
14
+ "for",
15
+ "of",
16
+ "with",
17
+ "by",
18
+ "in",
19
+ "on",
20
+ "and"
21
+ ]);
22
+ function searchKnowledge(index, query, limit = 10) {
23
+ const trimmed = query.trim();
24
+ if (trimmed === "") return [];
25
+ const tokenRanked = rankByTokens(index.pages, trimmed);
26
+ const graphRanked = rankByGraph(index.pages, tokenRanked);
27
+ const scores = reciprocalRankFusion([tokenRanked.map((p) => p.id), graphRanked.map((p) => p.id)]);
28
+ const byId = new Map(index.pages.map((page) => [page.id, page]));
29
+ const ranked = [...scores.entries()].map(([id, score]) => ({
30
+ page: byId.get(id),
31
+ score
32
+ })).filter((item) => Boolean(item.page)).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).slice(0, limit);
33
+ const topScore = ranked[0]?.score ?? 0;
34
+ return ranked.map((item, i) => ({
35
+ page: item.page,
36
+ score: item.score,
37
+ rrfScore: item.score,
38
+ normalizedScore: topScore > 0 ? item.score / topScore : 0,
39
+ rank: i + 1,
40
+ snippet: buildSnippet(item.page.text, trimmed),
41
+ reasons: reasonsFor(item.page, trimmed)
42
+ }));
43
+ }
44
+ function tokenizeQuery(query) {
45
+ const raw = query.toLowerCase().split(/[\s,,。!?、;:""''()()\-_/\\·~~…]+/).filter((token) => token.length > 1 && !STOP_WORDS.has(token));
46
+ const tokens = [];
47
+ for (const token of raw) {
48
+ if (/[\u4e00-\u9fff\u3400-\u4dbf]/.test(token) && token.length > 2) {
49
+ const chars = [...token];
50
+ for (let i = 0; i < chars.length - 1; i++) tokens.push(chars[i] + chars[i + 1]);
51
+ tokens.push(...chars);
52
+ }
53
+ tokens.push(token);
54
+ }
55
+ return [...new Set(tokens)];
56
+ }
57
+ function reciprocalRankFusion(rankLists, k = RRF_K) {
58
+ const scores = /* @__PURE__ */ new Map();
59
+ for (const list of rankLists) list.forEach((id, idx) => {
60
+ scores.set(id, (scores.get(id) ?? 0) + 1 / (k + idx + 1));
61
+ });
62
+ return scores;
63
+ }
64
+ function rankByTokens(pages, query) {
65
+ const tokens = tokenizeQuery(query);
66
+ const effective = tokens.length > 0 ? tokens : [query.toLowerCase()];
67
+ return pages.map((page) => ({
68
+ page,
69
+ score: tokenScore(page, query, effective)
70
+ })).filter((item) => item.score > 0).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).map((item) => item.page);
71
+ }
72
+ function rankByGraph(pages, tokenRanked) {
73
+ if (tokenRanked.length === 0) return [];
74
+ const seeds = new Set(tokenRanked.slice(0, 5).map((page) => page.id));
75
+ return pages.map((page) => ({
76
+ page,
77
+ score: page.outLinks.filter((link) => seeds.has(link)).length + page.sourceIds.filter((source) => tokenRanked.some((seed) => seed.sourceIds.includes(source))).length
78
+ })).filter((item) => item.score > 0).sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path)).map((item) => item.page);
79
+ }
80
+ function tokenScore(page, query, tokens) {
81
+ const title = page.title.toLowerCase();
82
+ const path = page.path.toLowerCase();
83
+ const body = page.text.toLowerCase();
84
+ const phrase = query.toLowerCase();
85
+ let score = 0;
86
+ if (path.endsWith(`${phrase}.md`) || title === phrase) score += 200;
87
+ if (title.includes(phrase)) score += 50;
88
+ if (body.includes(phrase)) score += 20;
89
+ for (const token of tokens) {
90
+ if (title.includes(token)) score += 5;
91
+ if (body.includes(token)) score += 1;
92
+ if (path.includes(token)) score += 3;
93
+ }
94
+ return score;
95
+ }
96
+ function buildSnippet(text, query) {
97
+ const compact = text.replace(/\s+/g, " ").trim();
98
+ const idx = compact.toLowerCase().indexOf(query.toLowerCase());
99
+ if (idx < 0) return compact.slice(0, 180);
100
+ return compact.slice(Math.max(0, idx - 80), Math.min(compact.length, idx + query.length + 100));
101
+ }
102
+ function reasonsFor(page, query) {
103
+ const lower = `${page.title}\n${page.text}`.toLowerCase();
104
+ const reasons = [];
105
+ if (lower.includes(query.toLowerCase())) reasons.push("phrase");
106
+ if (page.sourceIds.length > 0) reasons.push("sourced");
107
+ if (page.outLinks.length > 0) reasons.push("linked");
108
+ return reasons;
109
+ }
110
+ //#endregion
111
+ export { searchKnowledge as n, tokenizeQuery as r, reciprocalRankFusion as t };
112
+
113
+ //# sourceMappingURL=search-CP0QtBJZ.js.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"search-CP0QtBJZ.js","names":[],"sources":["../src/search.ts"],"sourcesContent":["import type { KnowledgeIndex, KnowledgePage, KnowledgeSearchResult } from './types'\n\nconst RRF_K = 60\nconst STOP_WORDS = new Set([\n 'the',\n 'is',\n 'a',\n 'an',\n 'what',\n 'how',\n 'are',\n 'was',\n 'were',\n 'to',\n 'for',\n 'of',\n 'with',\n 'by',\n 'in',\n 'on',\n 'and',\n])\n\nexport function searchKnowledge(\n index: KnowledgeIndex,\n query: string,\n limit = 10,\n): KnowledgeSearchResult[] {\n const trimmed = query.trim()\n if (trimmed === '') return []\n const tokenRanked = rankByTokens(index.pages, trimmed)\n const graphRanked = rankByGraph(index.pages, tokenRanked)\n const scores = reciprocalRankFusion([tokenRanked.map((p) => p.id), graphRanked.map((p) => p.id)])\n const byId = new Map(index.pages.map((page) => [page.id, page]))\n\n const ranked = [...scores.entries()]\n .map(([id, score]) => ({ page: byId.get(id), score }))\n .filter((item): item is { page: KnowledgePage; score: number } => Boolean(item.page))\n .sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path))\n .slice(0, limit)\n\n // Normalize against the top hit so callers can compare against natural\n // [0, 1] thresholds. Raw RRF values are typically ~0.016, which reads as\n // \"no relevance\" to humans even when the result is the best available.\n // The top hit becomes 1 by definition; lower-ranked hits scale linearly.\n const topScore = ranked[0]?.score ?? 0\n\n return ranked.map((item, i) => ({\n page: item.page,\n score: item.score,\n rrfScore: item.score,\n normalizedScore: topScore > 0 ? item.score / topScore : 0,\n rank: i + 1,\n snippet: buildSnippet(item.page.text, trimmed),\n reasons: reasonsFor(item.page, trimmed),\n }))\n}\n\nexport function tokenizeQuery(query: string): string[] {\n const raw = query\n .toLowerCase()\n .split(/[\\s,,。!?、;:\"\"''()()\\-_/\\\\·~~…]+/)\n .filter((token) => token.length > 1 && !STOP_WORDS.has(token))\n const tokens: string[] = []\n for (const token of raw) {\n if (/[\\u4e00-\\u9fff\\u3400-\\u4dbf]/.test(token) && token.length > 2) {\n const chars = [...token]\n for (let i = 0; i < chars.length - 1; i++) tokens.push(chars[i]! + chars[i + 1]!)\n tokens.push(...chars)\n }\n tokens.push(token)\n }\n return [...new Set(tokens)]\n}\n\nexport function reciprocalRankFusion(rankLists: string[][], k = RRF_K): Map<string, number> {\n const scores = new Map<string, number>()\n for (const list of rankLists) {\n list.forEach((id, idx) => {\n scores.set(id, (scores.get(id) ?? 0) + 1 / (k + idx + 1))\n })\n }\n return scores\n}\n\nfunction rankByTokens(pages: KnowledgePage[], query: string): KnowledgePage[] {\n const tokens = tokenizeQuery(query)\n const effective = tokens.length > 0 ? tokens : [query.toLowerCase()]\n return pages\n .map((page) => ({ page, score: tokenScore(page, query, effective) }))\n .filter((item) => item.score > 0)\n .sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path))\n .map((item) => item.page)\n}\n\nfunction rankByGraph(pages: KnowledgePage[], tokenRanked: KnowledgePage[]): KnowledgePage[] {\n if (tokenRanked.length === 0) return []\n const seeds = new Set(tokenRanked.slice(0, 5).map((page) => page.id))\n return pages\n .map((page) => ({\n page,\n score:\n page.outLinks.filter((link) => seeds.has(link)).length +\n page.sourceIds.filter((source) =>\n tokenRanked.some((seed) => seed.sourceIds.includes(source)),\n ).length,\n }))\n .filter((item) => item.score > 0)\n .sort((a, b) => b.score - a.score || a.page.path.localeCompare(b.page.path))\n .map((item) => item.page)\n}\n\nfunction tokenScore(page: KnowledgePage, query: string, tokens: string[]): number {\n const title = page.title.toLowerCase()\n const path = page.path.toLowerCase()\n const body = page.text.toLowerCase()\n const phrase = query.toLowerCase()\n let score = 0\n if (path.endsWith(`${phrase}.md`) || title === phrase) score += 200\n if (title.includes(phrase)) score += 50\n if (body.includes(phrase)) score += 20\n for (const token of tokens) {\n if (title.includes(token)) score += 5\n if (body.includes(token)) score += 1\n if (path.includes(token)) score += 3\n }\n return score\n}\n\nfunction buildSnippet(text: string, query: string): string {\n const compact = text.replace(/\\s+/g, ' ').trim()\n const idx = compact.toLowerCase().indexOf(query.toLowerCase())\n if (idx < 0) return compact.slice(0, 180)\n return compact.slice(Math.max(0, idx - 80), Math.min(compact.length, idx + query.length + 100))\n}\n\nfunction reasonsFor(page: KnowledgePage, query: string): string[] {\n const lower = `${page.title}\\n${page.text}`.toLowerCase()\n const reasons: string[] = []\n if (lower.includes(query.toLowerCase())) reasons.push('phrase')\n if (page.sourceIds.length > 0) reasons.push('sourced')\n if (page.outLinks.length > 0) reasons.push('linked')\n return reasons\n}\n"],"mappings":";AAEA,MAAM,QAAQ;AACd,MAAM,6BAAa,IAAI,IAAI;CACzB;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;CACA;AACF,CAAC;AAED,SAAgB,gBACd,OACA,OACA,QAAQ,IACiB;CACzB,MAAM,UAAU,MAAM,KAAK;CAC3B,IAAI,YAAY,IAAI,OAAO,CAAC;CAC5B,MAAM,cAAc,aAAa,MAAM,OAAO,OAAO;CACrD,MAAM,cAAc,YAAY,MAAM,OAAO,WAAW;CACxD,MAAM,SAAS,qBAAqB,CAAC,YAAY,KAAK,MAAM,EAAE,EAAE,GAAG,YAAY,KAAK,MAAM,EAAE,EAAE,CAAC,CAAC;CAChG,MAAM,OAAO,IAAI,IAAI,MAAM,MAAM,KAAK,SAAS,CAAC,KAAK,IAAI,IAAI,CAAC,CAAC;CAE/D,MAAM,SAAS,CAAC,GAAG,OAAO,QAAQ,CAAC,CAAC,CACjC,KAAK,CAAC,IAAI,YAAY;EAAE,MAAM,KAAK,IAAI,EAAE;EAAG;CAAM,EAAE,CAAC,CACrD,QAAQ,SAAyD,QAAQ,KAAK,IAAI,CAAC,CAAC,CACpF,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,SAAS,EAAE,KAAK,KAAK,cAAc,EAAE,KAAK,IAAI,CAAC,CAAC,CAC3E,MAAM,GAAG,KAAK;CAMjB,MAAM,WAAW,OAAO,EAAE,EAAE,SAAS;CAErC,OAAO,OAAO,KAAK,MAAM,OAAO;EAC9B,MAAM,KAAK;EACX,OAAO,KAAK;EACZ,UAAU,KAAK;EACf,iBAAiB,WAAW,IAAI,KAAK,QAAQ,WAAW;EACxD,MAAM,IAAI;EACV,SAAS,aAAa,KAAK,KAAK,MAAM,OAAO;EAC7C,SAAS,WAAW,KAAK,MAAM,OAAO;CACxC,EAAE;AACJ;AAEA,SAAgB,cAAc,OAAyB;CACrD,MAAM,MAAM,MACT,YAAY,CAAC,CACb,MAAM,iCAAiC,CAAC,CACxC,QAAQ,UAAU,MAAM,SAAS,KAAK,CAAC,WAAW,IAAI,KAAK,CAAC;CAC/D,MAAM,SAAmB,CAAC;CAC1B,KAAK,MAAM,SAAS,KAAK;EACvB,IAAI,+BAA+B,KAAK,KAAK,KAAK,MAAM,SAAS,GAAG;GAClE,MAAM,QAAQ,CAAC,GAAG,KAAK;GACvB,KAAK,IAAI,IAAI,GAAG,IAAI,MAAM,SAAS,GAAG,KAAK,OAAO,KAAK,MAAM,KAAM,MAAM,IAAI,EAAG;GAChF,OAAO,KAAK,GAAG,KAAK;EACtB;EACA,OAAO,KAAK,KAAK;CACnB;CACA,OAAO,CAAC,GAAG,IAAI,IAAI,MAAM,CAAC;AAC5B;AAEA,SAAgB,qBAAqB,WAAuB,IAAI,OAA4B;CAC1F,MAAM,yBAAS,IAAI,IAAoB;CACvC,KAAK,MAAM,QAAQ,WACjB,KAAK,SAAS,IAAI,QAAQ;EACxB,OAAO,IAAI,KAAK,OAAO,IAAI,EAAE,KAAK,KAAK,KAAK,IAAI,MAAM,EAAE;CAC1D,CAAC;CAEH,OAAO;AACT;AAEA,SAAS,aAAa,OAAwB,OAAgC;CAC5E,MAAM,SAAS,cAAc,KAAK;CAClC,MAAM,YAAY,OAAO,SAAS,IAAI,SAAS,CAAC,MAAM,YAAY,CAAC;CACnE,OAAO,MACJ,KAAK,UAAU;EAAE;EAAM,OAAO,WAAW,MAAM,OAAO,SAAS;CAAE,EAAE,CAAC,CACpE,QAAQ,SAAS,KAAK,QAAQ,CAAC,CAAC,CAChC,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,SAAS,EAAE,KAAK,KAAK,cAAc,EAAE,KAAK,IAAI,CAAC,CAAC,CAC3E,KAAK,SAAS,KAAK,IAAI;AAC5B;AAEA,SAAS,YAAY,OAAwB,aAA+C;CAC1F,IAAI,YAAY,WAAW,GAAG,OAAO,CAAC;CACtC,MAAM,QAAQ,IAAI,IAAI,YAAY,MAAM,GAAG,CAAC,CAAC,CAAC,KAAK,SAAS,KAAK,EAAE,CAAC;CACpE,OAAO,MACJ,KAAK,UAAU;EACd;EACA,OACE,KAAK,SAAS,QAAQ,SAAS,MAAM,IAAI,IAAI,CAAC,CAAC,CAAC,SAChD,KAAK,UAAU,QAAQ,WACrB,YAAY,MAAM,SAAS,KAAK,UAAU,SAAS,MAAM,CAAC,CAC5D,CAAC,CAAC;CACN,EAAE,CAAC,CACF,QAAQ,SAAS,KAAK,QAAQ,CAAC,CAAC,CAChC,MAAM,GAAG,MAAM,EAAE,QAAQ,EAAE,SAAS,EAAE,KAAK,KAAK,cAAc,EAAE,KAAK,IAAI,CAAC,CAAC,CAC3E,KAAK,SAAS,KAAK,IAAI;AAC5B;AAEA,SAAS,WAAW,MAAqB,OAAe,QAA0B;CAChF,MAAM,QAAQ,KAAK,MAAM,YAAY;CACrC,MAAM,OAAO,KAAK,KAAK,YAAY;CACnC,MAAM,OAAO,KAAK,KAAK,YAAY;CACnC,MAAM,SAAS,MAAM,YAAY;CACjC,IAAI,QAAQ;CACZ,IAAI,KAAK,SAAS,GAAG,OAAO,IAAI,KAAK,UAAU,QAAQ,SAAS;CAChE,IAAI,MAAM,SAAS,MAAM,GAAG,SAAS;CACrC,IAAI,KAAK,SAAS,MAAM,GAAG,SAAS;CACpC,KAAK,MAAM,SAAS,QAAQ;EAC1B,IAAI,MAAM,SAAS,KAAK,GAAG,SAAS;EACpC,IAAI,KAAK,SAAS,KAAK,GAAG,SAAS;EACnC,IAAI,KAAK,SAAS,KAAK,GAAG,SAAS;CACrC;CACA,OAAO;AACT;AAEA,SAAS,aAAa,MAAc,OAAuB;CACzD,MAAM,UAAU,KAAK,QAAQ,QAAQ,GAAG,CAAC,CAAC,KAAK;CAC/C,MAAM,MAAM,QAAQ,YAAY,CAAC,CAAC,QAAQ,MAAM,YAAY,CAAC;CAC7D,IAAI,MAAM,GAAG,OAAO,QAAQ,MAAM,GAAG,GAAG;CACxC,OAAO,QAAQ,MAAM,KAAK,IAAI,GAAG,MAAM,EAAE,GAAG,KAAK,IAAI,QAAQ,QAAQ,MAAM,MAAM,SAAS,GAAG,CAAC;AAChG;AAEA,SAAS,WAAW,MAAqB,OAAyB;CAChE,MAAM,QAAQ,GAAG,KAAK,MAAM,IAAI,KAAK,OAAO,YAAY;CACxD,MAAM,UAAoB,CAAC;CAC3B,IAAI,MAAM,SAAS,MAAM,YAAY,CAAC,GAAG,QAAQ,KAAK,QAAQ;CAC9D,IAAI,KAAK,UAAU,SAAS,GAAG,QAAQ,KAAK,SAAS;CACrD,IAAI,KAAK,SAAS,SAAS,GAAG,QAAQ,KAAK,QAAQ;CACnD,OAAO;AACT"}
@@ -1,3 +1,4 @@
1
+ //#region src/sources/types.d.ts
1
2
  /**
2
3
  * Pluggable knowledge source contract.
3
4
  *
@@ -30,27 +31,27 @@
30
31
  * injected for deterministic tests of change-detection windows.
31
32
  */
32
33
  interface FetchOpts {
33
- /** Abort signal forwarded to the underlying HTTP fetcher. */
34
- signal?: AbortSignal;
35
- /** Absolute path under which the source may cache raw bytes. */
36
- cacheDir?: string;
37
- /** Clock injection for deterministic tests. */
38
- now?: () => Date;
39
- /**
40
- * Maximum number of authority pages the source should fetch in this call.
41
- * Sources MUST respect this bound — exhaustively crawling Cornell LII on
42
- * every cron tick would be both rude and slow. Default is source-specific.
43
- */
44
- limit?: number;
45
- /**
46
- * Source-specific selector string. Examples:
47
- * - cornell-lii: `'uscode/text/18/1836'` or `'wex/non-compete'`
48
- * - irs-publications: `'index'` or `'p15'`
49
- * - state-sos: opaque, see `StateSosSourceConfig`
50
- *
51
- * Sources that don't need a selector ignore this field.
52
- */
53
- selector?: string;
34
+ /** Abort signal forwarded to the underlying HTTP fetcher. */
35
+ signal?: AbortSignal;
36
+ /** Absolute path under which the source may cache raw bytes. */
37
+ cacheDir?: string;
38
+ /** Clock injection for deterministic tests. */
39
+ now?: () => Date;
40
+ /**
41
+ * Maximum number of authority pages the source should fetch in this call.
42
+ * Sources MUST respect this bound — exhaustively crawling Cornell LII on
43
+ * every cron tick would be both rude and slow. Default is source-specific.
44
+ */
45
+ limit?: number;
46
+ /**
47
+ * Source-specific selector string. Examples:
48
+ * - cornell-lii: `'uscode/text/18/1836'` or `'wex/non-compete'`
49
+ * - irs-publications: `'index'` or `'p15'`
50
+ * - state-sos: opaque, see `StateSosSourceConfig`
51
+ *
52
+ * Sources that don't need a selector ignore this field.
53
+ */
54
+ selector?: string;
54
55
  }
55
56
  /**
56
57
  * The standard provenance shape every fragment carries. Kept separate from
@@ -58,68 +59,68 @@ interface FetchOpts {
58
59
  * also dragging the body text.
59
60
  */
60
61
  interface FragmentProvenance {
61
- /** Canonical URL the fragment was extracted from. */
62
- url: string;
63
- /**
64
- * Source-attested timestamp: the time the AUTHORITY last updated this
65
- * content, as reported by the source (Last-Modified header, in-page
66
- * effective date, registry generated-at, etc). Falls back to the fetch
67
- * time only when the authority publishes no timestamp.
68
- */
69
- sourceUpdatedAt: string;
70
- /** ISO timestamp the fragment was fetched. */
71
- fetchedAt: string;
72
- /**
73
- * Jurisdiction the content is binding within, if applicable. Use ISO
74
- * country code, US state abbreviation, or 'US-FED' for federal scope.
75
- * Statute sources MUST populate this; reference / encyclopedia sources
76
- * MAY leave it undefined.
77
- */
78
- jurisdiction?: string;
79
- /**
80
- * True iff the configured URL returned an acceptable response and the
81
- * expected content was extracted. False on a block page, rate-limit
82
- * response, 4xx/5xx, or selector miss. This is not publisher authentication
83
- * or cryptographic content verification. Consumers MUST refuse to promote
84
- * `verifiable: false` fragments into citable knowledge.
85
- */
86
- verifiable: boolean;
87
- /** If `verifiable === false`, the reason — surfaced to operators. */
88
- unverifiableReason?: string;
62
+ /** Canonical URL the fragment was extracted from. */
63
+ url: string;
64
+ /**
65
+ * Source-attested timestamp: the time the AUTHORITY last updated this
66
+ * content, as reported by the source (Last-Modified header, in-page
67
+ * effective date, registry generated-at, etc). Falls back to the fetch
68
+ * time only when the authority publishes no timestamp.
69
+ */
70
+ sourceUpdatedAt: string;
71
+ /** ISO timestamp the fragment was fetched. */
72
+ fetchedAt: string;
73
+ /**
74
+ * Jurisdiction the content is binding within, if applicable. Use ISO
75
+ * country code, US state abbreviation, or 'US-FED' for federal scope.
76
+ * Statute sources MUST populate this; reference / encyclopedia sources
77
+ * MAY leave it undefined.
78
+ */
79
+ jurisdiction?: string;
80
+ /**
81
+ * True iff the configured URL returned an acceptable response and the
82
+ * expected content was extracted. False on a block page, rate-limit
83
+ * response, 4xx/5xx, or selector miss. This is not publisher authentication
84
+ * or cryptographic content verification. Consumers MUST refuse to promote
85
+ * `verifiable: false` fragments into citable knowledge.
86
+ */
87
+ verifiable: boolean;
88
+ /** If `verifiable === false`, the reason — surfaced to operators. */
89
+ unverifiableReason?: string;
89
90
  }
90
91
  /**
91
92
  * One unit of authoritative content. Stable hash on `(id, body)` lets change
92
93
  * detection reason about identity across snapshots.
93
94
  */
94
95
  interface KnowledgeFragment {
95
- /**
96
- * Stable identity within (sourceId, selector-space). Two fetches against
97
- * the same authority section MUST produce the same `id`. The (sourceId,
98
- * id) pair is the primary key for change detection.
99
- */
100
- id: string;
101
- /** Free-form title — section heading, publication name, etc. */
102
- title: string;
103
- /** Body text, normalised: no HTML tags, line breaks preserved. */
104
- body: string;
105
- /** SHA-256 of `body`. Pre-computed so consumers don't re-hash on diff. */
106
- bodyHash: string;
107
- provenance: FragmentProvenance;
108
- /**
109
- * Eval dimensions an agent-eval campaign should re-score when this
110
- * fragment changes. Examples: `citation_hygiene`, `jurisdictional_accuracy`,
111
- * `tax_compliance`, `regulatory_currency`. The eval cron treats this as a
112
- * set, not a contract — adding a new dimension is non-breaking.
113
- *
114
- * This is the load-bearing field for the continuous-ingestion story: a
115
- * Ryan-LLC-style ruling vacates the FTC non-compete rule → the source
116
- * returns a fragment with `jurisdictional_accuracy` in this list →
117
- * `detectChanges()` emits a `KnowledgeChange` carrying that hint → the
118
- * cron knows exactly which agent-eval campaigns to re-run.
119
- */
120
- dimensionHints: string[];
121
- /** Arbitrary source-specific metadata for debugging / connector wiring. */
122
- metadata?: Record<string, unknown>;
96
+ /**
97
+ * Stable identity within (sourceId, selector-space). Two fetches against
98
+ * the same authority section MUST produce the same `id`. The (sourceId,
99
+ * id) pair is the primary key for change detection.
100
+ */
101
+ id: string;
102
+ /** Free-form title — section heading, publication name, etc. */
103
+ title: string;
104
+ /** Body text, normalised: no HTML tags, line breaks preserved. */
105
+ body: string;
106
+ /** SHA-256 of `body`. Pre-computed so consumers don't re-hash on diff. */
107
+ bodyHash: string;
108
+ provenance: FragmentProvenance;
109
+ /**
110
+ * Eval dimensions an agent-eval campaign should re-score when this
111
+ * fragment changes. Examples: `citation_hygiene`, `jurisdictional_accuracy`,
112
+ * `tax_compliance`, `regulatory_currency`. The eval cron treats this as a
113
+ * set, not a contract — adding a new dimension is non-breaking.
114
+ *
115
+ * This is the load-bearing field for the continuous-ingestion story: a
116
+ * Ryan-LLC-style ruling vacates the FTC non-compete rule → the source
117
+ * returns a fragment with `jurisdictional_accuracy` in this list →
118
+ * `detectChanges()` emits a `KnowledgeChange` carrying that hint → the
119
+ * cron knows exactly which agent-eval campaigns to re-run.
120
+ */
121
+ dimensionHints: string[];
122
+ /** Arbitrary source-specific metadata for debugging / connector wiring. */
123
+ metadata?: Record<string, unknown>;
123
124
  }
124
125
  /**
125
126
  * One pluggable knowledge source.
@@ -130,48 +131,49 @@ interface KnowledgeFragment {
130
131
  * the per-tenant isolation contract; see README).
131
132
  */
132
133
  interface KnowledgeSource {
133
- /** Stable id — used to key freshness state. MUST NOT change once shipped. */
134
- id: string;
135
- /** Human-readable name for dashboards. */
136
- name: string;
137
- /** One-sentence description: what authority + scope. */
138
- description: string;
139
- /**
140
- * Pull fragments for this source. Sources MUST:
141
- * - rate-limit themselves (>=1 req/sec per source by convention)
142
- * - send a polite User-Agent
143
- * - cache to disk when `opts.cacheDir` is set
144
- * - mark `verifiable: false` rather than throwing on parse/block
145
- * - honour `opts.signal`
146
- * - honour `opts.limit`
147
- */
148
- fetch(opts: FetchOpts): Promise<KnowledgeFragment[]>;
134
+ /** Stable id — used to key freshness state. MUST NOT change once shipped. */
135
+ id: string;
136
+ /** Human-readable name for dashboards. */
137
+ name: string;
138
+ /** One-sentence description: what authority + scope. */
139
+ description: string;
140
+ /**
141
+ * Pull fragments for this source. Sources MUST:
142
+ * - rate-limit themselves (>=1 req/sec per source by convention)
143
+ * - send a polite User-Agent
144
+ * - cache to disk when `opts.cacheDir` is set
145
+ * - mark `verifiable: false` rather than throwing on parse/block
146
+ * - honour `opts.signal`
147
+ * - honour `opts.limit`
148
+ */
149
+ fetch(opts: FetchOpts): Promise<KnowledgeFragment[]>;
149
150
  }
150
-
151
+ //#endregion
152
+ //#region src/sources/cornell-lii.d.ts
151
153
  interface CornellLiiSelector {
152
- /** Either 'uscode' or 'wex'. */
153
- kind: 'uscode' | 'wex';
154
- /**
155
- * For `uscode`: `<title>/<section>` (e.g. `'18/1836'` for DTSA).
156
- * For `wex`: the slug (e.g. `'non-compete'`).
157
- */
158
- path: string;
159
- /**
160
- * Optional pre-declared eval dimensions affected by this section. If
161
- * omitted, defaults are chosen from `kind` + path heuristics.
162
- */
163
- dimensionHints?: string[];
154
+ /** Either 'uscode' or 'wex'. */
155
+ kind: 'uscode' | 'wex';
156
+ /**
157
+ * For `uscode`: `<title>/<section>` (e.g. `'18/1836'` for DTSA).
158
+ * For `wex`: the slug (e.g. `'non-compete'`).
159
+ */
160
+ path: string;
161
+ /**
162
+ * Optional pre-declared eval dimensions affected by this section. If
163
+ * omitted, defaults are chosen from `kind` + path heuristics.
164
+ */
165
+ dimensionHints?: string[];
164
166
  }
165
167
  interface CornellLiiSourceOptions {
166
- /**
167
- * Selectors to fetch on each `fetch()` call. The caller (a per-tenant
168
- * workspace config, typically) lists exactly the authorities they need
169
- * tracked. There is no auto-discovery; that would crawl Cornell at
170
- * cron speed, which is what the polite-fetch contract exists to avoid.
171
- */
172
- selectors: CornellLiiSelector[];
173
- /** Source id override; default is `'cornell-lii'`. */
174
- id?: string;
168
+ /**
169
+ * Selectors to fetch on each `fetch()` call. The caller (a per-tenant
170
+ * workspace config, typically) lists exactly the authorities they need
171
+ * tracked. There is no auto-discovery; that would crawl Cornell at
172
+ * cron speed, which is what the polite-fetch contract exists to avoid.
173
+ */
174
+ selectors: CornellLiiSelector[];
175
+ /** Source id override; default is `'cornell-lii'`. */
176
+ id?: string;
175
177
  }
176
178
  /**
177
179
  * Build a Cornell LII source for the listed selectors.
@@ -187,7 +189,8 @@ interface CornellLiiSourceOptions {
187
189
  * ```
188
190
  */
189
191
  declare function createCornellLiiSource(options: CornellLiiSourceOptions): KnowledgeSource;
190
-
192
+ //#endregion
193
+ //#region src/sources/html.d.ts
191
194
  /**
192
195
  * Minimal HTML helpers used by the shipped sources.
193
196
  *
@@ -218,10 +221,11 @@ declare function innerHtmlById(html: string, id: string): string | undefined;
218
221
  * Returns absolute URLs by resolving against `baseUrl`.
219
222
  */
220
223
  declare function extractLinks(html: string, hrefPattern: RegExp, baseUrl: string): {
221
- href: string;
222
- text: string;
224
+ href: string;
225
+ text: string;
223
226
  }[];
224
-
227
+ //#endregion
228
+ //#region src/sources/http.d.ts
225
229
  /**
226
230
  * Polite HTTP fetcher shared by remote sources.
227
231
  *
@@ -238,40 +242,40 @@ declare const MIN_REQUEST_GAP_MS = 1000;
238
242
  /** Maximum response body we will buffer in memory (bytes). */
239
243
  declare const MAX_RESPONSE_BYTES: number;
240
244
  interface PoliteFetchOptions {
241
- signal?: AbortSignal;
242
- cacheDir?: string;
243
- /**
244
- * Cache age beyond which we re-fetch. Default 1 hour, long enough to
245
- * batch a cron sweep across many selectors, short enough that hourly
246
- * authoritative-page changes get picked up next tick.
247
- */
248
- cacheTtlMs?: number;
249
- /**
250
- * Extra request headers. The fetcher always sets `User-Agent` and
251
- * `Accept`; callers can add `Accept-Language` etc.
252
- */
253
- headers?: Record<string, string>;
245
+ signal?: AbortSignal;
246
+ cacheDir?: string;
247
+ /**
248
+ * Cache age beyond which we re-fetch. Default 1 hour, long enough to
249
+ * batch a cron sweep across many selectors, short enough that hourly
250
+ * authoritative-page changes get picked up next tick.
251
+ */
252
+ cacheTtlMs?: number;
253
+ /**
254
+ * Extra request headers. The fetcher always sets `User-Agent` and
255
+ * `Accept`; callers can add `Accept-Language` etc.
256
+ */
257
+ headers?: Record<string, string>;
254
258
  }
255
259
  interface PoliteFetchResult {
256
- url: string;
257
- status: number;
258
- /** Decoded UTF-8 body. Truncated to `MAX_RESPONSE_BYTES`. */
259
- body: string;
260
- /**
261
- * Best-effort source-attested timestamp. Reads `Last-Modified`,
262
- * falling back to `Date`, falling back to fetch time. Always ISO 8601.
263
- */
264
- sourceUpdatedAt: string;
265
- fetchedAt: string;
266
- /** True iff the response was satisfied from disk cache. */
267
- fromCache: boolean;
268
- /**
269
- * False on: non-2xx status, captcha/block page heuristic match, or
270
- * decoded body below 200 chars from a host known to serve real content
271
- * (Cornell, IRS, state SOS). `unverifiableReason` carries the why.
272
- */
273
- verifiable: boolean;
274
- unverifiableReason?: string;
260
+ url: string;
261
+ status: number;
262
+ /** Decoded UTF-8 body. Truncated to `MAX_RESPONSE_BYTES`. */
263
+ body: string;
264
+ /**
265
+ * Best-effort source-attested timestamp. Reads `Last-Modified`,
266
+ * falling back to `Date`, falling back to fetch time. Always ISO 8601.
267
+ */
268
+ sourceUpdatedAt: string;
269
+ fetchedAt: string;
270
+ /** True iff the response was satisfied from disk cache. */
271
+ fromCache: boolean;
272
+ /**
273
+ * False on: non-2xx status, captcha/block page heuristic match, or
274
+ * decoded body below 200 chars from a host known to serve real content
275
+ * (Cornell, IRS, state SOS). `unverifiableReason` carries the why.
276
+ */
277
+ verifiable: boolean;
278
+ unverifiableReason?: string;
275
279
  }
276
280
  /**
277
281
  * Fetch one URL with per-host throttling, on-disk cache, and block-page
@@ -287,27 +291,29 @@ declare function politeFetch(url: string, options?: PoliteFetchOptions): Promise
287
291
  declare function __resetHttpThrottle(): void;
288
292
  /** Cheap heuristic that catches CAPTCHA, WAF block pages, and "Just a moment" interstitials. */
289
293
  declare function looksLikeBlockPage(body: string): boolean;
290
-
294
+ //#endregion
295
+ //#region src/sources/irs-publications.d.ts
291
296
  interface IrsPublicationsSourceOptions {
292
- /**
293
- * Specific publication slugs to fetch (e.g. `['p15', 'p17', 'p463']`).
294
- * When `includeIndex` is true (default), the publications index page is
295
- * also fetched as a single fragment so change detection can notice
296
- * year/revision shifts across the whole catalogue.
297
- */
298
- publications?: string[];
299
- /**
300
- * Revenue procedure paths to fetch (e.g. `['/irb/2024-31_IRB']`). The
301
- * caller passes the exact path; this source does not auto-discover.
302
- */
303
- revenueProcedures?: string[];
304
- includeIndex?: boolean;
305
- id?: string;
297
+ /**
298
+ * Specific publication slugs to fetch (e.g. `['p15', 'p17', 'p463']`).
299
+ * When `includeIndex` is true (default), the publications index page is
300
+ * also fetched as a single fragment so change detection can notice
301
+ * year/revision shifts across the whole catalogue.
302
+ */
303
+ publications?: string[];
304
+ /**
305
+ * Revenue procedure paths to fetch (e.g. `['/irb/2024-31_IRB']`). The
306
+ * caller passes the exact path; this source does not auto-discover.
307
+ */
308
+ revenueProcedures?: string[];
309
+ includeIndex?: boolean;
310
+ id?: string;
306
311
  }
307
312
  /** Default eval dimensions for IRS-sourced fragments. */
308
313
  declare const IRS_DIMENSION_HINTS: string[];
309
314
  declare function createIrsPublicationsSource(options?: IrsPublicationsSourceOptions): KnowledgeSource;
310
-
315
+ //#endregion
316
+ //#region src/sources/state-sos.d.ts
311
317
  /**
312
318
  * Generic Secretary-of-State source.
313
319
  *
@@ -326,45 +332,46 @@ declare function createIrsPublicationsSource(options?: IrsPublicationsSourceOpti
326
332
  * @experimental Interface will likely grow as we add more state coverage.
327
333
  */
328
334
  interface StateSosEntity {
329
- /** Stable id for this fragment within the state (e.g. 'llc-formation', 'corp-formation'). */
330
- id: string;
331
- /** Path under the configured `baseUrl` for this entity. */
332
- path: string;
333
- /**
334
- * Extraction selector. Choose one:
335
- * - `{ kind: 'id', value: 'main-content' }` — innermost match of element with that id
336
- * - `{ kind: 'class', value: 'field--name-body' }` — innermost match of element with that class
337
- * - `{ kind: 'regex', value: /<article[\s\S]*?<\/article>/i }` — raw regex
338
- * - `{ kind: 'whole' }` — full body, tags stripped (fallback for unstructured pages)
339
- */
340
- selector: {
341
- kind: 'id';
342
- value: string;
343
- } | {
344
- kind: 'class';
345
- value: string;
346
- } | {
347
- kind: 'regex';
348
- value: RegExp;
349
- } | {
350
- kind: 'whole';
351
- };
352
- title: string;
353
- /** Eval dimensions this entity feeds. */
354
- dimensionHints?: string[];
335
+ /** Stable id for this fragment within the state (e.g. 'llc-formation', 'corp-formation'). */
336
+ id: string;
337
+ /** Path under the configured `baseUrl` for this entity. */
338
+ path: string;
339
+ /**
340
+ * Extraction selector. Choose one:
341
+ * - `{ kind: 'id', value: 'main-content' }` — innermost match of element with that id
342
+ * - `{ kind: 'class', value: 'field--name-body' }` — innermost match of element with that class
343
+ * - `{ kind: 'regex', value: /<article[\s\S]*?<\/article>/i }` — raw regex
344
+ * - `{ kind: 'whole' }` — full body, tags stripped (fallback for unstructured pages)
345
+ */
346
+ selector: {
347
+ kind: 'id';
348
+ value: string;
349
+ } | {
350
+ kind: 'class';
351
+ value: string;
352
+ } | {
353
+ kind: 'regex';
354
+ value: RegExp;
355
+ } | {
356
+ kind: 'whole';
357
+ };
358
+ title: string;
359
+ /** Eval dimensions this entity feeds. */
360
+ dimensionHints?: string[];
355
361
  }
356
362
  interface StateSosSourceConfig {
357
- /** US state postal code, e.g. 'CA', 'DE', 'TX'. */
358
- state: string;
359
- /** Base URL for the state SOS — e.g. 'https://www.sos.ca.gov'. */
360
- baseUrl: string;
361
- /** Entities this state exposes (LLC, Corp, etc). */
362
- entities: StateSosEntity[];
363
- /** Source id; default `state-sos:<state>`. */
364
- id?: string;
365
- /** Display name; default `<state> Secretary of State`. */
366
- name?: string;
363
+ /** US state postal code, e.g. 'CA', 'DE', 'TX'. */
364
+ state: string;
365
+ /** Base URL for the state SOS — e.g. 'https://www.sos.ca.gov'. */
366
+ baseUrl: string;
367
+ /** Entities this state exposes (LLC, Corp, etc). */
368
+ entities: StateSosEntity[];
369
+ /** Source id; default `state-sos:<state>`. */
370
+ id?: string;
371
+ /** Display name; default `<state> Secretary of State`. */
372
+ name?: string;
367
373
  }
368
374
  declare function createStateSosSource(config: StateSosSourceConfig): KnowledgeSource;
369
-
370
- export { type CornellLiiSelector, type CornellLiiSourceOptions, type FetchOpts, type FragmentProvenance, IRS_DIMENSION_HINTS, type IrsPublicationsSourceOptions, type KnowledgeFragment, type KnowledgeSource, MAX_RESPONSE_BYTES, MIN_REQUEST_GAP_MS, POLITE_USER_AGENT, type PoliteFetchOptions, type PoliteFetchResult, type StateSosEntity, type StateSosSourceConfig, __resetHttpThrottle, createCornellLiiSource, createIrsPublicationsSource, createStateSosSource, extractLinks, firstMatch, htmlToText, innerHtmlById, looksLikeBlockPage, politeFetch };
375
+ //#endregion
376
+ export { CornellLiiSelector, CornellLiiSourceOptions, FetchOpts, FragmentProvenance, IRS_DIMENSION_HINTS, IrsPublicationsSourceOptions, KnowledgeFragment, KnowledgeSource, MAX_RESPONSE_BYTES, MIN_REQUEST_GAP_MS, POLITE_USER_AGENT, PoliteFetchOptions, PoliteFetchResult, StateSosEntity, StateSosSourceConfig, __resetHttpThrottle, createCornellLiiSource, createIrsPublicationsSource, createStateSosSource, extractLinks, firstMatch, htmlToText, innerHtmlById, looksLikeBlockPage, politeFetch };
377
+ //# sourceMappingURL=index.d.ts.map
@@ -0,0 +1 @@
1
+ {"version":3,"file":"index.d.ts","names":[],"sources":["../../src/sources/types.ts","../../src/sources/cornell-lii.ts","../../src/sources/html.ts","../../src/sources/http.ts","../../src/sources/irs-publications.ts","../../src/sources/state-sos.ts"],"mappings":";;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;;UAgCiB;;EAEf,SAAS;;EAET;;EAEA,YAAY;;;;;;EAMZ;;;;;;;;;EASA;;;;;;;UAQe;;EAEf;;;;;;;EAOA;;EAEA;;;;;;;EAOA;;;;;;;;EAQA;;EAEA;;;;;;UAOe;;;;;;EAMf;;EAEA;;EAEA;;EAEA;EACA,YAAY;;;;;;;;;;;;;EAaZ;;EAEA,WAAW;;;;;;;;;;UAWI;;EAEf;;EAEA;;EAEA;;;;;;;;;;EAUA,MAAM,MAAM,YAAY,QAAQ;;;;UCrIjB;;EAEf;;;;;EAKA;;;;;EAKA;;UAGe;;;;;;;EAOf,WAAW;;EAEX;;;;;;;;;;;;;;;iBAgBc,uBAAuB,SAAS,0BAA0B;;;;;;;;;;;;;;;;;;;;;;;iBCrC1D,WAAW;;iBA4BX,WAAW,cAAc,SAAS;;iBAKlC,cAAc,cAAc;;;;;iBAa5B,aACd,cACA,aAAa,QACb;EACG;EAAc;;;;;;;;;;;;;;cCxDN;;cAIA;;cAGA;UAII;EACf,SAAS;EACT;;;;;;EAMA;;;;;EAKA,UAAU;;UAGK;EACf;EACA;;EAEA;;;;;EAKA;EACA;;EAEA;;;;;;EAMA;EACA;;;;;;;;;;;iBAYoB,YACpB,aACA,UAAS,qBACR,QAAQ;;iBAoEK;;iBA6DA,mBAAmB;;;UChLlB;;;;;;;EAOf;;;;;EAKA;EACA;EACA;;;cAIW;iBAEG,4BACd,UAAS,+BACR;;;;;;;;;;;;;;;;;;;;UC5Bc;;EAEf;;EAEA;;;;;;;;EAQA;IACM;IAAY;;IACZ;IAAe;;IACf;IAAe,OAAO;;IACtB;;EACN;;EAEA;;UAGe;;EAEf;;EAEA;;EAEA,UAAU;;EAEV;;EAEA;;iBAGc,qBAAqB,QAAQ,uBAAuB"}