@hanhnd/agent-kit 1.0.27 → 1.0.29

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,147 @@
1
+ import { createHash } from 'crypto';
2
+ export function computeChunkId(content) {
3
+ return createHash('sha256').update(content).digest('hex').slice(0, 16);
4
+ }
5
+ export function cleanContentForEmbedding(text) {
6
+ return text.replace(/<!--[\s\S]*?-->/g, '');
7
+ }
8
+ const HEADING_RE = /^(#{1,6}) (.+)$/;
9
+ const SENTENCE_END_RE = /([.!?。!?])\s+(?![0-9])/g;
10
+ const URL_RE = /https?:\/\/\S+/g;
11
+ function splitAtParagraph(text, maxLen) {
12
+ const paragraphs = text.split(/\n\n+/);
13
+ const parts = [];
14
+ let current = '';
15
+ for (const para of paragraphs) {
16
+ const candidate = current ? `${current}\n\n${para}` : para;
17
+ if (candidate.length <= maxLen) {
18
+ current = candidate;
19
+ }
20
+ else {
21
+ if (current)
22
+ parts.push(current);
23
+ current = para.length <= maxLen ? para : splitAtLine(para, maxLen).join('\n\n');
24
+ }
25
+ }
26
+ if (current)
27
+ parts.push(current);
28
+ return parts.filter(Boolean);
29
+ }
30
+ function splitAtLine(text, maxLen) {
31
+ const lines = text.split('\n');
32
+ const parts = [];
33
+ let current = '';
34
+ for (const line of lines) {
35
+ const candidate = current ? `${current}\n${line}` : line;
36
+ if (candidate.length <= maxLen) {
37
+ current = candidate;
38
+ }
39
+ else {
40
+ if (current)
41
+ parts.push(current);
42
+ current = line.length <= maxLen ? line : splitAtSentence(line, maxLen).join(' ');
43
+ }
44
+ }
45
+ if (current)
46
+ parts.push(current);
47
+ return parts.filter(Boolean);
48
+ }
49
+ function splitAtSentence(text, maxLen) {
50
+ const sanitized = text.replace(URL_RE, '[URL]');
51
+ const sentences = [];
52
+ let lastIndex = 0;
53
+ for (const match of sanitized.matchAll(SENTENCE_END_RE)) {
54
+ const end = (match.index ?? 0) + match[0].length;
55
+ sentences.push(text.slice(lastIndex, end).trim());
56
+ lastIndex = end;
57
+ }
58
+ if (lastIndex < text.length) {
59
+ sentences.push(text.slice(lastIndex).trim());
60
+ }
61
+ if (sentences.length === 0)
62
+ return [text.slice(0, maxLen)];
63
+ const parts = [];
64
+ let current = '';
65
+ for (const sentence of sentences) {
66
+ const candidate = current ? `${current} ${sentence}` : sentence;
67
+ if (candidate.length <= maxLen) {
68
+ current = candidate;
69
+ }
70
+ else {
71
+ if (current)
72
+ parts.push(current);
73
+ current = sentence.slice(0, maxLen);
74
+ }
75
+ }
76
+ if (current)
77
+ parts.push(current);
78
+ return parts.filter(Boolean);
79
+ }
80
+ function splitSection(text, maxLen) {
81
+ if (text.length <= maxLen)
82
+ return [text];
83
+ const byPara = splitAtParagraph(text, maxLen);
84
+ if (byPara.every((p) => p.length <= maxLen))
85
+ return byPara;
86
+ return byPara.flatMap((p) => (p.length <= maxLen ? [p] : splitAtLine(p, maxLen)));
87
+ }
88
+ export function chunkMarkdown(text, source, config) {
89
+ if (!text || !text.trim())
90
+ return [];
91
+ const { chunkSize, overlapLines } = config;
92
+ const lines = text.split('\n');
93
+ const sections = [];
94
+ let current = { heading: '', headingLevel: 0, lines: [] };
95
+ for (let i = 0; i < lines.length; i++) {
96
+ const line = lines[i];
97
+ const headingMatch = line.match(HEADING_RE);
98
+ if (headingMatch) {
99
+ if (current.lines.length > 0)
100
+ sections.push(current);
101
+ current = {
102
+ heading: headingMatch[2].trim(),
103
+ headingLevel: headingMatch[1].length,
104
+ lines: [{ text: line, lineNum: i + 1 }],
105
+ };
106
+ }
107
+ else {
108
+ current.lines.push({ text: line, lineNum: i + 1 });
109
+ }
110
+ }
111
+ if (current.lines.length > 0)
112
+ sections.push(current);
113
+ const chunks = [];
114
+ let prevOverlapLines = [];
115
+ for (const section of sections) {
116
+ const sectionText = section.lines.map((l) => l.text).join('\n');
117
+ if (!sectionText.trim())
118
+ continue;
119
+ const fullText = prevOverlapLines.length > 0
120
+ ? `${prevOverlapLines.join('\n')}\n${sectionText}`
121
+ : sectionText;
122
+ const parts = splitSection(fullText, chunkSize);
123
+ for (let pi = 0; pi < parts.length; pi++) {
124
+ const part = parts[pi];
125
+ if (!part.trim())
126
+ continue;
127
+ const partLines = part.split('\n');
128
+ const firstLine = section.lines[0]?.lineNum ?? 1;
129
+ const lastLine = section.lines[section.lines.length - 1]?.lineNum ?? firstLine;
130
+ const content = cleanContentForEmbedding(part);
131
+ chunks.push({
132
+ id: computeChunkId(content),
133
+ source,
134
+ heading: section.heading,
135
+ headingLevel: section.headingLevel,
136
+ content,
137
+ lineStart: firstLine,
138
+ lineEnd: lastLine,
139
+ });
140
+ // Update overlap for next iteration (last N lines of this chunk)
141
+ if (pi === parts.length - 1) {
142
+ prevOverlapLines = partLines.slice(-overlapLines);
143
+ }
144
+ }
145
+ }
146
+ return chunks;
147
+ }
@@ -0,0 +1,13 @@
1
+ export declare class EmbedderError extends Error {
2
+ constructor(message: string, cause?: unknown);
3
+ }
4
+ export declare class Embedder {
5
+ private readonly modelName;
6
+ private pipeline;
7
+ private _dimension;
8
+ constructor(modelName: string);
9
+ embed(texts: string[]): Promise<Float32Array[]>;
10
+ get dimension(): number | undefined;
11
+ isReady(): boolean;
12
+ private loadModel;
13
+ }
@@ -0,0 +1,68 @@
1
+ import * as os from 'os';
2
+ import * as path from 'path';
3
+ const DEFAULT_CACHE_DIR = path.join(os.homedir(), '.cache', 'fastembed');
4
+ export class EmbedderError extends Error {
5
+ constructor(message, cause) {
6
+ super(message);
7
+ this.name = 'EmbedderError';
8
+ if (cause instanceof Error)
9
+ this.cause = cause;
10
+ }
11
+ }
12
+ export class Embedder {
13
+ modelName;
14
+ pipeline = null;
15
+ _dimension;
16
+ constructor(modelName) {
17
+ this.modelName = modelName;
18
+ }
19
+ async embed(texts) {
20
+ if (texts.length === 0)
21
+ return [];
22
+ if (!this.pipeline) {
23
+ await this.loadModel();
24
+ }
25
+ try {
26
+ const results = await this.pipeline(texts);
27
+ return results.map((vec) => {
28
+ const arr = new Float32Array(vec);
29
+ if (!this._dimension)
30
+ this._dimension = arr.length;
31
+ return arr;
32
+ });
33
+ }
34
+ catch (cause) {
35
+ throw new EmbedderError(`Embedding failed for batch of ${texts.length} texts`, cause);
36
+ }
37
+ }
38
+ get dimension() {
39
+ return this._dimension;
40
+ }
41
+ isReady() {
42
+ return this.pipeline !== null;
43
+ }
44
+ async loadModel() {
45
+ try {
46
+ const { EmbeddingModel, FlagEmbedding } = await import('fastembed');
47
+ const modelMap = {
48
+ 'Xenova/bge-small-en-v1.5': EmbeddingModel.BGESmallENV15,
49
+ 'Xenova/bge-base-en-v1.5': EmbeddingModel.BGEBaseENV15,
50
+ 'Xenova/all-MiniLM-L6-v2': EmbeddingModel.AllMiniLML6V2,
51
+ };
52
+ const model = modelMap[this.modelName] ?? EmbeddingModel.BGESmallENV15;
53
+ const embedder = await FlagEmbedding.init({ model: model, cacheDir: DEFAULT_CACHE_DIR, showDownloadProgress: false });
54
+ this.pipeline = async (texts) => {
55
+ const result = [];
56
+ for await (const batch of embedder.embed(texts, texts.length)) {
57
+ for (const vec of batch) {
58
+ result.push(Array.from(vec));
59
+ }
60
+ }
61
+ return result;
62
+ };
63
+ }
64
+ catch (cause) {
65
+ throw new EmbedderError(`Failed to load embedding model: ${this.modelName}`, cause);
66
+ }
67
+ }
68
+ }
@@ -0,0 +1,5 @@
1
+ export { MemoryStore, StoreError } from './store.js';
2
+ export { Embedder, EmbedderError } from './embedder.js';
3
+ export { MemoryIndexer } from './indexer.js';
4
+ export type { MemoryChunk, SearchResult, MemoryConfig, IndexStats } from './types.js';
5
+ export { DEFAULT_MEMORY_CONFIG } from './types.js';
@@ -0,0 +1,4 @@
1
+ export { MemoryStore, StoreError } from './store.js';
2
+ export { Embedder, EmbedderError } from './embedder.js';
3
+ export { MemoryIndexer } from './indexer.js';
4
+ export { DEFAULT_MEMORY_CONFIG } from './types.js';
@@ -0,0 +1,20 @@
1
+ import type { Embedder } from './embedder.js';
2
+ import type { MemoryStore } from './store.js';
3
+ import type { IndexStats, MemoryConfig, SearchResult } from './types.js';
4
+ export declare class MemoryIndexer {
5
+ private readonly store;
6
+ private readonly embedder;
7
+ private readonly config;
8
+ private _ready;
9
+ constructor(store: MemoryStore, embedder: Embedder, config: MemoryConfig);
10
+ indexFile(absolutePath: string): Promise<IndexStats>;
11
+ private indexFileRelativeTo;
12
+ indexDirectory(rootDir: string, opts?: {
13
+ relativeBase?: string;
14
+ excludeFiles?: string[];
15
+ }): Promise<IndexStats>;
16
+ search(query: string, topK: number): Promise<SearchResult[]>;
17
+ save(content: string): Promise<IndexStats>;
18
+ startupIndex(): Promise<void>;
19
+ get ready(): boolean;
20
+ }
@@ -0,0 +1,222 @@
1
+ import * as fs from 'fs';
2
+ import * as path from 'path';
3
+ import { chunkMarkdown } from './chunker.js';
4
+ const LOCK_RETRY_MS = 50;
5
+ const LOCK_TIMEOUT_MS = 500;
6
+ async function acquireLock(lockPath) {
7
+ const deadline = Date.now() + LOCK_TIMEOUT_MS;
8
+ while (Date.now() < deadline) {
9
+ try {
10
+ fs.writeFileSync(lockPath, String(process.pid), { flag: 'wx' });
11
+ return true;
12
+ }
13
+ catch {
14
+ await new Promise((r) => setTimeout(r, LOCK_RETRY_MS));
15
+ }
16
+ }
17
+ return false;
18
+ }
19
+ function releaseLock(lockPath) {
20
+ try {
21
+ fs.unlinkSync(lockPath);
22
+ }
23
+ catch {
24
+ // ignore
25
+ }
26
+ }
27
+ export class MemoryIndexer {
28
+ store;
29
+ embedder;
30
+ config;
31
+ _ready = false;
32
+ constructor(store, embedder, config) {
33
+ this.store = store;
34
+ this.embedder = embedder;
35
+ this.config = config;
36
+ }
37
+ async indexFile(absolutePath) {
38
+ return this.indexFileRelativeTo(absolutePath, this.config.wikiDir);
39
+ }
40
+ async indexFileRelativeTo(absolutePath, sourceRoot) {
41
+ const source = path.relative(sourceRoot, absolutePath);
42
+ const existingHashes = this.store.hashesBySource(source);
43
+ let text;
44
+ try {
45
+ text = fs.readFileSync(absolutePath, 'utf8');
46
+ }
47
+ catch (err) {
48
+ console.warn(`[memory-indexer] Cannot read file ${absolutePath}:`, err);
49
+ return { indexed: 0, deleted: 0, skipped: existingHashes.size };
50
+ }
51
+ const allChunks = chunkMarkdown(text, source, this.config);
52
+ const newHashes = new Set(allChunks.map((c) => c.id));
53
+ const toIndex = allChunks.filter((c) => !existingHashes.has(c.id));
54
+ const toDelete = [...existingHashes].filter((h) => !newHashes.has(h));
55
+ let embeddings = [];
56
+ if (toIndex.length > 0) {
57
+ try {
58
+ embeddings = await this.embedder.embed(toIndex.map((c) => c.content));
59
+ }
60
+ catch (err) {
61
+ console.warn(`[memory-indexer] Embedding failed for ${absolutePath}:`, err);
62
+ return { indexed: 0, deleted: 0, skipped: existingHashes.size };
63
+ }
64
+ }
65
+ if (toIndex.length > 0) {
66
+ this.store.upsert(toIndex, embeddings);
67
+ }
68
+ if (toDelete.length > 0) {
69
+ this.store.deleteByIds(toDelete);
70
+ }
71
+ return {
72
+ indexed: toIndex.length,
73
+ deleted: toDelete.length,
74
+ skipped: allChunks.length - toIndex.length,
75
+ };
76
+ }
77
+ async indexDirectory(rootDir, opts = {}) {
78
+ const totals = { indexed: 0, deleted: 0, skipped: 0 };
79
+ if (!fs.existsSync(rootDir))
80
+ return totals;
81
+ const relativeBase = opts.relativeBase ?? this.config.wikiDir;
82
+ const excludeFiles = new Set(opts.excludeFiles ?? []);
83
+ const files = [];
84
+ const walk = (dirPath) => {
85
+ let entries;
86
+ try {
87
+ entries = fs.readdirSync(dirPath, { withFileTypes: true });
88
+ }
89
+ catch (err) {
90
+ console.warn(`[memory-indexer] Cannot scan directory ${dirPath}:`, err);
91
+ return;
92
+ }
93
+ for (const entry of entries) {
94
+ const entryPath = path.join(dirPath, entry.name);
95
+ if (entry.isDirectory()) {
96
+ walk(entryPath);
97
+ continue;
98
+ }
99
+ if (entry.isFile() && /\.md$/i.test(entry.name) && !excludeFiles.has(entry.name)) {
100
+ files.push(entryPath);
101
+ }
102
+ }
103
+ };
104
+ walk(rootDir);
105
+ const currentSources = new Set(files.map((file) => path.relative(relativeBase, file)));
106
+ // Remove stale sources (deleted files or pre-migration daily-file sources)
107
+ for (const stale of this.store.indexedSources()) {
108
+ if (!currentSources.has(stale)) {
109
+ this.store.deleteBySource(stale);
110
+ totals.deleted += 1;
111
+ }
112
+ }
113
+ for (const file of files) {
114
+ const stats = await this.indexFileRelativeTo(file, relativeBase);
115
+ totals.indexed += stats.indexed;
116
+ totals.deleted += stats.deleted;
117
+ totals.skipped += stats.skipped;
118
+ }
119
+ return totals;
120
+ }
121
+ async search(query, topK) {
122
+ const fetchLimit = topK * 2;
123
+ let denseResults = [];
124
+ if (this.store.vecAvailable) {
125
+ try {
126
+ const queryEmbedding = await this.embedder.embed([query]);
127
+ denseResults = this.store.searchDense(queryEmbedding[0], fetchLimit);
128
+ }
129
+ catch (err) {
130
+ console.warn('[memory-indexer] Dense search embedding failed:', err);
131
+ }
132
+ }
133
+ const bm25Results = this.store.searchBm25(query, fetchLimit);
134
+ // RRF fusion (k=60)
135
+ const k = 60;
136
+ const scoreMap = new Map();
137
+ denseResults.forEach((r, rank) => {
138
+ scoreMap.set(r.id, { dense: 1 / (k + rank + 1), bm25: 0 });
139
+ });
140
+ bm25Results.forEach((r, rank) => {
141
+ const existing = scoreMap.get(r.id);
142
+ if (existing) {
143
+ existing.bm25 = 1 / (k + rank + 1);
144
+ }
145
+ else {
146
+ scoreMap.set(r.id, { dense: 0, bm25: 1 / (k + rank + 1) });
147
+ }
148
+ });
149
+ const hasDense = this.store.vecAvailable && denseResults.length > 0;
150
+ const numRetrievers = hasDense ? 2 : 1;
151
+ const maxScore = numRetrievers / 61;
152
+ const ranked = [...scoreMap.entries()]
153
+ .map(([id, scores]) => ({
154
+ id,
155
+ totalScore: scores.dense + scores.bm25,
156
+ hasDense: scores.dense > 0,
157
+ hasBm25: scores.bm25 > 0,
158
+ }))
159
+ .sort((a, b) => b.totalScore - a.totalScore);
160
+ const chunks = this.store.getChunksByIds(ranked.map((r) => r.id));
161
+ const chunkMap = new Map(chunks.map((c) => [c.id, c]));
162
+ const results = [];
163
+ const seenSources = new Set();
164
+ for (const r of ranked) {
165
+ if (results.length >= topK)
166
+ break;
167
+ const chunk = chunkMap.get(r.id);
168
+ if (!chunk)
169
+ continue;
170
+ if (seenSources.has(chunk.source))
171
+ continue;
172
+ seenSources.add(chunk.source);
173
+ const normalizedScore = maxScore > 0 ? r.totalScore / maxScore : 0;
174
+ const retriever = r.hasDense && r.hasBm25 ? 'both' : r.hasDense ? 'dense' : 'bm25';
175
+ let contentSource = 'file';
176
+ try {
177
+ chunk.content = fs.readFileSync(path.join(this.config.wikiDir, chunk.source), 'utf8');
178
+ }
179
+ catch {
180
+ contentSource = 'fallback';
181
+ }
182
+ results.push({ chunk, score: normalizedScore, retriever, contentSource });
183
+ }
184
+ return results;
185
+ }
186
+ async save(content) {
187
+ const datePart = new Date().toISOString().slice(0, 10);
188
+ const rawDir = path.join(this.config.wikiDir, 'raw');
189
+ const savePath = path.join(rawDir, `conv_save_${datePart}.md`);
190
+ const lockPath = `${savePath}.lock`;
191
+ fs.mkdirSync(rawDir, { recursive: true });
192
+ const acquired = await acquireLock(lockPath);
193
+ if (!acquired) {
194
+ console.warn('[memory-indexer] Could not acquire write lock, writing anyway (fail-open)');
195
+ }
196
+ try {
197
+ fs.appendFileSync(savePath, `\n${content}\n`, 'utf8');
198
+ }
199
+ finally {
200
+ if (acquired)
201
+ releaseLock(lockPath);
202
+ }
203
+ return { indexed: 0, deleted: 0, skipped: 0 };
204
+ }
205
+ async startupIndex() {
206
+ try {
207
+ await this.indexDirectory(path.join(this.config.wikiDir, 'compiled'), {
208
+ relativeBase: this.config.wikiDir,
209
+ excludeFiles: ['index.md', 'log.md'],
210
+ });
211
+ }
212
+ catch (err) {
213
+ console.warn('[memory-indexer] Startup indexing failed:', err);
214
+ }
215
+ finally {
216
+ this._ready = true;
217
+ }
218
+ }
219
+ get ready() {
220
+ return this._ready;
221
+ }
222
+ }
@@ -0,0 +1,29 @@
1
+ import type { MemoryChunk, MemoryConfig } from './types.js';
2
+ export declare class StoreError extends Error {
3
+ constructor(message: string, cause?: unknown);
4
+ }
5
+ export declare class MemoryStore {
6
+ private readonly config;
7
+ private db;
8
+ private _vecAvailable;
9
+ constructor(dbPath: string, config: MemoryConfig);
10
+ private openDatabase;
11
+ private loadVecExtension;
12
+ private createSchema;
13
+ upsert(chunks: MemoryChunk[], embeddings: Float32Array[]): void;
14
+ hashesBySource(source: string): Set<string>;
15
+ indexedSources(): string[];
16
+ deleteBySource(source: string): void;
17
+ deleteByIds(ids: string[]): void;
18
+ searchDense(embedding: Float32Array, limit: number): Array<{
19
+ id: string;
20
+ score: number;
21
+ }>;
22
+ searchBm25(query: string, limit: number): Array<{
23
+ id: string;
24
+ score: number;
25
+ }>;
26
+ getChunksByIds(ids: string[]): MemoryChunk[];
27
+ get vecAvailable(): boolean;
28
+ close(): void;
29
+ }