@andreprado/agentkit 0.1.0-alpha.5 → 0.1.0-alpha.7
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +9 -0
- package/docs/guides/add-channel.md +25 -0
- package/docs/guides/add-knowledge.md +134 -0
- package/docs/guides/agentkit-skills-architecture.md +471 -0
- package/docs/guides/channels-production-handoff.md +2 -0
- package/docs/guides/connect-telegram.md +17 -0
- package/docs/guides/connect-whatsapp-zapster.md +16 -0
- package/docs/guides/create-agent.md +10 -1
- package/docs/guides/run-evals.md +36 -1
- package/docs/llms-full.txt +90 -1
- package/docs/llms.txt +9 -2
- package/package.json +2 -1
- package/src/cli/cloud-client.ts +10 -2
- package/src/cli/commands/channels.ts +67 -3
- package/src/cli/commands/knowledge.ts +136 -0
- package/src/cli/deploy-chat-ui.ts +7 -0
- package/src/cli/deploy-readiness.ts +19 -0
- package/src/cli/help.ts +26 -2
- package/src/cli/index.ts +140 -8
- package/src/cloud/artifact.ts +92 -1
- package/src/cloud/contracts.ts +16 -0
- package/src/create-project.ts +38 -6
- package/src/index.ts +142 -1
- package/src/providers/pi.ts +1 -1
- package/src/providers/test.ts +1 -1
- package/src/runtime/channel-buffer.ts +30 -0
- package/src/runtime/channels.ts +1 -0
- package/src/runtime/chat.ts +21 -2
- package/src/runtime/config.ts +175 -0
- package/src/runtime/core/manifest.ts +37 -0
- package/src/runtime/database.ts +93 -2
- package/src/runtime/db-commands.ts +9 -0
- package/src/runtime/deploy-readiness.ts +12 -0
- package/src/runtime/dev-server.ts +201 -11
- package/src/runtime/evals.ts +210 -20
- package/src/runtime/inspect.ts +39 -0
- package/src/runtime/knowledge/chunk.ts +333 -0
- package/src/runtime/knowledge/config.ts +135 -0
- package/src/runtime/knowledge/embeddings.ts +133 -0
- package/src/runtime/knowledge/ingest.ts +521 -0
- package/src/runtime/knowledge/prompt-policy.ts +30 -0
- package/src/runtime/knowledge/retrieve.ts +283 -0
- package/src/runtime/knowledge/schema.ts +56 -0
- package/src/runtime/knowledge/tool.ts +64 -0
- package/src/runtime/knowledge/vector.ts +258 -0
- package/src/runtime/spec.ts +152 -0
- package/src/runtime/sync.ts +144 -0
- package/src/runtime/targets/cloudflare/build.ts +514 -4
- package/src/runtime/tools.ts +121 -1
- package/src/runtime/traces.ts +41 -0
- package/src/storage/sqlite.ts +141 -0
- package/src/templates/blank.ts +16 -5
- package/src/templates/dentista.ts +17 -2
- package/src/templates/skills/agentkit-build-agent/SKILL.md +51 -0
- package/src/templates/skills/agentkit-build-agent/templates/appointment-intake.instructions.md +20 -0
- package/src/templates/skills/agentkit-build-agent/templates/sales-qualifier.instructions.md +17 -0
- package/src/templates/skills/agentkit-build-agent/templates/support-agent.instructions.md +16 -0
- package/src/templates/skills/agentkit-capsule/SKILL.md +62 -0
- package/src/templates/skills/agentkit-capsule/references/docs-router.md +15 -0
- package/src/templates/skills/agentkit-channels/SKILL.md +62 -0
- package/src/templates/skills/agentkit-channels/references/channel-buffering.md +58 -0
- package/src/templates/skills/agentkit-channels/references/channel-debugging.md +37 -0
- package/src/templates/skills/agentkit-channels/references/telegram.md +37 -0
- package/src/templates/skills/agentkit-channels/references/whatsapp-zapster.md +37 -0
- package/src/templates/skills/agentkit-database/SKILL.md +45 -0
- package/src/templates/skills/agentkit-database/templates/appointments.schema.sql +15 -0
- package/src/templates/skills/agentkit-database/templates/leads.schema.sql +17 -0
- package/src/templates/skills/agentkit-deploy/SKILL.md +44 -0
- package/src/templates/skills/agentkit-evals/SKILL.md +60 -0
- package/src/templates/skills/agentkit-evals/templates/multi-turn.eval.md +22 -0
- package/src/templates/skills/agentkit-evals/templates/no-leak.eval.md +14 -0
- package/src/templates/skills/agentkit-evals/templates/smoke.eval.md +14 -0
- package/src/templates/skills/agentkit-evals/templates/tool-call.eval.md +18 -0
- package/src/templates/skills/agentkit-knowledge/SKILL.md +40 -0
- package/src/templates/skills/agentkit-knowledge/templates/faq.md +14 -0
- package/src/templates/skills/agentkit-knowledge/templates/policies.md +14 -0
- package/src/templates/skills/agentkit-knowledge/templates/prices.csv +3 -0
- package/src/templates/skills/agentkit-prompts/SKILL.md +45 -0
- package/src/templates/skills/agentkit-prompts/templates/knowledge-grounded-faq.instructions.md +11 -0
- package/src/templates/skills/agentkit-provider/SKILL.md +57 -0
- package/src/templates/skills/agentkit-security/SKILL.md +55 -0
- package/src/templates/skills/agentkit-tools/SKILL.md +36 -0
- package/src/templates/skills/agentkit-tools/examples/database-write.tool.md +35 -0
- package/src/templates/skills/agentkit-tools/examples/eval-safe-external-action.tool.md +37 -0
- package/src/templates/skills/agentkit-tools/examples/lookup-order.tool.md +46 -0
- package/src/templates/skills/agentkit-troubleshooting/SKILL.md +52 -0
- package/src/templates/support.ts +15 -4
|
@@ -0,0 +1,333 @@
|
|
|
1
|
+
import { basename, extname } from "node:path";
|
|
2
|
+
|
|
3
|
+
export type KnowledgeChunkMetadata = {
|
|
4
|
+
sourcePath: string;
|
|
5
|
+
title: string;
|
|
6
|
+
section: string | null;
|
|
7
|
+
locator: string | null;
|
|
8
|
+
};
|
|
9
|
+
|
|
10
|
+
export type KnowledgeChunk = {
|
|
11
|
+
ordinal: number;
|
|
12
|
+
content: string;
|
|
13
|
+
tokenCount: number;
|
|
14
|
+
metadata: KnowledgeChunkMetadata;
|
|
15
|
+
};
|
|
16
|
+
|
|
17
|
+
export type ChunkKnowledgeSourceInput = {
|
|
18
|
+
sourcePath: string;
|
|
19
|
+
title?: string | null;
|
|
20
|
+
content: string;
|
|
21
|
+
};
|
|
22
|
+
|
|
23
|
+
const DEFAULT_MAX_WORDS = 720;
|
|
24
|
+
const DEFAULT_OVERLAP_WORDS = 80;
|
|
25
|
+
|
|
26
|
+
export function chunkKnowledgeSource(input: ChunkKnowledgeSourceInput): KnowledgeChunk[] {
|
|
27
|
+
const extension = extname(input.sourcePath).toLowerCase();
|
|
28
|
+
|
|
29
|
+
if (extension === ".csv") {
|
|
30
|
+
return chunkCsv(input);
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
if (extension === ".md" || extension === ".markdown") {
|
|
34
|
+
return chunkMarkdown(input);
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
return chunkText(input);
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
function chunkMarkdown(input: ChunkKnowledgeSourceInput): KnowledgeChunk[] {
|
|
41
|
+
const title = input.title ?? inferTitle(input.sourcePath, input.content);
|
|
42
|
+
const sections = markdownSections(input.content, title);
|
|
43
|
+
const chunks: KnowledgeChunk[] = [];
|
|
44
|
+
|
|
45
|
+
for (const section of sections) {
|
|
46
|
+
const contextualContent = [section.heading, section.body].filter(Boolean).join("\n\n").trim();
|
|
47
|
+
chunks.push(
|
|
48
|
+
...splitContentIntoChunks({
|
|
49
|
+
sourcePath: input.sourcePath,
|
|
50
|
+
title,
|
|
51
|
+
section: section.heading || null,
|
|
52
|
+
locator: section.heading ? `section:${section.heading}` : null,
|
|
53
|
+
content: contextualContent,
|
|
54
|
+
startingOrdinal: chunks.length,
|
|
55
|
+
}),
|
|
56
|
+
);
|
|
57
|
+
}
|
|
58
|
+
|
|
59
|
+
return ensureAtLeastOneChunk(chunks, input, title);
|
|
60
|
+
}
|
|
61
|
+
|
|
62
|
+
function chunkText(input: ChunkKnowledgeSourceInput): KnowledgeChunk[] {
|
|
63
|
+
const title = input.title ?? inferTitle(input.sourcePath, input.content);
|
|
64
|
+
const paragraphs = input.content
|
|
65
|
+
.split(/\n{2,}/)
|
|
66
|
+
.map((paragraph) => paragraph.trim())
|
|
67
|
+
.filter(Boolean);
|
|
68
|
+
const content = paragraphs.length > 0 ? paragraphs.join("\n\n") : input.content.trim();
|
|
69
|
+
|
|
70
|
+
return ensureAtLeastOneChunk(
|
|
71
|
+
splitContentIntoChunks({
|
|
72
|
+
sourcePath: input.sourcePath,
|
|
73
|
+
title,
|
|
74
|
+
section: null,
|
|
75
|
+
locator: null,
|
|
76
|
+
content,
|
|
77
|
+
startingOrdinal: 0,
|
|
78
|
+
}),
|
|
79
|
+
input,
|
|
80
|
+
title,
|
|
81
|
+
);
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
function chunkCsv(input: ChunkKnowledgeSourceInput): KnowledgeChunk[] {
|
|
85
|
+
const title = input.title ?? inferTitle(input.sourcePath, input.content);
|
|
86
|
+
const rows = parseCsv(input.content);
|
|
87
|
+
|
|
88
|
+
if (rows.length === 0) {
|
|
89
|
+
return ensureAtLeastOneChunk([], input, title);
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
const [headers, ...records] = rows;
|
|
93
|
+
|
|
94
|
+
if (headers.length === 0 || records.length === 0) {
|
|
95
|
+
return chunkText(input);
|
|
96
|
+
}
|
|
97
|
+
|
|
98
|
+
const chunks: KnowledgeChunk[] = [];
|
|
99
|
+
|
|
100
|
+
for (const [index, record] of records.entries()) {
|
|
101
|
+
const content = headers
|
|
102
|
+
.map((header, columnIndex) => {
|
|
103
|
+
const name = header.trim() || `column_${columnIndex + 1}`;
|
|
104
|
+
const value = record[columnIndex]?.trim() ?? "";
|
|
105
|
+
return `${name}: ${value}`;
|
|
106
|
+
})
|
|
107
|
+
.join("\n")
|
|
108
|
+
.trim();
|
|
109
|
+
|
|
110
|
+
if (!content) {
|
|
111
|
+
continue;
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
chunks.push({
|
|
115
|
+
ordinal: chunks.length,
|
|
116
|
+
content,
|
|
117
|
+
tokenCount: estimateTokenCount(content),
|
|
118
|
+
metadata: {
|
|
119
|
+
sourcePath: input.sourcePath,
|
|
120
|
+
title,
|
|
121
|
+
section: null,
|
|
122
|
+
locator: `row:${index + 2}`,
|
|
123
|
+
},
|
|
124
|
+
});
|
|
125
|
+
}
|
|
126
|
+
|
|
127
|
+
return ensureAtLeastOneChunk(chunks, input, title);
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
function markdownSections(content: string, fallbackTitle: string): Array<{ heading: string | null; body: string }> {
|
|
131
|
+
const sections: Array<{ heading: string | null; body: string[] }> = [];
|
|
132
|
+
let current: { heading: string | null; body: string[] } = { heading: null, body: [] };
|
|
133
|
+
const headingStack: string[] = [];
|
|
134
|
+
|
|
135
|
+
for (const line of content.split(/\r?\n/)) {
|
|
136
|
+
const heading = line.match(/^(#{1,6})\s+(.+?)\s*#*\s*$/);
|
|
137
|
+
|
|
138
|
+
if (heading) {
|
|
139
|
+
if (current.body.join("\n").trim()) {
|
|
140
|
+
sections.push(current);
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
const level = heading[1].length;
|
|
144
|
+
headingStack.splice(level - 1);
|
|
145
|
+
headingStack[level - 1] = heading[2].trim();
|
|
146
|
+
current = {
|
|
147
|
+
heading: headingStack.filter(Boolean).join(" > "),
|
|
148
|
+
body: [],
|
|
149
|
+
};
|
|
150
|
+
continue;
|
|
151
|
+
}
|
|
152
|
+
|
|
153
|
+
current.body.push(line);
|
|
154
|
+
}
|
|
155
|
+
|
|
156
|
+
if (current.body.join("\n").trim()) {
|
|
157
|
+
sections.push(current);
|
|
158
|
+
}
|
|
159
|
+
|
|
160
|
+
if (sections.length === 0) {
|
|
161
|
+
return [{ heading: fallbackTitle, body: content.trim() }];
|
|
162
|
+
}
|
|
163
|
+
|
|
164
|
+
return sections.map((section) => ({
|
|
165
|
+
heading: section.heading,
|
|
166
|
+
body: section.body.join("\n").trim(),
|
|
167
|
+
}));
|
|
168
|
+
}
|
|
169
|
+
|
|
170
|
+
function splitContentIntoChunks(input: {
|
|
171
|
+
sourcePath: string;
|
|
172
|
+
title: string;
|
|
173
|
+
section: string | null;
|
|
174
|
+
locator: string | null;
|
|
175
|
+
content: string;
|
|
176
|
+
startingOrdinal: number;
|
|
177
|
+
}): KnowledgeChunk[] {
|
|
178
|
+
const words = wordsOf(input.content);
|
|
179
|
+
|
|
180
|
+
if (words.length <= DEFAULT_MAX_WORDS) {
|
|
181
|
+
const content = input.content.trim();
|
|
182
|
+
|
|
183
|
+
if (!content) {
|
|
184
|
+
return [];
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
return [
|
|
188
|
+
{
|
|
189
|
+
ordinal: input.startingOrdinal,
|
|
190
|
+
content,
|
|
191
|
+
tokenCount: estimateTokenCount(content),
|
|
192
|
+
metadata: {
|
|
193
|
+
sourcePath: input.sourcePath,
|
|
194
|
+
title: input.title,
|
|
195
|
+
section: input.section,
|
|
196
|
+
locator: input.locator,
|
|
197
|
+
},
|
|
198
|
+
},
|
|
199
|
+
];
|
|
200
|
+
}
|
|
201
|
+
|
|
202
|
+
const chunks: KnowledgeChunk[] = [];
|
|
203
|
+
let start = 0;
|
|
204
|
+
|
|
205
|
+
while (start < words.length) {
|
|
206
|
+
const end = Math.min(start + DEFAULT_MAX_WORDS, words.length);
|
|
207
|
+
const content = words.slice(start, end).join(" ");
|
|
208
|
+
|
|
209
|
+
chunks.push({
|
|
210
|
+
ordinal: input.startingOrdinal + chunks.length,
|
|
211
|
+
content,
|
|
212
|
+
tokenCount: estimateTokenCount(content),
|
|
213
|
+
metadata: {
|
|
214
|
+
sourcePath: input.sourcePath,
|
|
215
|
+
title: input.title,
|
|
216
|
+
section: input.section,
|
|
217
|
+
locator: input.locator ? `${input.locator}:part:${chunks.length + 1}` : `part:${chunks.length + 1}`,
|
|
218
|
+
},
|
|
219
|
+
});
|
|
220
|
+
|
|
221
|
+
if (end === words.length) {
|
|
222
|
+
break;
|
|
223
|
+
}
|
|
224
|
+
|
|
225
|
+
start = Math.max(end - DEFAULT_OVERLAP_WORDS, start + 1);
|
|
226
|
+
}
|
|
227
|
+
|
|
228
|
+
return chunks;
|
|
229
|
+
}
|
|
230
|
+
|
|
231
|
+
function ensureAtLeastOneChunk(
|
|
232
|
+
chunks: KnowledgeChunk[],
|
|
233
|
+
input: ChunkKnowledgeSourceInput,
|
|
234
|
+
title: string,
|
|
235
|
+
): KnowledgeChunk[] {
|
|
236
|
+
if (chunks.length > 0) {
|
|
237
|
+
return chunks.map((chunk, index) => ({ ...chunk, ordinal: index }));
|
|
238
|
+
}
|
|
239
|
+
|
|
240
|
+
const content = input.content.trim();
|
|
241
|
+
|
|
242
|
+
if (!content) {
|
|
243
|
+
return [];
|
|
244
|
+
}
|
|
245
|
+
|
|
246
|
+
return [
|
|
247
|
+
{
|
|
248
|
+
ordinal: 0,
|
|
249
|
+
content,
|
|
250
|
+
tokenCount: estimateTokenCount(content),
|
|
251
|
+
metadata: {
|
|
252
|
+
sourcePath: input.sourcePath,
|
|
253
|
+
title,
|
|
254
|
+
section: null,
|
|
255
|
+
locator: null,
|
|
256
|
+
},
|
|
257
|
+
},
|
|
258
|
+
];
|
|
259
|
+
}
|
|
260
|
+
|
|
261
|
+
function parseCsv(content: string): string[][] {
|
|
262
|
+
const rows: string[][] = [];
|
|
263
|
+
let row: string[] = [];
|
|
264
|
+
let field = "";
|
|
265
|
+
let quoted = false;
|
|
266
|
+
|
|
267
|
+
for (let index = 0; index < content.length; index += 1) {
|
|
268
|
+
const char = content[index];
|
|
269
|
+
const next = content[index + 1];
|
|
270
|
+
|
|
271
|
+
if (quoted) {
|
|
272
|
+
if (char === '"' && next === '"') {
|
|
273
|
+
field += '"';
|
|
274
|
+
index += 1;
|
|
275
|
+
continue;
|
|
276
|
+
}
|
|
277
|
+
|
|
278
|
+
if (char === '"') {
|
|
279
|
+
quoted = false;
|
|
280
|
+
continue;
|
|
281
|
+
}
|
|
282
|
+
|
|
283
|
+
field += char;
|
|
284
|
+
continue;
|
|
285
|
+
}
|
|
286
|
+
|
|
287
|
+
if (char === '"') {
|
|
288
|
+
quoted = true;
|
|
289
|
+
continue;
|
|
290
|
+
}
|
|
291
|
+
|
|
292
|
+
if (char === ",") {
|
|
293
|
+
row.push(field);
|
|
294
|
+
field = "";
|
|
295
|
+
continue;
|
|
296
|
+
}
|
|
297
|
+
|
|
298
|
+
if (char === "\n") {
|
|
299
|
+
row.push(field.replace(/\r$/, ""));
|
|
300
|
+
rows.push(row);
|
|
301
|
+
row = [];
|
|
302
|
+
field = "";
|
|
303
|
+
continue;
|
|
304
|
+
}
|
|
305
|
+
|
|
306
|
+
field += char;
|
|
307
|
+
}
|
|
308
|
+
|
|
309
|
+
row.push(field.replace(/\r$/, ""));
|
|
310
|
+
if (row.some((value) => value.length > 0)) {
|
|
311
|
+
rows.push(row);
|
|
312
|
+
}
|
|
313
|
+
|
|
314
|
+
return rows;
|
|
315
|
+
}
|
|
316
|
+
|
|
317
|
+
function inferTitle(sourcePath: string, content: string): string {
|
|
318
|
+
const markdownTitle = content.match(/^#\s+(.+)$/m)?.[1]?.trim();
|
|
319
|
+
|
|
320
|
+
if (markdownTitle) {
|
|
321
|
+
return markdownTitle;
|
|
322
|
+
}
|
|
323
|
+
|
|
324
|
+
return basename(sourcePath).replace(/\.[^.]+$/, "") || sourcePath;
|
|
325
|
+
}
|
|
326
|
+
|
|
327
|
+
function wordsOf(content: string): string[] {
|
|
328
|
+
return content.replace(/\s+/g, " ").trim().split(" ").filter(Boolean);
|
|
329
|
+
}
|
|
330
|
+
|
|
331
|
+
export function estimateTokenCount(content: string): number {
|
|
332
|
+
return Math.max(1, wordsOf(content).length);
|
|
333
|
+
}
|
|
@@ -0,0 +1,135 @@
|
|
|
1
|
+
import type {
|
|
2
|
+
AgentConfig,
|
|
3
|
+
AgentKnowledgeConfig,
|
|
4
|
+
AgentKnowledgeEmbeddingProvider,
|
|
5
|
+
AgentKnowledgeSource,
|
|
6
|
+
} from "../../index";
|
|
7
|
+
|
|
8
|
+
export type ResolvedKnowledgeSource =
|
|
9
|
+
| {
|
|
10
|
+
kind: "file";
|
|
11
|
+
value: string;
|
|
12
|
+
title: string | null;
|
|
13
|
+
}
|
|
14
|
+
| {
|
|
15
|
+
kind: "url";
|
|
16
|
+
value: string;
|
|
17
|
+
title: string | null;
|
|
18
|
+
};
|
|
19
|
+
|
|
20
|
+
export type ResolvedKnowledgeConfig = {
|
|
21
|
+
enabled: boolean;
|
|
22
|
+
sources: ResolvedKnowledgeSource[];
|
|
23
|
+
embedding: {
|
|
24
|
+
provider: AgentKnowledgeEmbeddingProvider;
|
|
25
|
+
model: string | null;
|
|
26
|
+
dimensions: number | null;
|
|
27
|
+
secret: string | null;
|
|
28
|
+
};
|
|
29
|
+
retrieval: {
|
|
30
|
+
topK: number;
|
|
31
|
+
hybrid: boolean;
|
|
32
|
+
};
|
|
33
|
+
};
|
|
34
|
+
|
|
35
|
+
const DEFAULT_TOP_K = 8;
|
|
36
|
+
const DEFAULT_TEST_EMBEDDING_DIMENSIONS = 32;
|
|
37
|
+
const DEFAULT_OPENAI_EMBEDDING_MODEL = "text-embedding-3-small";
|
|
38
|
+
const DEFAULT_OPENAI_EMBEDDING_DIMENSIONS = 1536;
|
|
39
|
+
|
|
40
|
+
export function resolveKnowledgeConfig(config: Pick<AgentConfig, "knowledge">): ResolvedKnowledgeConfig {
|
|
41
|
+
const knowledge = config.knowledge;
|
|
42
|
+
|
|
43
|
+
if (!knowledge) {
|
|
44
|
+
return {
|
|
45
|
+
enabled: false,
|
|
46
|
+
sources: [],
|
|
47
|
+
embedding: {
|
|
48
|
+
provider: "none",
|
|
49
|
+
model: null,
|
|
50
|
+
dimensions: null,
|
|
51
|
+
secret: null,
|
|
52
|
+
},
|
|
53
|
+
retrieval: {
|
|
54
|
+
topK: DEFAULT_TOP_K,
|
|
55
|
+
hybrid: false,
|
|
56
|
+
},
|
|
57
|
+
};
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
const provider = knowledge.embedding?.provider ?? "none";
|
|
61
|
+
const embedding = resolveEmbeddingConfig(provider, knowledge);
|
|
62
|
+
const hybrid = knowledge.retrieval?.hybrid ?? provider !== "none";
|
|
63
|
+
|
|
64
|
+
return {
|
|
65
|
+
enabled: true,
|
|
66
|
+
sources: (knowledge.sources ?? []).map(resolveKnowledgeSource),
|
|
67
|
+
embedding,
|
|
68
|
+
retrieval: {
|
|
69
|
+
topK: knowledge.retrieval?.topK ?? DEFAULT_TOP_K,
|
|
70
|
+
hybrid,
|
|
71
|
+
},
|
|
72
|
+
};
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
function resolveEmbeddingConfig(
|
|
76
|
+
provider: AgentKnowledgeEmbeddingProvider,
|
|
77
|
+
knowledge: AgentKnowledgeConfig,
|
|
78
|
+
): ResolvedKnowledgeConfig["embedding"] {
|
|
79
|
+
if (provider === "openai") {
|
|
80
|
+
return {
|
|
81
|
+
provider,
|
|
82
|
+
model: knowledge.embedding?.model ?? DEFAULT_OPENAI_EMBEDDING_MODEL,
|
|
83
|
+
dimensions: knowledge.embedding?.dimensions ?? DEFAULT_OPENAI_EMBEDDING_DIMENSIONS,
|
|
84
|
+
secret: knowledge.embedding?.secret ?? "OPENAI_API_KEY",
|
|
85
|
+
};
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
if (provider === "test") {
|
|
89
|
+
return {
|
|
90
|
+
provider,
|
|
91
|
+
model: knowledge.embedding?.model ?? "fake",
|
|
92
|
+
dimensions: knowledge.embedding?.dimensions ?? DEFAULT_TEST_EMBEDDING_DIMENSIONS,
|
|
93
|
+
secret: knowledge.embedding?.secret ?? null,
|
|
94
|
+
};
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
return {
|
|
98
|
+
provider: "none",
|
|
99
|
+
model: knowledge.embedding?.model ?? null,
|
|
100
|
+
dimensions: knowledge.embedding?.dimensions ?? null,
|
|
101
|
+
secret: knowledge.embedding?.secret ?? null,
|
|
102
|
+
};
|
|
103
|
+
}
|
|
104
|
+
|
|
105
|
+
function resolveKnowledgeSource(source: AgentKnowledgeSource): ResolvedKnowledgeSource {
|
|
106
|
+
if (typeof source === "string") {
|
|
107
|
+
if (/^https?:\/\//i.test(source)) {
|
|
108
|
+
return {
|
|
109
|
+
kind: "url",
|
|
110
|
+
value: source,
|
|
111
|
+
title: null,
|
|
112
|
+
};
|
|
113
|
+
}
|
|
114
|
+
|
|
115
|
+
return {
|
|
116
|
+
kind: "file",
|
|
117
|
+
value: source,
|
|
118
|
+
title: null,
|
|
119
|
+
};
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
if ("url" in source) {
|
|
123
|
+
return {
|
|
124
|
+
kind: "url",
|
|
125
|
+
value: source.url,
|
|
126
|
+
title: source.title ?? null,
|
|
127
|
+
};
|
|
128
|
+
}
|
|
129
|
+
|
|
130
|
+
return {
|
|
131
|
+
kind: "file",
|
|
132
|
+
value: source.path,
|
|
133
|
+
title: source.title ?? null,
|
|
134
|
+
};
|
|
135
|
+
}
|
|
@@ -0,0 +1,133 @@
|
|
|
1
|
+
import { createHash } from "node:crypto";
|
|
2
|
+
|
|
3
|
+
import type { ResolvedKnowledgeConfig } from "./config";
|
|
4
|
+
import { AgentKitError } from "../errors";
|
|
5
|
+
|
|
6
|
+
export type EmbeddingProvider = {
|
|
7
|
+
embed(input: string): Promise<number[]>;
|
|
8
|
+
};
|
|
9
|
+
|
|
10
|
+
export type CreateEmbeddingProviderOptions = {
|
|
11
|
+
config: ResolvedKnowledgeConfig["embedding"];
|
|
12
|
+
env?: Record<string, string | undefined>;
|
|
13
|
+
fetch?: typeof fetch;
|
|
14
|
+
};
|
|
15
|
+
|
|
16
|
+
export function createEmbeddingProvider(options: CreateEmbeddingProviderOptions): EmbeddingProvider | null {
|
|
17
|
+
if (options.config.provider === "none") {
|
|
18
|
+
return null;
|
|
19
|
+
}
|
|
20
|
+
|
|
21
|
+
if (options.config.provider === "test") {
|
|
22
|
+
return {
|
|
23
|
+
async embed(input) {
|
|
24
|
+
return fakeEmbedding(input, options.config.dimensions ?? 32);
|
|
25
|
+
},
|
|
26
|
+
};
|
|
27
|
+
}
|
|
28
|
+
|
|
29
|
+
if (options.config.provider === "openai") {
|
|
30
|
+
return {
|
|
31
|
+
async embed(input) {
|
|
32
|
+
return openAiEmbedding(input, options);
|
|
33
|
+
},
|
|
34
|
+
};
|
|
35
|
+
}
|
|
36
|
+
|
|
37
|
+
throw new AgentKitError("knowledge_embedding_provider_unsupported", "Unsupported knowledge embedding provider.");
|
|
38
|
+
}
|
|
39
|
+
|
|
40
|
+
export function fakeEmbedding(input: string, dimensions: number): number[] {
|
|
41
|
+
const values = Array.from({ length: dimensions }, () => 0);
|
|
42
|
+
const tokens = input.toLowerCase().match(/[\p{L}\p{N}_-]+/gu) ?? [];
|
|
43
|
+
|
|
44
|
+
for (const token of tokens) {
|
|
45
|
+
const hash = createHash("sha256").update(token).digest();
|
|
46
|
+
const index = hash[0] % dimensions;
|
|
47
|
+
const sign = hash[1] % 2 === 0 ? 1 : -1;
|
|
48
|
+
const weight = 1 + (hash[2] % 7) / 10;
|
|
49
|
+
values[index] += sign * weight;
|
|
50
|
+
}
|
|
51
|
+
|
|
52
|
+
return normalizeVector(values);
|
|
53
|
+
}
|
|
54
|
+
|
|
55
|
+
export function cosineSimilarity(left: number[], right: number[]): number {
|
|
56
|
+
if (left.length !== right.length || left.length === 0) {
|
|
57
|
+
return 0;
|
|
58
|
+
}
|
|
59
|
+
|
|
60
|
+
let dot = 0;
|
|
61
|
+
let leftNorm = 0;
|
|
62
|
+
let rightNorm = 0;
|
|
63
|
+
|
|
64
|
+
for (let index = 0; index < left.length; index += 1) {
|
|
65
|
+
dot += left[index] * right[index];
|
|
66
|
+
leftNorm += left[index] * left[index];
|
|
67
|
+
rightNorm += right[index] * right[index];
|
|
68
|
+
}
|
|
69
|
+
|
|
70
|
+
if (leftNorm === 0 || rightNorm === 0) {
|
|
71
|
+
return 0;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
return dot / (Math.sqrt(leftNorm) * Math.sqrt(rightNorm));
|
|
75
|
+
}
|
|
76
|
+
|
|
77
|
+
async function openAiEmbedding(input: string, options: CreateEmbeddingProviderOptions): Promise<number[]> {
|
|
78
|
+
const secretName = options.config.secret ?? "OPENAI_API_KEY";
|
|
79
|
+
const apiKey = options.env?.[secretName] ?? process.env[secretName];
|
|
80
|
+
|
|
81
|
+
if (!apiKey) {
|
|
82
|
+
throw new AgentKitError(
|
|
83
|
+
"knowledge_embedding_secret_missing",
|
|
84
|
+
`Knowledge embeddings require missing secret "${secretName}". Set it in .env or the process environment.`,
|
|
85
|
+
);
|
|
86
|
+
}
|
|
87
|
+
|
|
88
|
+
const fetchImpl = options.fetch ?? fetch;
|
|
89
|
+
const response = await fetchImpl("https://api.openai.com/v1/embeddings", {
|
|
90
|
+
method: "POST",
|
|
91
|
+
headers: {
|
|
92
|
+
"content-type": "application/json",
|
|
93
|
+
authorization: `Bearer ${apiKey}`,
|
|
94
|
+
},
|
|
95
|
+
body: JSON.stringify({
|
|
96
|
+
model: options.config.model ?? "text-embedding-3-small",
|
|
97
|
+
input,
|
|
98
|
+
encoding_format: "float",
|
|
99
|
+
...(options.config.dimensions ? { dimensions: options.config.dimensions } : {}),
|
|
100
|
+
}),
|
|
101
|
+
});
|
|
102
|
+
|
|
103
|
+
if (!response.ok) {
|
|
104
|
+
const text = await response.text().catch(() => "");
|
|
105
|
+
throw new AgentKitError(
|
|
106
|
+
"knowledge_embedding_failed",
|
|
107
|
+
`OpenAI embedding request failed with HTTP ${response.status}.${text ? ` ${truncate(text, 240)}` : ""}`,
|
|
108
|
+
);
|
|
109
|
+
}
|
|
110
|
+
|
|
111
|
+
const payload = await response.json();
|
|
112
|
+
const embedding = payload?.data?.[0]?.embedding;
|
|
113
|
+
|
|
114
|
+
if (!Array.isArray(embedding) || embedding.some((value) => typeof value !== "number")) {
|
|
115
|
+
throw new AgentKitError("knowledge_embedding_failed", "OpenAI embedding response did not contain a numeric vector.");
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
return embedding;
|
|
119
|
+
}
|
|
120
|
+
|
|
121
|
+
function normalizeVector(values: number[]): number[] {
|
|
122
|
+
const norm = Math.sqrt(values.reduce((sum, value) => sum + value * value, 0));
|
|
123
|
+
|
|
124
|
+
if (norm === 0) {
|
|
125
|
+
return values;
|
|
126
|
+
}
|
|
127
|
+
|
|
128
|
+
return values.map((value) => Number((value / norm).toFixed(8)));
|
|
129
|
+
}
|
|
130
|
+
|
|
131
|
+
function truncate(value: string, maxLength: number): string {
|
|
132
|
+
return value.length <= maxLength ? value : `${value.slice(0, maxLength - 3)}...`;
|
|
133
|
+
}
|