@xano-sdk/vector 1.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/AGENTS.md +68 -0
- package/LICENSE +21 -0
- package/README.md +210 -0
- package/dist/index.d.ts +1189 -0
- package/dist/index.js +1429 -0
- package/llms.txt +62 -0
- package/package.json +85 -0
package/dist/index.js
ADDED
|
@@ -0,0 +1,1429 @@
|
|
|
1
|
+
// src/options.ts
|
|
2
|
+
var CHUNK_STRATEGIES = [
|
|
3
|
+
"fixed",
|
|
4
|
+
"paragraph",
|
|
5
|
+
"sentence",
|
|
6
|
+
"markdown",
|
|
7
|
+
"custom"
|
|
8
|
+
];
|
|
9
|
+
var DOCUMENT_STATUSES = [
|
|
10
|
+
"pending",
|
|
11
|
+
"indexing",
|
|
12
|
+
"indexed",
|
|
13
|
+
"failed"
|
|
14
|
+
];
|
|
15
|
+
var GEMINI_EMBEDDING_DIMENSIONS = 768;
|
|
16
|
+
var DEFAULT_GEMINI_MODEL = "gemini-embedding-2";
|
|
17
|
+
var DEFAULT_API_KEY_ENV = "GEMINI_API_KEY";
|
|
18
|
+
var DEFAULT_CHUNK_SIZE = 500;
|
|
19
|
+
var DEFAULT_CHUNK_OVERLAP = 50;
|
|
20
|
+
var DEFAULT_SEARCH_LIMIT = 10;
|
|
21
|
+
var DEFAULT_SEARCH_THRESHOLD = 0;
|
|
22
|
+
var DEFAULT_STRATEGY = "paragraph";
|
|
23
|
+
var DEFAULT_NAMES = {
|
|
24
|
+
document: "vector_document",
|
|
25
|
+
chunk: "vector_chunk",
|
|
26
|
+
embedFn: "vector/generate_embedding",
|
|
27
|
+
chunkFn: "vector/chunk_text",
|
|
28
|
+
ingestFn: "vector/ingest_document",
|
|
29
|
+
searchFn: "vector/search_vectors",
|
|
30
|
+
searchTool: "vector_search",
|
|
31
|
+
apiGroup: "Vector"
|
|
32
|
+
};
|
|
33
|
+
var CANONICAL_PATTERN = /^[a-zA-Z0-9_-]+$/;
|
|
34
|
+
function inferTableKeyType(table3) {
|
|
35
|
+
if (!table3 || typeof table3 === "string") return "int";
|
|
36
|
+
if (table3.idType === "uuid") return "uuid";
|
|
37
|
+
return "int";
|
|
38
|
+
}
|
|
39
|
+
function resolveOptions(opts = {}) {
|
|
40
|
+
const authenticated = opts.authenticated ?? opts.authTable !== void 0;
|
|
41
|
+
if (authenticated && opts.authTable === void 0) {
|
|
42
|
+
throw new Error(
|
|
43
|
+
"resolveOptions: `authTable` is required when `authenticated` is true."
|
|
44
|
+
);
|
|
45
|
+
}
|
|
46
|
+
const userIdType = opts.userIdType ?? inferTableKeyType(opts.authTable);
|
|
47
|
+
if (userIdType !== "int" && userIdType !== "uuid") {
|
|
48
|
+
throw new Error(
|
|
49
|
+
`resolveOptions: \`userIdType\` must be "int" or "uuid", got "${userIdType}".`
|
|
50
|
+
);
|
|
51
|
+
}
|
|
52
|
+
const model = opts.model ?? DEFAULT_GEMINI_MODEL;
|
|
53
|
+
if (typeof model !== "string" || !model.trim()) {
|
|
54
|
+
throw new Error("resolveOptions: `model` must be a non-empty string.");
|
|
55
|
+
}
|
|
56
|
+
const apiKeyEnv = opts.apiKeyEnv ?? DEFAULT_API_KEY_ENV;
|
|
57
|
+
if (typeof apiKeyEnv !== "string" || !apiKeyEnv.trim()) {
|
|
58
|
+
throw new Error("resolveOptions: `apiKeyEnv` must be a non-empty string.");
|
|
59
|
+
}
|
|
60
|
+
const defaultStrategy = opts.defaultStrategy ?? DEFAULT_STRATEGY;
|
|
61
|
+
if (!CHUNK_STRATEGIES.includes(defaultStrategy)) {
|
|
62
|
+
throw new Error(
|
|
63
|
+
`resolveOptions: \`defaultStrategy\` must be one of [${CHUNK_STRATEGIES.join(", ")}], got "${defaultStrategy}".`
|
|
64
|
+
);
|
|
65
|
+
}
|
|
66
|
+
const defaultChunkSize = opts.defaultChunkSize ?? DEFAULT_CHUNK_SIZE;
|
|
67
|
+
if (typeof defaultChunkSize !== "number" || !Number.isInteger(defaultChunkSize) || defaultChunkSize < 20 || defaultChunkSize > 1e4) {
|
|
68
|
+
throw new Error(
|
|
69
|
+
`resolveOptions: \`defaultChunkSize\` must be an integer between 20 and 10000, got ${defaultChunkSize}.`
|
|
70
|
+
);
|
|
71
|
+
}
|
|
72
|
+
const defaultChunkOverlap = opts.defaultChunkOverlap ?? DEFAULT_CHUNK_OVERLAP;
|
|
73
|
+
if (typeof defaultChunkOverlap !== "number" || !Number.isInteger(defaultChunkOverlap) || defaultChunkOverlap < 0 || defaultChunkOverlap >= defaultChunkSize) {
|
|
74
|
+
throw new Error(
|
|
75
|
+
`resolveOptions: \`defaultChunkOverlap\` must be an integer >= 0 and < defaultChunkSize, got ${defaultChunkOverlap}.`
|
|
76
|
+
);
|
|
77
|
+
}
|
|
78
|
+
const searchLimit = opts.searchLimit ?? DEFAULT_SEARCH_LIMIT;
|
|
79
|
+
if (typeof searchLimit !== "number" || !Number.isInteger(searchLimit) || searchLimit < 1 || searchLimit > 100) {
|
|
80
|
+
throw new Error(
|
|
81
|
+
`resolveOptions: \`searchLimit\` must be an integer between 1 and 100, got ${searchLimit}.`
|
|
82
|
+
);
|
|
83
|
+
}
|
|
84
|
+
const searchThreshold = opts.searchThreshold ?? DEFAULT_SEARCH_THRESHOLD;
|
|
85
|
+
if (typeof searchThreshold !== "number" || searchThreshold < 0 || searchThreshold > 1) {
|
|
86
|
+
throw new Error(
|
|
87
|
+
`resolveOptions: \`searchThreshold\` must be a number between 0.0 and 1.0, got ${searchThreshold}.`
|
|
88
|
+
);
|
|
89
|
+
}
|
|
90
|
+
const taskTypeDocument = opts.taskTypeDocument ?? "RETRIEVAL_DOCUMENT";
|
|
91
|
+
const taskTypeQuery = opts.taskTypeQuery ?? "RETRIEVAL_QUERY";
|
|
92
|
+
const citationFormat = opts.citationFormat ?? "markdown";
|
|
93
|
+
if (opts.names) {
|
|
94
|
+
for (const [k, v] of Object.entries(opts.names)) {
|
|
95
|
+
if (typeof v !== "string" || !v.trim()) {
|
|
96
|
+
throw new Error(`resolveOptions: names.${k} must be a non-empty string.`);
|
|
97
|
+
}
|
|
98
|
+
}
|
|
99
|
+
}
|
|
100
|
+
let routePrefix = opts.routePrefix ?? "vector";
|
|
101
|
+
if (opts.canonical !== void 0) {
|
|
102
|
+
const trimmed = opts.canonical.trim();
|
|
103
|
+
if (!trimmed || !CANONICAL_PATTERN.test(trimmed)) {
|
|
104
|
+
throw new Error(
|
|
105
|
+
`resolveOptions: \`canonical\` must match ${CANONICAL_PATTERN}, got "${opts.canonical}".`
|
|
106
|
+
);
|
|
107
|
+
}
|
|
108
|
+
routePrefix = trimmed;
|
|
109
|
+
}
|
|
110
|
+
routePrefix = routePrefix.replace(/^\/+|\/+$/g, "");
|
|
111
|
+
const names = {
|
|
112
|
+
document: opts.names?.document ?? (opts.canonical ? `${opts.canonical}_document` : DEFAULT_NAMES.document),
|
|
113
|
+
chunk: opts.names?.chunk ?? (opts.canonical ? `${opts.canonical}_chunk` : DEFAULT_NAMES.chunk),
|
|
114
|
+
embedFn: opts.names?.embedFn ?? (opts.canonical ? `${opts.canonical}/generate_embedding` : DEFAULT_NAMES.embedFn),
|
|
115
|
+
chunkFn: opts.names?.chunkFn ?? (opts.canonical ? `${opts.canonical}/chunk_text` : DEFAULT_NAMES.chunkFn),
|
|
116
|
+
ingestFn: opts.names?.ingestFn ?? (opts.canonical ? `${opts.canonical}/ingest_document` : DEFAULT_NAMES.ingestFn),
|
|
117
|
+
searchFn: opts.names?.searchFn ?? (opts.canonical ? `${opts.canonical}/search_vectors` : DEFAULT_NAMES.searchFn),
|
|
118
|
+
searchTool: opts.names?.searchTool ?? DEFAULT_NAMES.searchTool,
|
|
119
|
+
apiGroup: opts.names?.apiGroup ?? (opts.canonical ? `Vector (${opts.canonical})` : DEFAULT_NAMES.apiGroup)
|
|
120
|
+
};
|
|
121
|
+
const tags = opts.tags ?? ["vector", "ai", "search"];
|
|
122
|
+
return {
|
|
123
|
+
authTable: opts.authTable,
|
|
124
|
+
authenticated,
|
|
125
|
+
userIdType,
|
|
126
|
+
model,
|
|
127
|
+
apiKeyEnv,
|
|
128
|
+
dimensions: GEMINI_EMBEDDING_DIMENSIONS,
|
|
129
|
+
defaultStrategy,
|
|
130
|
+
defaultChunkSize,
|
|
131
|
+
defaultChunkOverlap,
|
|
132
|
+
searchLimit,
|
|
133
|
+
searchThreshold,
|
|
134
|
+
taskTypeDocument,
|
|
135
|
+
taskTypeQuery,
|
|
136
|
+
citationFormat,
|
|
137
|
+
routePrefix,
|
|
138
|
+
canonical: opts.canonical,
|
|
139
|
+
names,
|
|
140
|
+
tags
|
|
141
|
+
};
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
// src/tables/document.ts
|
|
145
|
+
import { table, f } from "@xano/sdk";
|
|
146
|
+
function documentTable(opts) {
|
|
147
|
+
const schema = {
|
|
148
|
+
title: f.text({
|
|
149
|
+
required: true,
|
|
150
|
+
methods: ["trim", "min:1"],
|
|
151
|
+
description: "Human-readable title or filename for the document."
|
|
152
|
+
}),
|
|
153
|
+
content: f.text({
|
|
154
|
+
required: false,
|
|
155
|
+
description: "Raw textual content of the document."
|
|
156
|
+
}),
|
|
157
|
+
media_data: f.text({
|
|
158
|
+
required: false,
|
|
159
|
+
description: "Base64-encoded media data for multimodal documents (image, audio, video)."
|
|
160
|
+
}),
|
|
161
|
+
mime_type: f.text({
|
|
162
|
+
default: "text/plain",
|
|
163
|
+
description: "MIME type (e.g. text/markdown, image/png, image/jpeg, audio/mp3, video/mp4)."
|
|
164
|
+
}),
|
|
165
|
+
metadata: f.json({
|
|
166
|
+
description: "Arbitrary structured metadata (tags, author, source URL, dimensions, etc.)."
|
|
167
|
+
}),
|
|
168
|
+
status: f.enum(DOCUMENT_STATUSES, {
|
|
169
|
+
default: "pending",
|
|
170
|
+
description: "Current indexing lifecycle status: pending, indexing, indexed, or failed."
|
|
171
|
+
}),
|
|
172
|
+
chunk_count: f.int({
|
|
173
|
+
default: 0,
|
|
174
|
+
description: "Total number of vector chunks generated from this document."
|
|
175
|
+
}),
|
|
176
|
+
strategy: f.enum(CHUNK_STRATEGIES, {
|
|
177
|
+
default: opts.defaultStrategy,
|
|
178
|
+
description: "Chunking strategy used during indexing."
|
|
179
|
+
}),
|
|
180
|
+
chunk_size: f.int({
|
|
181
|
+
default: opts.defaultChunkSize,
|
|
182
|
+
description: "Target character chunk size used during indexing."
|
|
183
|
+
}),
|
|
184
|
+
chunk_overlap: f.int({
|
|
185
|
+
default: opts.defaultChunkOverlap,
|
|
186
|
+
description: "Chunk character overlap used during indexing."
|
|
187
|
+
}),
|
|
188
|
+
error_message: f.text({
|
|
189
|
+
description: "Error description if status is failed."
|
|
190
|
+
})
|
|
191
|
+
};
|
|
192
|
+
if (opts.authenticated && opts.authTable) {
|
|
193
|
+
schema.user_id = f.tableRef(opts.authTable, {
|
|
194
|
+
required: false,
|
|
195
|
+
description: "The user who owns or uploaded this document."
|
|
196
|
+
});
|
|
197
|
+
}
|
|
198
|
+
const indexes = [
|
|
199
|
+
{
|
|
200
|
+
type: "btree",
|
|
201
|
+
fields: [{ name: "status", op: "asc" }]
|
|
202
|
+
}
|
|
203
|
+
];
|
|
204
|
+
if (opts.authenticated && opts.authTable) {
|
|
205
|
+
indexes.push({
|
|
206
|
+
type: "btree",
|
|
207
|
+
fields: [{ name: "user_id", op: "asc" }]
|
|
208
|
+
});
|
|
209
|
+
}
|
|
210
|
+
return table({
|
|
211
|
+
name: opts.names.document,
|
|
212
|
+
description: "Source documents and media assets managed by the vector pipeline.",
|
|
213
|
+
auth: false,
|
|
214
|
+
useXdo: true,
|
|
215
|
+
tags: opts.tags,
|
|
216
|
+
schema,
|
|
217
|
+
index: indexes
|
|
218
|
+
});
|
|
219
|
+
}
|
|
220
|
+
var PUBLIC_DOCUMENT_FIELDS = [
|
|
221
|
+
"id",
|
|
222
|
+
"created_at",
|
|
223
|
+
"title",
|
|
224
|
+
"content",
|
|
225
|
+
"media_data",
|
|
226
|
+
"mime_type",
|
|
227
|
+
"metadata",
|
|
228
|
+
"status",
|
|
229
|
+
"chunk_count",
|
|
230
|
+
"strategy",
|
|
231
|
+
"chunk_size",
|
|
232
|
+
"chunk_overlap",
|
|
233
|
+
"error_message"
|
|
234
|
+
];
|
|
235
|
+
|
|
236
|
+
// src/tables/chunk.ts
|
|
237
|
+
import { table as table2, f as f2 } from "@xano/sdk";
|
|
238
|
+
function chunkTable(opts, document) {
|
|
239
|
+
return table2({
|
|
240
|
+
name: opts.names.chunk,
|
|
241
|
+
description: "Document chunks and multimodal embeddings for semantic vector search.",
|
|
242
|
+
auth: false,
|
|
243
|
+
useXdo: true,
|
|
244
|
+
tags: opts.tags,
|
|
245
|
+
schema: {
|
|
246
|
+
document_id: f2.tableRef(document, {
|
|
247
|
+
required: true,
|
|
248
|
+
description: "The document this chunk was generated from."
|
|
249
|
+
}),
|
|
250
|
+
document_title: f2.text({
|
|
251
|
+
required: false,
|
|
252
|
+
description: "Title of the parent document."
|
|
253
|
+
}),
|
|
254
|
+
chunk_index: f2.int({
|
|
255
|
+
required: true,
|
|
256
|
+
description: "0-based sequence index of the chunk within the document."
|
|
257
|
+
}),
|
|
258
|
+
content: f2.text({
|
|
259
|
+
required: false,
|
|
260
|
+
description: "The text content or caption of this chunk."
|
|
261
|
+
}),
|
|
262
|
+
mime_type: f2.text({
|
|
263
|
+
default: "text/plain",
|
|
264
|
+
description: "MIME type of the chunk content (e.g. text/plain, image/png, audio/mp3)."
|
|
265
|
+
}),
|
|
266
|
+
embedding: f2.vector(GEMINI_EMBEDDING_DIMENSIONS, {
|
|
267
|
+
description: "Google Gemini 768-dimensional multimodal vector embedding."
|
|
268
|
+
}),
|
|
269
|
+
metadata: f2.json({
|
|
270
|
+
description: "Metadata specific to this chunk (e.g. section title, timestamps, bounding box)."
|
|
271
|
+
}),
|
|
272
|
+
char_count: f2.int({
|
|
273
|
+
default: 0,
|
|
274
|
+
description: "Character length of the text content."
|
|
275
|
+
})
|
|
276
|
+
},
|
|
277
|
+
index: [
|
|
278
|
+
{
|
|
279
|
+
type: "vector",
|
|
280
|
+
fields: [{ name: "embedding", op: "vector_cosine_ops" }]
|
|
281
|
+
},
|
|
282
|
+
{
|
|
283
|
+
type: "btree",
|
|
284
|
+
fields: [
|
|
285
|
+
{ name: "document_id", op: "asc" },
|
|
286
|
+
{ name: "chunk_index", op: "asc" }
|
|
287
|
+
]
|
|
288
|
+
}
|
|
289
|
+
]
|
|
290
|
+
});
|
|
291
|
+
}
|
|
292
|
+
var PUBLIC_CHUNK_FIELDS = [
|
|
293
|
+
"id",
|
|
294
|
+
"created_at",
|
|
295
|
+
"document_id",
|
|
296
|
+
"document_title",
|
|
297
|
+
"chunk_index",
|
|
298
|
+
"content",
|
|
299
|
+
"mime_type",
|
|
300
|
+
"metadata",
|
|
301
|
+
"char_count"
|
|
302
|
+
];
|
|
303
|
+
|
|
304
|
+
// src/functions/embed.ts
|
|
305
|
+
import {
|
|
306
|
+
defineFunction,
|
|
307
|
+
input,
|
|
308
|
+
s,
|
|
309
|
+
setVar,
|
|
310
|
+
c,
|
|
311
|
+
inp,
|
|
312
|
+
ref,
|
|
313
|
+
expr,
|
|
314
|
+
or,
|
|
315
|
+
obj,
|
|
316
|
+
env,
|
|
317
|
+
withFilters,
|
|
318
|
+
fl
|
|
319
|
+
} from "@xano/sdk";
|
|
320
|
+
var BUILD_PARTS_LAMBDA_JS = `
|
|
321
|
+
const parts = [];
|
|
322
|
+
const textVal = String($var.raw_text || "").trim();
|
|
323
|
+
const mediaVal = String($var.raw_media || "").trim();
|
|
324
|
+
const mimeVal = String($var.raw_mime || "image/png").trim();
|
|
325
|
+
const modelId = String($var.model_identifier || "models/gemini-embedding-2");
|
|
326
|
+
const taskTypeVal = String($var.raw_task_type || "").trim();
|
|
327
|
+
|
|
328
|
+
if (textVal) {
|
|
329
|
+
parts.push({ text: textVal });
|
|
330
|
+
}
|
|
331
|
+
if (mediaVal) {
|
|
332
|
+
parts.push({
|
|
333
|
+
inlineData: {
|
|
334
|
+
mimeType: mimeVal,
|
|
335
|
+
data: mediaVal,
|
|
336
|
+
},
|
|
337
|
+
});
|
|
338
|
+
}
|
|
339
|
+
|
|
340
|
+
const req = {
|
|
341
|
+
model: modelId,
|
|
342
|
+
content: { parts },
|
|
343
|
+
outputDimensionality: 768,
|
|
344
|
+
};
|
|
345
|
+
|
|
346
|
+
if (taskTypeVal) {
|
|
347
|
+
req.taskType = taskTypeVal;
|
|
348
|
+
}
|
|
349
|
+
|
|
350
|
+
return req;
|
|
351
|
+
`.trim();
|
|
352
|
+
function generateEmbeddingFn(opts) {
|
|
353
|
+
return defineFunction({
|
|
354
|
+
name: opts.names.embedFn,
|
|
355
|
+
description: "Generates a 768-dimensional multimodal vector embedding for text, images, audio, or video using Google Gemini Embeddings 2.",
|
|
356
|
+
tags: opts.tags,
|
|
357
|
+
input: {
|
|
358
|
+
text: input.text({
|
|
359
|
+
required: false,
|
|
360
|
+
description: "Text content to generate a vector embedding for."
|
|
361
|
+
}),
|
|
362
|
+
media_data: input.text({
|
|
363
|
+
required: false,
|
|
364
|
+
description: "Base64-encoded media data (image, audio, or video bytes)."
|
|
365
|
+
}),
|
|
366
|
+
mime_type: input.text({
|
|
367
|
+
default: "text/plain",
|
|
368
|
+
description: "MIME type for media data (e.g. image/png, image/jpeg, audio/mp3, video/mp4)."
|
|
369
|
+
}),
|
|
370
|
+
task_type: input.text({
|
|
371
|
+
required: false,
|
|
372
|
+
description: "Gemini embedding task type (RETRIEVAL_DOCUMENT, RETRIEVAL_QUERY, SEMANTIC_SIMILARITY, etc.)."
|
|
373
|
+
}),
|
|
374
|
+
model: input.text({
|
|
375
|
+
default: opts.model,
|
|
376
|
+
description: "Google Gemini embedding model name (default: gemini-embedding-2)."
|
|
377
|
+
}),
|
|
378
|
+
api_key: input.text({
|
|
379
|
+
required: false,
|
|
380
|
+
description: "Optional Google API key override. If omitted, uses the configured environment variable."
|
|
381
|
+
})
|
|
382
|
+
},
|
|
383
|
+
stack: [
|
|
384
|
+
setVar("raw_text", inp("text")),
|
|
385
|
+
setVar("raw_media", inp("media_data")),
|
|
386
|
+
setVar("raw_mime", inp("mime_type")),
|
|
387
|
+
setVar("raw_task_type", inp("task_type")),
|
|
388
|
+
s.precondition({
|
|
389
|
+
expr: or(
|
|
390
|
+
expr(ref("raw_text"), "!=", c.text("")),
|
|
391
|
+
expr(ref("raw_media"), "!=", c.text(""))
|
|
392
|
+
),
|
|
393
|
+
error_type: "badrequest",
|
|
394
|
+
error: c.text("generate_embedding: at least one of `text` or `media_data` must be provided.")
|
|
395
|
+
}),
|
|
396
|
+
setVar(
|
|
397
|
+
"active_key",
|
|
398
|
+
withFilters(inp("api_key"), [fl.first_notempty(env(opts.apiKeyEnv))])
|
|
399
|
+
),
|
|
400
|
+
s.precondition({
|
|
401
|
+
expr: expr(ref("active_key"), "!=", c.text("")),
|
|
402
|
+
error_type: "badrequest",
|
|
403
|
+
error: c.text(
|
|
404
|
+
`generate_embedding: missing Gemini API key. Set \`${opts.apiKeyEnv}\` in your environment or pass \`api_key\` explicitly.`
|
|
405
|
+
)
|
|
406
|
+
}),
|
|
407
|
+
setVar(
|
|
408
|
+
"endpoint_url",
|
|
409
|
+
withFilters(c.text("https://generativelanguage.googleapis.com/v1beta/models/"), [
|
|
410
|
+
fl.concat(inp("model")),
|
|
411
|
+
fl.concat(c.text(":embedContent?key=")),
|
|
412
|
+
fl.concat(ref("active_key"))
|
|
413
|
+
])
|
|
414
|
+
),
|
|
415
|
+
setVar(
|
|
416
|
+
"model_identifier",
|
|
417
|
+
withFilters(c.text("models/"), [fl.concat(inp("model"))])
|
|
418
|
+
),
|
|
419
|
+
s.lambda({
|
|
420
|
+
as: "request_body",
|
|
421
|
+
code: c.text(BUILD_PARTS_LAMBDA_JS)
|
|
422
|
+
}),
|
|
423
|
+
s.api.request({
|
|
424
|
+
url: ref("endpoint_url"),
|
|
425
|
+
method: "POST",
|
|
426
|
+
headers: {
|
|
427
|
+
"Content-Type": "application/json"
|
|
428
|
+
},
|
|
429
|
+
params: ref("request_body"),
|
|
430
|
+
as: "gemini_res"
|
|
431
|
+
}),
|
|
432
|
+
s.precondition({
|
|
433
|
+
expr: expr(ref("gemini_res.response.status"), "=", c.int(200)),
|
|
434
|
+
error_type: "badrequest",
|
|
435
|
+
error: withFilters(c.text("Gemini API error: "), [
|
|
436
|
+
fl.concat(ref("gemini_res.response.status")),
|
|
437
|
+
fl.concat(c.text(" - ")),
|
|
438
|
+
fl.concat(ref("gemini_res.response.result.error.message"))
|
|
439
|
+
])
|
|
440
|
+
}),
|
|
441
|
+
setVar(
|
|
442
|
+
"embedding_values",
|
|
443
|
+
ref("gemini_res.response.result.embedding.values")
|
|
444
|
+
)
|
|
445
|
+
],
|
|
446
|
+
response: obj({
|
|
447
|
+
embedding: ref("embedding_values"),
|
|
448
|
+
dimensions: c.int(GEMINI_EMBEDDING_DIMENSIONS)
|
|
449
|
+
}),
|
|
450
|
+
responseShape: {}
|
|
451
|
+
});
|
|
452
|
+
}
|
|
453
|
+
|
|
454
|
+
// src/functions/chunk.ts
|
|
455
|
+
import {
|
|
456
|
+
defineFunction as defineFunction2,
|
|
457
|
+
input as input2,
|
|
458
|
+
s as s2,
|
|
459
|
+
setVar as setVar2,
|
|
460
|
+
c as c2,
|
|
461
|
+
inp as inp2,
|
|
462
|
+
ref as ref2,
|
|
463
|
+
obj as obj2,
|
|
464
|
+
withFilters as withFilters2,
|
|
465
|
+
fl as fl2
|
|
466
|
+
} from "@xano/sdk";
|
|
467
|
+
var CHUNKER_LAMBDA_JS = `
|
|
468
|
+
const rawContent = String($var.raw_text || "").trim();
|
|
469
|
+
const strat = String($var.active_strategy || "paragraph");
|
|
470
|
+
const sizeLimit = Math.max(20, Number($var.target_size) || 500);
|
|
471
|
+
const overlapSize = Math.max(0, Math.min(sizeLimit - 1, Number($var.target_overlap) || 50));
|
|
472
|
+
|
|
473
|
+
if (!rawContent) {
|
|
474
|
+
return [];
|
|
475
|
+
}
|
|
476
|
+
|
|
477
|
+
const chunkList = [];
|
|
478
|
+
|
|
479
|
+
const buildChunk = (chunkIndex, textSegment, extraMeta = {}) => ({
|
|
480
|
+
index: chunkIndex,
|
|
481
|
+
content: textSegment.trim(),
|
|
482
|
+
char_count: textSegment.trim().length,
|
|
483
|
+
metadata: { strategy: strat, ...extraMeta },
|
|
484
|
+
});
|
|
485
|
+
|
|
486
|
+
if (strat === "fixed") {
|
|
487
|
+
const stepSize = Math.max(1, sizeLimit - overlapSize);
|
|
488
|
+
let currentIdx = 0;
|
|
489
|
+
for (let charPos = 0; charPos < rawContent.length; charPos += stepSize) {
|
|
490
|
+
const segment = rawContent.slice(charPos, charPos + sizeLimit);
|
|
491
|
+
if (segment.trim().length > 0) {
|
|
492
|
+
chunkList.push(buildChunk(currentIdx++, segment, { start_char: charPos, end_char: charPos + segment.length }));
|
|
493
|
+
}
|
|
494
|
+
}
|
|
495
|
+
} else if (strat === "sentence") {
|
|
496
|
+
const sentenceList = rawContent.split(/(?<=[.?!])\\s+/);
|
|
497
|
+
let currentAccumulator = "";
|
|
498
|
+
let currentIdx = 0;
|
|
499
|
+
for (const singleSentence of sentenceList) {
|
|
500
|
+
const trimmedSentence = singleSentence.trim();
|
|
501
|
+
if (!trimmedSentence) continue;
|
|
502
|
+
if (currentAccumulator && (currentAccumulator.length + 1 + trimmedSentence.length > sizeLimit)) {
|
|
503
|
+
chunkList.push(buildChunk(currentIdx++, currentAccumulator));
|
|
504
|
+
if (overlapSize > 0 && currentAccumulator.length > overlapSize) {
|
|
505
|
+
currentAccumulator = currentAccumulator.slice(currentAccumulator.length - overlapSize).trim() + " " + trimmedSentence;
|
|
506
|
+
} else {
|
|
507
|
+
currentAccumulator = trimmedSentence;
|
|
508
|
+
}
|
|
509
|
+
} else {
|
|
510
|
+
currentAccumulator = currentAccumulator ? currentAccumulator + " " + trimmedSentence : trimmedSentence;
|
|
511
|
+
}
|
|
512
|
+
}
|
|
513
|
+
if (currentAccumulator.trim().length > 0) {
|
|
514
|
+
chunkList.push(buildChunk(currentIdx++, currentAccumulator));
|
|
515
|
+
}
|
|
516
|
+
} else if (strat === "markdown") {
|
|
517
|
+
const sectionList = rawContent.split(/(?=(?:^|\\n)#{1,6}\\s+)/);
|
|
518
|
+
let currentIdx = 0;
|
|
519
|
+
for (const singleSection of sectionList) {
|
|
520
|
+
const trimmedSection = singleSection.trim();
|
|
521
|
+
if (!trimmedSection) continue;
|
|
522
|
+
const headerMatch = trimmedSection.match(/^#{1,6}\\s+(.+)$/m);
|
|
523
|
+
const headerTitle = headerMatch ? headerMatch[1] : "";
|
|
524
|
+
if (trimmedSection.length <= sizeLimit) {
|
|
525
|
+
chunkList.push(buildChunk(currentIdx++, trimmedSection, { header: headerTitle }));
|
|
526
|
+
} else {
|
|
527
|
+
const stepSize = Math.max(1, sizeLimit - overlapSize);
|
|
528
|
+
for (let charPos = 0; charPos < trimmedSection.length; charPos += stepSize) {
|
|
529
|
+
const segment = trimmedSection.slice(charPos, charPos + sizeLimit);
|
|
530
|
+
if (segment.trim().length > 0) {
|
|
531
|
+
chunkList.push(buildChunk(currentIdx++, segment, { header: headerTitle, start_char: charPos }));
|
|
532
|
+
}
|
|
533
|
+
}
|
|
534
|
+
}
|
|
535
|
+
}
|
|
536
|
+
} else if (strat === "custom") {
|
|
537
|
+
const customBlocks = rawContent.split(/\\n\\n+/);
|
|
538
|
+
let currentIdx = 0;
|
|
539
|
+
for (const customBlock of customBlocks) {
|
|
540
|
+
const trimmedBlock = customBlock.trim();
|
|
541
|
+
if (trimmedBlock.length > 0) {
|
|
542
|
+
chunkList.push(buildChunk(currentIdx++, trimmedBlock));
|
|
543
|
+
}
|
|
544
|
+
}
|
|
545
|
+
} else {
|
|
546
|
+
const paragraphList = rawContent.split(/\\n\\s*\\n+/);
|
|
547
|
+
let currentAccumulator = "";
|
|
548
|
+
let currentIdx = 0;
|
|
549
|
+
for (const singleParagraph of paragraphList) {
|
|
550
|
+
const trimmedParagraph = singleParagraph.trim();
|
|
551
|
+
if (!trimmedParagraph) continue;
|
|
552
|
+
if (currentAccumulator && (currentAccumulator.length + 2 + trimmedParagraph.length > sizeLimit)) {
|
|
553
|
+
chunkList.push(buildChunk(currentIdx++, currentAccumulator));
|
|
554
|
+
if (overlapSize > 0 && currentAccumulator.length > overlapSize) {
|
|
555
|
+
currentAccumulator = currentAccumulator.slice(currentAccumulator.length - overlapSize).trim() + "\\n\\n" + trimmedParagraph;
|
|
556
|
+
} else {
|
|
557
|
+
currentAccumulator = trimmedParagraph;
|
|
558
|
+
}
|
|
559
|
+
} else {
|
|
560
|
+
currentAccumulator = currentAccumulator ? currentAccumulator + "\\n\\n" + trimmedParagraph : trimmedParagraph;
|
|
561
|
+
}
|
|
562
|
+
}
|
|
563
|
+
if (currentAccumulator.trim().length > 0) {
|
|
564
|
+
chunkList.push(buildChunk(currentIdx++, currentAccumulator));
|
|
565
|
+
}
|
|
566
|
+
}
|
|
567
|
+
|
|
568
|
+
return chunkList;
|
|
569
|
+
`.trim();
|
|
570
|
+
function chunkTextFn(opts) {
|
|
571
|
+
return defineFunction2({
|
|
572
|
+
name: opts.names.chunkFn,
|
|
573
|
+
description: "Segments input text into structured chunks based on the chosen strategy (fixed, paragraph, sentence, markdown, custom).",
|
|
574
|
+
tags: opts.tags,
|
|
575
|
+
input: {
|
|
576
|
+
content: input2.text({
|
|
577
|
+
required: false,
|
|
578
|
+
description: "The raw text content to segment into chunks."
|
|
579
|
+
}),
|
|
580
|
+
strategy: input2.enum(CHUNK_STRATEGIES, {
|
|
581
|
+
default: opts.defaultStrategy,
|
|
582
|
+
description: "Chunking algorithm: fixed, paragraph, sentence, markdown, or custom."
|
|
583
|
+
}),
|
|
584
|
+
chunk_size: input2.int({
|
|
585
|
+
default: opts.defaultChunkSize,
|
|
586
|
+
description: "Target maximum character count per chunk."
|
|
587
|
+
}),
|
|
588
|
+
chunk_overlap: input2.int({
|
|
589
|
+
default: opts.defaultChunkOverlap,
|
|
590
|
+
description: "Character overlap preserved between consecutive chunks."
|
|
591
|
+
})
|
|
592
|
+
},
|
|
593
|
+
stack: [
|
|
594
|
+
setVar2("target_size", inp2("chunk_size")),
|
|
595
|
+
setVar2("target_overlap", inp2("chunk_overlap")),
|
|
596
|
+
setVar2("active_strategy", inp2("strategy")),
|
|
597
|
+
setVar2("raw_text", inp2("content")),
|
|
598
|
+
s2.lambda({
|
|
599
|
+
as: "chunks",
|
|
600
|
+
code: c2.text(CHUNKER_LAMBDA_JS)
|
|
601
|
+
})
|
|
602
|
+
],
|
|
603
|
+
response: obj2({
|
|
604
|
+
chunks: ref2("chunks"),
|
|
605
|
+
count: withFilters2(ref2("chunks"), [fl2.count()])
|
|
606
|
+
}),
|
|
607
|
+
responseShape: {}
|
|
608
|
+
});
|
|
609
|
+
}
|
|
610
|
+
|
|
611
|
+
// src/functions/ingest.ts
|
|
612
|
+
import {
|
|
613
|
+
defineFunction as defineFunction3,
|
|
614
|
+
input as input3,
|
|
615
|
+
s as s3,
|
|
616
|
+
setVar as setVar3,
|
|
617
|
+
c as c3,
|
|
618
|
+
inp as inp3,
|
|
619
|
+
ref as ref3,
|
|
620
|
+
col,
|
|
621
|
+
expr as expr2,
|
|
622
|
+
and,
|
|
623
|
+
obj as obj3,
|
|
624
|
+
withFilters as withFilters3,
|
|
625
|
+
fl as fl3
|
|
626
|
+
} from "@xano/sdk";
|
|
627
|
+
function ingestDocumentFn(opts, document, chunk, embedFn, chunkFn) {
|
|
628
|
+
return defineFunction3({
|
|
629
|
+
name: opts.names.ingestFn,
|
|
630
|
+
description: "Fetches a document/media record, chunks text (or processes media), generates Gemini embeddings, and saves chunks.",
|
|
631
|
+
tags: opts.tags,
|
|
632
|
+
input: {
|
|
633
|
+
document_id: input3.int({
|
|
634
|
+
required: true,
|
|
635
|
+
description: "The primary key ID of the document to ingest and index."
|
|
636
|
+
}),
|
|
637
|
+
api_key: input3.text({
|
|
638
|
+
required: false,
|
|
639
|
+
description: "Optional Google API key override."
|
|
640
|
+
})
|
|
641
|
+
},
|
|
642
|
+
stack: [
|
|
643
|
+
s3.db.get({
|
|
644
|
+
table: document,
|
|
645
|
+
fieldValue: inp3("document_id"),
|
|
646
|
+
as: "doc"
|
|
647
|
+
}),
|
|
648
|
+
s3.precondition({
|
|
649
|
+
expr: expr2(ref3("doc"), "!=", c3.null()),
|
|
650
|
+
error_type: "notfound",
|
|
651
|
+
error: c3.text("ingest_document: document not found.")
|
|
652
|
+
}),
|
|
653
|
+
s3.db.edit({
|
|
654
|
+
table: document,
|
|
655
|
+
fieldValue: ref3("doc.id"),
|
|
656
|
+
row: {
|
|
657
|
+
status: c3.text("indexing"),
|
|
658
|
+
error_message: c3.null()
|
|
659
|
+
}
|
|
660
|
+
}),
|
|
661
|
+
s3.db.bulk.delete({
|
|
662
|
+
table: chunk,
|
|
663
|
+
where: expr2(col("document_id"), "=", ref3("doc.id"))
|
|
664
|
+
}),
|
|
665
|
+
s3.conditional({
|
|
666
|
+
when: and(
|
|
667
|
+
expr2(ref3("doc.media_data"), "!=", c3.null()),
|
|
668
|
+
expr2(ref3("doc.media_data"), "!=", c3.text(""))
|
|
669
|
+
),
|
|
670
|
+
then: [
|
|
671
|
+
setVar3(
|
|
672
|
+
"media_text_label",
|
|
673
|
+
withFilters3(ref3("doc.content"), [fl3.first_notempty(ref3("doc.title"))])
|
|
674
|
+
),
|
|
675
|
+
s3.function.run({
|
|
676
|
+
fn: embedFn,
|
|
677
|
+
input: {
|
|
678
|
+
text: ref3("media_text_label"),
|
|
679
|
+
media_data: ref3("doc.media_data"),
|
|
680
|
+
mime_type: ref3("doc.mime_type"),
|
|
681
|
+
task_type: c3.text(opts.taskTypeDocument),
|
|
682
|
+
api_key: inp3("api_key")
|
|
683
|
+
},
|
|
684
|
+
as: "single_embed_res"
|
|
685
|
+
}),
|
|
686
|
+
s3.db.add({
|
|
687
|
+
table: chunk,
|
|
688
|
+
row: {
|
|
689
|
+
document_id: ref3("doc.id"),
|
|
690
|
+
document_title: ref3("doc.title"),
|
|
691
|
+
chunk_index: c3.int(0),
|
|
692
|
+
content: ref3("media_text_label"),
|
|
693
|
+
mime_type: ref3("doc.mime_type"),
|
|
694
|
+
embedding: ref3("single_embed_res.embedding"),
|
|
695
|
+
metadata: ref3("doc.metadata"),
|
|
696
|
+
char_count: c3.int(0)
|
|
697
|
+
}
|
|
698
|
+
}),
|
|
699
|
+
setVar3("final_chunk_count", c3.int(1))
|
|
700
|
+
],
|
|
701
|
+
else: [
|
|
702
|
+
s3.function.run({
|
|
703
|
+
fn: chunkFn,
|
|
704
|
+
input: {
|
|
705
|
+
content: ref3("doc.content"),
|
|
706
|
+
strategy: ref3("doc.strategy"),
|
|
707
|
+
chunk_size: ref3("doc.chunk_size"),
|
|
708
|
+
chunk_overlap: ref3("doc.chunk_overlap")
|
|
709
|
+
},
|
|
710
|
+
as: "chunks_res"
|
|
711
|
+
}),
|
|
712
|
+
s3.foreach({
|
|
713
|
+
list: ref3("chunks_res.chunks"),
|
|
714
|
+
as: "item",
|
|
715
|
+
body: [
|
|
716
|
+
s3.function.run({
|
|
717
|
+
fn: embedFn,
|
|
718
|
+
input: {
|
|
719
|
+
text: ref3("item.content"),
|
|
720
|
+
task_type: c3.text(opts.taskTypeDocument),
|
|
721
|
+
api_key: inp3("api_key")
|
|
722
|
+
},
|
|
723
|
+
as: "embed_res"
|
|
724
|
+
}),
|
|
725
|
+
s3.db.add({
|
|
726
|
+
table: chunk,
|
|
727
|
+
row: {
|
|
728
|
+
document_id: ref3("doc.id"),
|
|
729
|
+
document_title: ref3("doc.title"),
|
|
730
|
+
chunk_index: ref3("item.index"),
|
|
731
|
+
content: ref3("item.content"),
|
|
732
|
+
mime_type: ref3("doc.mime_type"),
|
|
733
|
+
embedding: ref3("embed_res.embedding"),
|
|
734
|
+
metadata: ref3("item.metadata"),
|
|
735
|
+
char_count: ref3("item.char_count")
|
|
736
|
+
}
|
|
737
|
+
})
|
|
738
|
+
]
|
|
739
|
+
}),
|
|
740
|
+
setVar3("final_chunk_count", ref3("chunks_res.count"))
|
|
741
|
+
]
|
|
742
|
+
}),
|
|
743
|
+
s3.db.edit({
|
|
744
|
+
table: document,
|
|
745
|
+
fieldValue: ref3("doc.id"),
|
|
746
|
+
row: {
|
|
747
|
+
status: c3.text("indexed"),
|
|
748
|
+
chunk_count: ref3("final_chunk_count"),
|
|
749
|
+
error_message: c3.null()
|
|
750
|
+
}
|
|
751
|
+
})
|
|
752
|
+
],
|
|
753
|
+
response: obj3({
|
|
754
|
+
document_id: ref3("doc.id"),
|
|
755
|
+
status: c3.text("indexed"),
|
|
756
|
+
chunk_count: ref3("final_chunk_count")
|
|
757
|
+
}),
|
|
758
|
+
responseShape: {}
|
|
759
|
+
});
|
|
760
|
+
}
|
|
761
|
+
|
|
762
|
+
// src/functions/search.ts
|
|
763
|
+
import {
|
|
764
|
+
defineFunction as defineFunction4,
|
|
765
|
+
input as input4,
|
|
766
|
+
s as s4,
|
|
767
|
+
setVar as setVar4,
|
|
768
|
+
c as c4,
|
|
769
|
+
inp as inp4,
|
|
770
|
+
ref as ref4,
|
|
771
|
+
expr as expr3,
|
|
772
|
+
or as or2,
|
|
773
|
+
obj as obj4,
|
|
774
|
+
withFilters as withFilters4,
|
|
775
|
+
fl as fl4
|
|
776
|
+
} from "@xano/sdk";
|
|
777
|
+
function searchVectorsFn(opts, chunk, embedFn) {
|
|
778
|
+
return defineFunction4({
|
|
779
|
+
name: opts.names.searchFn,
|
|
780
|
+
description: "Performs semantic similarity search over document chunks using Gemini 768-dim embeddings and pgvector.",
|
|
781
|
+
tags: opts.tags,
|
|
782
|
+
input: {
|
|
783
|
+
query: input4.text({
|
|
784
|
+
required: false,
|
|
785
|
+
description: "Natural language query string. If provided, embedded via Gemini."
|
|
786
|
+
}),
|
|
787
|
+
query_media_data: input4.text({
|
|
788
|
+
required: false,
|
|
789
|
+
description: "Base64-encoded media data for visual/audio/multimodal search queries."
|
|
790
|
+
}),
|
|
791
|
+
query_mime_type: input4.text({
|
|
792
|
+
default: "text/plain",
|
|
793
|
+
description: "MIME type for query media data (e.g. image/png, audio/mp3)."
|
|
794
|
+
}),
|
|
795
|
+
query_embedding: input4.vector(GEMINI_EMBEDDING_DIMENSIONS, {
|
|
796
|
+
required: false,
|
|
797
|
+
description: "Pre-computed 768-dimensional query vector embedding."
|
|
798
|
+
}),
|
|
799
|
+
limit: input4.int({
|
|
800
|
+
default: opts.searchLimit,
|
|
801
|
+
description: "Maximum number of search results to return."
|
|
802
|
+
}),
|
|
803
|
+
threshold: input4.decimal({
|
|
804
|
+
default: opts.searchThreshold,
|
|
805
|
+
description: "Minimum cosine similarity score threshold (0.0 to 1.0)."
|
|
806
|
+
}),
|
|
807
|
+
api_key: input4.text({
|
|
808
|
+
required: false,
|
|
809
|
+
description: "Optional Google API key override."
|
|
810
|
+
})
|
|
811
|
+
},
|
|
812
|
+
stack: [
|
|
813
|
+
setVar4("target_embedding", inp4("query_embedding")),
|
|
814
|
+
s4.conditional({
|
|
815
|
+
when: expr3(ref4("target_embedding"), "=", c4.null()),
|
|
816
|
+
then: [
|
|
817
|
+
s4.precondition({
|
|
818
|
+
expr: or2(
|
|
819
|
+
expr3(inp4("query"), "!=", c4.text("")),
|
|
820
|
+
expr3(inp4("query_media_data"), "!=", c4.text(""))
|
|
821
|
+
),
|
|
822
|
+
error_type: "badrequest",
|
|
823
|
+
error: c4.text("search_vectors: either `query` text, `query_media_data`, or `query_embedding` vector must be provided.")
|
|
824
|
+
}),
|
|
825
|
+
s4.function.run({
|
|
826
|
+
fn: embedFn,
|
|
827
|
+
input: {
|
|
828
|
+
text: inp4("query"),
|
|
829
|
+
media_data: inp4("query_media_data"),
|
|
830
|
+
mime_type: inp4("query_mime_type"),
|
|
831
|
+
task_type: c4.text(opts.taskTypeQuery),
|
|
832
|
+
api_key: inp4("api_key")
|
|
833
|
+
},
|
|
834
|
+
as: "embed_query_res"
|
|
835
|
+
}),
|
|
836
|
+
setVar4("target_embedding", ref4("embed_query_res.embedding"))
|
|
837
|
+
]
|
|
838
|
+
}),
|
|
839
|
+
s4.db.query({
|
|
840
|
+
table: chunk,
|
|
841
|
+
eval: [
|
|
842
|
+
{
|
|
843
|
+
name: "embedding",
|
|
844
|
+
as: "distance",
|
|
845
|
+
filters: [
|
|
846
|
+
{
|
|
847
|
+
name: "vector_cos_distance",
|
|
848
|
+
arg: [ref4("target_embedding")]
|
|
849
|
+
}
|
|
850
|
+
]
|
|
851
|
+
}
|
|
852
|
+
],
|
|
853
|
+
sort: [{ sortBy: "distance", dir: "asc" }],
|
|
854
|
+
paging: { per_page: inp4("limit") },
|
|
855
|
+
as: "hits"
|
|
856
|
+
})
|
|
857
|
+
],
|
|
858
|
+
response: obj4({
|
|
859
|
+
results: ref4("hits.items"),
|
|
860
|
+
count: withFilters4(ref4("hits.items"), [fl4.count()])
|
|
861
|
+
}),
|
|
862
|
+
responseShape: {}
|
|
863
|
+
});
|
|
864
|
+
}
|
|
865
|
+
|
|
866
|
+
// src/tool/vector-search-tool.ts
|
|
867
|
+
import { tool, input as input5, s as s5, inp as inp5, ref as ref5 } from "@xano/sdk";
|
|
868
|
+
function vectorSearchTool(opts, searchFn) {
|
|
869
|
+
const citationGuide = opts.citationFormat === "markdown" ? " When citing or referencing any retrieved documents, images, audio, or files, format links to the document as `[Document Title](#doc-<document_id>)`." : opts.citationFormat === "numeric" ? " When citing or referencing any retrieved documents, cite the document as `(Document ID <document_id>)`." : "";
|
|
870
|
+
return tool({
|
|
871
|
+
name: opts.names.searchTool,
|
|
872
|
+
description: "Search the indexed document and media knowledge base using semantic vector similarity with Gemini multimodal embeddings.",
|
|
873
|
+
instructions: "Use this tool to search and retrieve facts, documentation, notes, specifications, images, and media files from the user's knowledge base to answer questions." + citationGuide,
|
|
874
|
+
tags: opts.tags,
|
|
875
|
+
input: {
|
|
876
|
+
query: input5.text({
|
|
877
|
+
required: false,
|
|
878
|
+
description: "Natural language search query to locate relevant context or media."
|
|
879
|
+
}),
|
|
880
|
+
media_data: input5.text({
|
|
881
|
+
required: false,
|
|
882
|
+
description: "Optional base64 media data for visual/audio search queries."
|
|
883
|
+
}),
|
|
884
|
+
mime_type: input5.text({
|
|
885
|
+
default: "text/plain",
|
|
886
|
+
description: "MIME type for media data if provided (e.g. image/png, audio/mp3)."
|
|
887
|
+
}),
|
|
888
|
+
limit: input5.int({
|
|
889
|
+
default: 5,
|
|
890
|
+
description: "Maximum number of relevant chunks to retrieve (default: 5)."
|
|
891
|
+
})
|
|
892
|
+
},
|
|
893
|
+
stack: [
|
|
894
|
+
s5.function.run({
|
|
895
|
+
fn: searchFn,
|
|
896
|
+
input: {
|
|
897
|
+
query: inp5("query"),
|
|
898
|
+
query_media_data: inp5("media_data"),
|
|
899
|
+
query_mime_type: inp5("mime_type"),
|
|
900
|
+
limit: inp5("limit")
|
|
901
|
+
},
|
|
902
|
+
as: "search_res"
|
|
903
|
+
})
|
|
904
|
+
],
|
|
905
|
+
response: ref5("search_res.results")
|
|
906
|
+
});
|
|
907
|
+
}
|
|
908
|
+
|
|
909
|
+
// src/api/group.ts
|
|
910
|
+
import { apiGroup } from "@xano/sdk";
|
|
911
|
+
function vectorGroup(opts) {
|
|
912
|
+
return apiGroup({
|
|
913
|
+
name: opts.names.apiGroup,
|
|
914
|
+
canonical: opts.canonical,
|
|
915
|
+
description: "Vector embedding pipeline, document ingestion, and similarity search endpoints.",
|
|
916
|
+
tags: opts.tags
|
|
917
|
+
});
|
|
918
|
+
}
|
|
919
|
+
|
|
920
|
+
// src/api/documents.ts
|
|
921
|
+
import {
|
|
922
|
+
query,
|
|
923
|
+
input as input6,
|
|
924
|
+
s as s6,
|
|
925
|
+
setVar as setVar5,
|
|
926
|
+
c as c5,
|
|
927
|
+
inp as inp6,
|
|
928
|
+
ref as ref6,
|
|
929
|
+
auth,
|
|
930
|
+
col as col2,
|
|
931
|
+
expr as expr4,
|
|
932
|
+
or as or3,
|
|
933
|
+
obj as obj5,
|
|
934
|
+
withFilters as withFilters5,
|
|
935
|
+
fl as fl5
|
|
936
|
+
} from "@xano/sdk";
|
|
937
|
+
function authWhere(opts) {
|
|
938
|
+
if (opts.authenticated && opts.authTable) {
|
|
939
|
+
return expr4(col2("user_id"), "=", auth("id"));
|
|
940
|
+
}
|
|
941
|
+
return void 0;
|
|
942
|
+
}
|
|
943
|
+
function documentQueries(opts, group, document, chunk, ingestFn) {
|
|
944
|
+
const prefix = opts.routePrefix;
|
|
945
|
+
const authGate = opts.authenticated ? opts.authTable : false;
|
|
946
|
+
const createEndpoint = query({
|
|
947
|
+
name: `${prefix}/documents/create`,
|
|
948
|
+
verb: "POST",
|
|
949
|
+
apiGroup: group,
|
|
950
|
+
auth: authGate,
|
|
951
|
+
description: "Create a new document or media asset and immediately chunk and index it.",
|
|
952
|
+
tags: opts.tags,
|
|
953
|
+
input: {
|
|
954
|
+
title: input6.text({
|
|
955
|
+
required: true,
|
|
956
|
+
description: "Title or filename for the document."
|
|
957
|
+
}),
|
|
958
|
+
content: input6.text({
|
|
959
|
+
required: false,
|
|
960
|
+
description: "Raw text content of the document."
|
|
961
|
+
}),
|
|
962
|
+
media_data: input6.text({
|
|
963
|
+
required: false,
|
|
964
|
+
description: "Base64-encoded media data (images, audio, video)."
|
|
965
|
+
}),
|
|
966
|
+
mime_type: input6.text({
|
|
967
|
+
default: "text/plain",
|
|
968
|
+
description: "MIME type (e.g. text/markdown, image/png, audio/mp3)."
|
|
969
|
+
}),
|
|
970
|
+
metadata: input6.json({
|
|
971
|
+
required: false,
|
|
972
|
+
description: "Arbitrary structured metadata."
|
|
973
|
+
}),
|
|
974
|
+
strategy: input6.enum(CHUNK_STRATEGIES, {
|
|
975
|
+
default: opts.defaultStrategy,
|
|
976
|
+
description: "Chunking algorithm for text content."
|
|
977
|
+
}),
|
|
978
|
+
chunk_size: input6.int({
|
|
979
|
+
default: opts.defaultChunkSize,
|
|
980
|
+
description: "Target maximum chunk size in characters."
|
|
981
|
+
}),
|
|
982
|
+
chunk_overlap: input6.int({
|
|
983
|
+
default: opts.defaultChunkOverlap,
|
|
984
|
+
description: "Chunk character overlap."
|
|
985
|
+
})
|
|
986
|
+
},
|
|
987
|
+
stack: [
|
|
988
|
+
s6.precondition({
|
|
989
|
+
expr: or3(
|
|
990
|
+
expr4(inp6("content"), "!=", c5.text("")),
|
|
991
|
+
expr4(inp6("media_data"), "!=", c5.text(""))
|
|
992
|
+
),
|
|
993
|
+
error_type: "badrequest",
|
|
994
|
+
error: c5.text("create_document: either `content` text or `media_data` must be provided.")
|
|
995
|
+
}),
|
|
996
|
+
s6.db.add({
|
|
997
|
+
table: document,
|
|
998
|
+
row: {
|
|
999
|
+
title: inp6("title"),
|
|
1000
|
+
content: inp6("content"),
|
|
1001
|
+
media_data: inp6("media_data"),
|
|
1002
|
+
mime_type: inp6("mime_type"),
|
|
1003
|
+
metadata: inp6("metadata"),
|
|
1004
|
+
status: c5.text("pending"),
|
|
1005
|
+
chunk_count: c5.int(0),
|
|
1006
|
+
strategy: inp6("strategy"),
|
|
1007
|
+
chunk_size: inp6("chunk_size"),
|
|
1008
|
+
chunk_overlap: inp6("chunk_overlap"),
|
|
1009
|
+
...opts.authenticated && opts.authTable ? { user_id: auth("id") } : {}
|
|
1010
|
+
},
|
|
1011
|
+
as: "created_doc"
|
|
1012
|
+
}),
|
|
1013
|
+
s6.function.run({
|
|
1014
|
+
fn: ingestFn,
|
|
1015
|
+
input: {
|
|
1016
|
+
document_id: ref6("created_doc.id")
|
|
1017
|
+
},
|
|
1018
|
+
as: "ingest_res"
|
|
1019
|
+
}),
|
|
1020
|
+
s6.db.get({
|
|
1021
|
+
table: document,
|
|
1022
|
+
fieldValue: ref6("created_doc.id"),
|
|
1023
|
+
as: "final_doc"
|
|
1024
|
+
})
|
|
1025
|
+
],
|
|
1026
|
+
response: obj5({
|
|
1027
|
+
document: ref6("final_doc"),
|
|
1028
|
+
chunk_count: ref6("ingest_res.chunk_count"),
|
|
1029
|
+
status: ref6("ingest_res.status")
|
|
1030
|
+
}),
|
|
1031
|
+
responseShape: {}
|
|
1032
|
+
});
|
|
1033
|
+
const listEndpoint = query({
|
|
1034
|
+
name: `${prefix}/documents`,
|
|
1035
|
+
verb: "GET",
|
|
1036
|
+
apiGroup: group,
|
|
1037
|
+
auth: authGate,
|
|
1038
|
+
description: "List uploaded documents and media assets with pagination.",
|
|
1039
|
+
tags: opts.tags,
|
|
1040
|
+
input: {
|
|
1041
|
+
page: input6.int({
|
|
1042
|
+
default: 1,
|
|
1043
|
+
description: "Page number (1-based)."
|
|
1044
|
+
}),
|
|
1045
|
+
per_page: input6.int({
|
|
1046
|
+
default: 20,
|
|
1047
|
+
description: "Number of documents per page."
|
|
1048
|
+
})
|
|
1049
|
+
},
|
|
1050
|
+
stack: [
|
|
1051
|
+
s6.db.query({
|
|
1052
|
+
table: document,
|
|
1053
|
+
where: authWhere(opts),
|
|
1054
|
+
sort: [{ sortBy: "created_at", dir: "desc" }],
|
|
1055
|
+
paging: {
|
|
1056
|
+
page: inp6("page"),
|
|
1057
|
+
per_page: inp6("per_page")
|
|
1058
|
+
},
|
|
1059
|
+
as: "docs"
|
|
1060
|
+
})
|
|
1061
|
+
],
|
|
1062
|
+
response: ref6("docs"),
|
|
1063
|
+
responseShape: {}
|
|
1064
|
+
});
|
|
1065
|
+
const getEndpoint = query({
|
|
1066
|
+
name: `${prefix}/documents/{id}`,
|
|
1067
|
+
verb: "GET",
|
|
1068
|
+
apiGroup: group,
|
|
1069
|
+
auth: authGate,
|
|
1070
|
+
description: "Get a specific document record along with all its vector chunks.",
|
|
1071
|
+
tags: opts.tags,
|
|
1072
|
+
input: {
|
|
1073
|
+
id: input6.int({
|
|
1074
|
+
required: true,
|
|
1075
|
+
description: "Document ID."
|
|
1076
|
+
})
|
|
1077
|
+
},
|
|
1078
|
+
stack: [
|
|
1079
|
+
s6.db.get({
|
|
1080
|
+
table: document,
|
|
1081
|
+
fieldValue: inp6("id"),
|
|
1082
|
+
as: "doc"
|
|
1083
|
+
}),
|
|
1084
|
+
s6.precondition({
|
|
1085
|
+
expr: expr4(ref6("doc"), "!=", c5.null()),
|
|
1086
|
+
error_type: "notfound",
|
|
1087
|
+
error: c5.text("Document not found.")
|
|
1088
|
+
}),
|
|
1089
|
+
...opts.authenticated && opts.authTable ? [
|
|
1090
|
+
s6.precondition({
|
|
1091
|
+
expr: expr4(ref6("doc.user_id", { safe: true }), "=", auth("id")),
|
|
1092
|
+
error_type: "notfound",
|
|
1093
|
+
error: c5.text("Document not found.")
|
|
1094
|
+
})
|
|
1095
|
+
] : [],
|
|
1096
|
+
s6.db.query({
|
|
1097
|
+
table: chunk,
|
|
1098
|
+
where: expr4(col2("document_id"), "=", ref6("doc.id")),
|
|
1099
|
+
sort: [{ sortBy: "chunk_index", dir: "asc" }],
|
|
1100
|
+
as: "chunks"
|
|
1101
|
+
})
|
|
1102
|
+
],
|
|
1103
|
+
response: obj5({
|
|
1104
|
+
document: ref6("doc"),
|
|
1105
|
+
chunks: ref6("chunks")
|
|
1106
|
+
}),
|
|
1107
|
+
responseShape: {}
|
|
1108
|
+
});
|
|
1109
|
+
const deleteEndpoint = query({
|
|
1110
|
+
name: `${prefix}/documents/{id}/delete`,
|
|
1111
|
+
verb: "DELETE",
|
|
1112
|
+
apiGroup: group,
|
|
1113
|
+
auth: authGate,
|
|
1114
|
+
description: "Delete a document and all of its associated vector chunks.",
|
|
1115
|
+
tags: opts.tags,
|
|
1116
|
+
input: {
|
|
1117
|
+
id: input6.int({
|
|
1118
|
+
required: true,
|
|
1119
|
+
description: "Document ID to delete."
|
|
1120
|
+
})
|
|
1121
|
+
},
|
|
1122
|
+
stack: [
|
|
1123
|
+
s6.db.get({
|
|
1124
|
+
table: document,
|
|
1125
|
+
fieldValue: inp6("id"),
|
|
1126
|
+
as: "doc"
|
|
1127
|
+
}),
|
|
1128
|
+
s6.precondition({
|
|
1129
|
+
expr: expr4(ref6("doc"), "!=", c5.null()),
|
|
1130
|
+
error_type: "notfound",
|
|
1131
|
+
error: c5.text("Document not found.")
|
|
1132
|
+
}),
|
|
1133
|
+
...opts.authenticated && opts.authTable ? [
|
|
1134
|
+
s6.precondition({
|
|
1135
|
+
expr: expr4(ref6("doc.user_id", { safe: true }), "=", auth("id")),
|
|
1136
|
+
error_type: "notfound",
|
|
1137
|
+
error: c5.text("Document not found.")
|
|
1138
|
+
})
|
|
1139
|
+
] : [],
|
|
1140
|
+
s6.db.bulk.delete({
|
|
1141
|
+
table: chunk,
|
|
1142
|
+
where: expr4(col2("document_id"), "=", ref6("doc.id"))
|
|
1143
|
+
}),
|
|
1144
|
+
s6.db.del({
|
|
1145
|
+
table: document,
|
|
1146
|
+
fieldValue: ref6("doc.id")
|
|
1147
|
+
})
|
|
1148
|
+
],
|
|
1149
|
+
response: obj5({
|
|
1150
|
+
deleted: c5.bool(true),
|
|
1151
|
+
id: ref6("doc.id")
|
|
1152
|
+
}),
|
|
1153
|
+
responseShape: {}
|
|
1154
|
+
});
|
|
1155
|
+
const reindexEndpoint = query({
|
|
1156
|
+
name: `${prefix}/documents/{id}/reindex`,
|
|
1157
|
+
verb: "POST",
|
|
1158
|
+
apiGroup: group,
|
|
1159
|
+
auth: authGate,
|
|
1160
|
+
description: "Re-chunk and re-embed an existing document with updated chunking parameters.",
|
|
1161
|
+
tags: opts.tags,
|
|
1162
|
+
input: {
|
|
1163
|
+
id: input6.int({
|
|
1164
|
+
required: true,
|
|
1165
|
+
description: "Document ID to reindex."
|
|
1166
|
+
}),
|
|
1167
|
+
strategy: input6.enum(CHUNK_STRATEGIES, {
|
|
1168
|
+
required: false,
|
|
1169
|
+
description: "New chunking algorithm (fixed, paragraph, sentence, markdown, custom)."
|
|
1170
|
+
}),
|
|
1171
|
+
chunk_size: input6.int({
|
|
1172
|
+
required: false,
|
|
1173
|
+
description: "Updated target chunk size."
|
|
1174
|
+
}),
|
|
1175
|
+
chunk_overlap: input6.int({
|
|
1176
|
+
required: false,
|
|
1177
|
+
description: "Updated chunk overlap."
|
|
1178
|
+
})
|
|
1179
|
+
},
|
|
1180
|
+
stack: [
|
|
1181
|
+
s6.db.get({
|
|
1182
|
+
table: document,
|
|
1183
|
+
fieldValue: inp6("id"),
|
|
1184
|
+
as: "doc"
|
|
1185
|
+
}),
|
|
1186
|
+
s6.precondition({
|
|
1187
|
+
expr: expr4(ref6("doc"), "!=", c5.null()),
|
|
1188
|
+
error_type: "notfound",
|
|
1189
|
+
error: c5.text("Document not found.")
|
|
1190
|
+
}),
|
|
1191
|
+
...opts.authenticated && opts.authTable ? [
|
|
1192
|
+
s6.precondition({
|
|
1193
|
+
expr: expr4(ref6("doc.user_id", { safe: true }), "=", auth("id")),
|
|
1194
|
+
error_type: "notfound",
|
|
1195
|
+
error: c5.text("Document not found.")
|
|
1196
|
+
})
|
|
1197
|
+
] : [],
|
|
1198
|
+
setVar5(
|
|
1199
|
+
"new_strategy",
|
|
1200
|
+
withFilters5(inp6("strategy"), [fl5.first_notempty(ref6("doc.strategy"))])
|
|
1201
|
+
),
|
|
1202
|
+
setVar5(
|
|
1203
|
+
"new_size",
|
|
1204
|
+
withFilters5(inp6("chunk_size"), [fl5.first_notempty(ref6("doc.chunk_size"))])
|
|
1205
|
+
),
|
|
1206
|
+
setVar5(
|
|
1207
|
+
"new_overlap",
|
|
1208
|
+
withFilters5(inp6("chunk_overlap"), [fl5.first_notempty(ref6("doc.chunk_overlap"))])
|
|
1209
|
+
),
|
|
1210
|
+
s6.db.edit({
|
|
1211
|
+
table: document,
|
|
1212
|
+
fieldValue: ref6("doc.id"),
|
|
1213
|
+
row: {
|
|
1214
|
+
strategy: ref6("new_strategy"),
|
|
1215
|
+
chunk_size: ref6("new_size"),
|
|
1216
|
+
chunk_overlap: ref6("new_overlap")
|
|
1217
|
+
}
|
|
1218
|
+
}),
|
|
1219
|
+
s6.function.run({
|
|
1220
|
+
fn: ingestFn,
|
|
1221
|
+
input: {
|
|
1222
|
+
document_id: ref6("doc.id")
|
|
1223
|
+
},
|
|
1224
|
+
as: "ingest_res"
|
|
1225
|
+
})
|
|
1226
|
+
],
|
|
1227
|
+
response: ref6("ingest_res"),
|
|
1228
|
+
responseShape: {}
|
|
1229
|
+
});
|
|
1230
|
+
return {
|
|
1231
|
+
create: createEndpoint,
|
|
1232
|
+
createDocument: createEndpoint,
|
|
1233
|
+
list: listEndpoint,
|
|
1234
|
+
listDocuments: listEndpoint,
|
|
1235
|
+
get: getEndpoint,
|
|
1236
|
+
getDocument: getEndpoint,
|
|
1237
|
+
delete: deleteEndpoint,
|
|
1238
|
+
deleteDocument: deleteEndpoint,
|
|
1239
|
+
reindex: reindexEndpoint,
|
|
1240
|
+
reindexDocument: reindexEndpoint,
|
|
1241
|
+
all: [
|
|
1242
|
+
createEndpoint,
|
|
1243
|
+
listEndpoint,
|
|
1244
|
+
getEndpoint,
|
|
1245
|
+
deleteEndpoint,
|
|
1246
|
+
reindexEndpoint
|
|
1247
|
+
]
|
|
1248
|
+
};
|
|
1249
|
+
}
|
|
1250
|
+
|
|
1251
|
+
// src/api/search.ts
|
|
1252
|
+
import { query as query2, input as input7, s as s7, inp as inp7, ref as ref7, obj as obj6 } from "@xano/sdk";
|
|
1253
|
+
function searchQueries(opts, group, searchFn, embedFn) {
|
|
1254
|
+
const prefix = opts.routePrefix;
|
|
1255
|
+
const authGate = opts.authenticated ? opts.authTable : false;
|
|
1256
|
+
const searchEndpoint = query2({
|
|
1257
|
+
name: `${prefix}/search`,
|
|
1258
|
+
verb: "POST",
|
|
1259
|
+
apiGroup: group,
|
|
1260
|
+
auth: authGate,
|
|
1261
|
+
description: "Search document chunks and media using semantic similarity via 768-dim Gemini vector embeddings.",
|
|
1262
|
+
tags: opts.tags,
|
|
1263
|
+
input: {
|
|
1264
|
+
query: input7.text({
|
|
1265
|
+
required: false,
|
|
1266
|
+
description: "Natural language query string."
|
|
1267
|
+
}),
|
|
1268
|
+
query_media_data: input7.text({
|
|
1269
|
+
required: false,
|
|
1270
|
+
description: "Base64-encoded media data for visual/audio search queries."
|
|
1271
|
+
}),
|
|
1272
|
+
query_mime_type: input7.text({
|
|
1273
|
+
default: "text/plain",
|
|
1274
|
+
description: "MIME type for media data if provided."
|
|
1275
|
+
}),
|
|
1276
|
+
query_embedding: input7.vector(GEMINI_EMBEDDING_DIMENSIONS, {
|
|
1277
|
+
required: false,
|
|
1278
|
+
description: "Pre-computed 768-dimensional query vector."
|
|
1279
|
+
}),
|
|
1280
|
+
limit: input7.int({
|
|
1281
|
+
default: opts.searchLimit,
|
|
1282
|
+
description: "Maximum number of search results to return."
|
|
1283
|
+
}),
|
|
1284
|
+
threshold: input7.decimal({
|
|
1285
|
+
default: opts.searchThreshold,
|
|
1286
|
+
description: "Minimum similarity threshold (0.0 to 1.0)."
|
|
1287
|
+
})
|
|
1288
|
+
},
|
|
1289
|
+
stack: [
|
|
1290
|
+
s7.function.run({
|
|
1291
|
+
fn: searchFn,
|
|
1292
|
+
input: {
|
|
1293
|
+
query: inp7("query"),
|
|
1294
|
+
query_media_data: inp7("query_media_data"),
|
|
1295
|
+
query_mime_type: inp7("query_mime_type"),
|
|
1296
|
+
query_embedding: inp7("query_embedding"),
|
|
1297
|
+
limit: inp7("limit"),
|
|
1298
|
+
threshold: inp7("threshold")
|
|
1299
|
+
},
|
|
1300
|
+
as: "search_res"
|
|
1301
|
+
})
|
|
1302
|
+
],
|
|
1303
|
+
response: obj6({
|
|
1304
|
+
results: ref7("search_res.results"),
|
|
1305
|
+
count: ref7("search_res.count")
|
|
1306
|
+
}),
|
|
1307
|
+
responseShape: {}
|
|
1308
|
+
});
|
|
1309
|
+
const embedEndpoint = query2({
|
|
1310
|
+
name: `${prefix}/embed`,
|
|
1311
|
+
verb: "POST",
|
|
1312
|
+
apiGroup: group,
|
|
1313
|
+
auth: authGate,
|
|
1314
|
+
description: "Directly generates a 768-dimensional vector embedding for text or media using Google Gemini API.",
|
|
1315
|
+
tags: opts.tags,
|
|
1316
|
+
input: {
|
|
1317
|
+
text: input7.text({
|
|
1318
|
+
required: false,
|
|
1319
|
+
description: "Text content to generate an embedding vector for."
|
|
1320
|
+
}),
|
|
1321
|
+
media_data: input7.text({
|
|
1322
|
+
required: false,
|
|
1323
|
+
description: "Base64 media data (image, audio, video)."
|
|
1324
|
+
}),
|
|
1325
|
+
mime_type: input7.text({
|
|
1326
|
+
default: "text/plain",
|
|
1327
|
+
description: "MIME type for media data if provided."
|
|
1328
|
+
}),
|
|
1329
|
+
model: input7.text({
|
|
1330
|
+
default: opts.model,
|
|
1331
|
+
description: "Google Gemini embedding model name."
|
|
1332
|
+
})
|
|
1333
|
+
},
|
|
1334
|
+
stack: [
|
|
1335
|
+
s7.function.run({
|
|
1336
|
+
fn: embedFn,
|
|
1337
|
+
input: {
|
|
1338
|
+
text: inp7("text"),
|
|
1339
|
+
media_data: inp7("media_data"),
|
|
1340
|
+
mime_type: inp7("mime_type"),
|
|
1341
|
+
model: inp7("model")
|
|
1342
|
+
},
|
|
1343
|
+
as: "embed_res"
|
|
1344
|
+
})
|
|
1345
|
+
],
|
|
1346
|
+
response: ref7("embed_res"),
|
|
1347
|
+
responseShape: {}
|
|
1348
|
+
});
|
|
1349
|
+
return {
|
|
1350
|
+
search: searchEndpoint,
|
|
1351
|
+
embed: embedEndpoint,
|
|
1352
|
+
all: [searchEndpoint, embedEndpoint]
|
|
1353
|
+
};
|
|
1354
|
+
}
|
|
1355
|
+
|
|
1356
|
+
// src/register.ts
|
|
1357
|
+
var installed = /* @__PURE__ */ new WeakSet();
|
|
1358
|
+
function createVector(opts = {}) {
|
|
1359
|
+
const options = resolveOptions(opts);
|
|
1360
|
+
const document = documentTable(options);
|
|
1361
|
+
const chunk = chunkTable(options, document);
|
|
1362
|
+
const embedFn = generateEmbeddingFn(options);
|
|
1363
|
+
const chunkFn = chunkTextFn(options);
|
|
1364
|
+
const ingestFn = ingestDocumentFn(options, document, chunk, embedFn, chunkFn);
|
|
1365
|
+
const searchFn = searchVectorsFn(options, chunk, embedFn);
|
|
1366
|
+
const searchTool = vectorSearchTool(options, searchFn);
|
|
1367
|
+
const group = vectorGroup(options);
|
|
1368
|
+
const documents = documentQueries(options, group, document, chunk, ingestFn);
|
|
1369
|
+
const search = searchQueries(options, group, searchFn, embedFn);
|
|
1370
|
+
return {
|
|
1371
|
+
options,
|
|
1372
|
+
document,
|
|
1373
|
+
chunk,
|
|
1374
|
+
embedFn,
|
|
1375
|
+
chunkFn,
|
|
1376
|
+
ingestFn,
|
|
1377
|
+
searchFn,
|
|
1378
|
+
searchTool,
|
|
1379
|
+
group,
|
|
1380
|
+
documents,
|
|
1381
|
+
search,
|
|
1382
|
+
queries: [...documents.all, ...search.all]
|
|
1383
|
+
};
|
|
1384
|
+
}
|
|
1385
|
+
function registerVector(xano, opts = {}) {
|
|
1386
|
+
if (installed.has(xano)) {
|
|
1387
|
+
throw new Error(
|
|
1388
|
+
'registerVector: already called on this Xano instance. Register the vector set once \u2014 a second registration duplicates every def, which the SDK cannot catch (the two sets are distinct objects sharing names) and which surfaces at export() as "Duplicate object guid \u2026 shared by \\"dbo/vector_document\\" and \\"dbo/vector_document\\"". To run TWO vector pipelines in one workspace, build the second with `createVector({ canonical, names: { searchTool } })` and register its defs yourself.'
|
|
1389
|
+
);
|
|
1390
|
+
}
|
|
1391
|
+
const vec = createVector(opts);
|
|
1392
|
+
xano.registerTables([vec.document, vec.chunk]).registerFunctions([
|
|
1393
|
+
vec.embedFn,
|
|
1394
|
+
vec.chunkFn,
|
|
1395
|
+
vec.ingestFn,
|
|
1396
|
+
vec.searchFn
|
|
1397
|
+
]).registerTools([vec.searchTool]).registerApiGroups([vec.group]).registerQueries(vec.queries);
|
|
1398
|
+
installed.add(xano);
|
|
1399
|
+
return { ...vec, xano };
|
|
1400
|
+
}
|
|
1401
|
+
export {
|
|
1402
|
+
CHUNK_STRATEGIES,
|
|
1403
|
+
DEFAULT_API_KEY_ENV,
|
|
1404
|
+
DEFAULT_CHUNK_OVERLAP,
|
|
1405
|
+
DEFAULT_CHUNK_SIZE,
|
|
1406
|
+
DEFAULT_GEMINI_MODEL,
|
|
1407
|
+
DEFAULT_NAMES,
|
|
1408
|
+
DEFAULT_SEARCH_LIMIT,
|
|
1409
|
+
DEFAULT_SEARCH_THRESHOLD,
|
|
1410
|
+
DEFAULT_STRATEGY,
|
|
1411
|
+
DOCUMENT_STATUSES,
|
|
1412
|
+
GEMINI_EMBEDDING_DIMENSIONS,
|
|
1413
|
+
PUBLIC_CHUNK_FIELDS,
|
|
1414
|
+
PUBLIC_DOCUMENT_FIELDS,
|
|
1415
|
+
chunkTable,
|
|
1416
|
+
chunkTextFn,
|
|
1417
|
+
createVector,
|
|
1418
|
+
documentQueries,
|
|
1419
|
+
documentTable,
|
|
1420
|
+
generateEmbeddingFn,
|
|
1421
|
+
ingestDocumentFn,
|
|
1422
|
+
registerVector,
|
|
1423
|
+
resolveOptions,
|
|
1424
|
+
searchQueries,
|
|
1425
|
+
searchVectorsFn,
|
|
1426
|
+
vectorGroup,
|
|
1427
|
+
vectorSearchTool
|
|
1428
|
+
};
|
|
1429
|
+
//# sourceMappingURL=index.js.map
|