@lunora/ai 1.0.0-alpha.7 → 1.0.0-alpha.71
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +3 -1
- package/dist/index.d.mts +88 -27
- package/dist/index.d.ts +88 -27
- package/dist/index.mjs +1 -3
- package/dist/packem_shared/AI_DEFAULT_EMBEDDING_MODEL_ENV-mi9Aaq1z.mjs +1 -0
- package/dist/packem_shared/DEFAULT_MODEL_PRICES-Q8uxdiuV.mjs +1 -0
- package/dist/packem_shared/VECTORIZE_CAPABILITIES-CUQDoxis.mjs +1 -0
- package/dist/packem_shared/batchReranker-Bc38FBLH.mjs +1 -0
- package/dist/packem_shared/bm25-9q0Avwi-.mjs +1 -0
- package/dist/packem_shared/bm25LexicalStore-DMUzAL0O.mjs +1 -0
- package/dist/packem_shared/concurrent-C6nqBv41.mjs +1 -0
- package/dist/packem_shared/contentHash-BIn6ECP8.mjs +1 -0
- package/dist/packem_shared/createAi-CwY7eL7P.mjs +1 -0
- package/dist/packem_shared/defineRag-wBDjkuHP.mjs +8 -0
- package/dist/packem_shared/defineRagSource-Q3f3niU8.mjs +1 -0
- package/dist/packem_shared/fixedWindowChunks-C461ahRE.mjs +1 -0
- package/dist/packem_shared/hybridRank-DejmVw2I.mjs +1 -0
- package/dist/packem_shared/markdownChunker-Bcv56GEz.mjs +5 -0
- package/dist/packem_shared/matchesMetadataFilter-BbIOyA5g.mjs +1 -0
- package/dist/packem_shared/ragSyncTriggers-DPqzBNFw.mjs +1 -0
- package/dist/packem_shared/sql-D5aqEMCY.mjs +1 -0
- package/dist/packem_shared/sqlLexicalStore-4C_cIwef.mjs +1 -0
- package/dist/packem_shared/sqliteVectorStore-D32l9lP0.mjs +1 -0
- package/dist/packem_shared/types.d-eaM7juQg.d.mts +271 -0
- package/dist/packem_shared/types.d-eaM7juQg.d.ts +271 -0
- package/dist/rag/index.d.mts +1200 -0
- package/dist/rag/index.d.ts +1200 -0
- package/dist/rag/index.mjs +1 -0
- package/package.json +12 -6
- package/dist/packem_shared/createAi-Bq_4LMcp.mjs +0 -54
|
@@ -0,0 +1,1200 @@
|
|
|
1
|
+
import { Tool } from 'ai';
|
|
2
|
+
import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-eaM7juQg.mjs";
|
|
3
|
+
/**
|
|
4
|
+
* Built-in fixed-window chunker: split into `size`-char windows overlapping by
|
|
5
|
+
* `overlap` chars. Deliberately simple and deterministic — the zero-config
|
|
6
|
+
* default. Token-aware / sentence / semantic strategies plug in via
|
|
7
|
+
* `RagConfig.chunk`.
|
|
8
|
+
* @experimental
|
|
9
|
+
*/
|
|
10
|
+
declare const fixedWindowChunks: (text: string, size: number, overlap: number) => ReadonlyArray<string>;
|
|
11
|
+
/**
|
|
12
|
+
* Options shared by the character-budgeted chunkers. `overlap` is a **floor**
|
|
13
|
+
* in characters: enough trailing atoms are carried into the next window to
|
|
14
|
+
* cover it, so the realised overlap lands on an atom boundary and is usually a
|
|
15
|
+
* little larger.
|
|
16
|
+
*/
|
|
17
|
+
interface ChunkerOptions {
|
|
18
|
+
/** Overlap floor between adjacent chunks, in characters. Default 200. Must be `< size`. */
|
|
19
|
+
overlap?: number;
|
|
20
|
+
/** Maximum chunk size, in characters. Default 1000. */
|
|
21
|
+
size?: number;
|
|
22
|
+
}
|
|
23
|
+
/**
|
|
24
|
+
* Options for {@link tokenChunker}.
|
|
25
|
+
*
|
|
26
|
+
* `countTokens` is **required and injected**: a real token count needs the
|
|
27
|
+
* model's tokenizer, and `@lunora/ai` will not add one as a dependency or
|
|
28
|
+
* pretend a characters-per-token constant is a token count — a "token-aware"
|
|
29
|
+
* chunker built on an estimate is a character chunker with extra steps, and it
|
|
30
|
+
* silently overshoots on code and non-Latin scripts, which is exactly where the
|
|
31
|
+
* budget matters. Pass `js-tiktoken`, `gpt-tokenizer`, or your provider's
|
|
32
|
+
* counter.
|
|
33
|
+
*/
|
|
34
|
+
interface TokenChunkerOptions {
|
|
35
|
+
/** Token counter for the target model — e.g. `(text) => encoding.encode(text).length`. */
|
|
36
|
+
countTokens: (text: string) => number;
|
|
37
|
+
/** Maximum chunk size, in tokens. Default 256. */
|
|
38
|
+
maxTokens?: number;
|
|
39
|
+
/** Overlap floor between adjacent chunks, in tokens. Default 0. Must be `< maxTokens`. */
|
|
40
|
+
overlapTokens?: number;
|
|
41
|
+
}
|
|
42
|
+
/**
|
|
43
|
+
* Split on sentence boundaries, then greedily pack whole sentences into
|
|
44
|
+
* `size`-bounded chunks. Prefer this over the fixed window for prose: chunks
|
|
45
|
+
* start and end where the author ended a thought, which is what the embedding
|
|
46
|
+
* model was trained on.
|
|
47
|
+
* @experimental
|
|
48
|
+
*/
|
|
49
|
+
declare const sentenceChunker: (options?: ChunkerOptions) => ((text: string) => ReadonlyArray<string>);
|
|
50
|
+
/**
|
|
51
|
+
* Split a Markdown document at its ATX headings, then pack each section's
|
|
52
|
+
* sentences into `size`-bounded chunks.
|
|
53
|
+
*
|
|
54
|
+
* Every emitted chunk is prefixed with its **heading trail** (`# Guide > ##
|
|
55
|
+
* Auth > ### OAuth`), so a chunk taken from deep inside a long document still
|
|
56
|
+
* carries the context that says what it is about. That prefix is what makes a
|
|
57
|
+
* mid-document chunk retrievable by a query naming its section rather than its
|
|
58
|
+
* prose, and it is why this beats {@link sentenceChunker} on structured docs.
|
|
59
|
+
* @experimental
|
|
60
|
+
*/
|
|
61
|
+
declare const markdownChunker: (options?: ChunkerOptions) => ((text: string) => ReadonlyArray<string>);
|
|
62
|
+
/**
|
|
63
|
+
* Pack sentences into chunks bounded by a real **token** count rather than a
|
|
64
|
+
* character count. Use this when the embedding model's context window is the
|
|
65
|
+
* binding constraint — a 512-token model silently truncates anything longer, so
|
|
66
|
+
* the tail of an over-long chunk is embedded as if it were never written.
|
|
67
|
+
* @experimental
|
|
68
|
+
*/
|
|
69
|
+
declare const tokenChunker: (options: TokenChunkerOptions) => ((text: string) => ReadonlyArray<string>);
|
|
70
|
+
/**
|
|
71
|
+
* What a store can hold and return. Every field is a hard limit `defineRag`
|
|
72
|
+
* enforces locally, so a breach fails here — naming the store and the limit —
|
|
73
|
+
* rather than at the backend with nothing saying why.
|
|
74
|
+
*/
|
|
75
|
+
interface RagVectorStoreCapabilities {
|
|
76
|
+
/**
|
|
77
|
+
* Ceiling on embedding dimensionality, or `false` for no limit.
|
|
78
|
+
*
|
|
79
|
+
* Vectorize stores at most 1536 at 32-bit precision, which rules out
|
|
80
|
+
* `text-embedding-3-large` (3072) and Qwen3-Embedding (4096). A store with
|
|
81
|
+
* no such limit declares `false` and those models just work.
|
|
82
|
+
*/
|
|
83
|
+
maxDimensions: number | false;
|
|
84
|
+
/**
|
|
85
|
+
* Ceiling on a vector's id, in bytes, or `false` for no limit.
|
|
86
|
+
*
|
|
87
|
+
* Vectorize allows 64, and chunk ids are derived from the namespace and the
|
|
88
|
+
* caller's source id — a bucket key like
|
|
89
|
+
* `handbook/engineering/onboarding/day-one.md` under a uuid namespace is
|
|
90
|
+
* already past it. Without this the upsert is rejected at the far side with
|
|
91
|
+
* nothing naming the cause, the same failure {@link RagVectorStoreCapabilities.maxMetadataBytes}
|
|
92
|
+
* exists to replace.
|
|
93
|
+
*/
|
|
94
|
+
maxIdBytes: number | false;
|
|
95
|
+
/**
|
|
96
|
+
* Ceiling on the serialized metadata object per vector, in bytes, or
|
|
97
|
+
* `false` for no limit. Covers the whole object — chunk text included when
|
|
98
|
+
* no `textStore` moves it out.
|
|
99
|
+
*/
|
|
100
|
+
maxMetadataBytes: number | false;
|
|
101
|
+
/** Ceiling on `topK` when only indexed metadata is requested (text-store mode). */
|
|
102
|
+
maxTopK: number;
|
|
103
|
+
/** Ceiling on `topK` when the query asks for full metadata (the default mode). */
|
|
104
|
+
maxTopKWithMetadata: number;
|
|
105
|
+
}
|
|
106
|
+
/**
|
|
107
|
+
* The storage operations `defineRag` needs. Deliberately the same four
|
|
108
|
+
* operations `RagVectors` already exposes — this is a capability-carrying
|
|
109
|
+
* wrapper, not a new protocol, so adapting an existing implementation is a
|
|
110
|
+
* one-liner.
|
|
111
|
+
*/
|
|
112
|
+
interface RagVectorStore {
|
|
113
|
+
capabilities: RagVectorStoreCapabilities;
|
|
114
|
+
deleteByIds: (ids: ReadonlyArray<string>, namespace?: string) => Promise<unknown>;
|
|
115
|
+
getByIds: (ids: ReadonlyArray<string>, namespace?: string) => Promise<ReadonlyArray<RagVectorRecord>>;
|
|
116
|
+
query: (input: RagVectorQueryInput) => Promise<RagVectorMatches>;
|
|
117
|
+
upsert: (input: RagVectorUpsertInput) => Promise<unknown>;
|
|
118
|
+
}
|
|
119
|
+
/**
|
|
120
|
+
* Vectorize's documented limits.
|
|
121
|
+
*
|
|
122
|
+
* `maxTopKWithMetadata` is 50, matching Vectorize V2. **Legacy V1 indexes cap
|
|
123
|
+
* at 20** and reject a larger `topK` remotely; a binding handle does not expose
|
|
124
|
+
* its index version, so this cannot branch on it.
|
|
125
|
+
*/
|
|
126
|
+
declare const VECTORIZE_CAPABILITIES: RagVectorStoreCapabilities;
|
|
127
|
+
/**
|
|
128
|
+
* Wrap a `ctx.vectors` facade (or any {@link RagVectors}) as a store declaring
|
|
129
|
+
* Vectorize's limits. This is the default when no `store` is configured, so the
|
|
130
|
+
* behaviour of an existing `defineRag` is unchanged.
|
|
131
|
+
* @experimental
|
|
132
|
+
*/
|
|
133
|
+
declare const vectorizeStore: (vectors: RagVectors, indexName: string) => RagVectorStore;
|
|
134
|
+
/**
|
|
135
|
+
* `(text) => vector` — the embedder shape `ctx.vectors` accepts on both its
|
|
136
|
+
* write (`upsert`) and read (`query`) inputs. Matches `@lunora/server`'s
|
|
137
|
+
* `VectorEmbedder` and `@lunora/bindings/vectors`' `EmbedFunction<string>`.
|
|
138
|
+
* @experimental
|
|
139
|
+
*/
|
|
140
|
+
type RagEmbedder = (input: string) => Promise<ReadonlyArray<number>> | ReadonlyArray<number>;
|
|
141
|
+
/**
|
|
142
|
+
* `RagVectorMatch` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
143
|
+
* @experimental
|
|
144
|
+
*/
|
|
145
|
+
interface RagVectorMatch {
|
|
146
|
+
id: string;
|
|
147
|
+
metadata?: Record<string, unknown>;
|
|
148
|
+
score: number;
|
|
149
|
+
}
|
|
150
|
+
/**
|
|
151
|
+
* `RagVectorMatches` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
152
|
+
* @experimental
|
|
153
|
+
*/
|
|
154
|
+
interface RagVectorMatches {
|
|
155
|
+
count: number;
|
|
156
|
+
matches: ReadonlyArray<RagVectorMatch>;
|
|
157
|
+
}
|
|
158
|
+
/**
|
|
159
|
+
* `RagVectorQueryInput` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
160
|
+
* @experimental
|
|
161
|
+
*/
|
|
162
|
+
interface RagVectorQueryInput {
|
|
163
|
+
/** Embedder used to vectorize `input`. */
|
|
164
|
+
embed?: RagEmbedder;
|
|
165
|
+
filter?: Record<string, unknown>;
|
|
166
|
+
/** Natural-language query text, embedded via `embed`. */
|
|
167
|
+
input?: string;
|
|
168
|
+
namespace?: string;
|
|
169
|
+
/**
|
|
170
|
+
* How much stored metadata to return on matches. The runtime honours it even
|
|
171
|
+
* though `@lunora/server`'s ctx type does not declare it — the helper relies
|
|
172
|
+
* on it to read chunk text back in metadata mode.
|
|
173
|
+
*/
|
|
174
|
+
returnMetadata?: "all" | "indexed" | "none";
|
|
175
|
+
topK?: number;
|
|
176
|
+
}
|
|
177
|
+
/**
|
|
178
|
+
* `RagVectorRecord` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
179
|
+
* @experimental
|
|
180
|
+
*/
|
|
181
|
+
interface RagVectorRecord {
|
|
182
|
+
id: string;
|
|
183
|
+
metadata?: Record<string, unknown>;
|
|
184
|
+
}
|
|
185
|
+
/**
|
|
186
|
+
* `RagVectorUpsertInput` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
187
|
+
* @experimental
|
|
188
|
+
*/
|
|
189
|
+
interface RagVectorUpsertInput {
|
|
190
|
+
/** Embedder used to vectorize `input`. Optional — omitted for text-search indexes. */
|
|
191
|
+
embed?: RagEmbedder;
|
|
192
|
+
id: string;
|
|
193
|
+
input: string;
|
|
194
|
+
metadata?: Record<string, unknown>;
|
|
195
|
+
namespace?: string;
|
|
196
|
+
}
|
|
197
|
+
/**
|
|
198
|
+
* Structural subset of the vector surface the RAG helper needs. Both the
|
|
199
|
+
* `ctx.vectors` facade on Mutation/Action ctx (`@lunora/server`'s
|
|
200
|
+
* `VectorSearch`) and the raw `@lunora/bindings/vectors` `LunoraVectors`
|
|
201
|
+
* satisfy it — declared here so `@lunora/ai` depends on neither package.
|
|
202
|
+
* @experimental
|
|
203
|
+
*/
|
|
204
|
+
interface RagVectors {
|
|
205
|
+
deleteByIds: (indexName: string, ids: ReadonlyArray<string>, namespace?: string) => Promise<unknown>;
|
|
206
|
+
getByIds: (indexName: string, ids: ReadonlyArray<string>, namespace?: string) => Promise<ReadonlyArray<RagVectorRecord>>;
|
|
207
|
+
query: (indexName: string, input: RagVectorQueryInput) => Promise<RagVectorMatches>;
|
|
208
|
+
upsert: (indexName: string, input: RagVectorUpsertInput) => Promise<unknown>;
|
|
209
|
+
}
|
|
210
|
+
/**
|
|
211
|
+
* The two facades `defineRag` binds. An `ActionCtx` satisfies this directly
|
|
212
|
+
* (`ctx.ai` is action-only, so RAG methods run inside actions); any object
|
|
213
|
+
* carrying the two facades works in tests.
|
|
214
|
+
* @experimental
|
|
215
|
+
*/
|
|
216
|
+
interface RagContext {
|
|
217
|
+
/**
|
|
218
|
+
* Resolves a Workers AI embedding-model id (or the omitted default) — an
|
|
219
|
+
* `ActionCtx`'s `ctx.ai` satisfies it. OPTIONAL: when
|
|
220
|
+
* {@link RagConfig.embeddingModel} is a direct AI SDK `EmbeddingModel` object
|
|
221
|
+
* (bring-your-own embeddings, e.g. `@ai-sdk/openai`), the helper uses that
|
|
222
|
+
* object as-is and never reads `ai`, so a hand-built context may omit it and
|
|
223
|
+
* no `env.AI` binding is needed. A model-id string (or an omitted model) with
|
|
224
|
+
* no `ai` present throws a directed error.
|
|
225
|
+
*/
|
|
226
|
+
ai?: Pick<LunoraAi, "embeddingModel">;
|
|
227
|
+
/**
|
|
228
|
+
* The verified retrieval identity, read by {@link RagConfig.rlsFilter} to
|
|
229
|
+
* derive a per-request row filter. An `ActionCtx` carrying `ctx.auth`
|
|
230
|
+
* satisfies this structurally, so `docs(ctx)` picks the identity up
|
|
231
|
+
* automatically; tests pass any value. `unknown` on purpose — `@lunora/ai`
|
|
232
|
+
* stays decoupled from `@lunora/server`'s identity type; `rlsFilter` narrows.
|
|
233
|
+
*/
|
|
234
|
+
auth?: unknown;
|
|
235
|
+
/**
|
|
236
|
+
* Optional conversation / session id. When set (and `trace` is present), each
|
|
237
|
+
* embedding-model span additionally carries `gen_ai.conversation.id`, so a
|
|
238
|
+
* RAG embed done inside a multi-turn conversation groups with that
|
|
239
|
+
* conversation's other generation spans in the trace store. Omitted → the
|
|
240
|
+
* attribute is absent (backward-compatible).
|
|
241
|
+
*/
|
|
242
|
+
conversationId?: string;
|
|
243
|
+
/**
|
|
244
|
+
* Optional `ctx.trace` span factory — an `ActionCtx`'s `ctx.trace` satisfies
|
|
245
|
+
* it structurally. When present, `defineRag` wraps each embedding-model
|
|
246
|
+
* call in a `generation` span carrying `gen_ai.operation.name: "embeddings"`
|
|
247
|
+
* and `gen_ai.request.model` up front, plus — attached post-hoc through the
|
|
248
|
+
* span handle the tracer hands the body — `gen_ai.usage.input_tokens` (from
|
|
249
|
+
* the embed result's token usage) and `gen_ai.usage.cost` (probed from the
|
|
250
|
+
* embed result's provider metadata, e.g. AI Gateway) when those are present.
|
|
251
|
+
* So the embed shows up on the trace waterfall with its usage like any other
|
|
252
|
+
* instrumented model call. `unknown` on purpose — the same decoupling
|
|
253
|
+
* rationale as `auth`: `defineRag` narrows it to a callable and runs embeds
|
|
254
|
+
* untraced when it is absent (a hand-built context / test).
|
|
255
|
+
*/
|
|
256
|
+
trace?: unknown;
|
|
257
|
+
/**
|
|
258
|
+
* The Vectorize facade backing the default store — an `ActionCtx`'s
|
|
259
|
+
* `ctx.vectors` satisfies it.
|
|
260
|
+
*
|
|
261
|
+
* OPTIONAL, because it is never read when {@link RagConfig.store} is set,
|
|
262
|
+
* and codegen only emits `ctx.vectors` for a schema that declares a vector
|
|
263
|
+
* index. Requiring it made "the store that needs no Vectorize" impossible
|
|
264
|
+
* to type-check in an app with no Vectorize index. Absent with no `store`
|
|
265
|
+
* configured throws a directed error when the RAG is bound.
|
|
266
|
+
*/
|
|
267
|
+
vectors?: RagVectors;
|
|
268
|
+
}
|
|
269
|
+
/**
|
|
270
|
+
* Pluggable chunk-text storage. By default chunk text is stored in vector
|
|
271
|
+
* metadata (`__ragText`), which forces `returnMetadata: "all"` on retrieval and
|
|
272
|
+
* caps `topK` at 50 (the Vectorize full-metadata ceiling) — and each vector's
|
|
273
|
+
* metadata must stay under the ~10 KiB Vectorize cap. Supplying a text store
|
|
274
|
+
* (a DO table, KV, …) moves the text out of metadata: retrieval queries with
|
|
275
|
+
* `returnMetadata: "indexed"` (topK up to 100) and hydrates text by chunk id.
|
|
276
|
+
* @experimental
|
|
277
|
+
*/
|
|
278
|
+
interface RagTextStore {
|
|
279
|
+
/** Fetch chunk texts by id, aligned with the input order; `undefined` for misses. */
|
|
280
|
+
getMany: (ids: ReadonlyArray<string>, options: {
|
|
281
|
+
namespace?: string;
|
|
282
|
+
}) => Promise<ReadonlyArray<string | undefined>>;
|
|
283
|
+
/** Persist chunk texts. Must be idempotent by chunk `id` (re-index re-puts). */
|
|
284
|
+
put: (chunks: ReadonlyArray<StoredRagChunk>, options: {
|
|
285
|
+
namespace?: string;
|
|
286
|
+
}) => Promise<void>;
|
|
287
|
+
/** Optional cleanup hook, invoked when a source's chunks are deleted. */
|
|
288
|
+
remove?: (ids: ReadonlyArray<string>, options: {
|
|
289
|
+
namespace?: string;
|
|
290
|
+
}) => Promise<void>;
|
|
291
|
+
}
|
|
292
|
+
/**
|
|
293
|
+
* A chunk handed to {@link RagTextStore.put} / {@link RagLexicalStore.index}.
|
|
294
|
+
* @experimental
|
|
295
|
+
*/
|
|
296
|
+
interface StoredRagChunk {
|
|
297
|
+
chunkIndex: number;
|
|
298
|
+
id: string;
|
|
299
|
+
/**
|
|
300
|
+
* The caller `metadata` attached to this chunk's source at index time
|
|
301
|
+
* (internal `__rag*` keys excluded).
|
|
302
|
+
*
|
|
303
|
+
* Present so a {@link RagLexicalStore} can evaluate the same metadata
|
|
304
|
+
* filter the vector leg gets. Without it a lexical store has nothing to
|
|
305
|
+
* filter on and must fail closed on every filtered query — which made
|
|
306
|
+
* hybrid search and metadata-based RLS mutually exclusive.
|
|
307
|
+
*/
|
|
308
|
+
metadata?: Record<string, unknown>;
|
|
309
|
+
sourceId: string;
|
|
310
|
+
text: string;
|
|
311
|
+
}
|
|
312
|
+
/**
|
|
313
|
+
* One lexical (BM25) hit returned by {@link RagLexicalStore.search}.
|
|
314
|
+
* @experimental
|
|
315
|
+
*/
|
|
316
|
+
interface LexicalMatch {
|
|
317
|
+
/** The chunk vector id — the same id scheme the vector leg uses, so RRF can fuse the two. */
|
|
318
|
+
id: string;
|
|
319
|
+
/** BM25 relevance score (higher = better). Used only for the leg's internal ranking; RRF fuses by rank. */
|
|
320
|
+
score: number;
|
|
321
|
+
/** The chunk text, returned so a fused lexical-only hit needs no extra hydration round-trip. */
|
|
322
|
+
text: string;
|
|
323
|
+
}
|
|
324
|
+
/**
|
|
325
|
+
* Pluggable lexical (BM25 / keyword) store — the production seam for hybrid
|
|
326
|
+
* retrieval. When {@link RagConfig.lexicalStore} is set, `index()` mirrors each
|
|
327
|
+
* chunk's text here and `retrieve()` fuses this store's keyword ranking with the
|
|
328
|
+
* vector store's semantic ranking via Reciprocal Rank Fusion. Mirrors the
|
|
329
|
+
* {@link RagTextStore} shape (idempotent by chunk `id`, namespace-partitioned).
|
|
330
|
+
*
|
|
331
|
+
* `@lunora/ai/rag` ships `bm25LexicalStore()`, an in-memory reference adapter;
|
|
332
|
+
* production deployments plug a durable one (DO SQLite inverted index, D1,
|
|
333
|
+
* Vectorize-adjacent search service, …) behind this same interface.
|
|
334
|
+
* @experimental
|
|
335
|
+
*/
|
|
336
|
+
interface RagLexicalStore {
|
|
337
|
+
/** Index chunk texts for keyword search. Must be idempotent by chunk `id` (re-index re-puts). */
|
|
338
|
+
index: (chunks: ReadonlyArray<StoredRagChunk>, options: {
|
|
339
|
+
namespace?: string;
|
|
340
|
+
}) => Promise<void>;
|
|
341
|
+
/** Optional cleanup hook, invoked when a source's chunks are deleted or a re-index shrinks it. */
|
|
342
|
+
remove?: (ids: ReadonlyArray<string>, options: {
|
|
343
|
+
namespace?: string;
|
|
344
|
+
}) => Promise<void>;
|
|
345
|
+
/**
|
|
346
|
+
* Rank chunks by lexical relevance to `query`. `filter` carries the same
|
|
347
|
+
* (RLS-merged) metadata predicate handed to the vector leg — a store that
|
|
348
|
+
* indexes metadata MUST honour it so hybrid retrieval can't surface a row
|
|
349
|
+
* the RLS filter would exclude. The shipped `bm25LexicalStore` does — it
|
|
350
|
+
* evaluates the predicate against each document's stored `metadata`. A store
|
|
351
|
+
* that indexes no metadata has nothing to filter on and must fail CLOSED on
|
|
352
|
+
* every filtered query rather than ignore the predicate.
|
|
353
|
+
*/
|
|
354
|
+
search: (query: string, options: {
|
|
355
|
+
filter?: Record<string, unknown>;
|
|
356
|
+
namespace?: string;
|
|
357
|
+
topK: number;
|
|
358
|
+
}) => Promise<ReadonlyArray<LexicalMatch>>;
|
|
359
|
+
}
|
|
360
|
+
/**
|
|
361
|
+
* A pre-defined, reusable filter expression. Declared on `RagConfig.filters`
|
|
362
|
+
* (keyed by name) and referenced by name from `RetrieveOptions.filter` — avoids
|
|
363
|
+
* repeating the same tenant/RBAC filter shape across every retrieval site.
|
|
364
|
+
* @example
|
|
365
|
+
* ```ts
|
|
366
|
+
* const docs = defineRag({
|
|
367
|
+
* index: "docs",
|
|
368
|
+
* filters: {
|
|
369
|
+
* published: { filter: { status: "published", deleted: false }, description: "Only published content" },
|
|
370
|
+
* },
|
|
371
|
+
* });
|
|
372
|
+
* // Later — reference by name:
|
|
373
|
+
* docs(ctx).retrieve("query", { filter: "published" });
|
|
374
|
+
* ```
|
|
375
|
+
* @experimental
|
|
376
|
+
*/
|
|
377
|
+
interface RagNamedFilter {
|
|
378
|
+
/** Optional human-readable description for observability / Studio display. */
|
|
379
|
+
description?: string;
|
|
380
|
+
/** The filter expression passed verbatim to Vectorize's `filter` parameter. */
|
|
381
|
+
filter: Record<string, unknown>;
|
|
382
|
+
}
|
|
383
|
+
/**
|
|
384
|
+
* Re-score retrieved candidates against the query. Returns the chunks in their
|
|
385
|
+
* new order; may drop chunks. See {@link RagConfig.rerank}.
|
|
386
|
+
* @experimental
|
|
387
|
+
*/
|
|
388
|
+
type RagReranker = (query: string, chunks: ReadonlyArray<RetrievedChunk>) => Promise<ReadonlyArray<RetrievedChunk>> | ReadonlyArray<RetrievedChunk>;
|
|
389
|
+
/**
|
|
390
|
+
* Rewrite a query, or expand it into several. See {@link RagConfig.transformQuery}.
|
|
391
|
+
* @experimental
|
|
392
|
+
*/
|
|
393
|
+
type RagQueryTransform = (query: string, info: {
|
|
394
|
+
conversationId?: string;
|
|
395
|
+
namespace?: string;
|
|
396
|
+
}) => Promise<ReadonlyArray<string> | string> | ReadonlyArray<string> | string;
|
|
397
|
+
/**
|
|
398
|
+
* `RagConfig` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
399
|
+
* @experimental
|
|
400
|
+
*/
|
|
401
|
+
interface RagConfig {
|
|
402
|
+
/**
|
|
403
|
+
* Suppress the one-time dev warning emitted when `index`/`retrieve` run
|
|
404
|
+
* without a `namespace`. Only appropriate for genuinely single-tenant apps —
|
|
405
|
+
* Vectorize indexes are account-global, so a namespace-less index shares
|
|
406
|
+
* vectors across every tenant.
|
|
407
|
+
*/
|
|
408
|
+
allowSharedNamespace?: boolean;
|
|
409
|
+
/**
|
|
410
|
+
* Retain up to this many embeddings per bound context, keyed by text, so a
|
|
411
|
+
* repeated `retrieve()` of the same question does not re-embed it.
|
|
412
|
+
*
|
|
413
|
+
* Default 0 (retain nothing beyond one call). Indexing always batches its
|
|
414
|
+
* embeds regardless of this setting — the batch lives in its own
|
|
415
|
+
* request-scoped map, released when `index()` returns, so it neither needs
|
|
416
|
+
* this budget nor evicts what is held in it.
|
|
417
|
+
*
|
|
418
|
+
* Sized in entries, not bytes, but budget in bytes: one 1536-dimension
|
|
419
|
+
* embedding is ~12 KB, so 100 entries is over a megabyte held in the
|
|
420
|
+
* isolate. Keep it small.
|
|
421
|
+
*/
|
|
422
|
+
cacheEmbeddings?: number;
|
|
423
|
+
/**
|
|
424
|
+
* How many chunks each retrieval leg fetches **before** fusion and
|
|
425
|
+
* reranking trim the result to `topK`.
|
|
426
|
+
*
|
|
427
|
+
* Defaults to `topK * 4` whenever anything downstream reorders — a
|
|
428
|
+
* `lexicalStore`, a multi-query `transformQuery`, or a `rerank` — and to
|
|
429
|
+
* plain `topK` otherwise. Bounded by the store ceiling (50 in metadata
|
|
430
|
+
* mode, 100 with a `textStore`).
|
|
431
|
+
*
|
|
432
|
+
* This is the knob that decides how much recall the reordering has to work
|
|
433
|
+
* with. Fetching only `topK` per leg defeats the point of having two: the
|
|
434
|
+
* whole reason to run a lexical leg is to surface a chunk the vector leg
|
|
435
|
+
* ranked *below* `topK`, and it cannot do that if it was never asked for
|
|
436
|
+
* more than `topK`.
|
|
437
|
+
*/
|
|
438
|
+
candidates?: number;
|
|
439
|
+
/** Custom chunker; overrides the built-in fixed-window splitter. */
|
|
440
|
+
chunk?: (text: string) => ReadonlyArray<string>;
|
|
441
|
+
/** Overlap (chars) between adjacent chunks. Default 200. Must be < `chunkSize`. */
|
|
442
|
+
chunkOverlap?: number;
|
|
443
|
+
/** Target chunk size (chars). Default 1000. */
|
|
444
|
+
chunkSize?: number;
|
|
445
|
+
/**
|
|
446
|
+
* Embedding model, declared once so index + retrieve embed identically: a
|
|
447
|
+
* Workers AI id (e.g. `@cf/baai/bge-base-en-v1.5`) or any AI SDK
|
|
448
|
+
* `EmbeddingModel`.
|
|
449
|
+
*
|
|
450
|
+
* Omitting it resolves through `ai.embeddingModel(undefined)`, which falls
|
|
451
|
+
* back to `createAi`'s `defaultEmbeddingModel` — NOT `defaultModel`, which
|
|
452
|
+
* this said before and which would be a language-model id in an
|
|
453
|
+
* embedding-model slot. On the generated `ctx.ai` that default comes from
|
|
454
|
+
* `LUNORA_AI_DEFAULT_EMBEDDING_MODEL` in the Worker env; with neither set,
|
|
455
|
+
* the first index/retrieve throws.
|
|
456
|
+
*/
|
|
457
|
+
embeddingModel?: EmbeddingModelInput;
|
|
458
|
+
/**
|
|
459
|
+
* Embedding-model version tag — an opt-in discriminator that partitions the
|
|
460
|
+
* vector space so a model swap can never silently return garbage. Vectors
|
|
461
|
+
* embedded by one model live in a different space from another's, and
|
|
462
|
+
* querying across the two returns meaningless neighbours. When set, the tag
|
|
463
|
+
* is folded into the effective Vectorize namespace (and chunk-id prefix) of
|
|
464
|
+
* every index/retrieve/remove, so bumping it re-partitions cleanly: old
|
|
465
|
+
* vectors become unreachable to new queries (empty ≫ wrong) until sources
|
|
466
|
+
* are re-indexed under the new tag.
|
|
467
|
+
*
|
|
468
|
+
* Set + bump this whenever you change {@link RagConfig.embeddingModel} (or
|
|
469
|
+
* its dimensions). Opt-in and non-breaking — omitting it keeps the exact
|
|
470
|
+
* chunk-id/namespace scheme of un-versioned indexes. Must match
|
|
471
|
+
* `^[A-Za-z0-9._-]{1,40}$` (e.g. `"bge-v1.5"`, `"v2"`).
|
|
472
|
+
*/
|
|
473
|
+
embeddingModelVersion?: string;
|
|
474
|
+
/**
|
|
475
|
+
* Pre-defined named filter expressions. Each key is a filter name users
|
|
476
|
+
* pass through `RetrieveOptions.filter`. Throws at retrieve-time if the
|
|
477
|
+
* name is not found here — catches spelling mistakes early.
|
|
478
|
+
*/
|
|
479
|
+
filters?: Record<string, RagNamedFilter>;
|
|
480
|
+
/** The Vectorize index name (a `ctx.vectors` index binding key). */
|
|
481
|
+
index: string;
|
|
482
|
+
/**
|
|
483
|
+
* Pluggable lexical (BM25) store for hybrid retrieval. When set, `index()`
|
|
484
|
+
* mirrors chunk text into it and `retrieve()` fuses the vector (semantic)
|
|
485
|
+
* and lexical (keyword) rankings via Reciprocal Rank Fusion — recovering the
|
|
486
|
+
* exact-term / rare-token matches a pure-embedding search misses. Use the
|
|
487
|
+
* shipped `bm25LexicalStore()` reference adapter or plug your own durable
|
|
488
|
+
* one. See {@link RagLexicalStore}.
|
|
489
|
+
*/
|
|
490
|
+
lexicalStore?: RagLexicalStore;
|
|
491
|
+
/** Retrieval depth for the lexical leg of hybrid search. Defaults to the effective `topK`. */
|
|
492
|
+
lexicalTopK?: number;
|
|
493
|
+
/**
|
|
494
|
+
* Ceiling on the embedding model's dimensionality, checked once per bound
|
|
495
|
+
* context against the first embedding actually produced. Defaults to
|
|
496
|
+
* **1536** — Vectorize's per-vector limit at 32-bit precision.
|
|
497
|
+
*
|
|
498
|
+
* The default rules out most current large embedding models
|
|
499
|
+
* (`text-embedding-3-large` and Gemini embedding at 3072,
|
|
500
|
+
* Qwen3-Embedding at 4096). Without the check they fail at Vectorize with
|
|
501
|
+
* nothing naming the cause; with it they fail at the first embed, naming
|
|
502
|
+
* the ceiling and both escapes — truncate via the provider's Matryoshka
|
|
503
|
+
* `dimensions` option, or set this to `false` when the index is not
|
|
504
|
+
* Vectorize-backed.
|
|
505
|
+
*
|
|
506
|
+
* `false` disables the check entirely: the right setting for a
|
|
507
|
+
* {@link RagVectors} implementation with a different (or no) ceiling.
|
|
508
|
+
*/
|
|
509
|
+
maxEmbeddingDimensions?: number | false;
|
|
510
|
+
/**
|
|
511
|
+
* Enforce tenant isolation: throw (instead of the one-time dev warning)
|
|
512
|
+
* when `index`/`retrieve`/`remove` run without a `namespace`. Recommended
|
|
513
|
+
* for every multi-tenant app — Vectorize indexes are account-global, and
|
|
514
|
+
* in metadata mode the leaked payload includes raw chunk text.
|
|
515
|
+
*/
|
|
516
|
+
requireNamespace?: boolean;
|
|
517
|
+
/**
|
|
518
|
+
* Re-score the retrieved candidates against the query before they are
|
|
519
|
+
* trimmed to `topK`. **Injected, not bundled** — a reranker is a model call,
|
|
520
|
+
* and `@lunora/ai` takes no provider dependency to make one. Adapt yours
|
|
521
|
+
* with `scoreReranker`/`batchReranker`, or write the two-line hook yourself.
|
|
522
|
+
*
|
|
523
|
+
* This is the standard quality step that vector search alone cannot do:
|
|
524
|
+
* an embedding is computed without the query, so it cannot know which of
|
|
525
|
+
* two topically-similar passages actually answers *this* question. A
|
|
526
|
+
* cross-encoder sees both at once and orders them accordingly.
|
|
527
|
+
*
|
|
528
|
+
* Retrieval fetches {@link RagConfig.candidates} chunks, hands them
|
|
529
|
+
* here, and keeps the first `topK` of whatever comes back — so the hook may
|
|
530
|
+
* reorder and drop, but its output order is final. Runs after hybrid fusion
|
|
531
|
+
* and before `chunkContext` expansion.
|
|
532
|
+
*/
|
|
533
|
+
rerank?: RagReranker;
|
|
534
|
+
/**
|
|
535
|
+
* Row-level-security filter derived from the retrieval identity. Called once
|
|
536
|
+
* per `retrieve()` with {@link RagContext.auth} (the bound ctx's `auth`); the
|
|
537
|
+
* returned Vectorize metadata filter is merged over the caller's `filter`
|
|
538
|
+
* with **RLS keys winning** (a caller can never widen past what RLS allows),
|
|
539
|
+
* then applied to both the vector and the lexical legs. Return `undefined` to
|
|
540
|
+
* add no constraint (e.g. an admin identity). Runs on retrieval only —
|
|
541
|
+
* indexing is a trusted server path.
|
|
542
|
+
* @example
|
|
543
|
+
* ```ts
|
|
544
|
+
* const docs = defineRag({
|
|
545
|
+
* index: "docs",
|
|
546
|
+
* // only ever return the caller's own org, whatever else they ask for:
|
|
547
|
+
* rlsFilter: (auth) => ({ orgId: (auth as { orgId: string }).orgId }),
|
|
548
|
+
* });
|
|
549
|
+
* ```
|
|
550
|
+
*/
|
|
551
|
+
rlsFilter?: (auth: unknown) => Promise<Record<string, unknown> | undefined> | Record<string, unknown> | undefined;
|
|
552
|
+
/**
|
|
553
|
+
* Back this RAG with a different vector store.
|
|
554
|
+
*
|
|
555
|
+
* Called once per bound context with that context, so a store needing
|
|
556
|
+
* per-request state (a Hyperdrive/pgvector connection from `ctx.sql`, a
|
|
557
|
+
* shard's own SQLite) can build itself from it. Defaults to wrapping
|
|
558
|
+
* `context.vectors` as a Vectorize-backed store.
|
|
559
|
+
*
|
|
560
|
+
* The store declares its own limits, and `defineRag` reads them instead of
|
|
561
|
+
* assuming Vectorize's — so a pgvector index is not held to a
|
|
562
|
+
* 1536-dimension ceiling or a 10 KiB metadata budget it does not have.
|
|
563
|
+
*/
|
|
564
|
+
store?: (context: RagContext) => RagVectorStore;
|
|
565
|
+
/** Chunk-text storage override — see {@link RagTextStore}. */
|
|
566
|
+
textStore?: RagTextStore;
|
|
567
|
+
/** Default retrieval depth. Default 5. Capped by the store: 50 (metadata mode) / 100 (text-store mode) on Vectorize. */
|
|
568
|
+
topK?: number;
|
|
569
|
+
/**
|
|
570
|
+
* Rewrite or expand the query before it is embedded.
|
|
571
|
+
*
|
|
572
|
+
* The raw user query is often the worst possible search string: a
|
|
573
|
+
* conversational follow-up ("what about the other one?") carries its
|
|
574
|
+
* meaning in the preceding turns, and a short question shares few terms
|
|
575
|
+
* with the long passage that answers it.
|
|
576
|
+
*
|
|
577
|
+
* Return **one** string to rewrite, or **several** to run multi-query
|
|
578
|
+
* retrieval — each is embedded and searched independently and the rankings
|
|
579
|
+
* are fused with RRF, which recovers passages any single phrasing would
|
|
580
|
+
* miss. Returning the query unchanged is a no-op.
|
|
581
|
+
*
|
|
582
|
+
* **Injected, not bundled**, for the same reason as {@link RagConfig.rerank}:
|
|
583
|
+
* every useful strategy (HyDE, multi-query expansion, follow-up rewriting)
|
|
584
|
+
* needs a language model, and this package does not pick one for you. The
|
|
585
|
+
* lexical leg searches the first returned query.
|
|
586
|
+
*/
|
|
587
|
+
transformQuery?: RagQueryTransform;
|
|
588
|
+
}
|
|
589
|
+
/**
|
|
590
|
+
* `IndexInput` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
591
|
+
* @experimental
|
|
592
|
+
*/
|
|
593
|
+
interface IndexInput {
|
|
594
|
+
/**
|
|
595
|
+
* When `false`, throws if the source text produces zero chunks (e.g. empty
|
|
596
|
+
* or whitespace-only text). Default `true` (silently produces zero chunks).
|
|
597
|
+
*/
|
|
598
|
+
allowEmptySources?: boolean;
|
|
599
|
+
/** Source document id — chunk ids derive from it as `${id}#${chunkIndex}`. */
|
|
600
|
+
id: string;
|
|
601
|
+
/**
|
|
602
|
+
* Relative weight in `[0, 1]` multiplied into this source's match scores at
|
|
603
|
+
* retrieval time (default 1). Lets canonical docs outrank incidental ones.
|
|
604
|
+
*/
|
|
605
|
+
importance?: number;
|
|
606
|
+
/** Source metadata copied onto every chunk vector (e.g. title, url). */
|
|
607
|
+
metadata?: Record<string, unknown>;
|
|
608
|
+
/** Tenant/shard key. Required for multi-tenant apps — Vectorize is account-global. */
|
|
609
|
+
namespace?: string;
|
|
610
|
+
/**
|
|
611
|
+
* Called after each chunk is successfully upserted. Useful for progress
|
|
612
|
+
* tracking during large indexing operations — e.g. updating a UI progress
|
|
613
|
+
* bar or logging per-chunk status.
|
|
614
|
+
*/
|
|
615
|
+
onChunk?: (info: {
|
|
616
|
+
chunkIndex: number;
|
|
617
|
+
id: string;
|
|
618
|
+
text: string;
|
|
619
|
+
total: number;
|
|
620
|
+
}) => void;
|
|
621
|
+
/**
|
|
622
|
+
* Index this source even when its identity hash (`text` + `metadata` +
|
|
623
|
+
* `importance`) is unchanged.
|
|
624
|
+
*
|
|
625
|
+
* The hash short-circuit skips chunking, embedding and every write — which
|
|
626
|
+
* is what makes a cron re-sync cheap, and also what makes attaching a
|
|
627
|
+
* `textStore` or `lexicalStore` to an ALREADY-indexed corpus a silent no-op:
|
|
628
|
+
* the new store is never written, so the keyword leg returns nothing
|
|
629
|
+
* forever with no error. Set this for the one pass that backfills it.
|
|
630
|
+
*
|
|
631
|
+
* It re-embeds, so it is not a setting to leave on.
|
|
632
|
+
*/
|
|
633
|
+
reindex?: boolean;
|
|
634
|
+
/** The document body to chunk + embed + upsert. */
|
|
635
|
+
text: string;
|
|
636
|
+
}
|
|
637
|
+
/**
|
|
638
|
+
* `IndexResult` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
639
|
+
* @experimental
|
|
640
|
+
*/
|
|
641
|
+
interface IndexResult {
|
|
642
|
+
/** Number of chunks the source is indexed into. */
|
|
643
|
+
chunks: number;
|
|
644
|
+
/** The deterministic chunk vector ids, in chunk order. */
|
|
645
|
+
ids: ReadonlyArray<string>;
|
|
646
|
+
/**
|
|
647
|
+
* True when the source's identity hash — its `text`, `metadata` and
|
|
648
|
+
* `importance` together — matched the previously indexed one, so
|
|
649
|
+
* chunking/embedding/upserts were skipped entirely (a no-op re-sync).
|
|
650
|
+
* Changing `metadata` alone (a tenant move, an ACL correction) therefore
|
|
651
|
+
* re-indexes: the old values are what `rlsFilter` scopes retrieval on.
|
|
652
|
+
*/
|
|
653
|
+
unchanged: boolean;
|
|
654
|
+
}
|
|
655
|
+
/**
|
|
656
|
+
* `RemoveInput` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
657
|
+
* @experimental
|
|
658
|
+
*/
|
|
659
|
+
interface RemoveInput {
|
|
660
|
+
/** The source document id whose chunks are removed. */
|
|
661
|
+
id: string;
|
|
662
|
+
namespace?: string;
|
|
663
|
+
}
|
|
664
|
+
/**
|
|
665
|
+
* `RetrieveOptions` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
666
|
+
* @experimental
|
|
667
|
+
*/
|
|
668
|
+
interface RetrieveOptions {
|
|
669
|
+
/**
|
|
670
|
+
* Also return this many neighbouring chunks around each match (fetched by
|
|
671
|
+
* deterministic id, not re-queried) — "embed small, retrieve big". Neighbour
|
|
672
|
+
* text is stitched into the chunk's `text` in document order. Best combined
|
|
673
|
+
* with `chunkOverlap: 0`, since overlapping windows repeat boundary text.
|
|
674
|
+
*/
|
|
675
|
+
chunkContext?: {
|
|
676
|
+
after?: number;
|
|
677
|
+
before?: number;
|
|
678
|
+
};
|
|
679
|
+
/**
|
|
680
|
+
* Vectorize filter expression — or the name of a pre-defined filter declared
|
|
681
|
+
* in `RagConfig.filters`. Passing a name that is not registered throws at
|
|
682
|
+
* call time, catching spelling mistakes early.
|
|
683
|
+
*/
|
|
684
|
+
filter?: Record<string, unknown> | string;
|
|
685
|
+
/**
|
|
686
|
+
* Drop matches whose (importance-adjusted) score falls below this threshold.
|
|
687
|
+
*
|
|
688
|
+
* Applied to the VECTOR leg, where the score is still the cosine scale this
|
|
689
|
+
* option is documented against — every fusion below replaces `score` with an
|
|
690
|
+
* RRF score, and thresholding that against a cosine number keeps or drops
|
|
691
|
+
* chunks essentially at random. A chunk the vector leg rejected here stays
|
|
692
|
+
* rejected even if the lexical leg also ranks it; a lexical-only hit the
|
|
693
|
+
* vector leg never scored is NOT gated, since its BM25 score is not on this
|
|
694
|
+
* scale (see `hybridRank`).
|
|
695
|
+
*/
|
|
696
|
+
minScore?: number;
|
|
697
|
+
namespace?: string;
|
|
698
|
+
/**
|
|
699
|
+
* Fires after retrieval completes, before chunk expansion. Useful for
|
|
700
|
+
* observability — logging query latency, hit counts, etc.
|
|
701
|
+
*/
|
|
702
|
+
onRetrieve?: (info: {
|
|
703
|
+
matches: number;
|
|
704
|
+
query: string;
|
|
705
|
+
}) => void;
|
|
706
|
+
/**
|
|
707
|
+
* Set `false` to skip {@link RagConfig.rerank} for this call — the escape
|
|
708
|
+
* hatch for a latency-sensitive path (typeahead, an agent's inner loop)
|
|
709
|
+
* that cannot afford the extra model round-trip.
|
|
710
|
+
*/
|
|
711
|
+
rerank?: false;
|
|
712
|
+
topK?: number;
|
|
713
|
+
/** Set `false` to skip {@link RagConfig.transformQuery} for this call. */
|
|
714
|
+
transformQuery?: false;
|
|
715
|
+
}
|
|
716
|
+
/**
|
|
717
|
+
* `RetrievedChunk` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
718
|
+
* @experimental
|
|
719
|
+
*/
|
|
720
|
+
interface RetrievedChunk {
|
|
721
|
+
chunkIndex: number;
|
|
722
|
+
id: string;
|
|
723
|
+
/**
|
|
724
|
+
* The source-level importance weight that was multiplied into this chunk's
|
|
725
|
+
* score. `1` when no importance was set at index time.
|
|
726
|
+
*/
|
|
727
|
+
importance: number;
|
|
728
|
+
/** Caller metadata stored on the vector (internal `__rag*` keys stripped). */
|
|
729
|
+
metadata?: Record<string, unknown>;
|
|
730
|
+
/** Cosine similarity, multiplied by the source's `importance` when one was set. */
|
|
731
|
+
score: number;
|
|
732
|
+
sourceId: string;
|
|
733
|
+
text: string;
|
|
734
|
+
}
|
|
735
|
+
/**
|
|
736
|
+
* `RagSource` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
737
|
+
* @experimental
|
|
738
|
+
*/
|
|
739
|
+
interface RagSource {
|
|
740
|
+
id: string;
|
|
741
|
+
/** Caller metadata from the source's first-seen chunk (internal keys stripped). */
|
|
742
|
+
metadata?: Record<string, unknown>;
|
|
743
|
+
/**
|
|
744
|
+
* The source's importance weight (the `importance` value passed at index
|
|
745
|
+
* time, default 1), propagated so downstream consumers can factor it into
|
|
746
|
+
* their own ranking or UI.
|
|
747
|
+
*/
|
|
748
|
+
weight?: number;
|
|
749
|
+
}
|
|
750
|
+
/**
|
|
751
|
+
* The retrieve return shape — designed so an agent memory step consumes it directly.
|
|
752
|
+
* @experimental
|
|
753
|
+
*/
|
|
754
|
+
interface RetrieveResult {
|
|
755
|
+
/** Ranked chunks (best first). */
|
|
756
|
+
chunks: ReadonlyArray<RetrievedChunk>;
|
|
757
|
+
/** Ready-to-inject prompt context: chunks joined under `[source:<id>#<n>]` headers. */
|
|
758
|
+
context: string;
|
|
759
|
+
/** Deduped source references, in first-seen (best) order. */
|
|
760
|
+
sources: ReadonlyArray<RagSource>;
|
|
761
|
+
}
|
|
762
|
+
/**
|
|
763
|
+
* `RagToolOptions` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
764
|
+
* @experimental
|
|
765
|
+
*/
|
|
766
|
+
interface RagToolOptions {
|
|
767
|
+
/** Tool description shown to the model. Defaults to a search description naming the index. */
|
|
768
|
+
description?: string;
|
|
769
|
+
/** Namespace applied to every tool-invoked retrieval (the tenant key). */
|
|
770
|
+
namespace?: string;
|
|
771
|
+
/** Retrieval depth for tool-invoked retrievals. */
|
|
772
|
+
topK?: number;
|
|
773
|
+
}
|
|
774
|
+
/**
|
|
775
|
+
* The per-request RAG surface returned by binding a ctx: `docs(ctx)`.
|
|
776
|
+
* @experimental
|
|
777
|
+
*/
|
|
778
|
+
interface Rag {
|
|
779
|
+
/**
|
|
780
|
+
* Expose `retrieve` as an AI SDK tool (for `generateText`/`streamText`
|
|
781
|
+
* `tools:` maps), so a model can decide to search the index itself.
|
|
782
|
+
*/
|
|
783
|
+
asTool: (options?: RagToolOptions) => Tool<{
|
|
784
|
+
query: string;
|
|
785
|
+
}, RetrieveResult>;
|
|
786
|
+
/**
|
|
787
|
+
* Chunk + embed + upsert one source document. Re-indexing the same `id` is
|
|
788
|
+
* an atomic-enough replace: unchanged content short-circuits via content
|
|
789
|
+
* hash, and stale chunks beyond the new count are deleted automatically.
|
|
790
|
+
*/
|
|
791
|
+
index: (input: IndexInput) => Promise<IndexResult>;
|
|
792
|
+
/** Delete every chunk of a previously indexed source. */
|
|
793
|
+
remove: (input: RemoveInput) => Promise<void>;
|
|
794
|
+
/** Embed the query and return ranked chunks + prompt-ready context. */
|
|
795
|
+
retrieve: (query: string, options?: RetrieveOptions) => Promise<RetrieveResult>;
|
|
796
|
+
}
|
|
797
|
+
declare const defineRag: (config: RagConfig) => ((context: RagContext) => Rag);
|
|
798
|
+
/**
|
|
799
|
+
* Guess a MIME type from a file extension. Lowercases and strips a leading `.`
|
|
800
|
+
* from `extension`; returns `"application/octet-stream"` for unknown extensions.
|
|
801
|
+
*
|
|
802
|
+
* Covers the broad set of extensions users are likely to encounter in a web /
|
|
803
|
+
* document-processing context — images, video, audio, office docs, PDF, text,
|
|
804
|
+
* archives, and source code. Follows the same approach as Convex's
|
|
805
|
+
* `guessMimeType` helper.
|
|
806
|
+
* @experimental
|
|
807
|
+
*/
|
|
808
|
+
declare const guessMimeTypeFromExtension: (extension: string) => string;
|
|
809
|
+
/**
|
|
810
|
+
* SHA-256 hex digest of binary data. Accepts a `BufferSource` (`ArrayBuffer` or
|
|
811
|
+
* `ArrayBufferView` such as `Uint8Array`). Useful for content-addressable
|
|
812
|
+
* storage — pair with `IndexInput.text` to detect duplicates across re-indexes.
|
|
813
|
+
* @experimental
|
|
814
|
+
*/
|
|
815
|
+
declare const contentHash: (data: BufferSource) => Promise<string>;
|
|
816
|
+
/**
|
|
817
|
+
* Reciprocal Rank Fusion (RRF): merge two ranked lists of chunks by their
|
|
818
|
+
* _rank position_ rather than their absolute scores, which are not comparable
|
|
819
|
+
* across different search methods (cosine vs BM25).
|
|
820
|
+
*
|
|
821
|
+
* Each result set contributes `1 / (k + rank)` to each chunk's fused score,
|
|
822
|
+
* where `rank` is 0-based position in the list. The constant `k` (default 60)
|
|
823
|
+
* dampens the influence of high ranks.
|
|
824
|
+
*
|
|
825
|
+
* **The returned chunks carry the fused score in `score`**, multiplied by the
|
|
826
|
+
* chunk's `importance` so source weighting still applies. Writing it back is
|
|
827
|
+
* what makes the fusion survive: a caller that re-sorts by `score` — as
|
|
828
|
+
* `retrieve()` does, to apply importance weighting — would otherwise re-order
|
|
829
|
+
* by the incomparable inputs and discard the ranking this function computed.
|
|
830
|
+
* Since BM25 is unbounded while cosine is `[0, 1]`, that silently promoted
|
|
831
|
+
* every lexical-only hit above every vector hit.
|
|
832
|
+
*
|
|
833
|
+
* So in hybrid mode `RetrievedChunk.score` is an RRF score (small, ~`1/60`
|
|
834
|
+
* scale), not a cosine similarity. `retrieve()` applies `minScore` to the
|
|
835
|
+
* vector leg *before* fusion for exactly this reason — the option is documented
|
|
836
|
+
* against the cosine scale.
|
|
837
|
+
*
|
|
838
|
+
* Ties are broken by preferring the chunk ranked higher in the vector result,
|
|
839
|
+
* typically the more semantically accurate of the two methods.
|
|
840
|
+
*
|
|
841
|
+
* Callers MUST ensure every chunk in both lists carries a unique, comparable
|
|
842
|
+
* `id` — guaranteed by the chunk-id scheme `${sourceId}#${chunkIndex}`.
|
|
843
|
+
* @experimental
|
|
844
|
+
*/
|
|
845
|
+
declare const hybridRank: (vectorResults: ReadonlyArray<RetrievedChunk>, textResults: ReadonlyArray<RetrievedChunk>, k?: number) => ReadonlyArray<RetrievedChunk>;
|
|
846
|
+
/**
|
|
847
|
+
* An **in-memory** Okapi BM25 lexical store — the reference adapter behind
|
|
848
|
+
* `RagConfig.lexicalStore`, giving hybrid retrieval its keyword leg with zero
|
|
849
|
+
* infrastructure. State lives in the worker isolate: it is **not durable and not
|
|
850
|
+
* shared across isolates**, so it is intended for tests, local development, and
|
|
851
|
+
* single-isolate workloads. Production deployments plug a durable
|
|
852
|
+
* {@link RagLexicalStore} (a DO-SQLite inverted index, D1, or an external search
|
|
853
|
+
* service) behind the same seam.
|
|
854
|
+
*
|
|
855
|
+
* Tenant isolation is by `namespace` (each namespace keeps its own index), and
|
|
856
|
+
* each chunk's source `metadata` is stored alongside it so `search` evaluates
|
|
857
|
+
* the **same** metadata predicate the vector leg receives — including an
|
|
858
|
+
* `rlsFilter` result. A hit the filter excludes never reaches fusion.
|
|
859
|
+
*
|
|
860
|
+
* That matters because the filter carries the tenant/RBAC scope: a lexical leg
|
|
861
|
+
* that ignored it would leak excluded chunk text into the fused result no
|
|
862
|
+
* matter what the vector leg returned. This store previously had no metadata to
|
|
863
|
+
* check and so refused every filtered query, which made hybrid search and
|
|
864
|
+
* metadata-based RLS mutually exclusive.
|
|
865
|
+
* @experimental
|
|
866
|
+
*/
|
|
867
|
+
declare const bm25LexicalStore: () => RagLexicalStore;
|
|
868
|
+
/**
|
|
869
|
+
* True when `metadata` satisfies every clause of `filter`. An empty or absent
|
|
870
|
+
* filter matches everything; absent metadata satisfies only an empty filter.
|
|
871
|
+
*
|
|
872
|
+
* Clauses are ANDed, matching Vectorize.
|
|
873
|
+
*/
|
|
874
|
+
declare const matchesMetadataFilter: (metadata: Record<string, unknown> | undefined, filter: Record<string, unknown> | undefined) => boolean;
|
|
875
|
+
/** Options for {@link scoreReranker}. */
|
|
876
|
+
interface ScoreRerankerOptions {
|
|
877
|
+
/**
|
|
878
|
+
* Drop chunks scoring below this threshold, in the scorer's own scale.
|
|
879
|
+
* Omitted → nothing is dropped and the reranker only reorders.
|
|
880
|
+
*
|
|
881
|
+
* Worth setting: reranking's real value is not just promoting the best
|
|
882
|
+
* passage but *rejecting* the ones vector search only matched topically,
|
|
883
|
+
* and a chunk that survives to the prompt is a chunk the model may cite.
|
|
884
|
+
*/
|
|
885
|
+
minScore?: number;
|
|
886
|
+
/**
|
|
887
|
+
* Score one `(query, passage)` pair for relevance. Higher is better; the
|
|
888
|
+
* scale does not matter, only the ordering it induces.
|
|
889
|
+
*
|
|
890
|
+
* Called once per candidate chunk. Wrap a Workers AI reranker
|
|
891
|
+
* (`ctx.ai.run("@cf/baai/bge-reranker-base", …)`), a Cohere/Voyage rerank
|
|
892
|
+
* endpoint, or an LLM prompted to rate relevance.
|
|
893
|
+
*/
|
|
894
|
+
score: (query: string, text: string) => Promise<number> | number;
|
|
895
|
+
}
|
|
896
|
+
/**
|
|
897
|
+
* Batched variant — score every candidate in one call.
|
|
898
|
+
*/
|
|
899
|
+
interface BatchRerankerOptions {
|
|
900
|
+
/** Drop chunks scoring below this threshold. See {@link ScoreRerankerOptions.minScore}. */
|
|
901
|
+
minScore?: number;
|
|
902
|
+
/**
|
|
903
|
+
* Score all candidates at once, returning one number per passage **in the
|
|
904
|
+
* same order**. Prefer this over {@link ScoreRerankerOptions.score} when the
|
|
905
|
+
* provider has a batch endpoint: cross-encoder APIs are priced and rate
|
|
906
|
+
* limited per request, so N passages in one call beats N calls.
|
|
907
|
+
*
|
|
908
|
+
* A result whose length does not match the input is rejected rather than
|
|
909
|
+
* zipped against the wrong passages.
|
|
910
|
+
*/
|
|
911
|
+
scoreAll: (query: string, texts: ReadonlyArray<string>) => Promise<ReadonlyArray<number>> | ReadonlyArray<number>;
|
|
912
|
+
}
|
|
913
|
+
/**
|
|
914
|
+
* Adapt a per-passage relevance scorer into a {@link RagReranker}.
|
|
915
|
+
*
|
|
916
|
+
* The scorer is **injected** — `@lunora/ai` takes no provider dependency to
|
|
917
|
+
* make a model call. Scoring runs with bounded concurrency so a 50-candidate
|
|
918
|
+
* pool does not fan out 50 simultaneous subrequests.
|
|
919
|
+
*
|
|
920
|
+
* ```ts
|
|
921
|
+
* defineRag({
|
|
922
|
+
* index: "docs",
|
|
923
|
+
* rerank: scoreReranker({
|
|
924
|
+
* score: async (query, text) => {
|
|
925
|
+
* const result = await ctx.ai.run("@cf/baai/bge-reranker-base", { query, contexts: [{ text }] });
|
|
926
|
+
* return result.response[0].score;
|
|
927
|
+
* },
|
|
928
|
+
* }),
|
|
929
|
+
* });
|
|
930
|
+
* ```
|
|
931
|
+
* @experimental
|
|
932
|
+
*/
|
|
933
|
+
declare const scoreReranker: (options: ScoreRerankerOptions) => RagReranker;
|
|
934
|
+
/**
|
|
935
|
+
* Adapt a **batch** relevance scorer into a {@link RagReranker} — one call for
|
|
936
|
+
* the whole candidate pool. See {@link BatchRerankerOptions.scoreAll}.
|
|
937
|
+
* @experimental
|
|
938
|
+
*/
|
|
939
|
+
declare const batchReranker: (options: BatchRerankerOptions) => RagReranker;
|
|
940
|
+
/** One object listed by a {@link RagObjectSource}. */
|
|
941
|
+
interface RagSourceObject {
|
|
942
|
+
/** Content type, when the source knows it. Falls back to the key's extension. */
|
|
943
|
+
contentType?: string;
|
|
944
|
+
/** Stable key — becomes the indexed source id. */
|
|
945
|
+
key: string;
|
|
946
|
+
/** Metadata attached to every chunk of this object. */
|
|
947
|
+
metadata?: Record<string, unknown>;
|
|
948
|
+
}
|
|
949
|
+
/**
|
|
950
|
+
* Where documents come from. Two operations, both injected.
|
|
951
|
+
*
|
|
952
|
+
* `list` is an async iterable rather than an array so the caller decides how to
|
|
953
|
+
* page a large bucket rather than being forced to build one array up front.
|
|
954
|
+
* `sync` still collects the KEYS it yields — it needs the full current key set
|
|
955
|
+
* to work out what disappeared — but never more than one object's BODY at a
|
|
956
|
+
* time, which is where the memory actually is.
|
|
957
|
+
*/
|
|
958
|
+
interface RagObjectSource {
|
|
959
|
+
/** Fetch one object's raw text. Return `undefined` to skip it (unreadable, unsupported). */
|
|
960
|
+
get: (object: RagSourceObject) => Promise<string | undefined> | string | undefined;
|
|
961
|
+
/** Enumerate the objects to index. */
|
|
962
|
+
list: () => AsyncIterable<RagSourceObject> | Iterable<RagSourceObject>;
|
|
963
|
+
}
|
|
964
|
+
/** Turn a fetched object's raw text into indexable plain text. */
|
|
965
|
+
type RagExtractor = (raw: string, object: RagSourceObject) => Promise<string | undefined> | string | undefined;
|
|
966
|
+
/** Options for {@link defineRagSource}. */
|
|
967
|
+
interface RagSourceOptions {
|
|
968
|
+
/**
|
|
969
|
+
* How many objects to process at once. Default 4.
|
|
970
|
+
*
|
|
971
|
+
* Each one is a fetch plus an embed plus an upsert, so this multiplies into
|
|
972
|
+
* the subrequest budget — the default is deliberately low.
|
|
973
|
+
*/
|
|
974
|
+
concurrency?: number;
|
|
975
|
+
/**
|
|
976
|
+
* Extractors keyed by content type (`text/html`, `application/pdf`, …), or
|
|
977
|
+
* `"*"` as a fallback. A content type with no extractor and no `"*"` entry
|
|
978
|
+
* is skipped and counted in `skipped` — never indexed as raw bytes, which
|
|
979
|
+
* would fill the index with markup or binary noise that embeds to nothing
|
|
980
|
+
* meaningful.
|
|
981
|
+
*/
|
|
982
|
+
extractors?: Record<string, RagExtractor>;
|
|
983
|
+
/** Namespace (tenant/shard key) applied to every indexed object. */
|
|
984
|
+
namespace?: string;
|
|
985
|
+
/** Called after each object is handled — for progress reporting. */
|
|
986
|
+
onObject?: (info: {
|
|
987
|
+
chunks: number;
|
|
988
|
+
key: string;
|
|
989
|
+
status: "indexed" | "skipped" | "unchanged";
|
|
990
|
+
}) => void;
|
|
991
|
+
}
|
|
992
|
+
/** Per-pass options for {@link RagSourceSync.sync}. */
|
|
993
|
+
interface RagSyncPassOptions {
|
|
994
|
+
/**
|
|
995
|
+
* The keys this index is believed to already hold. Any of them missing from
|
|
996
|
+
* this pass's `list()` is deleted from the index.
|
|
997
|
+
*
|
|
998
|
+
* Pruning is what makes the index a mirror rather than an append-only pile:
|
|
999
|
+
* a document removed at the source but left in the index keeps being
|
|
1000
|
+
* retrieved and cited, which is worse than never having indexed it.
|
|
1001
|
+
*
|
|
1002
|
+
* It is the CALLER's set, and required, because nothing else can hold it
|
|
1003
|
+
* honestly. A previous pass's keys remembered in the instance would be
|
|
1004
|
+
* empty on the shape this helper is documented in — built per request, from
|
|
1005
|
+
* a per-request `rag(ctx)` — so a prune that defaulted on would in practice
|
|
1006
|
+
* never run and never say so. Persist the set (a table, a KV key, the
|
|
1007
|
+
* bucket listing itself) and hand it in.
|
|
1008
|
+
*
|
|
1009
|
+
* Omitted ⇒ nothing is pruned, and `pruned` comes back empty.
|
|
1010
|
+
*/
|
|
1011
|
+
knownKeys?: Iterable<string>;
|
|
1012
|
+
}
|
|
1013
|
+
/** What one {@link RagSourceSync.sync} pass did. */
|
|
1014
|
+
interface RagSyncReport {
|
|
1015
|
+
/** Keys indexed for the first time or re-indexed after a change. */
|
|
1016
|
+
indexed: string[];
|
|
1017
|
+
/** Keys deleted from the index because they no longer exist at the source. Empty unless `knownKeys` was supplied. */
|
|
1018
|
+
pruned: string[];
|
|
1019
|
+
/** Keys skipped — no extractor, or the source returned nothing. */
|
|
1020
|
+
skipped: string[];
|
|
1021
|
+
/** Keys whose content hash matched, so nothing was embedded. */
|
|
1022
|
+
unchanged: string[];
|
|
1023
|
+
}
|
|
1024
|
+
/** The bound ingestion surface. */
|
|
1025
|
+
interface RagSourceSync {
|
|
1026
|
+
/** Run one full pass over the source. */
|
|
1027
|
+
sync: (source: RagObjectSource, passOptions?: RagSyncPassOptions) => Promise<RagSyncReport>;
|
|
1028
|
+
}
|
|
1029
|
+
/**
|
|
1030
|
+
* Declare a bulk-ingestion pass over a {@link Rag}.
|
|
1031
|
+
*
|
|
1032
|
+
* ```ts
|
|
1033
|
+
* const ingest = defineRagSource(docs(ctx), { namespace: ctx.shardKey });
|
|
1034
|
+
*
|
|
1035
|
+
* const report = await ingest.sync(
|
|
1036
|
+
* {
|
|
1037
|
+
* list: async function* () {
|
|
1038
|
+
* for await (const object of bucket.list()) {
|
|
1039
|
+
* yield { key: object.key, metadata: { url: object.key } };
|
|
1040
|
+
* }
|
|
1041
|
+
* },
|
|
1042
|
+
* get: async (object) => (await bucket.get(object.key))?.text(),
|
|
1043
|
+
* },
|
|
1044
|
+
* // Pass what you already indexed to have deletions mirrored; omit it and
|
|
1045
|
+
* // nothing is pruned.
|
|
1046
|
+
* { knownKeys: await ctx.db.query("indexedDocs").collect().then((rows) => rows.map((row) => row.key)) },
|
|
1047
|
+
* );
|
|
1048
|
+
* ```
|
|
1049
|
+
* @experimental
|
|
1050
|
+
*/
|
|
1051
|
+
declare const defineRagSource: (rag: Rag, options?: RagSourceOptions) => RagSourceSync;
|
|
1052
|
+
/**
|
|
1053
|
+
* Run one statement and return its rows.
|
|
1054
|
+
*
|
|
1055
|
+
* Rows come back as plain objects keyed by column name — the shape
|
|
1056
|
+
* `SqlStorage#exec().toArray()`, D1's `.all().results`, and `postgres.js` all
|
|
1057
|
+
* already produce. A statement returning nothing yields an empty array.
|
|
1058
|
+
*
|
|
1059
|
+
* May be synchronous: a Durable Object's SQLite is, and forcing it through a
|
|
1060
|
+
* promise would add a microtask per row batch for nothing.
|
|
1061
|
+
*/
|
|
1062
|
+
type RagSqlExec = (sql: string, parameters: ReadonlyArray<unknown>) => Promise<ReadonlyArray<Record<string, unknown>>> | ReadonlyArray<Record<string, unknown>>;
|
|
1063
|
+
/** Options for {@link sqlLexicalStore}. */
|
|
1064
|
+
interface SqlLexicalStoreOptions {
|
|
1065
|
+
/** Execute one statement. See {@link RagSqlExec}. */
|
|
1066
|
+
exec: RagSqlExec;
|
|
1067
|
+
/**
|
|
1068
|
+
* Table-name prefix. Default `lunora_rag_lexical`; the postings table is
|
|
1069
|
+
* this plus `_terms`. Must be a bare SQL identifier.
|
|
1070
|
+
*/
|
|
1071
|
+
table?: string;
|
|
1072
|
+
}
|
|
1073
|
+
declare const sqlLexicalStore: (options: SqlLexicalStoreOptions) => RagLexicalStore;
|
|
1074
|
+
/** Options for {@link sqliteVectorStore}. */
|
|
1075
|
+
interface SqliteVectorStoreOptions {
|
|
1076
|
+
/** Execute one statement. See {@link RagSqlExec}. */
|
|
1077
|
+
exec: RagSqlExec;
|
|
1078
|
+
/**
|
|
1079
|
+
* Ceiling on embedding dimensionality. Defaults to `false` (no limit) —
|
|
1080
|
+
* vectors are stored as JSON, so nothing here cares how wide they are, and
|
|
1081
|
+
* inheriting Vectorize's 1536 would be inventing a constraint.
|
|
1082
|
+
*/
|
|
1083
|
+
maxDimensions?: number | false;
|
|
1084
|
+
/**
|
|
1085
|
+
* Upper bound on how many rows a single namespace may be scanned for.
|
|
1086
|
+
* Default 50,000.
|
|
1087
|
+
*
|
|
1088
|
+
* Search is linear, so this is the difference between a slow query and a
|
|
1089
|
+
* Worker that exceeds its CPU budget and is killed with nothing explaining
|
|
1090
|
+
* why. Exceeding it throws, naming the namespace and the count.
|
|
1091
|
+
*/
|
|
1092
|
+
maxScan?: number;
|
|
1093
|
+
/** Table name. Default `lunora_rag_vectors`. Must be a bare SQL identifier. */
|
|
1094
|
+
table?: string;
|
|
1095
|
+
}
|
|
1096
|
+
declare const sqliteVectorStore: (options: SqliteVectorStoreOptions) => RagVectorStore;
|
|
1097
|
+
/**
|
|
1098
|
+
* Keep a RAG index in step with a table, so the table IS the index.
|
|
1099
|
+
*
|
|
1100
|
+
* `rag.index(...)` is a manual call, which means an app that edits a document
|
|
1101
|
+
* has to remember to re-index it — and a forgotten call is invisible: retrieval
|
|
1102
|
+
* keeps answering, just from stale text. The fix is to hang the re-index off the
|
|
1103
|
+
* write itself.
|
|
1104
|
+
*
|
|
1105
|
+
* Embedding is network I/O, so it cannot run inside the mutation that wrote the
|
|
1106
|
+
* row. The bridge is the two seams that already exist: a table `.triggers()`
|
|
1107
|
+
* handler runs in the write path and can `ctx.scheduler.runAfter(...)`, so the
|
|
1108
|
+
* trigger records the intent and an internal ACTION does the embedding a moment
|
|
1109
|
+
* later. That is what {@link ragSyncTriggers} wires up.
|
|
1110
|
+
*
|
|
1111
|
+
* Re-indexing unchanged text is already cheap — `rag.index` short-circuits on a
|
|
1112
|
+
* content hash — but this skips scheduling entirely when an update didn't touch
|
|
1113
|
+
* the indexed text, so an unrelated column edit costs nothing at all.
|
|
1114
|
+
*/
|
|
1115
|
+
/**
|
|
1116
|
+
* A dispatchable function reference — the `internal.docs.reindex` you pass as
|
|
1117
|
+
* `action`. Typed structurally (rather than `unknown`) so passing the wrong
|
|
1118
|
+
* thing is a compile error: mis-wiring the action is the mistake this API is
|
|
1119
|
+
* most likely to see, and it would otherwise surface as a silent no-op.
|
|
1120
|
+
*/
|
|
1121
|
+
interface RagSyncActionReference {
|
|
1122
|
+
readonly __lunoraRef: string;
|
|
1123
|
+
}
|
|
1124
|
+
/** Structural slice of `ctx.scheduler` — enough to defer the re-index. */
|
|
1125
|
+
interface RagSyncScheduler {
|
|
1126
|
+
runAfter: (delayMs: number, target: unknown, args?: Record<string, unknown>) => Promise<string>;
|
|
1127
|
+
}
|
|
1128
|
+
/** Structural slice of the `TriggerCtx` a `.triggers()` handler receives. */
|
|
1129
|
+
interface RagSyncTriggerContext {
|
|
1130
|
+
readonly scheduler: RagSyncScheduler;
|
|
1131
|
+
}
|
|
1132
|
+
/** The trigger events this helper handles, narrowed to what it reads. */
|
|
1133
|
+
interface RagSyncEvent {
|
|
1134
|
+
readonly doc?: Record<string, unknown>;
|
|
1135
|
+
readonly id: string;
|
|
1136
|
+
readonly previous?: Record<string, unknown>;
|
|
1137
|
+
}
|
|
1138
|
+
/** What the scheduled action receives — one document to re-index, or one to drop. */
|
|
1139
|
+
interface RagSyncArgs extends Record<string, unknown> {
|
|
1140
|
+
/** `true` when the source row was deleted: call `rag.remove({ id })`. */
|
|
1141
|
+
deleted?: boolean;
|
|
1142
|
+
/** The source id — the same `id` you pass to `rag.index`/`rag.remove`. */
|
|
1143
|
+
id: string;
|
|
1144
|
+
/** The text to embed. Absent on a delete. */
|
|
1145
|
+
text?: string;
|
|
1146
|
+
}
|
|
1147
|
+
interface RagSyncOptions<Document extends Record<string, unknown> = Record<string, unknown>> {
|
|
1148
|
+
/**
|
|
1149
|
+
* The internal action to dispatch — it receives {@link RagSyncArgs} and calls
|
|
1150
|
+
* `rag.index` / `rag.remove`. An action, not a mutation: embedding is network
|
|
1151
|
+
* I/O and never runs in the deterministic write path.
|
|
1152
|
+
*/
|
|
1153
|
+
action: RagSyncActionReference;
|
|
1154
|
+
/**
|
|
1155
|
+
* How long to wait before re-indexing. A small delay coalesces nothing by
|
|
1156
|
+
* itself, but it keeps the embed off the write's own tail latency. Default
|
|
1157
|
+
* `0` — as soon as the mutation commits.
|
|
1158
|
+
*/
|
|
1159
|
+
delayMs?: number;
|
|
1160
|
+
/** The source id to index under. Defaults to the row's own id. */
|
|
1161
|
+
id?: (document: Document) => string;
|
|
1162
|
+
/** The text to embed. Return `undefined` to skip the row (a draft, an empty body). */
|
|
1163
|
+
text: (document: Document) => string | undefined;
|
|
1164
|
+
}
|
|
1165
|
+
/** One trigger definition, structurally — matches what `.triggers((t) => …)` returns. */
|
|
1166
|
+
type RagSyncHandler = (context: RagSyncTriggerContext, event: RagSyncEvent) => Promise<void>;
|
|
1167
|
+
/**
|
|
1168
|
+
* Build the three write-path handlers that keep a RAG index in step with a
|
|
1169
|
+
* table. Wire them into the table's `.triggers()`:
|
|
1170
|
+
*
|
|
1171
|
+
* ```ts
|
|
1172
|
+
* const sync = ragSyncTriggers({ action: internal.docs.reindex, text: (doc) => doc.body });
|
|
1173
|
+
*
|
|
1174
|
+
* export const schema = defineSchema({
|
|
1175
|
+
* docs: defineTable({ body: v.string(), title: v.string() }).triggers((t) => ({
|
|
1176
|
+
* ragDelete: t.afterDelete(sync.afterDelete),
|
|
1177
|
+
* ragInsert: t.afterInsert(sync.afterInsert),
|
|
1178
|
+
* ragUpdate: t.afterUpdate(sync.afterUpdate),
|
|
1179
|
+
* })),
|
|
1180
|
+
* });
|
|
1181
|
+
* ```
|
|
1182
|
+
*
|
|
1183
|
+
* The action on the other end is three lines:
|
|
1184
|
+
*
|
|
1185
|
+
* ```ts
|
|
1186
|
+
* export const reindex = internalAction.input({ deleted: v.optional(v.boolean()), id: v.string(), text: v.optional(v.string()) }).action(
|
|
1187
|
+
* async ({ args, ctx }) => {
|
|
1188
|
+
* const rag = docsRag(ctx);
|
|
1189
|
+
*
|
|
1190
|
+
* await (args.deleted === true || args.text === undefined ? rag.remove({ id: args.id }) : rag.index({ id: args.id, text: args.text }));
|
|
1191
|
+
* },
|
|
1192
|
+
* );
|
|
1193
|
+
* ```
|
|
1194
|
+
*/
|
|
1195
|
+
declare const ragSyncTriggers: <Document extends Record<string, unknown> = Record<string, unknown>>(options: RagSyncOptions<Document>) => {
|
|
1196
|
+
afterDelete: RagSyncHandler;
|
|
1197
|
+
afterInsert: RagSyncHandler;
|
|
1198
|
+
afterUpdate: RagSyncHandler;
|
|
1199
|
+
};
|
|
1200
|
+
export { type BatchRerankerOptions, type ChunkerOptions, type IndexInput, type IndexResult, type LexicalMatch, type Rag, type RagConfig, type RagContext, type RagEmbedder, type RagExtractor, type RagLexicalStore, type RagNamedFilter, type RagObjectSource, type RagQueryTransform, type RagReranker, type RagSource, type RagSourceObject, type RagSourceOptions, type RagSourceSync, type RagSqlExec, type RagSyncActionReference, type RagSyncArgs, type RagSyncOptions, type RagSyncPassOptions, type RagSyncReport, type RagTextStore, type RagToolOptions, type RagVectorMatch, type RagVectorMatches, type RagVectorQueryInput, type RagVectorRecord, type RagVectorStore, type RagVectorStoreCapabilities, type RagVectorUpsertInput, type RagVectors, type RemoveInput, type RetrieveOptions, type RetrieveResult, type RetrievedChunk, type ScoreRerankerOptions, type SqlLexicalStoreOptions, type SqliteVectorStoreOptions, type StoredRagChunk, type TokenChunkerOptions, VECTORIZE_CAPABILITIES, batchReranker, bm25LexicalStore, contentHash, defineRag, defineRagSource, fixedWindowChunks, guessMimeTypeFromExtension, hybridRank, markdownChunker, matchesMetadataFilter, ragSyncTriggers, scoreReranker, sentenceChunker, sqlLexicalStore, sqliteVectorStore, tokenChunker, vectorizeStore };
|