@lunora/ai 1.0.0-alpha.59 → 1.0.0-alpha.60

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -8,6 +8,129 @@ import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.
8
8
  * @experimental
9
9
  */
10
10
  declare const fixedWindowChunks: (text: string, size: number, overlap: number) => ReadonlyArray<string>;
11
+ /**
12
+ * Options shared by the character-budgeted chunkers. `overlap` is a **floor**
13
+ * in characters: enough trailing atoms are carried into the next window to
14
+ * cover it, so the realised overlap lands on an atom boundary and is usually a
15
+ * little larger.
16
+ */
17
+ interface ChunkerOptions {
18
+ /** Overlap floor between adjacent chunks, in characters. Default 200. Must be `< size`. */
19
+ overlap?: number;
20
+ /** Maximum chunk size, in characters. Default 1000. */
21
+ size?: number;
22
+ }
23
+ /**
24
+ * Options for {@link tokenChunker}.
25
+ *
26
+ * `countTokens` is **required and injected**: a real token count needs the
27
+ * model's tokenizer, and `@lunora/ai` will not add one as a dependency or
28
+ * pretend a characters-per-token constant is a token count — a "token-aware"
29
+ * chunker built on an estimate is a character chunker with extra steps, and it
30
+ * silently overshoots on code and non-Latin scripts, which is exactly where the
31
+ * budget matters. Pass `js-tiktoken`, `gpt-tokenizer`, or your provider's
32
+ * counter.
33
+ */
34
+ interface TokenChunkerOptions {
35
+ /** Token counter for the target model — e.g. `(text) => encoding.encode(text).length`. */
36
+ countTokens: (text: string) => number;
37
+ /** Maximum chunk size, in tokens. Default 256. */
38
+ maxTokens?: number;
39
+ /** Overlap floor between adjacent chunks, in tokens. Default 0. Must be `< maxTokens`. */
40
+ overlapTokens?: number;
41
+ }
42
+ /**
43
+ * Split on sentence boundaries, then greedily pack whole sentences into
44
+ * `size`-bounded chunks. Prefer this over the fixed window for prose: chunks
45
+ * start and end where the author ended a thought, which is what the embedding
46
+ * model was trained on.
47
+ * @experimental
48
+ */
49
+ declare const sentenceChunker: (options?: ChunkerOptions) => ((text: string) => ReadonlyArray<string>);
50
+ /**
51
+ * Split a Markdown document at its ATX headings, then pack each section's
52
+ * sentences into `size`-bounded chunks.
53
+ *
54
+ * Every emitted chunk is prefixed with its **heading trail** (`# Guide > ##
55
+ * Auth > ### OAuth`), so a chunk taken from deep inside a long document still
56
+ * carries the context that says what it is about. That prefix is what makes a
57
+ * mid-document chunk retrievable by a query naming its section rather than its
58
+ * prose, and it is why this beats {@link sentenceChunker} on structured docs.
59
+ * @experimental
60
+ */
61
+ declare const markdownChunker: (options?: ChunkerOptions) => ((text: string) => ReadonlyArray<string>);
62
+ /**
63
+ * Pack sentences into chunks bounded by a real **token** count rather than a
64
+ * character count. Use this when the embedding model's context window is the
65
+ * binding constraint — a 512-token model silently truncates anything longer, so
66
+ * the tail of an over-long chunk is embedded as if it were never written.
67
+ * @experimental
68
+ */
69
+ declare const tokenChunker: (options: TokenChunkerOptions) => ((text: string) => ReadonlyArray<string>);
70
+ /**
71
+ * What a store can hold and return. Every field is a hard limit `defineRag`
72
+ * enforces locally, so a breach fails here — naming the store and the limit —
73
+ * rather than at the backend with nothing saying why.
74
+ */
75
+ interface RagVectorStoreCapabilities {
76
+ /**
77
+ * Ceiling on embedding dimensionality, or `false` for no limit.
78
+ *
79
+ * Vectorize stores at most 1536 at 32-bit precision, which rules out
80
+ * `text-embedding-3-large` (3072) and Qwen3-Embedding (4096). A store with
81
+ * no such limit declares `false` and those models just work.
82
+ */
83
+ maxDimensions: number | false;
84
+ /**
85
+ * Ceiling on a vector's id, in bytes, or `false` for no limit.
86
+ *
87
+ * Vectorize allows 64, and chunk ids are derived from the namespace and the
88
+ * caller's source id — a bucket key like
89
+ * `handbook/engineering/onboarding/day-one.md` under a uuid namespace is
90
+ * already past it. Without this the upsert is rejected at the far side with
91
+ * nothing naming the cause, the same failure {@link RagVectorStoreCapabilities.maxMetadataBytes}
92
+ * exists to replace.
93
+ */
94
+ maxIdBytes: number | false;
95
+ /**
96
+ * Ceiling on the serialized metadata object per vector, in bytes, or
97
+ * `false` for no limit. Covers the whole object — chunk text included when
98
+ * no `textStore` moves it out.
99
+ */
100
+ maxMetadataBytes: number | false;
101
+ /** Ceiling on `topK` when only indexed metadata is requested (text-store mode). */
102
+ maxTopK: number;
103
+ /** Ceiling on `topK` when the query asks for full metadata (the default mode). */
104
+ maxTopKWithMetadata: number;
105
+ }
106
+ /**
107
+ * The storage operations `defineRag` needs. Deliberately the same four
108
+ * operations `RagVectors` already exposes — this is a capability-carrying
109
+ * wrapper, not a new protocol, so adapting an existing implementation is a
110
+ * one-liner.
111
+ */
112
+ interface RagVectorStore {
113
+ capabilities: RagVectorStoreCapabilities;
114
+ deleteByIds: (ids: ReadonlyArray<string>, namespace?: string) => Promise<unknown>;
115
+ getByIds: (ids: ReadonlyArray<string>, namespace?: string) => Promise<ReadonlyArray<RagVectorRecord>>;
116
+ query: (input: RagVectorQueryInput) => Promise<RagVectorMatches>;
117
+ upsert: (input: RagVectorUpsertInput) => Promise<unknown>;
118
+ }
119
+ /**
120
+ * Vectorize's documented limits.
121
+ *
122
+ * `maxTopKWithMetadata` is 50, matching Vectorize V2. **Legacy V1 indexes cap
123
+ * at 20** and reject a larger `topK` remotely; a binding handle does not expose
124
+ * its index version, so this cannot branch on it.
125
+ */
126
+ declare const VECTORIZE_CAPABILITIES: RagVectorStoreCapabilities;
127
+ /**
128
+ * Wrap a `ctx.vectors` facade (or any {@link RagVectors}) as a store declaring
129
+ * Vectorize's limits. This is the default when no `store` is configured, so the
130
+ * behaviour of an existing `defineRag` is unchanged.
131
+ * @experimental
132
+ */
133
+ declare const vectorizeStore: (vectors: RagVectors, indexName: string) => RagVectorStore;
11
134
  /**
12
135
  * `(text) => vector` — the embedder shape `ctx.vectors` accepts on both its
13
136
  * write (`upsert`) and read (`query`) inputs. Matches `@lunora/server`'s
@@ -131,12 +254,22 @@ interface RagContext {
131
254
  * untraced when it is absent (a hand-built context / test).
132
255
  */
133
256
  trace?: unknown;
134
- vectors: RagVectors;
257
+ /**
258
+ * The Vectorize facade backing the default store — an `ActionCtx`'s
259
+ * `ctx.vectors` satisfies it.
260
+ *
261
+ * OPTIONAL, because it is never read when {@link RagConfig.store} is set,
262
+ * and codegen only emits `ctx.vectors` for a schema that declares a vector
263
+ * index. Requiring it made "the store that needs no Vectorize" impossible
264
+ * to type-check in an app with no Vectorize index. Absent with no `store`
265
+ * configured throws a directed error when the RAG is bound.
266
+ */
267
+ vectors?: RagVectors;
135
268
  }
136
269
  /**
137
270
  * Pluggable chunk-text storage. By default chunk text is stored in vector
138
271
  * metadata (`__ragText`), which forces `returnMetadata: "all"` on retrieval and
139
- * caps `topK` at 20 (the Vectorize full-metadata ceiling) — and each vector's
272
+ * caps `topK` at 50 (the Vectorize full-metadata ceiling) — and each vector's
140
273
  * metadata must stay under the ~10 KiB Vectorize cap. Supplying a text store
141
274
  * (a DO table, KV, …) moves the text out of metadata: retrieval queries with
142
275
  * `returnMetadata: "indexed"` (topK up to 100) and hydrates text by chunk id.
@@ -163,6 +296,16 @@ interface RagTextStore {
163
296
  interface StoredRagChunk {
164
297
  chunkIndex: number;
165
298
  id: string;
299
+ /**
300
+ * The caller `metadata` attached to this chunk's source at index time
301
+ * (internal `__rag*` keys excluded).
302
+ *
303
+ * Present so a {@link RagLexicalStore} can evaluate the same metadata
304
+ * filter the vector leg gets. Without it a lexical store has nothing to
305
+ * filter on and must fail closed on every filtered query — which made
306
+ * hybrid search and metadata-based RLS mutually exclusive.
307
+ */
308
+ metadata?: Record<string, unknown>;
166
309
  sourceId: string;
167
310
  text: string;
168
311
  }
@@ -235,6 +378,20 @@ interface RagNamedFilter {
235
378
  /** The filter expression passed verbatim to Vectorize's `filter` parameter. */
236
379
  filter: Record<string, unknown>;
237
380
  }
381
+ /**
382
+ * Re-score retrieved candidates against the query. Returns the chunks in their
383
+ * new order; may drop chunks. See {@link RagConfig.rerank}.
384
+ * @experimental
385
+ */
386
+ type RagReranker = (query: string, chunks: ReadonlyArray<RetrievedChunk>) => Promise<ReadonlyArray<RetrievedChunk>> | ReadonlyArray<RetrievedChunk>;
387
+ /**
388
+ * Rewrite a query, or expand it into several. See {@link RagConfig.transformQuery}.
389
+ * @experimental
390
+ */
391
+ type RagQueryTransform = (query: string, info: {
392
+ conversationId?: string;
393
+ namespace?: string;
394
+ }) => Promise<ReadonlyArray<string> | string> | ReadonlyArray<string> | string;
238
395
  /**
239
396
  * `RagConfig` is part of the experimental `@lunora/ai` API and may change without a major version bump.
240
397
  * @experimental
@@ -247,6 +404,36 @@ interface RagConfig {
247
404
  * vectors across every tenant.
248
405
  */
249
406
  allowSharedNamespace?: boolean;
407
+ /**
408
+ * Retain up to this many embeddings per bound context, keyed by text, so a
409
+ * repeated `retrieve()` of the same question does not re-embed it.
410
+ *
411
+ * Default 0 (retain nothing beyond one call). Indexing always batches its
412
+ * embeds regardless of this setting — the batch lives in its own
413
+ * request-scoped map, released when `index()` returns, so it neither needs
414
+ * this budget nor evicts what is held in it.
415
+ *
416
+ * Sized in entries, not bytes, but budget in bytes: one 1536-dimension
417
+ * embedding is ~12 KB, so 100 entries is over a megabyte held in the
418
+ * isolate. Keep it small.
419
+ */
420
+ cacheEmbeddings?: number;
421
+ /**
422
+ * How many chunks each retrieval leg fetches **before** fusion and
423
+ * reranking trim the result to `topK`.
424
+ *
425
+ * Defaults to `topK * 4` whenever anything downstream reorders — a
426
+ * `lexicalStore`, a multi-query `transformQuery`, or a `rerank` — and to
427
+ * plain `topK` otherwise. Bounded by the store ceiling (50 in metadata
428
+ * mode, 100 with a `textStore`).
429
+ *
430
+ * This is the knob that decides how much recall the reordering has to work
431
+ * with. Fetching only `topK` per leg defeats the point of having two: the
432
+ * whole reason to run a lexical leg is to surface a chunk the vector leg
433
+ * ranked *below* `topK`, and it cannot do that if it was never asked for
434
+ * more than `topK`.
435
+ */
436
+ candidates?: number;
250
437
  /** Custom chunker; overrides the built-in fixed-window splitter. */
251
438
  chunk?: (text: string) => ReadonlyArray<string>;
252
439
  /** Overlap (chars) between adjacent chunks. Default 200. Must be < `chunkSize`. */
@@ -294,6 +481,23 @@ interface RagConfig {
294
481
  lexicalStore?: RagLexicalStore;
295
482
  /** Retrieval depth for the lexical leg of hybrid search. Defaults to the effective `topK`. */
296
483
  lexicalTopK?: number;
484
+ /**
485
+ * Ceiling on the embedding model's dimensionality, checked once per bound
486
+ * context against the first embedding actually produced. Defaults to
487
+ * **1536** — Vectorize's per-vector limit at 32-bit precision.
488
+ *
489
+ * The default rules out most current large embedding models
490
+ * (`text-embedding-3-large` and Gemini embedding at 3072,
491
+ * Qwen3-Embedding at 4096). Without the check they fail at Vectorize with
492
+ * nothing naming the cause; with it they fail at the first embed, naming
493
+ * the ceiling and both escapes — truncate via the provider's Matryoshka
494
+ * `dimensions` option, or set this to `false` when the index is not
495
+ * Vectorize-backed.
496
+ *
497
+ * `false` disables the check entirely: the right setting for a
498
+ * {@link RagVectors} implementation with a different (or no) ceiling.
499
+ */
500
+ maxEmbeddingDimensions?: number | false;
297
501
  /**
298
502
  * Enforce tenant isolation: throw (instead of the one-time dev warning)
299
503
  * when `index`/`retrieve`/`remove` run without a `namespace`. Recommended
@@ -301,6 +505,23 @@ interface RagConfig {
301
505
  * in metadata mode the leaked payload includes raw chunk text.
302
506
  */
303
507
  requireNamespace?: boolean;
508
+ /**
509
+ * Re-score the retrieved candidates against the query before they are
510
+ * trimmed to `topK`. **Injected, not bundled** — a reranker is a model call,
511
+ * and `@lunora/ai` takes no provider dependency to make one. Adapt yours
512
+ * with `scoreReranker`/`batchReranker`, or write the two-line hook yourself.
513
+ *
514
+ * This is the standard quality step that vector search alone cannot do:
515
+ * an embedding is computed without the query, so it cannot know which of
516
+ * two topically-similar passages actually answers *this* question. A
517
+ * cross-encoder sees both at once and orders them accordingly.
518
+ *
519
+ * Retrieval fetches {@link RagConfig.candidates} chunks, hands them
520
+ * here, and keeps the first `topK` of whatever comes back — so the hook may
521
+ * reorder and drop, but its output order is final. Runs after hybrid fusion
522
+ * and before `chunkContext` expansion.
523
+ */
524
+ rerank?: RagReranker;
304
525
  /**
305
526
  * Row-level-security filter derived from the retrieval identity. Called once
306
527
  * per `retrieve()` with {@link RagContext.auth} (the bound ctx's `auth`); the
@@ -319,10 +540,42 @@ interface RagConfig {
319
540
  * ```
320
541
  */
321
542
  rlsFilter?: (auth: unknown) => Promise<Record<string, unknown> | undefined> | Record<string, unknown> | undefined;
543
+ /**
544
+ * Back this RAG with a different vector store.
545
+ *
546
+ * Called once per bound context with that context, so a store needing
547
+ * per-request state (a Hyperdrive/pgvector connection from `ctx.sql`, a
548
+ * shard's own SQLite) can build itself from it. Defaults to wrapping
549
+ * `context.vectors` as a Vectorize-backed store.
550
+ *
551
+ * The store declares its own limits, and `defineRag` reads them instead of
552
+ * assuming Vectorize's — so a pgvector index is not held to a
553
+ * 1536-dimension ceiling or a 10 KiB metadata budget it does not have.
554
+ */
555
+ store?: (context: RagContext) => RagVectorStore;
322
556
  /** Chunk-text storage override — see {@link RagTextStore}. */
323
557
  textStore?: RagTextStore;
324
- /** Default retrieval depth. Default 5. Capped at 20 (metadata mode) / 100 (text-store mode). */
558
+ /** Default retrieval depth. Default 5. Capped by the store: 50 (metadata mode) / 100 (text-store mode) on Vectorize. */
325
559
  topK?: number;
560
+ /**
561
+ * Rewrite or expand the query before it is embedded.
562
+ *
563
+ * The raw user query is often the worst possible search string: a
564
+ * conversational follow-up ("what about the other one?") carries its
565
+ * meaning in the preceding turns, and a short question shares few terms
566
+ * with the long passage that answers it.
567
+ *
568
+ * Return **one** string to rewrite, or **several** to run multi-query
569
+ * retrieval — each is embedded and searched independently and the rankings
570
+ * are fused with RRF, which recovers passages any single phrasing would
571
+ * miss. Returning the query unchanged is a no-op.
572
+ *
573
+ * **Injected, not bundled**, for the same reason as {@link RagConfig.rerank}:
574
+ * every useful strategy (HyDE, multi-query expansion, follow-up rewriting)
575
+ * needs a language model, and this package does not pick one for you. The
576
+ * lexical leg searches the first returned query.
577
+ */
578
+ transformQuery?: RagQueryTransform;
326
579
  }
327
580
  /**
328
581
  * `IndexInput` is part of the experimental `@lunora/ai` API and may change without a major version bump.
@@ -356,6 +609,18 @@ interface IndexInput {
356
609
  text: string;
357
610
  total: number;
358
611
  }) => void;
612
+ /**
613
+ * Index this source even when its content hash is unchanged.
614
+ *
615
+ * The hash short-circuit skips chunking, embedding and every write — which
616
+ * is what makes a cron re-sync cheap, and also what makes attaching a
617
+ * `textStore` or `lexicalStore` to an ALREADY-indexed corpus a silent no-op:
618
+ * the new store is never written, so the keyword leg returns nothing
619
+ * forever with no error. Set this for the one pass that backfills it.
620
+ *
621
+ * It re-embeds, so it is not a setting to leave on.
622
+ */
623
+ reindex?: boolean;
359
624
  /** The document body to chunk + embed + upsert. */
360
625
  text: string;
361
626
  }
@@ -415,7 +680,15 @@ interface RetrieveOptions {
415
680
  matches: number;
416
681
  query: string;
417
682
  }) => void;
683
+ /**
684
+ * Set `false` to skip {@link RagConfig.rerank} for this call — the escape
685
+ * hatch for a latency-sensitive path (typeahead, an agent's inner loop)
686
+ * that cannot afford the extra model round-trip.
687
+ */
688
+ rerank?: false;
418
689
  topK?: number;
690
+ /** Set `false` to skip {@link RagConfig.transformQuery} for this call. */
691
+ transformQuery?: false;
419
692
  }
420
693
  /**
421
694
  * `RetrievedChunk` is part of the experimental `@lunora/ai` API and may change without a major version bump.
@@ -524,15 +797,26 @@ declare const contentHash: (data: BufferSource) => Promise<string>;
524
797
  *
525
798
  * Each result set contributes `1 / (k + rank)` to each chunk's fused score,
526
799
  * where `rank` is 0-based position in the list. The constant `k` (default 60)
527
- * dampens the influence of high ranks — the standard value from the RRF
528
- * literature that works well across domains.
800
+ * dampens the influence of high ranks.
529
801
  *
530
- * The fused list is sorted descending by fused score. Ties are broken by
531
- * preferring the chunk ranked higher in the vector search result (typically
532
- * the more semantically accurate of the two methods).
802
+ * **The returned chunks carry the fused score in `score`**, multiplied by the
803
+ * chunk's `importance` so source weighting still applies. Writing it back is
804
+ * what makes the fusion survive: a caller that re-sorts by `score` — as
805
+ * `retrieve()` does, to apply importance weighting — would otherwise re-order
806
+ * by the incomparable inputs and discard the ranking this function computed.
807
+ * Since BM25 is unbounded while cosine is `[0, 1]`, that silently promoted
808
+ * every lexical-only hit above every vector hit.
809
+ *
810
+ * So in hybrid mode `RetrievedChunk.score` is an RRF score (small, ~`1/60`
811
+ * scale), not a cosine similarity. `retrieve()` applies `minScore` to the
812
+ * vector leg *before* fusion for exactly this reason — the option is documented
813
+ * against the cosine scale.
814
+ *
815
+ * Ties are broken by preferring the chunk ranked higher in the vector result,
816
+ * typically the more semantically accurate of the two methods.
533
817
  *
534
818
  * Callers MUST ensure every chunk in both lists carries a unique, comparable
535
- * `id` — this is guaranteed by the chunk-id scheme `${sourceId}#${chunkIndex}`.
819
+ * `id` — guaranteed by the chunk-id scheme `${sourceId}#${chunkIndex}`.
536
820
  * @experimental
537
821
  */
538
822
  declare const hybridRank: (vectorResults: ReadonlyArray<RetrievedChunk>, textResults: ReadonlyArray<RetrievedChunk>, k?: number) => ReadonlyArray<RetrievedChunk>;
@@ -545,16 +829,260 @@ declare const hybridRank: (vectorResults: ReadonlyArray<RetrievedChunk>, textRes
545
829
  * {@link RagLexicalStore} (a DO-SQLite inverted index, D1, or an external search
546
830
  * service) behind the same seam.
547
831
  *
548
- * Tenant isolation is by `namespace` (each namespace keeps its own index). This
549
- * store holds **no metadata**, so it cannot evaluate a metadata `filter`
550
- * (including an `rlsFilter` result): when `search` is called with a non-empty
551
- * filter it **fails closed** — returns no lexical hits and warns once — rather
552
- * than risk surfacing a row the filter would exclude. If your RLS is
553
- * metadata-based (not namespace-based) and you want a lexical leg, fold the RLS
554
- * dimension into the `namespace` or plug a filter-aware store.
832
+ * Tenant isolation is by `namespace` (each namespace keeps its own index), and
833
+ * each chunk's source `metadata` is stored alongside it so `search` evaluates
834
+ * the **same** metadata predicate the vector leg receives — including an
835
+ * `rlsFilter` result. A hit the filter excludes never reaches fusion.
836
+ *
837
+ * That matters because the filter carries the tenant/RBAC scope: a lexical leg
838
+ * that ignored it would leak excluded chunk text into the fused result no
839
+ * matter what the vector leg returned. This store previously had no metadata to
840
+ * check and so refused every filtered query, which made hybrid search and
841
+ * metadata-based RLS mutually exclusive.
555
842
  * @experimental
556
843
  */
557
844
  declare const bm25LexicalStore: () => RagLexicalStore;
845
+ /**
846
+ * True when `metadata` satisfies every clause of `filter`. An empty or absent
847
+ * filter matches everything; absent metadata satisfies only an empty filter.
848
+ *
849
+ * Clauses are ANDed, matching Vectorize.
850
+ */
851
+ declare const matchesMetadataFilter: (metadata: Record<string, unknown> | undefined, filter: Record<string, unknown> | undefined) => boolean;
852
+ /** Options for {@link scoreReranker}. */
853
+ interface ScoreRerankerOptions {
854
+ /**
855
+ * Drop chunks scoring below this threshold, in the scorer's own scale.
856
+ * Omitted → nothing is dropped and the reranker only reorders.
857
+ *
858
+ * Worth setting: reranking's real value is not just promoting the best
859
+ * passage but *rejecting* the ones vector search only matched topically,
860
+ * and a chunk that survives to the prompt is a chunk the model may cite.
861
+ */
862
+ minScore?: number;
863
+ /**
864
+ * Score one `(query, passage)` pair for relevance. Higher is better; the
865
+ * scale does not matter, only the ordering it induces.
866
+ *
867
+ * Called once per candidate chunk. Wrap a Workers AI reranker
868
+ * (`ctx.ai.run("@cf/baai/bge-reranker-base", …)`), a Cohere/Voyage rerank
869
+ * endpoint, or an LLM prompted to rate relevance.
870
+ */
871
+ score: (query: string, text: string) => Promise<number> | number;
872
+ }
873
+ /**
874
+ * Batched variant — score every candidate in one call.
875
+ */
876
+ interface BatchRerankerOptions {
877
+ /** Drop chunks scoring below this threshold. See {@link ScoreRerankerOptions.minScore}. */
878
+ minScore?: number;
879
+ /**
880
+ * Score all candidates at once, returning one number per passage **in the
881
+ * same order**. Prefer this over {@link ScoreRerankerOptions.score} when the
882
+ * provider has a batch endpoint: cross-encoder APIs are priced and rate
883
+ * limited per request, so N passages in one call beats N calls.
884
+ *
885
+ * A result whose length does not match the input is rejected rather than
886
+ * zipped against the wrong passages.
887
+ */
888
+ scoreAll: (query: string, texts: ReadonlyArray<string>) => Promise<ReadonlyArray<number>> | ReadonlyArray<number>;
889
+ }
890
+ /**
891
+ * Adapt a per-passage relevance scorer into a {@link RagReranker}.
892
+ *
893
+ * The scorer is **injected** — `@lunora/ai` takes no provider dependency to
894
+ * make a model call. Scoring runs with bounded concurrency so a 50-candidate
895
+ * pool does not fan out 50 simultaneous subrequests.
896
+ *
897
+ * ```ts
898
+ * defineRag({
899
+ * index: "docs",
900
+ * rerank: scoreReranker({
901
+ * score: async (query, text) => {
902
+ * const result = await ctx.ai.run("@cf/baai/bge-reranker-base", { query, contexts: [{ text }] });
903
+ * return result.response[0].score;
904
+ * },
905
+ * }),
906
+ * });
907
+ * ```
908
+ * @experimental
909
+ */
910
+ declare const scoreReranker: (options: ScoreRerankerOptions) => RagReranker;
911
+ /**
912
+ * Adapt a **batch** relevance scorer into a {@link RagReranker} — one call for
913
+ * the whole candidate pool. See {@link BatchRerankerOptions.scoreAll}.
914
+ * @experimental
915
+ */
916
+ declare const batchReranker: (options: BatchRerankerOptions) => RagReranker;
917
+ /** One object listed by a {@link RagObjectSource}. */
918
+ interface RagSourceObject {
919
+ /** Content type, when the source knows it. Falls back to the key's extension. */
920
+ contentType?: string;
921
+ /** Stable key — becomes the indexed source id. */
922
+ key: string;
923
+ /** Metadata attached to every chunk of this object. */
924
+ metadata?: Record<string, unknown>;
925
+ }
926
+ /**
927
+ * Where documents come from. Two operations, both injected.
928
+ *
929
+ * `list` is an async iterable rather than an array so the caller decides how to
930
+ * page a large bucket rather than being forced to build one array up front.
931
+ * `sync` still collects the KEYS it yields — it needs the full current key set
932
+ * to work out what disappeared — but never more than one object's BODY at a
933
+ * time, which is where the memory actually is.
934
+ */
935
+ interface RagObjectSource {
936
+ /** Fetch one object's raw text. Return `undefined` to skip it (unreadable, unsupported). */
937
+ get: (object: RagSourceObject) => Promise<string | undefined> | string | undefined;
938
+ /** Enumerate the objects to index. */
939
+ list: () => AsyncIterable<RagSourceObject> | Iterable<RagSourceObject>;
940
+ }
941
+ /** Turn a fetched object's raw text into indexable plain text. */
942
+ type RagExtractor = (raw: string, object: RagSourceObject) => Promise<string | undefined> | string | undefined;
943
+ /** Options for {@link defineRagSource}. */
944
+ interface RagSourceOptions {
945
+ /**
946
+ * How many objects to process at once. Default 4.
947
+ *
948
+ * Each one is a fetch plus an embed plus an upsert, so this multiplies into
949
+ * the subrequest budget — the default is deliberately low.
950
+ */
951
+ concurrency?: number;
952
+ /**
953
+ * Extractors keyed by content type (`text/html`, `application/pdf`, …), or
954
+ * `"*"` as a fallback. A content type with no extractor and no `"*"` entry
955
+ * is skipped and counted in `skipped` — never indexed as raw bytes, which
956
+ * would fill the index with markup or binary noise that embeds to nothing
957
+ * meaningful.
958
+ */
959
+ extractors?: Record<string, RagExtractor>;
960
+ /** Namespace (tenant/shard key) applied to every indexed object. */
961
+ namespace?: string;
962
+ /** Called after each object is handled — for progress reporting. */
963
+ onObject?: (info: {
964
+ chunks: number;
965
+ key: string;
966
+ status: "indexed" | "skipped" | "unchanged";
967
+ }) => void;
968
+ }
969
+ /** Per-pass options for {@link RagSourceSync.sync}. */
970
+ interface RagSyncPassOptions {
971
+ /**
972
+ * The keys this index is believed to already hold. Any of them missing from
973
+ * this pass's `list()` is deleted from the index.
974
+ *
975
+ * Pruning is what makes the index a mirror rather than an append-only pile:
976
+ * a document removed at the source but left in the index keeps being
977
+ * retrieved and cited, which is worse than never having indexed it.
978
+ *
979
+ * It is the CALLER's set, and required, because nothing else can hold it
980
+ * honestly. A previous pass's keys remembered in the instance would be
981
+ * empty on the shape this helper is documented in — built per request, from
982
+ * a per-request `rag(ctx)` — so a prune that defaulted on would in practice
983
+ * never run and never say so. Persist the set (a table, a KV key, the
984
+ * bucket listing itself) and hand it in.
985
+ *
986
+ * Omitted ⇒ nothing is pruned, and `pruned` comes back empty.
987
+ */
988
+ knownKeys?: Iterable<string>;
989
+ }
990
+ /** What one {@link RagSourceSync.sync} pass did. */
991
+ interface RagSyncReport {
992
+ /** Keys indexed for the first time or re-indexed after a change. */
993
+ indexed: string[];
994
+ /** Keys deleted from the index because they no longer exist at the source. Empty unless `knownKeys` was supplied. */
995
+ pruned: string[];
996
+ /** Keys skipped — no extractor, or the source returned nothing. */
997
+ skipped: string[];
998
+ /** Keys whose content hash matched, so nothing was embedded. */
999
+ unchanged: string[];
1000
+ }
1001
+ /** The bound ingestion surface. */
1002
+ interface RagSourceSync {
1003
+ /** Run one full pass over the source. */
1004
+ sync: (source: RagObjectSource, passOptions?: RagSyncPassOptions) => Promise<RagSyncReport>;
1005
+ }
1006
+ /**
1007
+ * Declare a bulk-ingestion pass over a {@link Rag}.
1008
+ *
1009
+ * ```ts
1010
+ * const ingest = defineRagSource(docs(ctx), { namespace: ctx.shardKey });
1011
+ *
1012
+ * const report = await ingest.sync(
1013
+ * {
1014
+ * list: async function* () {
1015
+ * for await (const object of bucket.list()) {
1016
+ * yield { key: object.key, metadata: { url: object.key } };
1017
+ * }
1018
+ * },
1019
+ * get: async (object) => (await bucket.get(object.key))?.text(),
1020
+ * },
1021
+ * // Pass what you already indexed to have deletions mirrored; omit it and
1022
+ * // nothing is pruned.
1023
+ * { knownKeys: await ctx.db.query("indexedDocs").collect().then((rows) => rows.map((row) => row.key)) },
1024
+ * );
1025
+ * ```
1026
+ * @experimental
1027
+ */
1028
+ declare const defineRagSource: (rag: Rag, options?: RagSourceOptions) => RagSourceSync;
1029
+ /**
1030
+ * The injected SQL seam the durable RAG stores share.
1031
+ *
1032
+ * `@lunora/ai` depends on no database. A store is handed an executor instead,
1033
+ * so the same adapter runs on a Durable Object's SQLite, D1, `node:sqlite`, or
1034
+ * Postgres over Hyperdrive — and this package stays free of every one of them.
1035
+ *
1036
+ * The statements the stores emit are SQLite-shaped (`ON CONFLICT … DO UPDATE`,
1037
+ * `CREATE INDEX IF NOT EXISTS`, `?` placeholders), so the executor must front a
1038
+ * SQLite engine — a Durable Object, D1, or `node:sqlite`.
1039
+ * @experimental
1040
+ */
1041
+ /**
1042
+ * Run one statement and return its rows.
1043
+ *
1044
+ * Rows come back as plain objects keyed by column name — the shape
1045
+ * `SqlStorage#exec().toArray()`, D1's `.all().results`, and `postgres.js` all
1046
+ * already produce. A statement returning nothing yields an empty array.
1047
+ *
1048
+ * May be synchronous: a Durable Object's SQLite is, and forcing it through a
1049
+ * promise would add a microtask per row batch for nothing.
1050
+ */
1051
+ type RagSqlExec = (sql: string, parameters: ReadonlyArray<unknown>) => Promise<ReadonlyArray<Record<string, unknown>>> | ReadonlyArray<Record<string, unknown>>;
1052
+ /** Options for {@link sqlLexicalStore}. */
1053
+ interface SqlLexicalStoreOptions {
1054
+ /** Execute one statement. See {@link RagSqlExec}. */
1055
+ exec: RagSqlExec;
1056
+ /**
1057
+ * Table-name prefix. Default `lunora_rag_lexical`; the postings table is
1058
+ * this plus `_terms`. Must be a bare SQL identifier.
1059
+ */
1060
+ table?: string;
1061
+ }
1062
+ declare const sqlLexicalStore: (options: SqlLexicalStoreOptions) => RagLexicalStore;
1063
+ /** Options for {@link sqliteVectorStore}. */
1064
+ interface SqliteVectorStoreOptions {
1065
+ /** Execute one statement. See {@link RagSqlExec}. */
1066
+ exec: RagSqlExec;
1067
+ /**
1068
+ * Ceiling on embedding dimensionality. Defaults to `false` (no limit) —
1069
+ * vectors are stored as JSON, so nothing here cares how wide they are, and
1070
+ * inheriting Vectorize's 1536 would be inventing a constraint.
1071
+ */
1072
+ maxDimensions?: number | false;
1073
+ /**
1074
+ * Upper bound on how many rows a single namespace may be scanned for.
1075
+ * Default 50,000.
1076
+ *
1077
+ * Search is linear, so this is the difference between a slow query and a
1078
+ * Worker that exceeds its CPU budget and is killed with nothing explaining
1079
+ * why. Exceeding it throws, naming the namespace and the count.
1080
+ */
1081
+ maxScan?: number;
1082
+ /** Table name. Default `lunora_rag_vectors`. Must be a bare SQL identifier. */
1083
+ table?: string;
1084
+ }
1085
+ declare const sqliteVectorStore: (options: SqliteVectorStoreOptions) => RagVectorStore;
558
1086
  /**
559
1087
  * Keep a RAG index in step with a table, so the table IS the index.
560
1088
  *
@@ -658,4 +1186,4 @@ declare const ragSyncTriggers: <Document extends Record<string, unknown> = Recor
658
1186
  afterInsert: RagSyncHandler;
659
1187
  afterUpdate: RagSyncHandler;
660
1188
  };
661
- export { type IndexInput, type IndexResult, type LexicalMatch, type Rag, type RagConfig, type RagContext, type RagEmbedder, type RagLexicalStore, type RagNamedFilter, type RagSource, type RagSyncActionReference, type RagSyncArgs, type RagSyncOptions, type RagTextStore, type RagToolOptions, type RagVectorMatch, type RagVectorMatches, type RagVectorQueryInput, type RagVectorRecord, type RagVectorUpsertInput, type RagVectors, type RemoveInput, type RetrieveOptions, type RetrieveResult, type RetrievedChunk, type StoredRagChunk, bm25LexicalStore, contentHash, defineRag, fixedWindowChunks, guessMimeTypeFromExtension, hybridRank, ragSyncTriggers };
1189
+ export { type BatchRerankerOptions, type ChunkerOptions, type IndexInput, type IndexResult, type LexicalMatch, type Rag, type RagConfig, type RagContext, type RagEmbedder, type RagExtractor, type RagLexicalStore, type RagNamedFilter, type RagObjectSource, type RagQueryTransform, type RagReranker, type RagSource, type RagSourceObject, type RagSourceOptions, type RagSourceSync, type RagSqlExec, type RagSyncActionReference, type RagSyncArgs, type RagSyncOptions, type RagSyncPassOptions, type RagSyncReport, type RagTextStore, type RagToolOptions, type RagVectorMatch, type RagVectorMatches, type RagVectorQueryInput, type RagVectorRecord, type RagVectorStore, type RagVectorStoreCapabilities, type RagVectorUpsertInput, type RagVectors, type RemoveInput, type RetrieveOptions, type RetrieveResult, type RetrievedChunk, type ScoreRerankerOptions, type SqlLexicalStoreOptions, type SqliteVectorStoreOptions, type StoredRagChunk, type TokenChunkerOptions, VECTORIZE_CAPABILITIES, batchReranker, bm25LexicalStore, contentHash, defineRag, defineRagSource, fixedWindowChunks, guessMimeTypeFromExtension, hybridRank, markdownChunker, matchesMetadataFilter, ragSyncTriggers, scoreReranker, sentenceChunker, sqlLexicalStore, sqliteVectorStore, tokenChunker, vectorizeStore };