@lunora/bindings 1.0.0-alpha.8 → 1.0.0-alpha.81
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/dist/analytics/index.d.mts +81 -85
- package/dist/analytics/index.d.ts +81 -85
- package/dist/analytics/index.mjs +1 -2
- package/dist/images/index.d.mts +159 -174
- package/dist/images/index.d.ts +159 -174
- package/dist/images/index.mjs +1 -3
- package/dist/kv/index.d.mts +74 -151
- package/dist/kv/index.d.ts +74 -151
- package/dist/kv/index.mjs +1 -2
- package/dist/packem_shared/AnalyticsSqlError-CeQ3A5Eb.mjs +1 -0
- package/dist/packem_shared/R2SqlError-ZHgOWClB.mjs +1 -0
- package/dist/packem_shared/SelectBuilder-DJXNSdqC.mjs +1 -0
- package/dist/packem_shared/SetOperation-kiI5Wnlm.mjs +1 -0
- package/dist/packem_shared/Sql-BfnxRway.mjs +1 -0
- package/dist/packem_shared/WindowExpression-VX7EEV3h.mjs +1 -0
- package/dist/packem_shared/WindowFunction-CL4jYy2l.mjs +1 -0
- package/dist/packem_shared/asc-DP_WFiAE.mjs +1 -0
- package/dist/packem_shared/buildImageDeliveryUrl-Brqs-dcZ.mjs +1 -0
- package/dist/packem_shared/buildSignedImageUrl-BV-iSJKA.mjs +4 -0
- package/dist/packem_shared/cap-error-body-YBKO32BF.mjs +1 -0
- package/dist/packem_shared/concurrent-vRmSvRpF.mjs +1 -0
- package/dist/packem_shared/createAnalytics-BXTNc57d.mjs +1 -0
- package/dist/packem_shared/createContextVectors-DiyO3pZU.mjs +6 -0
- package/dist/packem_shared/createImages-D7JExfqF.mjs +1 -0
- package/dist/packem_shared/createKv-mAHanD5g.mjs +1 -0
- package/dist/packem_shared/createKvIntrospector-BpRiFRFQ.mjs +1 -0
- package/dist/packem_shared/createPipelines-CIvqrc7E.mjs +1 -0
- package/dist/packem_shared/createVectorAdminIntrospector-Ct8v6PxJ.mjs +1 -0
- package/dist/packem_shared/createVectors-Dzv0ilKE.mjs +1 -0
- package/dist/pipelines/index.d.mts +24 -24
- package/dist/pipelines/index.d.ts +24 -24
- package/dist/pipelines/index.mjs +1 -1
- package/dist/r2sql/index.d.mts +169 -125
- package/dist/r2sql/index.d.ts +169 -125
- package/dist/r2sql/index.mjs +1 -7
- package/dist/vectors/index.d.mts +299 -134
- package/dist/vectors/index.d.ts +299 -134
- package/dist/vectors/index.mjs +1 -3
- package/package.json +3 -2
- package/dist/packem_shared/AnalyticsSqlError-C2nz3jpH.mjs +0 -41
- package/dist/packem_shared/R2SqlError-drPKSCZ3.mjs +0 -65
- package/dist/packem_shared/SelectBuilder-BOqJQHEv.mjs +0 -168
- package/dist/packem_shared/SetOperation-DmPgUL8W.mjs +0 -81
- package/dist/packem_shared/Sql-B3zq2YGx.mjs +0 -74
- package/dist/packem_shared/WindowExpression-BT_uA6g1.mjs +0 -44
- package/dist/packem_shared/WindowFunction-DrnuZUF6.mjs +0 -82
- package/dist/packem_shared/asc-DZbQCxh1.mjs +0 -16
- package/dist/packem_shared/buildImageDeliveryUrl-qZ7XbqTL.mjs +0 -35
- package/dist/packem_shared/buildSignedImageUrl-DNUFfyGP.mjs +0 -130
- package/dist/packem_shared/concurrent-CkCEVwqP.mjs +0 -39
- package/dist/packem_shared/createAnalytics-CEEI69o9.mjs +0 -57
- package/dist/packem_shared/createContextVectors-DwZtnPeC.mjs +0 -140
- package/dist/packem_shared/createImages-BzRnsz3H.mjs +0 -85
- package/dist/packem_shared/createKv-C8Iyu5hD.mjs +0 -145
- package/dist/packem_shared/createKvIntrospector-Byk4GfsY.mjs +0 -77
- package/dist/packem_shared/createPipelines-CfyJ6VGu.mjs +0 -10
- package/dist/packem_shared/createVectorAdminIntrospector-DuSvcBa5.mjs +0 -53
- package/dist/packem_shared/createVectors-CTSrctiK.mjs +0 -95
package/dist/vectors/index.d.mts
CHANGED
|
@@ -1,67 +1,17 @@
|
|
|
1
|
+
import { VectorizeDeleteMutation, VectorizeIndexDetails, VectorizeVector, VectorizeMatches, VectorizeUpsertMutation, VectorizeIndexLike, VectorMetric } from '@lunora/platform';
|
|
2
|
+
export type { VectorMetric, VectorizeDeleteMutation, VectorizeIndexDetails, VectorizeIndexLike, VectorizeMatch, VectorizeMatches, VectorizeQueryOptions, VectorizeUpsertMutation, VectorizeVector } from '@lunora/platform';
|
|
1
3
|
/**
|
|
2
|
-
*
|
|
3
|
-
*
|
|
4
|
-
*
|
|
5
|
-
|
|
6
|
-
*/
|
|
7
|
-
interface VectorizeIndexLike {
|
|
8
|
-
deleteByIds: (ids: ReadonlyArray<string>) => Promise<VectorizeDeleteMutation>;
|
|
9
|
-
describe?: () => Promise<VectorizeIndexDetails>;
|
|
10
|
-
getByIds: (ids: ReadonlyArray<string>) => Promise<ReadonlyArray<VectorizeVector>>;
|
|
11
|
-
insert: (vectors: ReadonlyArray<VectorizeVector>) => Promise<VectorizeUpsertMutation>;
|
|
12
|
-
query: (vector: ReadonlyArray<number>, options?: VectorizeQueryOptions) => Promise<VectorizeMatches>;
|
|
13
|
-
upsert: (vectors: ReadonlyArray<VectorizeVector>) => Promise<VectorizeUpsertMutation>;
|
|
14
|
-
}
|
|
15
|
-
type VectorMetric = "cosine" | "euclidean" | "dot-product";
|
|
16
|
-
interface VectorizeVector {
|
|
17
|
-
id: string;
|
|
18
|
-
metadata?: Record<string, unknown>;
|
|
19
|
-
namespace?: string;
|
|
20
|
-
values: ReadonlyArray<number>;
|
|
21
|
-
}
|
|
22
|
-
interface VectorizeQueryOptions {
|
|
23
|
-
filter?: Record<string, unknown>;
|
|
24
|
-
namespace?: string;
|
|
25
|
-
returnMetadata?: "none" | "indexed" | "all";
|
|
26
|
-
returnValues?: boolean;
|
|
27
|
-
topK?: number;
|
|
28
|
-
}
|
|
29
|
-
interface VectorizeMatch {
|
|
30
|
-
id: string;
|
|
31
|
-
metadata?: Record<string, unknown>;
|
|
32
|
-
namespace?: string;
|
|
33
|
-
score: number;
|
|
34
|
-
values?: ReadonlyArray<number>;
|
|
35
|
-
}
|
|
36
|
-
interface VectorizeMatches {
|
|
37
|
-
count: number;
|
|
38
|
-
matches: ReadonlyArray<VectorizeMatch>;
|
|
39
|
-
}
|
|
40
|
-
interface VectorizeUpsertMutation {
|
|
41
|
-
mutationId: string;
|
|
42
|
-
}
|
|
43
|
-
interface VectorizeDeleteMutation {
|
|
44
|
-
count?: number;
|
|
45
|
-
mutationId: string;
|
|
46
|
-
}
|
|
47
|
-
interface VectorizeIndexDetails {
|
|
48
|
-
dimensions: number;
|
|
49
|
-
processedUpToDatetime?: string;
|
|
50
|
-
processedUpToMutation?: string;
|
|
51
|
-
vectorsCount: number;
|
|
52
|
-
}
|
|
53
|
-
/**
|
|
54
|
-
* Bring-your-own-embedder: a user-supplied async fn that converts a single
|
|
55
|
-
* source value (a row, a chunk, an arbitrary string) into a numeric vector.
|
|
56
|
-
* The runtime calls this at upsert time so we don't couple to any provider.
|
|
57
|
-
*/
|
|
4
|
+
* Bring-your-own-embedder: a user-supplied async fn that converts a single
|
|
5
|
+
* source value (a row, a chunk, an arbitrary string) into a numeric vector.
|
|
6
|
+
* The runtime calls this at upsert time so we don't couple to any provider.
|
|
7
|
+
*/
|
|
58
8
|
type EmbedFunction<TInput = unknown> = (input: TInput) => Promise<ReadonlyArray<number>> | ReadonlyArray<number>;
|
|
59
9
|
interface LunoraVectorsOptions {
|
|
60
10
|
/**
|
|
61
|
-
|
|
62
|
-
|
|
63
|
-
|
|
64
|
-
|
|
11
|
+
* Map of logical index name -> Vectorize binding. Most apps wire one
|
|
12
|
+
* binding per index; multi-index apps register all of them here so calls
|
|
13
|
+
* like `vectors.query("docs-body", ...)` can resolve to the right binding.
|
|
14
|
+
*/
|
|
65
15
|
indexes: Record<string, VectorizeIndexLike>;
|
|
66
16
|
}
|
|
67
17
|
interface UpsertInput<TInput = unknown> {
|
|
@@ -91,9 +41,9 @@ interface LunoraVectors {
|
|
|
91
41
|
upsertMany: <TInput>(indexName: string, inputs: ReadonlyArray<UpsertInput<TInput>>) => Promise<VectorizeUpsertMutation>;
|
|
92
42
|
}
|
|
93
43
|
/**
|
|
94
|
-
* `(input: string) => vector`. Matches `@lunora/server`'s `VectorEmbedder` so
|
|
95
|
-
* the bridged surface is assignable to the server's `VectorSearch` contract.
|
|
96
|
-
*/
|
|
44
|
+
* `(input: string) => vector`. Matches `@lunora/server`'s `VectorEmbedder` so
|
|
45
|
+
* the bridged surface is assignable to the server's `VectorSearch` contract.
|
|
46
|
+
*/
|
|
97
47
|
type VectorEmbedderLike = (input: string) => Promise<ReadonlyArray<number>> | ReadonlyArray<number>;
|
|
98
48
|
interface VectorMatchLike {
|
|
99
49
|
id: string;
|
|
@@ -107,6 +57,7 @@ interface VectorMatchesLike {
|
|
|
107
57
|
interface VectorRecordLike {
|
|
108
58
|
id: string;
|
|
109
59
|
metadata?: Record<string, unknown>;
|
|
60
|
+
namespace?: string;
|
|
110
61
|
values: ReadonlyArray<number>;
|
|
111
62
|
}
|
|
112
63
|
interface VectorQueryInputLike {
|
|
@@ -115,11 +66,11 @@ interface VectorQueryInputLike {
|
|
|
115
66
|
input?: string;
|
|
116
67
|
namespace?: string;
|
|
117
68
|
/**
|
|
118
|
-
|
|
119
|
-
|
|
120
|
-
|
|
121
|
-
|
|
122
|
-
|
|
69
|
+
* How much stored metadata to return on matches. Defaults to `"indexed"`
|
|
70
|
+
* (only fields declared as index metadata) rather than `"all"`, so a query
|
|
71
|
+
* never leaks arbitrary stored fields by default. Callers that genuinely
|
|
72
|
+
* need every field opt in with `"all"`; pass `"none"` to drop metadata.
|
|
73
|
+
*/
|
|
123
74
|
returnMetadata?: "none" | "indexed" | "all";
|
|
124
75
|
topK?: number;
|
|
125
76
|
vector?: ReadonlyArray<number>;
|
|
@@ -132,24 +83,127 @@ interface VectorUpsertInputLike {
|
|
|
132
83
|
namespace?: string;
|
|
133
84
|
}
|
|
134
85
|
/**
|
|
135
|
-
* Structural mirror of `@lunora/server`'s `VectorSearch`. Declared here so the
|
|
136
|
-
* adapter never imports `@lunora/server` (keeps the dependency edge one-way:
|
|
137
|
-
* the generated DO depends on both, neither depends on the other).
|
|
138
|
-
|
|
86
|
+
* Structural mirror of `@lunora/server`'s `VectorSearch`. Declared here so the
|
|
87
|
+
* adapter never imports `@lunora/server` (keeps the dependency edge one-way:
|
|
88
|
+
* the generated DO depends on both, neither depends on the other). `getByIds`/
|
|
89
|
+
* `deleteByIds` carry an optional trailing `namespace` — a pure addition (more
|
|
90
|
+
* general, not narrower) that stays assignable to `@lunora/server`'s
|
|
91
|
+
* `VectorSearchReader`/`VectorSearch`, whose own two-argument signatures are
|
|
92
|
+
* unchanged: a function accepting an extra OPTIONAL parameter is assignable
|
|
93
|
+
* wherever a function taking fewer parameters is expected.
|
|
94
|
+
*/
|
|
139
95
|
interface VectorSearchLike {
|
|
140
|
-
deleteByIds: (indexName: string, ids: ReadonlyArray<string
|
|
141
|
-
getByIds: (indexName: string, ids: ReadonlyArray<string
|
|
96
|
+
deleteByIds: (indexName: string, ids: ReadonlyArray<string>, namespace?: string) => Promise<void>;
|
|
97
|
+
getByIds: (indexName: string, ids: ReadonlyArray<string>, namespace?: string) => Promise<ReadonlyArray<VectorRecordLike>>;
|
|
142
98
|
query: (indexName: string, input: VectorQueryInputLike) => Promise<VectorMatchesLike>;
|
|
143
99
|
upsert: (indexName: string, input: VectorUpsertInputLike) => Promise<void>;
|
|
144
100
|
upsertNow: (indexName: string, input: VectorUpsertInputLike) => Promise<void>;
|
|
145
101
|
}
|
|
102
|
+
/** Options for {@link createContextVectors}. */
|
|
103
|
+
interface CreateContextVectorsOptions {
|
|
104
|
+
/**
|
|
105
|
+
* Hold `upsert`'s remote write until the caller's storage transaction has
|
|
106
|
+
* COMMITTED, running it at once when none is open. The shard host supplies
|
|
107
|
+
* `ShardDO.deferAfterCommit`; codegen wires it.
|
|
108
|
+
*
|
|
109
|
+
* This is what separates `upsert` from `upsertNow`. Vectorize is outside the
|
|
110
|
+
* shard's SQLite and cannot roll back, so an inline `ctx.vectors.upsert` in a
|
|
111
|
+
* mutation that later throws leaves a vector pointing at a row that does not
|
|
112
|
+
* exist — and a search surfaces it. Omitted, both methods write inline, which
|
|
113
|
+
* is correct for a caller that has no transaction to wait for (an action, a
|
|
114
|
+
* test, the `@lunora/ai` RAG helpers).
|
|
115
|
+
*/
|
|
116
|
+
deferAfterCommit?: (work: () => Promise<void>) => Promise<void>;
|
|
117
|
+
/**
|
|
118
|
+
* The DO's own shard/tenant key, applied as the default `namespace` for
|
|
119
|
+
* an operation against an index in `shardedIndexNames` that doesn't pass
|
|
120
|
+
* one explicitly. `undefined` means this instance HAS no shard key —
|
|
121
|
+
* always true for the root/default DO instance, since only a per-tenant
|
|
122
|
+
* instance owns one. See `shardedIndexNames` for what that implies per
|
|
123
|
+
* index, and {@link createContextVectors}'s docblock for the full
|
|
124
|
+
* root-instance rule.
|
|
125
|
+
*/
|
|
126
|
+
namespace?: string;
|
|
127
|
+
/**
|
|
128
|
+
* Vector index names sourced from a `.shardBy()`'d table — the ones
|
|
129
|
+
* `namespace` is a meaningful tenant scope for. `ctx.vectors` is a single
|
|
130
|
+
* flat facade over EVERY declared index (root-scoped and sharded tables
|
|
131
|
+
* alike — Vectorize indexes are account-global and `config.vectors(env)`
|
|
132
|
+
* registers them all in one flat map), reachable from ANY DO instance —
|
|
133
|
+
* so `namespace` can only be a safe default for the indexes actually
|
|
134
|
+
* listed here.
|
|
135
|
+
*
|
|
136
|
+
* An index NOT in this set (sourced from a root-scoped table) always
|
|
137
|
+
* stays namespace-less, regardless of `namespace` or which DO instance
|
|
138
|
+
* calls it — it has no tenant identity to begin with, so scoping it would
|
|
139
|
+
* silently return nothing for legitimate, intentionally shared data (and,
|
|
140
|
+
* called from a per-tenant instance, would wrongly search under that
|
|
141
|
+
* tenant's namespace even though nothing was ever written there under
|
|
142
|
+
* it). An index IN this set, called from a per-tenant DO instance
|
|
143
|
+
* (`namespace` is set), defaults to `namespace`, scoping correctly. An
|
|
144
|
+
* index IN this set, called from the root/default DO instance
|
|
145
|
+
* (`namespace` is `undefined`) with no explicit override, is unsafe to
|
|
146
|
+
* default at all — see {@link createContextVectors}'s docblock.
|
|
147
|
+
*
|
|
148
|
+
* Omitted (or empty) → no index is ever treated as sharded, i.e.
|
|
149
|
+
* `namespace` never applies as a default on any call — the unsharded-app,
|
|
150
|
+
* byte-identical-to-today case.
|
|
151
|
+
*/
|
|
152
|
+
shardedIndexNames?: ReadonlyArray<string>;
|
|
153
|
+
}
|
|
146
154
|
/**
|
|
147
|
-
* Bridge `LunoraVectors` (returns Vectorize mutation receipts) to the server's
|
|
148
|
-
* `VectorSearch` contract (void mutations, server match/record shapes).
|
|
149
|
-
*
|
|
150
|
-
*
|
|
151
|
-
|
|
152
|
-
|
|
155
|
+
* Bridge `LunoraVectors` (returns Vectorize mutation receipts) to the server's
|
|
156
|
+
* `VectorSearch` contract (void mutations, server match/record shapes).
|
|
157
|
+
*
|
|
158
|
+
* `upsert` vs `upsertNow` — IMPORTANT: with `options.deferAfterCommit` supplied
|
|
159
|
+
* (codegen wires the shard host's), `upsert` holds the remote write until the
|
|
160
|
+
* caller's transaction has committed and `upsertNow` writes inline, which is
|
|
161
|
+
* what `MutationCtx`'s contract documents. Without it both write inline: a
|
|
162
|
+
* caller with no transaction open has nothing to wait for. The NAMESPACE is
|
|
163
|
+
* resolved eagerly either way — before the deferral, not inside it — so a
|
|
164
|
+
* misconfiguration (the root-instance throw below) still reaches the handler
|
|
165
|
+
* that made the call instead of a post-commit log line nobody is holding.
|
|
166
|
+
*
|
|
167
|
+
* Tenant isolation (read side) — IMPORTANT: an explicit `namespace` argument
|
|
168
|
+
* on any call (`input.namespace` for `query`/`upsert`/`upsertNow`, the
|
|
169
|
+
* trailing `namespace` parameter for `getByIds`/`deleteByIds`) ALWAYS wins —
|
|
170
|
+
* this is a deliberate soft default, not a hard boundary: `ctx.vectors` is
|
|
171
|
+
* trusted server-side app code (the same trust level that lets `ctx.db` read
|
|
172
|
+
* any table), so a caller that explicitly names a namespace is trusted to
|
|
173
|
+
* mean it, including a legitimate cross-tenant admin read/write. Absent an
|
|
174
|
+
* explicit namespace, `options.namespace` (this DO instance's own shard key)
|
|
175
|
+
* is the DEFAULT for any index in `options.shardedIndexNames` — see that
|
|
176
|
+
* option's docblock for why the default is index-scoped rather than global.
|
|
177
|
+
*
|
|
178
|
+
* Root-instance rule — IMPORTANT: when an operation targets a sharded index
|
|
179
|
+
* (one in `shardedIndexNames`) and BOTH the explicit argument and
|
|
180
|
+
* `options.namespace` are absent (this is the root/default DO instance, which
|
|
181
|
+
* owns no shard key), there is no safe default and no override — this THROWS
|
|
182
|
+
* rather than silently resolving to "no namespace". A namespace-less
|
|
183
|
+
* query/getByIds/deleteByIds/upsert against a sharded index would reach or
|
|
184
|
+
* mutate EVERY tenant's vectors (Vectorize indexes are account-global), which
|
|
185
|
+
* is the exact cross-tenant leak this file exists to close; returning an
|
|
186
|
+
* empty result set instead would masquerade that same configuration problem
|
|
187
|
+
* as "no data", which is worse — a caller debugging it sees nothing rather
|
|
188
|
+
* than a directed error. This case is reachable in a MIXED schema (some
|
|
189
|
+
* vectorized tables `.shardBy()`'d, others root-scoped) whenever application
|
|
190
|
+
* code queries a sharded index's name from the root DO instance without an
|
|
191
|
+
* explicit namespace; it is not reachable from `createVectorSyncHook`'s own
|
|
192
|
+
* internal calls, which only ever process a table this DO instance owns (so
|
|
193
|
+
* a sharded table's write never reaches a root instance in the first place).
|
|
194
|
+
*
|
|
195
|
+
* Id path, unrelated axis — IMPORTANT: independent of the override/root rules
|
|
196
|
+
* above, `getByIds`/`deleteByIds` can't ask Vectorize to filter by namespace
|
|
197
|
+
* remotely at all (its id-based operations take no `namespace` option), so
|
|
198
|
+
* once a namespace IS resolved (explicit or defaulted) for these two methods,
|
|
199
|
+
* isolation is enforced client-side: `getByIds` drops any returned record
|
|
200
|
+
* whose `namespace` doesn't match (fail closed: a record with no `namespace`
|
|
201
|
+
* field is treated as a mismatch, never as "belongs to everyone"), and
|
|
202
|
+
* `deleteByIds` resolves ids via `getByIds` first and only deletes the subset
|
|
203
|
+
* that belongs to the resolved namespace — silently, by design (see the
|
|
204
|
+
* `deleteByIds` implementation for the no-signal tradeoff this makes).
|
|
205
|
+
*/
|
|
206
|
+
declare const createContextVectors: (lunora: LunoraVectors, options?: CreateContextVectorsOptions) => VectorSearchLike;
|
|
153
207
|
/** A single row mutation observed by the ctx-db, fed to {@link createVectorSyncHook}. */
|
|
154
208
|
interface WriteEvent {
|
|
155
209
|
doc?: Record<string, unknown>;
|
|
@@ -160,71 +214,182 @@ interface WriteEvent {
|
|
|
160
214
|
type WriteHook = (event: WriteEvent) => Promise<void>;
|
|
161
215
|
/** Inline vector index declared via `.vectorize(field, ...)` (DSL Shape A). */
|
|
162
216
|
interface TableVectorIndexLike {
|
|
217
|
+
dimensions?: number;
|
|
163
218
|
embed: VectorEmbedderLike;
|
|
164
219
|
field: string;
|
|
165
220
|
metadata?: ReadonlyArray<string>;
|
|
221
|
+
metric?: string;
|
|
222
|
+
/** Declared identifier of what `embed` produces; part of the backfill fingerprint. */
|
|
223
|
+
model?: string;
|
|
166
224
|
name: string;
|
|
167
225
|
}
|
|
168
226
|
interface TableDefinitionLike {
|
|
227
|
+
/** `.softDelete()` marker column. A row whose marker is set is hidden from `ctx.db`, so it must have no vector. */
|
|
228
|
+
softDeleteMode?: {
|
|
229
|
+
field: string;
|
|
230
|
+
};
|
|
169
231
|
vectorIndexes?: ReadonlyArray<TableVectorIndexLike>;
|
|
170
232
|
}
|
|
171
233
|
/** Standalone vector index declared via `defineVectorIndex(...)` (DSL Shape B). */
|
|
172
234
|
interface VectorIndexDefinitionLike {
|
|
235
|
+
dimensions?: number;
|
|
173
236
|
embed: VectorEmbedderLike;
|
|
174
237
|
metadata?: (row: Record<string, unknown>) => Record<string, unknown>;
|
|
238
|
+
metric?: string;
|
|
239
|
+
/** Declared identifier of what `embed` produces; part of the backfill fingerprint. */
|
|
240
|
+
model?: string;
|
|
175
241
|
select: (row: Record<string, unknown>) => string;
|
|
176
242
|
table: string;
|
|
177
243
|
}
|
|
178
244
|
/**
|
|
179
|
-
* Structural mirror of `@lunora/server`'s `Schema`, narrowed to the fields the
|
|
180
|
-
* sync hook reads. Carries live `embed`/`select` closures, so the hook must be
|
|
181
|
-
* built from the imported `schema` value — never a serialized descriptor.
|
|
182
|
-
*/
|
|
245
|
+
* Structural mirror of `@lunora/server`'s `Schema`, narrowed to the fields the
|
|
246
|
+
* sync hook reads. Carries live `embed`/`select` closures, so the hook must be
|
|
247
|
+
* built from the imported `schema` value — never a serialized descriptor.
|
|
248
|
+
*/
|
|
183
249
|
interface SchemaLike {
|
|
184
250
|
tables: Record<string, TableDefinitionLike>;
|
|
185
251
|
vectorIndexes: Record<string, VectorIndexDefinitionLike>;
|
|
186
252
|
}
|
|
187
253
|
/**
|
|
188
|
-
* Build a {@link WriteHook} that keeps Vectorize in sync with row writes. On
|
|
189
|
-
* insert/update it embeds each matching index's source (Shape A `row[field]`,
|
|
190
|
-
* Shape B `select(row)`) and upserts; on delete it removes the row's id from
|
|
191
|
-
* every index sourced from the table.
|
|
192
|
-
*
|
|
193
|
-
* Tenant isolation — IMPORTANT: Vectorize indexes are account-global and shared
|
|
194
|
-
* by every shard DO. Without a `namespace`, a multi-tenant sharded app has NO
|
|
195
|
-
* isolation between tenants in the vector index — one tenant's vectors are
|
|
196
|
-
* queryable by another (ids/scores leak existence + semantic similarity even
|
|
197
|
-
* when no metadata is indexed). The caller MUST pass `options.namespace` (the
|
|
198
|
-
* shard / tenant key) so upserts are scoped, and MUST apply the same namespace
|
|
199
|
-
* on the query side — query-side namespace filtering is mandatory, not optional.
|
|
200
|
-
* The namespace is threaded onto upserts here; pass it from the shard DO that
|
|
201
|
-
* owns this hook. Any namespace-less sync emits a one-time-per-index dev warning
|
|
202
|
-
* (regardless of whether metadata is present); a genuinely single-tenant app
|
|
203
|
-
* suppresses it with `allowSharedNamespace: true`.
|
|
204
|
-
*
|
|
205
|
-
*
|
|
206
|
-
*
|
|
207
|
-
*
|
|
208
|
-
*
|
|
209
|
-
*
|
|
210
|
-
*
|
|
211
|
-
*
|
|
212
|
-
*
|
|
213
|
-
*
|
|
214
|
-
*
|
|
215
|
-
|
|
254
|
+
* Build a {@link WriteHook} that keeps Vectorize in sync with row writes. On
|
|
255
|
+
* insert/update it embeds each matching index's source (Shape A `row[field]`,
|
|
256
|
+
* Shape B `select(row)`) and upserts; on delete it removes the row's id from
|
|
257
|
+
* every index sourced from the table. {@link planRowSync} makes the decision.
|
|
258
|
+
*
|
|
259
|
+
* Tenant isolation — IMPORTANT: Vectorize indexes are account-global and shared
|
|
260
|
+
* by every shard DO. Without a `namespace`, a multi-tenant sharded app has NO
|
|
261
|
+
* isolation between tenants in the vector index — one tenant's vectors are
|
|
262
|
+
* queryable by another (ids/scores leak existence + semantic similarity even
|
|
263
|
+
* when no metadata is indexed). The caller MUST pass `options.namespace` (the
|
|
264
|
+
* shard / tenant key) so upserts are scoped, and MUST apply the same namespace
|
|
265
|
+
* on the query side — query-side namespace filtering is mandatory, not optional.
|
|
266
|
+
* The namespace is threaded onto upserts here; pass it from the shard DO that
|
|
267
|
+
* owns this hook. Any namespace-less sync emits a one-time-per-index dev warning
|
|
268
|
+
* (regardless of whether metadata is present); a genuinely single-tenant app
|
|
269
|
+
* suppresses it with `allowSharedNamespace: true`.
|
|
270
|
+
*
|
|
271
|
+
* Since plan 255, codegen satisfies the query-side requirement automatically
|
|
272
|
+
* for a `.shardBy()`'d vectorized table: the `vectors` instance passed in
|
|
273
|
+
* `options` here is the SAME `createContextVectors(...)` instance exposed as
|
|
274
|
+
* `ctx.vectors`, constructed with the identical shard-key `namespace` default
|
|
275
|
+
* AND the identical `shardedIndexNames` — so `ctx.vectors.query`/`getByIds`/
|
|
276
|
+
* `deleteByIds` are scoped without any app code changes. One consequence of
|
|
277
|
+
* sharing that instance: this hook's own internal `deleteByIds` calls (on row
|
|
278
|
+
* delete and on a cleared inline field) now also go through the
|
|
279
|
+
* namespace-verifying path described on
|
|
280
|
+
* {@link createContextVectors} — an extra `getByIds` subrequest per
|
|
281
|
+
* delete-shaped write, not a behavior change (the row being deleted was
|
|
282
|
+
* written under this same shard's namespace, so the verification passes).
|
|
283
|
+
* This never hits {@link createContextVectors}'s root-instance throw: a write
|
|
284
|
+
* event only ever fires for a table THIS DO instance owns, so if this hook
|
|
285
|
+
* processes a write for a sharded index, this instance IS a real per-tenant
|
|
286
|
+
* shard (not root) — `namespace` here is never `undefined` for that index.
|
|
287
|
+
*
|
|
288
|
+
* Consistency — IMPORTANT: Vectorize is external and non-transactional, so this
|
|
289
|
+
* hook runs AFTER the mutation's transaction has committed, never inside it (the
|
|
290
|
+
* shard host holds it — `ShardDO.deferAfterCommit`). That ordering is what stops
|
|
291
|
+
* a rolled-back write from leaving a vector for a row that does not exist, and a
|
|
292
|
+
* rolled-back delete from leaving a live row with its vector already purged.
|
|
293
|
+
*
|
|
294
|
+
* Two commits to the same row do not race: the shard host drains one
|
|
295
|
+
* transaction's held work entirely before the next transaction's, so the hooks
|
|
296
|
+
* apply in COMMIT order even though each may take hundreds of milliseconds. Fan
|
|
297
|
+
* out within a single hook is still unordered — the indexes are independent.
|
|
298
|
+
*
|
|
299
|
+
* What remains is the opposite divergence, and it is the one worth having: the
|
|
300
|
+
* row is committed and this hook may still fail — fully, or partway through a
|
|
301
|
+
* fan-out that already applied to some indexes. The row is then indexed in some
|
|
302
|
+
* indexes and not others. Nothing is compensated, deliberately: the row SURVIVES
|
|
303
|
+
* a failure here, so purging the indexes that did apply would turn a partially
|
|
304
|
+
* indexed row into an unsearchable one. Upserts and deletes are idempotent
|
|
305
|
+
* (keyed by row id), so re-running the same write converges.
|
|
306
|
+
*/
|
|
216
307
|
declare const createVectorSyncHook: (options: {
|
|
217
308
|
allowSharedNamespace?: boolean;
|
|
218
309
|
namespace?: string;
|
|
219
310
|
schema: SchemaLike;
|
|
220
311
|
vectors: VectorSearchLike;
|
|
221
312
|
}) => WriteHook;
|
|
313
|
+
/** A row the backfill could not index, and why. */
|
|
314
|
+
interface VectorBackfillFailure {
|
|
315
|
+
error: unknown;
|
|
316
|
+
id: string;
|
|
317
|
+
}
|
|
318
|
+
/**
|
|
319
|
+
* Index one page of rows for the shard's vector backfill. Resolves with the rows
|
|
320
|
+
* that failed on their own; REJECTS when the failure is the service's rather than
|
|
321
|
+
* a row's, so the caller holds its cursor and retries the page — with a
|
|
322
|
+
* `SERVICE_UNAVAILABLE` `LunoraError` when the error shows the failure to be
|
|
323
|
+
* transient, and with the raw error when only the whole page failing suggests it.
|
|
324
|
+
*/
|
|
325
|
+
type VectorBackfillSync = (table: string, rows: ReadonlyArray<{
|
|
326
|
+
doc: Record<string, unknown>;
|
|
327
|
+
id: string;
|
|
328
|
+
}>) => Promise<ReadonlyArray<VectorBackfillFailure>>;
|
|
329
|
+
/**
|
|
330
|
+
* The backfill's counterpart to {@link createVectorSyncHook}: the same
|
|
331
|
+
* {@link planRowSync} decision for a whole page of rows, with the remote calls
|
|
332
|
+
* batched — every row is embedded (bounded concurrency), then each index takes
|
|
333
|
+
* ONE `upsertMany` and ONE `deleteByIds` per 1000 rows instead of a call per row.
|
|
334
|
+
* That is what keeps a page short enough to hold the shard's write-hook chain.
|
|
335
|
+
*
|
|
336
|
+
* Failures are split in two, because they need opposite handling.
|
|
337
|
+
*
|
|
338
|
+
* A ROW failure is deterministic and would fail on every retry: a non-string
|
|
339
|
+
* source, a `select()` that throws, text the model rejects, metadata Vectorize
|
|
340
|
+
* refuses. The row is reported and the page moves on — the live hook only logs
|
|
341
|
+
* these too, and a backfill that stopped on one would never finish.
|
|
342
|
+
*
|
|
343
|
+
* A SERVICE failure is transient: the embedder or Vectorize is unreachable. The
|
|
344
|
+
* call rejects, so the page is retried. It is recognised by the error where the
|
|
345
|
+
* error says (an HTTP 5xx/408/429 status, a timeout — see {@link classifyFailure};
|
|
346
|
+
* these reject as `SERVICE_UNAVAILABLE`), and otherwise as every attempt in a
|
|
347
|
+
* group of two or more failing, and as a failed `deleteByIds` (which has no row
|
|
348
|
+
* content to blame). A batch `upsertMany` that fails is retried one row at a
|
|
349
|
+
* time to tell the two apart. A group that fails whole on every retry — each
|
|
350
|
+
* row refused for the same reason, with no status to show it — rejects each
|
|
351
|
+
* time too; the backfill writes such a page off after a few consecutive tries.
|
|
352
|
+
*
|
|
353
|
+
* `upsertMany` is the raw binding call, so the namespace is passed explicitly —
|
|
354
|
+
* the same `namespace` the live hook scopes by.
|
|
355
|
+
*/
|
|
356
|
+
type BackfillSyncOptions = {
|
|
357
|
+
allowSharedNamespace?: boolean;
|
|
358
|
+
namespace?: string;
|
|
359
|
+
schema: SchemaLike;
|
|
360
|
+
upsertMany: LunoraVectors["upsertMany"];
|
|
361
|
+
vectors: VectorSearchLike;
|
|
362
|
+
};
|
|
363
|
+
declare const createVectorBackfillSync: (options: BackfillSyncOptions) => VectorBackfillSync;
|
|
364
|
+
/**
|
|
365
|
+
* Every table with a vector index sourced from it, each with a fingerprint of
|
|
366
|
+
* what its stored vectors were built from — the input to the shard's vector
|
|
367
|
+
* backfill, which re-walks a table whose fingerprint changed.
|
|
368
|
+
*
|
|
369
|
+
* Covers what the schema can see: index names, the inline source field,
|
|
370
|
+
* dimensions, metric, inline metadata fields, the declared `model`, and the
|
|
371
|
+
* table's `.softDelete()` field — a row hidden by a newly chosen marker keeps
|
|
372
|
+
* its vector until the table is walked again. A
|
|
373
|
+
* function (`embed`, a Shape B `select`/`metadata`) has no stable identity to
|
|
374
|
+
* fingerprint — its source text changes with unrelated rebuilds of the bundle,
|
|
375
|
+
* which would re-embed whole tables for nothing — so the declared `model` string
|
|
376
|
+
* stands in for `embed`, and any other change is announced by calling the
|
|
377
|
+
* backfill with `restart: true`.
|
|
378
|
+
*
|
|
379
|
+
* `model` joins a descriptor only when declared, and the soft-delete field only
|
|
380
|
+
* when the table has one, so an index without either keeps the fingerprint it
|
|
381
|
+
* was recorded under and is not re-embedded for it.
|
|
382
|
+
*/
|
|
383
|
+
declare const vectorBackfillTargets: (schema: SchemaLike) => {
|
|
384
|
+
profile: string;
|
|
385
|
+
table: string;
|
|
386
|
+
}[];
|
|
222
387
|
/**
|
|
223
|
-
* One vector index as the generated `LUNORA_VECTOR_INDEXES` registry describes
|
|
224
|
-
* it — the static schema shape, independent of any live binding. Structurally
|
|
225
|
-
* the codegen `LunoraVectorIndex`, restated here so this package stays free of a
|
|
226
|
-
* dependency on `@lunora/codegen`.
|
|
227
|
-
*/
|
|
388
|
+
* One vector index as the generated `LUNORA_VECTOR_INDEXES` registry describes
|
|
389
|
+
* it — the static schema shape, independent of any live binding. Structurally
|
|
390
|
+
* the codegen `LunoraVectorIndex`, restated here so this package stays free of a
|
|
391
|
+
* dependency on `@lunora/codegen`.
|
|
392
|
+
*/
|
|
228
393
|
interface VectorIndexRegistryEntry {
|
|
229
394
|
dimensions?: number;
|
|
230
395
|
field?: string;
|
|
@@ -245,9 +410,9 @@ interface VectorAdminQueryMatch {
|
|
|
245
410
|
score: number;
|
|
246
411
|
}
|
|
247
412
|
/**
|
|
248
|
-
* The admin introspector the worker passes to `createWorker({ vectorIntrospector })`.
|
|
249
|
-
* `queryIndex` is present only when at least one embedder is wired.
|
|
250
|
-
*/
|
|
413
|
+
* The admin introspector the worker passes to `createWorker({ vectorIntrospector })`.
|
|
414
|
+
* `queryIndex` is present only when at least one embedder is wired.
|
|
415
|
+
*/
|
|
251
416
|
interface VectorAdminIntrospector {
|
|
252
417
|
listIndexes: () => Promise<VectorAdminIndexSummary[]>;
|
|
253
418
|
queryIndex?: (options: {
|
|
@@ -260,11 +425,11 @@ interface VectorAdminIntrospector {
|
|
|
260
425
|
}
|
|
261
426
|
interface VectorAdminIntrospectorOptions {
|
|
262
427
|
/**
|
|
263
|
-
|
|
264
|
-
|
|
265
|
-
|
|
266
|
-
|
|
267
|
-
|
|
428
|
+
* Per-index embedder (text → vector), keyed by index name. Supply the
|
|
429
|
+
* schema's embedders to enable studio similarity queries; omit it (or leave
|
|
430
|
+
* an index out) and that index lists read-only — `queryIndex` is withheld
|
|
431
|
+
* entirely when no embedder is provided.
|
|
432
|
+
*/
|
|
268
433
|
embedders?: Record<string, EmbedFunction<string>>;
|
|
269
434
|
/** Live Vectorize bindings keyed by index name, from `env`. */
|
|
270
435
|
indexes: Record<string, VectorizeIndexLike>;
|
|
@@ -272,14 +437,14 @@ interface VectorAdminIntrospectorOptions {
|
|
|
272
437
|
registry: ReadonlyArray<VectorIndexRegistryEntry>;
|
|
273
438
|
}
|
|
274
439
|
/**
|
|
275
|
-
* Build the read-only Vectorize introspector backing the studio's vector
|
|
276
|
-
* browser. `listIndexes` returns the static registry, enriching each entry with
|
|
277
|
-
* live `describe()` stats when the matching binding is present (a binding that
|
|
278
|
-
* throws or lacks `describe` degrades to the static shape rather than failing
|
|
279
|
-
* the whole list). `queryIndex` embeds the query text via the index's embedder
|
|
280
|
-
* and runs an ANN search; it is omitted when no embedders are configured, so the
|
|
281
|
-
* worker reports `VECTOR_QUERY_UNSUPPORTED` rather than half-answering.
|
|
282
|
-
*/
|
|
440
|
+
* Build the read-only Vectorize introspector backing the studio's vector
|
|
441
|
+
* browser. `listIndexes` returns the static registry, enriching each entry with
|
|
442
|
+
* live `describe()` stats when the matching binding is present (a binding that
|
|
443
|
+
* throws or lacks `describe` degrades to the static shape rather than failing
|
|
444
|
+
* the whole list). `queryIndex` embeds the query text via the index's embedder
|
|
445
|
+
* and runs an ANN search; it is omitted when no embedders are configured, so the
|
|
446
|
+
* worker reports `VECTOR_QUERY_UNSUPPORTED` rather than half-answering.
|
|
447
|
+
*/
|
|
283
448
|
declare const createVectorAdminIntrospector: (options: VectorAdminIntrospectorOptions) => VectorAdminIntrospector;
|
|
284
449
|
declare const createVectors: (options: LunoraVectorsOptions) => LunoraVectors;
|
|
285
|
-
export { type EmbedFunction, type LunoraVectors, type LunoraVectorsOptions, type QueryInput, type SchemaLike, type TableDefinitionLike, type TableVectorIndexLike, type UpsertInput, type VectorAdminIndexSummary, type VectorAdminIntrospector, type VectorAdminIntrospectorOptions, type VectorAdminQueryMatch, type VectorEmbedderLike, type VectorIndexDefinitionLike, type VectorIndexRegistryEntry, type VectorMatchLike, type VectorMatchesLike, type
|
|
450
|
+
export { type EmbedFunction, type LunoraVectors, type LunoraVectorsOptions, type QueryInput, type SchemaLike, type TableDefinitionLike, type TableVectorIndexLike, type UpsertInput, type VectorAdminIndexSummary, type VectorAdminIntrospector, type VectorAdminIntrospectorOptions, type VectorAdminQueryMatch, type VectorBackfillFailure, type VectorBackfillSync, type VectorEmbedderLike, type VectorIndexDefinitionLike, type VectorIndexRegistryEntry, type VectorMatchLike, type VectorMatchesLike, type VectorQueryInputLike, type VectorRecordLike, type VectorSearchLike, type VectorUpsertInputLike, type WriteEvent, type WriteHook, createContextVectors, createVectorAdminIntrospector, createVectorBackfillSync, createVectorSyncHook, createVectors, vectorBackfillTargets };
|