@lunora/bindings 1.0.0-alpha.63 → 1.0.0-alpha.65

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
@@ -0,0 +1,6 @@
1
+ import{c as b,U as g}from"./concurrent-vRmSvRpF.mjs";const x=(s,c)=>{if(!c.includes("."))return s[c];let n=s;for(const d of c.split(".")){if(n===null||typeof n!="object"||Array.isArray(n))return;n=n[d]}return n},S=(s,c)=>{const n=c?.namespace,d=new Set(c?.shardedIndexNames),l=(t,a)=>{if(a!==void 0)return a;if(d.has(t)){if(n!==void 0)return n;throw new Error(`@lunora/bindings/vectors: index "${t}" belongs to a sharded table, but this DO instance has no shard key (it is the root/default DO) and no explicit namespace was given. A namespace-less operation here would reach every tenant's vectors — Vectorize indexes are account-global. Pass an explicit namespace, or issue this call from the sharded DO instance that owns the tenant.`)}},i=c?.deferAfterCommit,v=async(t,a,r)=>{await s.upsert(t,{embed:a.embed,id:a.id,input:a.input,metadata:a.metadata,namespace:r})},u=async(t,a)=>{await v(t,a,l(t,a.namespace))},f=i===void 0?u:async(t,a)=>{const r=l(t,a.namespace);await i(async()=>v(t,a,r))},h=async(t,a,r)=>{const o=await s.getByIds(t,a);return r===void 0?o:o.filter(m=>m.namespace===r)};return{deleteByIds:async(t,a,r)=>{const o=l(t,r);if(o===void 0){await s.deleteByIds(t,a);return}const m=await h(t,a,o);m.length!==0&&await s.deleteByIds(t,m.map(e=>e.id))},getByIds:async(t,a,r)=>{const o=l(t,r);return(await h(t,a,o)).map(e=>({id:e.id,metadata:e.metadata,namespace:e.namespace,values:[...e.values]}))},query:async(t,a)=>{const r=await s.query(t,{embed:a.embed,filter:a.filter,input:a.input,namespace:l(t,a.namespace),returnMetadata:a.returnMetadata??"indexed",topK:a.topK,vector:a.vector});return{count:r.count,matches:r.matches.map(o=>({id:o.id,metadata:o.metadata,score:o.score}))}},upsert:f,upsertNow:u}},w=new Set,y=s=>{w.has(s)||(w.add(s),console.warn(`[@lunora/bindings/vectors] index "${s}" syncs vectors without a namespace — in a
2
+ multi-tenant/sharded app this exposes one tenant's vectors (and any captured
3
+ metadata) to every other tenant, since Vectorize indexes are account-global.
4
+ Pass \`namespace\` (the shard/tenant key) on both write and query — query-side
5
+ namespace filtering is mandatory for multi-tenant apps. Single-tenant apps that
6
+ legitimately have no tenant key suppress this via { allowSharedNamespace: true }.`))},I=(s,c)=>{const n={};for(const d of c)d in s&&(n[d]=s[d]);return n},C=s=>{const{allowSharedNamespace:c,namespace:n,schema:d,vectors:l}=s;return async i=>{const u=d.tables[i.table]?.vectorIndexes??[],f=Object.entries(d.vectorIndexes).filter(([,e])=>e.table===i.table);if(u.length===0&&f.length===0)return;const h=[...u.map(e=>e.name),...f.map(([e])=>e)];if(i.op==="delete"){await Promise.all(h.map(e=>l.deleteByIds(e,[i.id])));return}const t=i.doc;if(!t)return;const a=u.map(e=>({index:e,value:x(t,e.field)})),r=a.filter(e=>e.value!==void 0&&e.value!==null),o=a.filter(e=>e.value===void 0||e.value===null);for(const{index:e,value:p}of r)if(typeof p!="string")throw new TypeError(`@lunora/bindings/vectors: inline index "${e.name}" expects a string source at "${e.field}" on table "${i.table}" (got ${typeof p}); use a standalone defineVectorIndex with a select() to derive text from non-string columns`);const m=[...o.map(e=>async()=>{await l.deleteByIds(e.index.name,[i.id])}),...r.map(e=>async()=>{!c&&n===void 0&&y(e.index.name),await l.upsertNow(e.index.name,{embed:e.index.embed,id:i.id,input:e.value,metadata:e.index.metadata?I(t,e.index.metadata):void 0,namespace:n})}),...f.map(([e,p])=>async()=>{!c&&n===void 0&&y(e),await l.upsertNow(e,{embed:p.embed,id:i.id,input:p.select(t),metadata:p.metadata?.(t),namespace:n})})];await b(m,g,async e=>e())}};export{S as createContextVectors,C as createVectorSyncHook};
@@ -101,6 +101,19 @@ interface VectorSearchLike {
101
101
  }
102
102
  /** Options for {@link createContextVectors}. */
103
103
  interface CreateContextVectorsOptions {
104
+ /**
105
+ * Hold `upsert`'s remote write until the caller's storage transaction has
106
+ * COMMITTED, running it at once when none is open. The shard host supplies
107
+ * `ShardDO.deferAfterCommit`; codegen wires it.
108
+ *
109
+ * This is what separates `upsert` from `upsertNow`. Vectorize is outside the
110
+ * shard's SQLite and cannot roll back, so an inline `ctx.vectors.upsert` in a
111
+ * mutation that later throws leaves a vector pointing at a row that does not
112
+ * exist — and a search surfaces it. Omitted, both methods write inline, which
113
+ * is correct for a caller that has no transaction to wait for (an action, a
114
+ * test, the `@lunora/ai` RAG helpers).
115
+ */
116
+ deferAfterCommit?: (work: () => Promise<void>) => Promise<void>;
104
117
  /**
105
118
  * The DO's own shard/tenant key, applied as the default `namespace` for
106
119
  * an operation against an index in `shardedIndexNames` that doesn't pass
@@ -140,9 +153,16 @@ interface CreateContextVectorsOptions {
140
153
  }
141
154
  /**
142
155
  * Bridge `LunoraVectors` (returns Vectorize mutation receipts) to the server's
143
- * `VectorSearch` contract (void mutations, server match/record shapes). Both
144
- * `upsert` and `upsertNow` write inline — this design has no post-commit queue,
145
- * so "now" and "deferred" collapse to the same synchronous call.
156
+ * `VectorSearch` contract (void mutations, server match/record shapes).
157
+ *
158
+ * `upsert` vs `upsertNow` — IMPORTANT: with `options.deferAfterCommit` supplied
159
+ * (codegen wires the shard host's), `upsert` holds the remote write until the
160
+ * caller's transaction has committed and `upsertNow` writes inline, which is
161
+ * what `MutationCtx`'s contract documents. Without it both write inline: a
162
+ * caller with no transaction open has nothing to wait for. The NAMESPACE is
163
+ * resolved eagerly either way — before the deferral, not inside it — so a
164
+ * misconfiguration (the root-instance throw below) still reaches the handler
165
+ * that made the call instead of a post-commit log line nobody is holding.
146
166
  *
147
167
  * Tenant isolation (read side) — IMPORTANT: an explicit `namespace` argument
148
168
  * on any call (`input.namespace` for `query`/`upsert`/`upsertNow`, the
@@ -222,7 +242,7 @@ interface SchemaLike {
222
242
  * Build a {@link WriteHook} that keeps Vectorize in sync with row writes. On
223
243
  * insert/update it embeds each matching index's source (Shape A `row[field]`,
224
244
  * Shape B `select(row)`) and upserts; on delete it removes the row's id from
225
- * every index sourced from the table. Runs inline within the write path.
245
+ * every index sourced from the table.
226
246
  *
227
247
  * Tenant isolation — IMPORTANT: Vectorize indexes are account-global and shared
228
248
  * by every shard DO. Without a `namespace`, a multi-tenant sharded app has NO
@@ -243,8 +263,8 @@ interface SchemaLike {
243
263
  * AND the identical `shardedIndexNames` — so `ctx.vectors.query`/`getByIds`/
244
264
  * `deleteByIds` are scoped without any app code changes. One consequence of
245
265
  * sharing that instance: this hook's own internal `deleteByIds` calls (on row
246
- * delete, on a cleared inline field, and on compensation after a failed
247
- * upsert) now also go through the namespace-verifying path described on
266
+ * delete and on a cleared inline field) now also go through the
267
+ * namespace-verifying path described on
248
268
  * {@link createContextVectors} — an extra `getByIds` subrequest per
249
269
  * delete-shaped write, not a behavior change (the row being deleted was
250
270
  * written under this same shard's namespace, so the verification passes).
@@ -253,18 +273,24 @@ interface SchemaLike {
253
273
  * processes a write for a sharded index, this instance IS a real per-tenant
254
274
  * shard (not root) — `namespace` here is never `undefined` for that index.
255
275
  *
256
- * Consistency — IMPORTANT: this hook runs inline within the mutation but talks
257
- * to Vectorize, which is external and non-transactional. The per-index calls
258
- * fan out; if one fails after others have already applied, the SQLite write may
259
- * roll back while the applied Vectorize mutations cannot — leaving SQLite and
260
- * Vectorize diverged. We mitigate, not eliminate: upserts/deletes are
261
- * idempotent (keyed by row id), so a retry of the same write converges; and on
262
- * a fan-out failure we attempt a best-effort compensating delete of the row's
263
- * id before re-throwing — from every index of the table on an insert, and on an
264
- * update from the indexes this fan-out actually wrote, since the rollback leaves
265
- * the untouched ones holding the prior row's still-correct vector. A delete
266
- * after a failed upsert can itself fail — this is best-effort, the authoritative
267
- * recovery is re-running the (idempotent) write.
276
+ * Consistency — IMPORTANT: Vectorize is external and non-transactional, so this
277
+ * hook runs AFTER the mutation's transaction has committed, never inside it (the
278
+ * shard host holds it — `ShardDO.deferAfterCommit`). That ordering is what stops
279
+ * a rolled-back write from leaving a vector for a row that does not exist, and a
280
+ * rolled-back delete from leaving a live row with its vector already purged.
281
+ *
282
+ * Two commits to the same row do not race: the shard host drains one
283
+ * transaction's held work entirely before the next transaction's, so the hooks
284
+ * apply in COMMIT order even though each may take hundreds of milliseconds. Fan
285
+ * out within a single hook is still unordered — the indexes are independent.
286
+ *
287
+ * What remains is the opposite divergence, and it is the one worth having: the
288
+ * row is committed and this hook may still fail — fully, or partway through a
289
+ * fan-out that already applied to some indexes. The row is then indexed in some
290
+ * indexes and not others. Nothing is compensated, deliberately: the row SURVIVES
291
+ * a failure here, so purging the indexes that did apply would turn a partially
292
+ * indexed row into an unsearchable one. Upserts and deletes are idempotent
293
+ * (keyed by row id), so re-running the same write converges.
268
294
  */
269
295
  declare const createVectorSyncHook: (options: {
270
296
  allowSharedNamespace?: boolean;
@@ -101,6 +101,19 @@ interface VectorSearchLike {
101
101
  }
102
102
  /** Options for {@link createContextVectors}. */
103
103
  interface CreateContextVectorsOptions {
104
+ /**
105
+ * Hold `upsert`'s remote write until the caller's storage transaction has
106
+ * COMMITTED, running it at once when none is open. The shard host supplies
107
+ * `ShardDO.deferAfterCommit`; codegen wires it.
108
+ *
109
+ * This is what separates `upsert` from `upsertNow`. Vectorize is outside the
110
+ * shard's SQLite and cannot roll back, so an inline `ctx.vectors.upsert` in a
111
+ * mutation that later throws leaves a vector pointing at a row that does not
112
+ * exist — and a search surfaces it. Omitted, both methods write inline, which
113
+ * is correct for a caller that has no transaction to wait for (an action, a
114
+ * test, the `@lunora/ai` RAG helpers).
115
+ */
116
+ deferAfterCommit?: (work: () => Promise<void>) => Promise<void>;
104
117
  /**
105
118
  * The DO's own shard/tenant key, applied as the default `namespace` for
106
119
  * an operation against an index in `shardedIndexNames` that doesn't pass
@@ -140,9 +153,16 @@ interface CreateContextVectorsOptions {
140
153
  }
141
154
  /**
142
155
  * Bridge `LunoraVectors` (returns Vectorize mutation receipts) to the server's
143
- * `VectorSearch` contract (void mutations, server match/record shapes). Both
144
- * `upsert` and `upsertNow` write inline — this design has no post-commit queue,
145
- * so "now" and "deferred" collapse to the same synchronous call.
156
+ * `VectorSearch` contract (void mutations, server match/record shapes).
157
+ *
158
+ * `upsert` vs `upsertNow` — IMPORTANT: with `options.deferAfterCommit` supplied
159
+ * (codegen wires the shard host's), `upsert` holds the remote write until the
160
+ * caller's transaction has committed and `upsertNow` writes inline, which is
161
+ * what `MutationCtx`'s contract documents. Without it both write inline: a
162
+ * caller with no transaction open has nothing to wait for. The NAMESPACE is
163
+ * resolved eagerly either way — before the deferral, not inside it — so a
164
+ * misconfiguration (the root-instance throw below) still reaches the handler
165
+ * that made the call instead of a post-commit log line nobody is holding.
146
166
  *
147
167
  * Tenant isolation (read side) — IMPORTANT: an explicit `namespace` argument
148
168
  * on any call (`input.namespace` for `query`/`upsert`/`upsertNow`, the
@@ -222,7 +242,7 @@ interface SchemaLike {
222
242
  * Build a {@link WriteHook} that keeps Vectorize in sync with row writes. On
223
243
  * insert/update it embeds each matching index's source (Shape A `row[field]`,
224
244
  * Shape B `select(row)`) and upserts; on delete it removes the row's id from
225
- * every index sourced from the table. Runs inline within the write path.
245
+ * every index sourced from the table.
226
246
  *
227
247
  * Tenant isolation — IMPORTANT: Vectorize indexes are account-global and shared
228
248
  * by every shard DO. Without a `namespace`, a multi-tenant sharded app has NO
@@ -243,8 +263,8 @@ interface SchemaLike {
243
263
  * AND the identical `shardedIndexNames` — so `ctx.vectors.query`/`getByIds`/
244
264
  * `deleteByIds` are scoped without any app code changes. One consequence of
245
265
  * sharing that instance: this hook's own internal `deleteByIds` calls (on row
246
- * delete, on a cleared inline field, and on compensation after a failed
247
- * upsert) now also go through the namespace-verifying path described on
266
+ * delete and on a cleared inline field) now also go through the
267
+ * namespace-verifying path described on
248
268
  * {@link createContextVectors} — an extra `getByIds` subrequest per
249
269
  * delete-shaped write, not a behavior change (the row being deleted was
250
270
  * written under this same shard's namespace, so the verification passes).
@@ -253,18 +273,24 @@ interface SchemaLike {
253
273
  * processes a write for a sharded index, this instance IS a real per-tenant
254
274
  * shard (not root) — `namespace` here is never `undefined` for that index.
255
275
  *
256
- * Consistency — IMPORTANT: this hook runs inline within the mutation but talks
257
- * to Vectorize, which is external and non-transactional. The per-index calls
258
- * fan out; if one fails after others have already applied, the SQLite write may
259
- * roll back while the applied Vectorize mutations cannot — leaving SQLite and
260
- * Vectorize diverged. We mitigate, not eliminate: upserts/deletes are
261
- * idempotent (keyed by row id), so a retry of the same write converges; and on
262
- * a fan-out failure we attempt a best-effort compensating delete of the row's
263
- * id before re-throwing — from every index of the table on an insert, and on an
264
- * update from the indexes this fan-out actually wrote, since the rollback leaves
265
- * the untouched ones holding the prior row's still-correct vector. A delete
266
- * after a failed upsert can itself fail — this is best-effort, the authoritative
267
- * recovery is re-running the (idempotent) write.
276
+ * Consistency — IMPORTANT: Vectorize is external and non-transactional, so this
277
+ * hook runs AFTER the mutation's transaction has committed, never inside it (the
278
+ * shard host holds it — `ShardDO.deferAfterCommit`). That ordering is what stops
279
+ * a rolled-back write from leaving a vector for a row that does not exist, and a
280
+ * rolled-back delete from leaving a live row with its vector already purged.
281
+ *
282
+ * Two commits to the same row do not race: the shard host drains one
283
+ * transaction's held work entirely before the next transaction's, so the hooks
284
+ * apply in COMMIT order even though each may take hundreds of milliseconds. Fan
285
+ * out within a single hook is still unordered — the indexes are independent.
286
+ *
287
+ * What remains is the opposite divergence, and it is the one worth having: the
288
+ * row is committed and this hook may still fail — fully, or partway through a
289
+ * fan-out that already applied to some indexes. The row is then indexed in some
290
+ * indexes and not others. Nothing is compensated, deliberately: the row SURVIVES
291
+ * a failure here, so purging the indexes that did apply would turn a partially
292
+ * indexed row into an unsearchable one. Upserts and deletes are idempotent
293
+ * (keyed by row id), so re-running the same write converges.
268
294
  */
269
295
  declare const createVectorSyncHook: (options: {
270
296
  allowSharedNamespace?: boolean;
@@ -1 +1 @@
1
- import{createContextVectors as t,createVectorSyncHook as o}from"../packem_shared/createContextVectors-BcfM30uc.mjs";import{createVectorAdminIntrospector as a}from"../packem_shared/createVectorAdminIntrospector-Ct8v6PxJ.mjs";import{default as m}from"../packem_shared/createVectors-Dzv0ilKE.mjs";export{t as createContextVectors,a as createVectorAdminIntrospector,o as createVectorSyncHook,m as createVectors};
1
+ import{createContextVectors as t,createVectorSyncHook as o}from"../packem_shared/createContextVectors-Bjr3VlJp.mjs";import{createVectorAdminIntrospector as a}from"../packem_shared/createVectorAdminIntrospector-Ct8v6PxJ.mjs";import{default as m}from"../packem_shared/createVectors-Dzv0ilKE.mjs";export{t as createContextVectors,a as createVectorAdminIntrospector,o as createVectorSyncHook,m as createVectors};
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lunora/bindings",
3
- "version": "1.0.0-alpha.63",
3
+ "version": "1.0.0-alpha.65",
4
4
  "description": "Lightweight Cloudflare binding helpers for Lunora — ctx.kv, ctx.images, ctx.analytics, ctx.pipelines, ctx.vectors, ctx.r2sql — one install, per-binding subpaths",
5
5
  "keywords": [
6
6
  "analytics",
@@ -1,6 +0,0 @@
1
- import{c as x,U as I}from"./concurrent-vRmSvRpF.mjs";const B=(s,c)=>{if(!c.includes("."))return s[c];let n=s;for(const d of c.split(".")){if(n===null||typeof n!="object"||Array.isArray(n))return;n=n[d]}return n},C=(s,c)=>{const n=c?.namespace,d=new Set(c?.shardedIndexNames),l=(t,a)=>{if(a!==void 0)return a;if(d.has(t)){if(n!==void 0)return n;throw new Error(`@lunora/bindings/vectors: index "${t}" belongs to a sharded table, but this DO instance has no shard key (it is the root/default DO) and no explicit namespace was given. A namespace-less operation here would reach every tenant's vectors — Vectorize indexes are account-global. Pass an explicit namespace, or issue this call from the sharded DO instance that owns the tenant.`)}},o=async(t,a)=>{await s.upsert(t,{embed:a.embed,id:a.id,input:a.input,metadata:a.metadata,namespace:l(t,a.namespace)})},h=async(t,a,i)=>{const r=await s.getByIds(t,a);return i===void 0?r:r.filter(m=>m.namespace===i)};return{deleteByIds:async(t,a,i)=>{const r=l(t,i);if(r===void 0){await s.deleteByIds(t,a);return}const m=await h(t,a,r);m.length!==0&&await s.deleteByIds(t,m.map(u=>u.id))},getByIds:async(t,a,i)=>{const r=l(t,i);return(await h(t,a,r)).map(u=>({id:u.id,metadata:u.metadata,namespace:u.namespace,values:[...u.values]}))},query:async(t,a)=>{const i=await s.query(t,{embed:a.embed,filter:a.filter,input:a.input,namespace:l(t,a.namespace),returnMetadata:a.returnMetadata??"indexed",topK:a.topK,vector:a.vector});return{count:i.count,matches:i.matches.map(r=>({id:r.id,metadata:r.metadata,score:r.score}))}},upsert:o,upsertNow:o}},v=new Set,w=s=>{v.has(s)||(v.add(s),console.warn(`[@lunora/bindings/vectors] index "${s}" syncs vectors without a namespace — in a
2
- multi-tenant/sharded app this exposes one tenant's vectors (and any captured
3
- metadata) to every other tenant, since Vectorize indexes are account-global.
4
- Pass \`namespace\` (the shard/tenant key) on both write and query — query-side
5
- namespace filtering is mandatory for multi-tenant apps. Single-tenant apps that
6
- legitimately have no tenant key suppress this via { allowSharedNamespace: true }.`))},S=(s,c)=>{const n={};for(const d of c)d in s&&(n[d]=s[d]);return n},E=s=>{const{allowSharedNamespace:c,namespace:n,schema:d,vectors:l}=s;return async o=>{const t=d.tables[o.table]?.vectorIndexes??[],a=Object.entries(d.vectorIndexes).filter(([,e])=>e.table===o.table);if(t.length===0&&a.length===0)return;const i=[...t.map(e=>e.name),...a.map(([e])=>e)];if(o.op==="delete"){await Promise.all(i.map(e=>l.deleteByIds(e,[o.id])));return}const r=o.doc;if(!r)return;const m=t.map(e=>({index:e,value:B(r,e.field)})),u=m.filter(e=>e.value!==void 0&&e.value!==null),y=m.filter(e=>e.value===void 0||e.value===null);for(const{index:e,value:p}of u)if(typeof p!="string")throw new TypeError(`@lunora/bindings/vectors: inline index "${e.name}" expects a string source at "${e.field}" on table "${o.table}" (got ${typeof p}); use a standalone defineVectorIndex with a select() to derive text from non-string columns`);const f=[],b=[...y.map(e=>async()=>{await l.deleteByIds(e.index.name,[o.id])}),...u.map(e=>async()=>{!c&&n===void 0&&w(e.index.name),await l.upsert(e.index.name,{embed:e.index.embed,id:o.id,input:e.value,metadata:e.index.metadata?S(r,e.index.metadata):void 0,namespace:n}),f.push(e.index.name)}),...a.map(([e,p])=>async()=>{!c&&n===void 0&&w(e),await l.upsert(e,{embed:p.embed,id:o.id,input:p.select(r),metadata:p.metadata?.(r),namespace:n}),f.push(e)})];try{await x(b,I,async e=>e())}catch(e){const p=o.op==="insert"?i:f;throw await Promise.allSettled(p.map(g=>l.deleteByIds(g,[o.id]))),e}}};export{C as createContextVectors,E as createVectorSyncHook};