@lunora/ai 1.0.0-alpha.102 → 1.0.0-alpha.104
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +8 -7
- package/dist/index.d.mts +7 -2
- package/dist/index.d.ts +7 -2
- package/dist/index.mjs +1 -1
- package/dist/packem_shared/AI_DEFAULT_EMBEDDING_MODEL_ENV-IwcHZiKj.mjs +1 -0
- package/dist/packem_shared/createAi-CdiXSCUA.mjs +1 -0
- package/dist/packem_shared/defineRag-CMTzKfS7.mjs +8 -0
- package/dist/packem_shared/sqliteVectorStore-SjOFKoHo.mjs +1 -0
- package/dist/packem_shared/{types.d-D3U2budn.d.mts → types.d-BNOWTjHf.d.mts} +69 -5
- package/dist/packem_shared/{types.d-D3U2budn.d.ts → types.d-BNOWTjHf.d.ts} +69 -5
- package/dist/packem_shared/usage-GPnTSqR8.mjs +1 -0
- package/dist/rag/index.d.mts +23 -1
- package/dist/rag/index.d.ts +23 -1
- package/dist/rag/index.mjs +1 -1
- package/package.json +3 -13
- package/dist/packem_shared/AI_DEFAULT_EMBEDDING_MODEL_ENV-B58f6yo6.mjs +0 -1
- package/dist/packem_shared/createAi-mXofcZid.mjs +0 -1
- package/dist/packem_shared/defineRag-Cs7m4xJJ.mjs +0 -8
- package/dist/packem_shared/sqliteVectorStore-D32l9lP0.mjs +0 -1
package/README.md
CHANGED
|
@@ -54,11 +54,11 @@ yarn add @lunora/ai
|
|
|
54
54
|
pnpm add @lunora/ai
|
|
55
55
|
```
|
|
56
56
|
|
|
57
|
-
|
|
57
|
+
Other providers need no extra install: `ctx.ai.model("anthropic/claude-sonnet-5")` routes through Cloudflare AI Gateway (see below).
|
|
58
58
|
|
|
59
59
|
## Usage
|
|
60
60
|
|
|
61
|
-
When a function uses AI, codegen wires a typed **`ctx.ai`** onto the action context (inference is an external call, so — like `ctx.fetch` — it lives on actions). Workers AI is the zero-config default;
|
|
61
|
+
When a function uses AI, codegen wires a typed **`ctx.ai`** onto the action context (inference is an external call, so — like `ctx.fetch` — it lives on actions). Workers AI is the zero-config default; a `"<provider>/<model>"` id reaches any other provider through Cloudflare AI Gateway, and any AI SDK model object passes straight through.
|
|
62
62
|
|
|
63
63
|
```ts
|
|
64
64
|
// lunora/summarize.ts — ctx.ai (codegen-wired), Workers AI by default
|
|
@@ -82,13 +82,14 @@ export const summarize = action.input({ text: v.string().max(20_000) }).action(a
|
|
|
82
82
|
```
|
|
83
83
|
|
|
84
84
|
```ts
|
|
85
|
-
//
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
const result = streamText({ model: openai("gpt-5"), messages });
|
|
85
|
+
// Any provider: change the string, nothing else. Routed through Cloudflare AI
|
|
86
|
+
// Gateway over the same `AI` binding — Unified Billing or keys stored on the
|
|
87
|
+
// gateway, so the app holds no provider API key.
|
|
88
|
+
const result = streamText({ model: ctx.ai.model("anthropic/claude-sonnet-5"), messages });
|
|
90
89
|
```
|
|
91
90
|
|
|
91
|
+
Every `ctx.ai.model(...)` call is traced and its tokens and cost are counted per function (`gen_ai.usage.*`), which Studio's **AI usage** page charts. `lunora ai gateway` creates a gateway for the app and sets `LUNORA_AI_GATEWAY_ID`; without it, calls use the account's `default` gateway.
|
|
92
|
+
|
|
92
93
|
```ts
|
|
93
94
|
// RAG: embed via ctx.ai, store/search with @lunora/bindings/vectors (ctx.vectors)
|
|
94
95
|
import { embed } from "@lunora/ai";
|
package/dist/index.d.mts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { L as LunoraAiOptions, a as LunoraAi } from "./packem_shared/types.d-
|
|
2
|
-
export { A as AI_DEFAULT_EMBEDDING_MODEL_ENV, b as AI_DEFAULT_MODEL_ENV, c as AI_GATEWAY_ACCOUNT_ID_ENV, d as AI_GATEWAY_ID_ENV, e as AI_GATEWAY_METADATA_MAX_KEYS, f as AI_GATEWAY_TAGS_ENV, g as AI_GATEWAY_TOKEN_ENV, type
|
|
1
|
+
import { L as LunoraAiOptions, a as LunoraAi } from "./packem_shared/types.d-BNOWTjHf.mjs";
|
|
2
|
+
export { A as AI_DEFAULT_EMBEDDING_MODEL_ENV, b as AI_DEFAULT_MODEL_ENV, c as AI_GATEWAY_ACCOUNT_ID_ENV, d as AI_GATEWAY_ID_ENV, e as AI_GATEWAY_METADATA_MAX_KEYS, f as AI_GATEWAY_TAGS_ENV, g as AI_GATEWAY_TOKEN_ENV, h as AI_PROXY_TOKEN_ENV, i as AI_PROXY_URL_ENV, type j as AiBindingLike, type k as AiGatewayMetadata, type l as AiGatewayOptions, type m as AiMetrics, type n as AiSpan, type o as AiTelemetry, type p as AiTracer, type E as EmbeddingModelInput, type M as ModelInput, type R as ResolvedAiGateway, type W as WorkersAiProviderLike, q as buildAiGatewayMetadataFields, r as readAiGatewayEnvTags, s as resolveAiGateway } from "./packem_shared/types.d-BNOWTjHf.mjs";
|
|
3
3
|
export { type EmbeddingModel, type LanguageModel, embed, embedMany, generateObject, generateText, hasToolCall, jsonSchema, streamObject, streamText, tool } from 'ai';
|
|
4
4
|
export { createWorkersAI } from 'workers-ai-provider';
|
|
5
5
|
/**
|
|
@@ -11,6 +11,11 @@ export { createWorkersAI } from 'workers-ai-provider';
|
|
|
11
11
|
* (`@ai-sdk/openai`, `@ai-sdk/anthropic`, OpenRouter, …), so apps are never
|
|
12
12
|
* locked to Workers AI. Pair `embed` with `@lunora/bindings/vectors` for RAG.
|
|
13
13
|
*
|
|
14
|
+
* Without a binding, a string id resolves only as a `"<provider>/<model>"` slug
|
|
15
|
+
* through {@link AI_PROXY_URL_ENV}; everything else that needs the binding
|
|
16
|
+
* throws a directed error when called, never at construction — so the
|
|
17
|
+
* generated `ctx.ai` is always this facade.
|
|
18
|
+
*
|
|
14
19
|
* Combine with the re-exported `generateText`/`streamText`/`generateObject`/
|
|
15
20
|
* `embed`/`tool` from this package:
|
|
16
21
|
*
|
package/dist/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
|
-
import { L as LunoraAiOptions, a as LunoraAi } from "./packem_shared/types.d-
|
|
2
|
-
export { A as AI_DEFAULT_EMBEDDING_MODEL_ENV, b as AI_DEFAULT_MODEL_ENV, c as AI_GATEWAY_ACCOUNT_ID_ENV, d as AI_GATEWAY_ID_ENV, e as AI_GATEWAY_METADATA_MAX_KEYS, f as AI_GATEWAY_TAGS_ENV, g as AI_GATEWAY_TOKEN_ENV, type
|
|
1
|
+
import { L as LunoraAiOptions, a as LunoraAi } from "./packem_shared/types.d-BNOWTjHf.js";
|
|
2
|
+
export { A as AI_DEFAULT_EMBEDDING_MODEL_ENV, b as AI_DEFAULT_MODEL_ENV, c as AI_GATEWAY_ACCOUNT_ID_ENV, d as AI_GATEWAY_ID_ENV, e as AI_GATEWAY_METADATA_MAX_KEYS, f as AI_GATEWAY_TAGS_ENV, g as AI_GATEWAY_TOKEN_ENV, h as AI_PROXY_TOKEN_ENV, i as AI_PROXY_URL_ENV, type j as AiBindingLike, type k as AiGatewayMetadata, type l as AiGatewayOptions, type m as AiMetrics, type n as AiSpan, type o as AiTelemetry, type p as AiTracer, type E as EmbeddingModelInput, type M as ModelInput, type R as ResolvedAiGateway, type W as WorkersAiProviderLike, q as buildAiGatewayMetadataFields, r as readAiGatewayEnvTags, s as resolveAiGateway } from "./packem_shared/types.d-BNOWTjHf.js";
|
|
3
3
|
export { type EmbeddingModel, type LanguageModel, embed, embedMany, generateObject, generateText, hasToolCall, jsonSchema, streamObject, streamText, tool } from 'ai';
|
|
4
4
|
export { createWorkersAI } from 'workers-ai-provider';
|
|
5
5
|
/**
|
|
@@ -11,6 +11,11 @@ export { createWorkersAI } from 'workers-ai-provider';
|
|
|
11
11
|
* (`@ai-sdk/openai`, `@ai-sdk/anthropic`, OpenRouter, …), so apps are never
|
|
12
12
|
* locked to Workers AI. Pair `embed` with `@lunora/bindings/vectors` for RAG.
|
|
13
13
|
*
|
|
14
|
+
* Without a binding, a string id resolves only as a `"<provider>/<model>"` slug
|
|
15
|
+
* through {@link AI_PROXY_URL_ENV}; everything else that needs the binding
|
|
16
|
+
* throws a directed error when called, never at construction — so the
|
|
17
|
+
* generated `ctx.ai` is always this facade.
|
|
18
|
+
*
|
|
14
19
|
* Combine with the re-exported `generateText`/`streamText`/`generateObject`/
|
|
15
20
|
* `embed`/`tool` from this package:
|
|
16
21
|
*
|
package/dist/index.mjs
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
import{default as _}from"./packem_shared/createAi-
|
|
1
|
+
import{default as _}from"./packem_shared/createAi-CdiXSCUA.mjs";import{AI_DEFAULT_EMBEDDING_MODEL_ENV as t,AI_DEFAULT_MODEL_ENV as a,AI_GATEWAY_ACCOUNT_ID_ENV as o,AI_GATEWAY_ID_ENV as r,AI_GATEWAY_METADATA_MAX_KEYS as T,AI_GATEWAY_TAGS_ENV as I,AI_GATEWAY_TOKEN_ENV as N,AI_PROXY_TOKEN_ENV as l,AI_PROXY_URL_ENV as m,buildAiGatewayMetadataFields as s,readAiGatewayEnvTags as D,resolveAiGateway as G}from"./packem_shared/AI_DEFAULT_EMBEDDING_MODEL_ENV-IwcHZiKj.mjs";import{DEFAULT_MODEL_PRICES as O,estimateModelCost as d,lookupModelPrice as i}from"./packem_shared/DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";import{embed as Y,embedMany as x,generateObject as L,generateText as c,hasToolCall as f,jsonSchema as p,streamObject as W,streamText as b,tool as n}from"ai";import{createWorkersAI as U}from"workers-ai-provider";export{t as AI_DEFAULT_EMBEDDING_MODEL_ENV,a as AI_DEFAULT_MODEL_ENV,o as AI_GATEWAY_ACCOUNT_ID_ENV,r as AI_GATEWAY_ID_ENV,T as AI_GATEWAY_METADATA_MAX_KEYS,I as AI_GATEWAY_TAGS_ENV,N as AI_GATEWAY_TOKEN_ENV,l as AI_PROXY_TOKEN_ENV,m as AI_PROXY_URL_ENV,O as DEFAULT_MODEL_PRICES,s as buildAiGatewayMetadataFields,_ as createAi,U as createWorkersAI,Y as embed,x as embedMany,d as estimateModelCost,L as generateObject,c as generateText,f as hasToolCall,p as jsonSchema,i as lookupModelPrice,D as readAiGatewayEnvTags,G as resolveAiGateway,W as streamObject,b as streamText,n as tool};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
const a=(t,n)=>{if(t===void 0)return;const e=t[n];return typeof e=="string"&&e.length>0?e:void 0};let _=!1,d=!1;const f=5,E=t=>{if(t===void 0)return;const n={};typeof t.functionPath=="string"&&t.functionPath.length>0&&(n.functionPath=t.functionPath),typeof t.traceId=="string"&&t.traceId.length>0&&(n.traceId=t.traceId);const e={};for(const[o,s]of Object.entries(t.tags??{}))typeof s=="string"&&s.length>0&&o.length>0&&!Object.hasOwn(n,o)&&(e[o]=s);Object.assign(e,n);const r=Object.keys(e);if(r.length===0)return;if(r.length<=f)return e;const i=r.slice(-f);return Object.fromEntries(i.map(o=>[o,e[o]]))},g=t=>{const n=E(t);return n===void 0?void 0:JSON.stringify(n)},h="LUNORA_AI_DEFAULT_MODEL",T="LUNORA_AI_DEFAULT_EMBEDDING_MODEL",l="LUNORA_AI_GATEWAY_ACCOUNT_ID",I="LUNORA_AI_GATEWAY_ID",c="LUNORA_AI_GATEWAY_TOKEN",u="LUNORA_AI_GATEWAY_TAGS",N="LUNORA_AI_PROXY_URL",y="LUNORA_AI_PROXY_TOKEN",v=t=>{const n=a(t,u);if(n!==void 0)try{const e=JSON.parse(n);if(typeof e!="object"||e===null||Array.isArray(e))throw new TypeError("expected a JSON object");const r={};for(const[i,o]of Object.entries(e))typeof o=="string"&&o.length>0&&(r[i]=o);return Object.keys(r).length>0?r:void 0}catch{d||(d=!0,console.warn(`[lunora:ai] ${u} is not a flat JSON object of strings — AI Gateway tags from it are ignored.`));return}},O=t=>{_||a(t,c)===void 0||(_=!0,console.warn(`[lunora:ai] ${c} is set, but the Workers AI binding cannot send a gateway auth token — Cloudflare's native gateway option has no authorization field. The token is ignored on this path; use a bring-your-own AI SDK provider (which sends cf-aig-authorization), or make the AI Gateway unauthenticated for Workers AI.`))},w=(t,n,e="byo-provider")=>{const r=a(t,l),i=a(t,I);if(r===void 0||i===void 0)return;const o=a(t,c),s={};o!==void 0&&(s["cf-aig-authorization"]=`Bearer ${o}`,e==="workers-ai-binding"&&O(t));const A=g(n);return A!==void 0&&(s["cf-aig-metadata"]=A),{accountId:r,baseURL:`https://gateway.ai.cloudflare.com/v1/${r}/${i}`,gatewayId:i,headers:s}};export{T as AI_DEFAULT_EMBEDDING_MODEL_ENV,h as AI_DEFAULT_MODEL_ENV,l as AI_GATEWAY_ACCOUNT_ID_ENV,I as AI_GATEWAY_ID_ENV,f as AI_GATEWAY_METADATA_MAX_KEYS,u as AI_GATEWAY_TAGS_ENV,c as AI_GATEWAY_TOKEN_ENV,y as AI_PROXY_TOKEN_ENV,N as AI_PROXY_URL_ENV,E as buildAiGatewayMetadataFields,v as readAiGatewayEnvTags,a as readEnv,w as resolveAiGateway,O as warnIgnoredBindingToken};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
import{createOpenAI as G}from"@ai-sdk/openai";import{LunoraError as w}from"@lunora/errors";import{createWorkersAI as $}from"workers-ai-provider";import{anthropic as S}from"workers-ai-provider/anthropic";import{openai as W}from"workers-ai-provider/openai";import{buildAiGatewayMetadataFields as T,readEnv as A,AI_DEFAULT_MODEL_ENV as M,AI_DEFAULT_EMBEDDING_MODEL_ENV as _,AI_PROXY_URL_ENV as b,AI_PROXY_TOKEN_ENV as R,readAiGatewayEnvTags as V,AI_GATEWAY_ID_ENV as F,warnIgnoredBindingToken as K}from"./AI_DEFAULT_EMBEDDING_MODEL_ENV-IwcHZiKj.mjs";import{wrapLanguageModel as q}from"ai";import{m as Y,r as I}from"./usage-GPnTSqR8.mjs";import"./DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";const j={setAttribute:()=>{},setAttributes:()=>{}},O=async(e,t)=>t(O,j),B=(e,{metrics:t,trace:a=O})=>{const o={"gen_ai.operation.name":"chat","gen_ai.request.model":e};return{wrapGenerate:async({doGenerate:r})=>a("ai.generate",async(c,v)=>{const f=await r();return I(e,f,v,t),f},o),wrapStream:async({doStream:r})=>{const c=(g,s)=>{const d=g.stream.getReader();let u={};return{...g,stream:new ReadableStream({cancel:async l=>{s({outcome:u}),await d.cancel(l)},pull:async l=>{let m;try{m=await d.read()}catch(E){s({error:E,failed:!0,outcome:u}),l.error(E);return}if(m.done){s({outcome:u}),l.close();return}m.value.type==="finish"&&(u={providerMetadata:m.value.providerMetadata,usage:m.value.usage}),l.enqueue(m.value)}})}},v=Promise.withResolvers(),f=Promise.withResolvers();return a("ai.stream",async(g,s)=>{v.resolve(c(await r(),f.resolve));const d=await f.promise;if(I(e,d.outcome,s,t),d.failed===!0)throw d.error},o).catch(v.reject),v.promise}}},L=(e,t,a)=>{let o=e;return t!==void 0&&typeof e!="string"&&(o=q({middleware:B(a??Y(e)??"unknown",t),model:e})),o},H=[W,S],N=e=>!e.startsWith("@")&&e.includes("/"),y=e=>{throw new w("INTERNAL",`@lunora/ai: ${e} needs the \`AI\` binding (env.AI). Add an \`ai\` binding to wrangler.jsonc, or set ${b} to an OpenAI-compatible proxy for "<provider>/<model>" slugs.`)},x=()=>y("this model id");x.textEmbeddingModel=()=>y("this embedding model id");const X=new Set(["127.0.0.1","[::1]","localhost"]),C=e=>{const t=A(e,b);if(t===void 0)return;const a=A(e,R),o=URL.canParse(t)?new URL(t):void 0;let r;if(o===void 0?r=`${b} is not a valid URL`:a!==void 0&&o.protocol!=="https:"&&!X.has(o.hostname)&&(r=`${b} (${o.origin}) is not HTTPS, so ${R} would travel in cleartext — use an https:// URL`),r!==void 0){const c=()=>{throw new w("INTERNAL",`@lunora/ai: ${r}`)};return{chat:c,embedding:c}}return G({apiKey:a??"",baseURL:t,name:"lunora-proxy"})},z=(e,t)=>{const a=e===void 0?void 0:V(e);return a===void 0?t:{...t,tags:{...a,...t?.tags}}},J=(e,t,a)=>{const o=T(a);if(e!==void 0)return o!==void 0&&e.metadata===void 0?{...e,metadata:o}:e;if(t===void 0)return;const r=A(t,F);if(r!==void 0)return K(t),o===void 0?{id:r}:{id:r,metadata:o}},ce=e=>{const{binding:t,defaultEmbeddingModel:a,defaultModel:o,env:r,gateway:c,metadata:v,provider:f,telemetry:g}=e,s=C(r),d=z(r,v),u=J(c,r,d),l=c?.metadata===void 0?T(d):void 0,m=o??A(r,M),E=a??A(r,_),p=f??(t?$({binding:t,gateway:u,providers:H}):x),D=n=>N(n)?s!==void 0?s.chat(n):l===void 0?p(n):p(n,{metadata:l}):p(n),P=n=>{const i=n??m;if(i===void 0||i==="")throw new w("INTERNAL",`@lunora/ai: no model supplied and no default configured — pass a model id, or set ${M} in the Worker env (wrangler \`vars\` / \`.dev.vars\`)`);return typeof i=="string"?L(D(i),g,i):L(i,g)},U=n=>{if(s!==void 0&&N(n))return s.embedding(n);const i=p.textEmbeddingModel;if(typeof i!="function")throw new w("INTERNAL","@lunora/ai: the Workers AI provider does not expose `textEmbeddingModel`; pass an AI SDK EmbeddingModel (e.g. from @ai-sdk/openai) to embed()");return i.call(p,n)};return{embeddingModel:n=>{if(typeof n=="object")return n;const i=n??E;if(!i)throw new w("INTERNAL",`@lunora/ai: no embedding model supplied and no default configured — pass an embedding model id or an AI SDK EmbeddingModel, or set ${_} in the Worker env (wrangler \`vars\` / \`.dev.vars\`)`);return U(i)},model:P,run:async(n,i,h)=>{if(!t)return y("ai.run");const k=u!==void 0&&h?.gateway===void 0?{...h,gateway:u}:h;return t.run(n,i,k)},workersai:p}};export{ce as default};
|
|
@@ -0,0 +1,8 @@
|
|
|
1
|
+
import{LunoraError as f,isLunoraError as Me}from"@lunora/errors";import{tool as Ae,jsonSchema as Ne,embedMany as De,embed as Re}from"ai";import{s as Ce}from"./stable-key-B_BlboiY.mjs";import{r as ie,m as Q}from"./usage-GPnTSqR8.mjs";import Be from"./fixedWindowChunks-C461ahRE.mjs";import{c as $e,I as Oe}from"./concurrent-C6nqBv41.mjs";import{contentHash as Ke}from"./contentHash-BIn6ECP8.mjs";import{hybridRank as Ue}from"./hybridRank-B4skyCLx.mjs";import{VECTORIZE_CAPABILITIES as de,vectorizeStore as Qe}from"./VECTORIZE_CAPABILITIES-CUQDoxis.mjs";const Ve=1e3,Fe=200,Le=5,ze=4,ce=de.maxMetadataBytes===!1?Number.POSITIVE_INFINITY:de.maxMetadataBytes,Pe=2*1024,he="__ragChunk",pe="__ragSource",B="__ragText",j="__ragHash",Y="__ragChunks",O="__ragImportance",ge="__ragModel",je=new Set([he,Y,j,O,ge,pe,B]),Ye=(e,l,h,w)=>{if(w===!1)return;const v=new TextEncoder().encode(JSON.stringify(e)).length;if(v<=w)return;const K=(typeof e[B]=="string"?new TextEncoder().encode(e[B]).length:0)*2>v?"lower `chunkSize` (it counts characters, not bytes — multibyte text costs up to 3 bytes each), or supply `textStore` to move chunk text out of metadata entirely":"attach less per-source `metadata`";throw new f("BAD_REQUEST",`@lunora/ai/rag: chunk ${String(l)} of "${h}" carries ${String(v)} bytes of metadata, over the store's ${String(w)}-byte per-vector ceiling — ${K}`)},He=(e,l,h)=>{if(h===!1)return;const w=new TextEncoder().encode(e).length;if(!(w<=h))throw new f("BAD_REQUEST",`@lunora/ai/rag: chunk id "${e}" for source "${l}" is ${String(w)} bytes, over the store's ${String(h)}-byte per-vector id ceiling — shorten the source id (hash long keys before indexing them) or shorten the \`namespace\`, which is prefixed onto every chunk id`)},qe=/^[\w.-]{1,40}$/,fe=e=>e===void 0?"":`${encodeURIComponent(e)}#`,C=(e,l,h)=>`${fe(e)}${l}#${String(h)}`,ue=(e,l)=>{const h=fe(l),w=h!==""&&e.startsWith(h)?e.slice(h.length):e,v=w.lastIndexOf("#"),M=v===-1?Number.NaN:Number(w.slice(v+1));return v===-1||!Number.isInteger(M)||M<0?{chunkIndex:0,sourceId:w}:{chunkIndex:M,sourceId:w.slice(0,v)}},We=async e=>Ke(new TextEncoder().encode(e)),Ze=e=>{try{return Ce([e.text,e.metadata,e.importance])}catch{return}},Xe=(e,l)=>{const h=[],w=[];for(const v of e)v.score>=l?h.push(v):w.push(v.id);return{kept:h,rejectedIds:w}},Je=e=>e!==void 0&&Object.keys(e).length>0,P=e=>{if(!e)return;const l=Object.entries(e).filter(([h])=>!je.has(h));return l.length>0?Object.fromEntries(l):void 0},le=new Set,Ge=e=>{le.has(e)||(le.add(e),console.warn(`[@lunora/ai/rag] index "${e}" is used without a namespace — in a multi-tenant/sharded
|
|
2
|
+
app this shares one tenant's chunks (text included) with every other tenant, since
|
|
3
|
+
Vectorize indexes are account-global. Pass \`namespace\` (the shard/tenant key) on both
|
|
4
|
+
index() and retrieve(). Single-tenant apps suppress this via { allowSharedNamespace: true }.`))},et=e=>e.map(l=>`[source:${l.sourceId}#${String(l.chunkIndex)}]
|
|
5
|
+
${l.text}`).join(`
|
|
6
|
+
|
|
7
|
+
`),me=(e,l)=>{if(typeof e=="object")return e;if(l===void 0)throw new f("INTERNAL","@lunora/ai/rag: `embeddingModel` is a Workers AI model id (or omitted) but the bound context has no `ai` (env.AI). Pass an AI SDK EmbeddingModel object to embed without Workers AI, or bind a context whose `ctx.ai` is wired.");return l.embeddingModel(e)},tt=e=>{if(e===void 0)throw new f("INTERNAL","@lunora/ai/rag: the bound context has no `vectors` (env.VECTORIZE) and no `store` is configured — bind a context whose `ctx.vectors` is wired, or configure `store` (e.g. `sqliteVectorStore`) to back this index without Vectorize.");return e},mt=e=>{if(typeof e.index!="string"||e.index.length===0)throw new f("BAD_REQUEST","@lunora/ai/rag: `index` must be a non-empty Vectorize index name");const l=e.chunkSize??Ve,h=e.chunkOverlap??Fe;if(!Number.isInteger(l)||l<1)throw new f("BAD_REQUEST","@lunora/ai/rag: `chunkSize` must be a positive integer");if(!Number.isInteger(h)||h<0||h>=l)throw new f("BAD_REQUEST","@lunora/ai/rag: `chunkOverlap` must be a non-negative integer smaller than `chunkSize`");const w=ce-Pe;if(!e.chunk&&!e.textStore&&!e.store&&l>w)throw new f("BAD_REQUEST",`@lunora/ai/rag: \`chunkSize\` of ${String(l)} leaves no room under Vectorize's ${String(ce)}-byte metadata limit, which also carries this chunk's bookkeeping and any \`metadata\` you attach — keep it under ${String(w)}, or supply \`textStore\` to move chunk text out of metadata entirely`);const v=e.topK??Le;if(!Number.isInteger(v)||v<1)throw new f("BAD_REQUEST","@lunora/ai/rag: `topK` must be a positive integer");if(e.maxEmbeddingDimensions!==void 0&&e.maxEmbeddingDimensions!==!1&&(!Number.isInteger(e.maxEmbeddingDimensions)||e.maxEmbeddingDimensions<1))throw new f("BAD_REQUEST","@lunora/ai/rag: `maxEmbeddingDimensions` must be a positive integer, or `false` to disable the check");if(e.embeddingModelVersion!==void 0&&!qe.test(e.embeddingModelVersion))throw new f("BAD_REQUEST",'@lunora/ai/rag: `embeddingModelVersion` must match /^[A-Za-z0-9._-]{1,40}$/ (a short, stable tag like "bge-v1.5")');if(e.candidates!==void 0&&(!Number.isInteger(e.candidates)||e.candidates<1))throw new f("BAD_REQUEST","@lunora/ai/rag: `candidates` must be a positive integer");if(e.cacheEmbeddings!==void 0&&(!Number.isInteger(e.cacheEmbeddings)||e.cacheEmbeddings<0))throw new f("BAD_REQUEST","@lunora/ai/rag: `cacheEmbeddings` must be a non-negative integer");const M=e.cacheEmbeddings??0,K=e.rerank,we=e.chunk??(p=>Be(p,l,h)),{textStore:k}=e,$=e.embeddingModelVersion,V=p=>$===void 0?p:p===void 0?$:`${$}::${p}`;return p=>{const E=e.store?e.store(p):Qe(tt(p.vectors),e.index),H=k?E.capabilities.maxTopK:E.capabilities.maxTopKWithMetadata,U=e.maxEmbeddingDimensions??E.capabilities.maxDimensions;let N;const q=typeof p.trace=="function"?p.trace:void 0,W=typeof p.metrics?.count=="function"?p.metrics:void 0;let Z=U===!1;const X=(t,r)=>{if(Z||(Z=!0,U===!1||t<=U))return;const n=Q(r);throw new f("BAD_REQUEST",`@lunora/ai/rag: embedding model${n===void 0?"":` "${n}"`} produces ${String(t)}-dimension vectors, over the ${String(U)}-dimension ceiling of index "${e.index}" — either truncate them with the provider's \`dimensions\` option (Matryoshka models such as text-embedding-3-large support this), or set \`maxEmbeddingDimensions: false\` if this index is not Vectorize-backed`)},D=new Map,be=(t,r)=>{if(M!==0)for(D.set(t,r);D.size>M;){const n=D.keys().next();if(n.done===!0)break;D.delete(n.value)}},J=async t=>{const r=D.get(t);if(r!==void 0)return r;N??=me(e.embeddingModel,p.ai);const n=N,o=async a=>{const{embedding:c,providerMetadata:b,usage:d}=await Re({model:n,value:t});return X(c.length,n),ie(Q(n),{providerMetadata:b,usage:{inputTokens:{total:d.tokens}}},a,W),be(t,c),c};if(q===void 0)return o();const s=Q(n),i=typeof p.conversationId=="string"&&p.conversationId.length>0?p.conversationId:void 0;return q("ai.embed",(a,c)=>o(c),{"gen_ai.operation.name":"embeddings",...s===void 0?{}:{"gen_ai.request.model":s},...i===void 0?{}:{"gen_ai.conversation.id":i}})},ve=async(t,r,n)=>{if(!e.transformQuery||r?.transformQuery===!1)return[t];const o=typeof p.conversationId=="string"&&p.conversationId.length>0?p.conversationId:void 0,s=await e.transformQuery(t,{conversationId:o,namespace:n}),i=(typeof s=="string"?[s]:[...s]).map(a=>a.trim()).filter(a=>a.length>0);return i.length>0?i:[t]},ye=async t=>{const r=new Map,n=[...new Set(t.filter(o=>!D.has(o)))];if(n.length<2)return r;N??=me(e.embeddingModel,p.ai);try{const{embeddings:o,providerMetadata:s,usage:i}=await De({model:N,values:n});if(ie(Q(N),{providerMetadata:s,usage:{inputTokens:{total:i.tokens}}},void 0,W),o.length!==n.length)return r;const[a]=o;a!==void 0&&X(a.length,N);for(const[c,b]of n.entries())r.set(b,o[c])}catch(o){if(Me(o))throw o}return r},F=t=>{if(t===void 0){if(e.requireNamespace)throw new f("BAD_REQUEST",`@lunora/ai/rag: index "${e.index}" requires a namespace (requireNamespace is set) — pass the tenant/shard key on index()/retrieve()/remove()`);e.allowSharedNamespace||Ge(e.index)}},G=async(t,r)=>{const[n]=await E.getByIds([C(r,t,0)],r),o=n?.metadata?.[j],s=n?.metadata?.[Y];return{chunks:typeof s=="number"&&Number.isInteger(s)&&s>0?s:void 0,hash:typeof o=="string"?o:void 0}},ee=async(t,r,n,o)=>{const s=Array.from({length:n-r},(i,a)=>C(o,t,r+a));s.length!==0&&(await E.deleteByIds(s,o),await k?.remove?.(s,{namespace:o}),await e.lexicalStore?.remove?.(s,{namespace:o}))},Ee=async t=>{if(F(t.namespace),t.importance!==void 0&&(typeof t.importance!="number"||t.importance<0||t.importance>1))throw new f("BAD_REQUEST","@lunora/ai/rag: `importance` must be a number in [0, 1]");const r=V(t.namespace),n=Ze(t),o=await We(n??t.text),s=await G(t.id,r);if(t.reindex!==!0&&n!==void 0&&s.hash===o&&s.chunks!==void 0)return{chunks:s.chunks,ids:Array.from({length:s.chunks},(m,u)=>C(r,t.id,u)),unchanged:!0};const i=we(t.text),a=i.map((m,u)=>C(r,t.id,u)),c=a.at(-1);if(c!==void 0&&He(c,t.id,E.capabilities.maxIdBytes),i.length===0&&t.allowEmptySources===!1)throw new f("BAD_REQUEST",`@lunora/ai/rag: source "${t.id}" produced zero chunks — set allowEmptySources: true to allow this`);if(i.length>0){const m=i.map((u,y)=>({chunkIndex:y,id:a[y],sourceId:t.id,text:u,...t.metadata===void 0?{}:{metadata:t.metadata}}));k&&await k.put(m,{namespace:r}),e.lexicalStore&&await e.lexicalStore.index(m,{namespace:r})}const b=await ye(i),d=async m=>b.get(m)??await J(m);return await $e(i,Oe,async(m,u)=>{const y=a[u],x={...t.metadata,[he]:u,[pe]:t.id};k||(x[B]=m),t.importance!==void 0&&(x[O]=t.importance),u===0&&(x[j]=o,x[Y]=i.length,$!==void 0&&(x[ge]=$)),Ye(x,u,t.id,E.capabilities.maxMetadataBytes),await E.upsert({embed:d,id:y,input:m,metadata:x,namespace:r}),t.onChunk?.({chunkIndex:u,id:y,text:m,total:i.length})}),s.chunks!==void 0&&s.chunks>i.length&&await ee(t.id,i.length,s.chunks,r),{chunks:i.length,ids:a,unchanged:!1}},xe=async t=>{F(t.namespace);const r=V(t.namespace),o=(await G(t.id,r)).chunks??1;await ee(t.id,0,o,r)},te=async(t,r)=>{const n=new Map;if(t.length===0)return n;if(k){const s=await k.getMany(t,{namespace:r});for(const[i,a]of t.entries()){const c=s[i];typeof c=="string"&&n.set(a,c)}return n}const o=await E.getByIds(t,r);for(const s of o){const i=s.metadata?.[B];typeof i=="string"&&n.set(s.id,i)}return n},Ie=async(t,r,n)=>{const o=r?.chunkContext?.before??0,s=r?.chunkContext?.after??0;if(o===0&&s===0)return t;if(!Number.isInteger(o)||o<0||!Number.isInteger(s)||s<0)throw new f("BAD_REQUEST","@lunora/ai/rag: `chunkContext.before`/`chunkContext.after` must be non-negative integers");const i=new Map(t.map(d=>[d.id,d.text])),a=new Set;for(const d of t)for(let m=-o;m<=s;m+=1){const u=d.chunkIndex+m,y=C(n,d.sourceId,u);m!==0&&u>=0&&!i.has(y)&&a.add(y)}const c=await te([...a],n),b=(d,m)=>{const u=C(n,d,m);return i.get(u)??c.get(u)};return t.map(d=>{const m=[];for(let u=-o;u<=s;u+=1){const y=u===0?d.text:b(d.sourceId,d.chunkIndex+u);y!==void 0&&m.push(y)}return{...d,text:m.join(`
|
|
8
|
+
`)}})},Se=t=>{if(typeof t=="string"){const r=e.filters!==void 0&&Object.hasOwn(e.filters,t)?e.filters[t]:void 0;if(!r)throw new f("NOT_FOUND",`@lunora/ai/rag: unknown named filter "${t}" — must be one of the keys declared in RagConfig.filters`);return r.filter}return t},ke=(t,r)=>t.matches.map(n=>{const o=n.metadata??{},s=ue(n.id,r),i=o[B],a=o[O],c=typeof a=="number"&&a>=0&&a<=1?a:1;return{chunkIndex:s.chunkIndex,id:n.id,importance:c,metadata:P(o),score:n.score*c,sourceId:s.sourceId,text:typeof i=="string"?i:""}}),_e=async(t,r)=>{if(!k)return t;const n=t.map(a=>a.id),[o,s]=await Promise.all([te(n,r),E.getByIds(n,r)]),i=new Map(s.map(a=>[a.id,a.metadata]));return t.flatMap(a=>{const c=o.get(a.id);if(c===void 0)return[];const b=i.get(a.id),d=b?.[O],m=typeof d=="number"&&d>=0&&d<=1?d:a.importance,y=(a.importance===0?0:a.score/a.importance)*m;return[{...a,importance:m,metadata:P(b)??a.metadata,score:y,text:c}]})},re=async(t,r,n)=>{const o=t.map(a=>a.id).filter(a=>!r.has(a)),s=o.length===0?[]:await E.getByIds(o,n),i=new Map(s.map(a=>[a.id,a.metadata]));return t.map(a=>{const c=ue(a.id,n),b=i.get(a.id),d=b?.[O];return{chunkIndex:c.chunkIndex,id:a.id,importance:typeof d=="number"&&d>=0&&d<=1?d:1,metadata:P(b),score:a.score,sourceId:c.sourceId,text:a.text}})},ne=async(t,r)=>{F(r?.namespace);const n=V(r?.namespace),o=Se(r?.filter),s=e.rlsFilter?await e.rlsFilter(p.auth):void 0,i=s?{...o,...s}:o,a=Math.min(r?.topK??v,H),c=await ve(t,r,n),b=c[0],d=K!==void 0&&r?.rerank!==!1,u=d||e.lexicalStore!==void 0||e.graphStore!==void 0||c.length>1?Math.min(e.candidates??a*ze,H):a,y=r?.minScore,x=new Set,ae=async g=>{const _=await E.query({embed:J,filter:i,input:g,namespace:n,returnMetadata:k?"indexed":"all",topK:u}),A=await _e(ke(_,n),n);if(y===void 0)return A;const{kept:S,rejectedIds:T}=Xe(A,y);for(const Te of T)x.add(Te);return S},R=[{chunks:await ae(b)}];for(const g of c.slice(1))R.push({chunks:await ae(g)});if(e.lexicalStore){const g=await e.lexicalStore.search(b,{filter:i,namespace:n,topK:e.lexicalTopK??u}),_=new Set(R.flatMap(S=>S.chunks).map(S=>S.id)),A=g.filter(S=>!x.has(S.id)||_.has(S.id));R.push({chunks:await re(A,_,n)})}const L=R.flatMap(g=>[...g.chunks]);if(e.graphStore&&L.length>0&&(e.graphStore.enforcesFilter||!Je(i))){const g=[...new Set(L.map(T=>T.sourceId))],_=await e.graphStore.related(g,{filter:i,namespace:n,topK:e.graphTopK??u}),A=new Set(L.map(T=>T.id)),S=_.filter(T=>!x.has(T.id)||A.has(T.id));R.push({chunks:await re(S,A,n),weight:"proximity"})}const z=R.filter(g=>g.chunks.length>0);let I=z.length>1?[...Ue(z)]:[...z[0]?.chunks??[]];I.sort((g,_)=>_.score-g.score),d&&(I=[...await K(b,I)]),I=I.slice(0,a),I=[...await Ie(I,r,n)];const se=[],oe=new Set;for(const g of I)oe.has(g.sourceId)||(oe.add(g.sourceId),se.push({id:g.sourceId,metadata:g.metadata,weight:g.importance}));return r?.onRetrieve?.({matches:I.length,query:t}),{chunks:I,context:et(I),sources:se}};return{asTool:t=>Ae({description:t?.description??`Search the "${e.index}" knowledge base for passages relevant to a natural-language query.`,execute:async({query:r})=>ne(r,{namespace:t?.namespace,topK:t?.topK}),inputSchema:Ne({properties:{query:{description:"The natural-language search query.",type:"string"}},required:["query"],type:"object"})}),index:Ee,remove:xe,retrieve:ne}}};export{mt as default};
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
import x from"./matchesMetadataFilter-BbIOyA5g.mjs";import{p as I,a as q,r as y,c as F,i as O}from"./sql-D5aqEMCY.mjs";import{LunoraError as _}from"@lunora/errors";const p=500,C=4096,R=(d,r)=>JSON.stringify([d,r]),D=d=>{const{dimensions:r,exec:o,table:m}=d,s=`${m}_ann`,E=`${m}_ann_ready`,u=(a,t)=>{if(a!==r)throw new _("RAG_DIMENSION_MISMATCH",`@lunora/ai/rag: ${t} is ${String(a)}-dimension but the \`ann\` index is ${String(r)}-dimension — set \`ann.dimensions\` to the embedding model's width, or reindex into a new \`table\` after changing models`)},g=async(a,t,i)=>{await o(`INSERT INTO ${s} (ref, namespace, embedding, id) VALUES (?, ?, ?, ?)`,[R(a,t),a,i,t])},S=async()=>{let a=0;for(;;){const t=await o(`SELECT rowid, id, namespace, vector FROM ${m} WHERE rowid > ? ORDER BY rowid LIMIT ?`,[a,p]);for(const i of t)u(JSON.parse(String(i.vector)).length,`stored vector "${String(i.id)}"`),await g(String(i.namespace),String(i.id),String(i.vector));if(t.length<p)return;a=Number(t.at(-1)?.rowid)}};return{ensure:async()=>{if((await o("SELECT name FROM sqlite_master WHERE name = ?",[E])).length>0){const[t]=await o(`SELECT dimensions FROM ${E}`,[]);u(Number(t?.dimensions),"the existing index");return}await o(`DROP TABLE IF EXISTS ${s}`,[]);try{await o(`CREATE VIRTUAL TABLE ${s} USING vec0(ref TEXT PRIMARY KEY, namespace TEXT PARTITION KEY, embedding FLOAT[${String(r)}] distance_metric=cosine, +id TEXT)`,[])}catch(t){throw t instanceof Error&&t.message.includes("no such module: vec0")?new Error("@lunora/ai/rag: sqliteVectorStore `ann` needs the sqlite-vec extension in this SQLite (`no such module: vec0`) — on celld set the `sqlite_vec` compatibility flag, on node:sqlite load the extension",{cause:t}):t}await S(),await o(`CREATE TABLE ${E} (dimensions INTEGER NOT NULL)`,[]),await o(`INSERT INTO ${E} (dimensions) VALUES (?)`,[r])},nearest:async(a,t,i)=>(u(t.length,"the query embedding"),(await o(`SELECT id, distance FROM ${s} WHERE embedding MATCH ? AND k = ? AND namespace = ?`,[JSON.stringify([...t]),Math.min(i,C),a])).map(e=>({distance:Number(e.distance),id:String(e.id)}))),put:async(a,t,i)=>{u(i.length,`the embedding for "${t}"`),await o(`DELETE FROM ${s} WHERE ref = ?`,[R(a,t)]),await g(a,t,JSON.stringify([...i]))},remove:async(a,t)=>{await o(`DELETE FROM ${s} WHERE ref IN (${I(t.length)})`,t.map(i=>R(a,i)))}}},B="lunora_rag_vectors",U=5e4,$=100,V=null,h=d=>d??"",W=d=>{if(typeof d.exec!="function")throw new TypeError("@lunora/ai/rag: sqliteVectorStore requires an `exec` function");const r=q(d.table??B,"sqliteVectorStore `table`"),o=d.maxScan??U,{ann:m,exec:s}=d;if(m!==void 0&&(!Number.isInteger(m.dimensions)||m.dimensions<=0))throw new TypeError("@lunora/ai/rag: sqliteVectorStore `ann.dimensions` must be a positive integer");const E=m===void 0?void 0:D({dimensions:m.dimensions,exec:s,table:r}),u={maxDimensions:d.maxDimensions??m?.dimensions??!1,maxIdBytes:!1,maxMetadataBytes:!1,maxTopK:$,maxTopKWithMetadata:$};let g;const S=async()=>{g??=(async()=>{await s(`CREATE TABLE IF NOT EXISTS ${r} (id TEXT NOT NULL, namespace TEXT NOT NULL DEFAULT '', vector TEXT NOT NULL, metadata TEXT, PRIMARY KEY (namespace, id))`,[]),await s(`CREATE INDEX IF NOT EXISTS ${r}_namespace ON ${r} (namespace)`,[]),await E?.ensure()})().catch(e=>{throw g=void 0,e}),await g},b=async e=>{if(await S(),!e.embed)throw new TypeError("@lunora/ai/rag: sqliteVectorStore requires an `embed` function on upsert");const c=await e.embed(e.input),n=h(e.namespace);await E?.put(n,e.id,c),await s(`INSERT INTO ${r} (id, namespace, vector, metadata) VALUES (${I(4)}) ON CONFLICT(namespace, id) DO UPDATE SET vector = excluded.vector, metadata = excluded.metadata`,[e.id,n,JSON.stringify([...c]),e.metadata===void 0?V:JSON.stringify(e.metadata)])},a=async(e,c)=>{if(await S(),e.length===0)return[];const n=[];for(const w of O(e)){const f=await s(`SELECT id, metadata FROM ${r} WHERE namespace = ? AND id IN (${I(w.length)})`,[h(c),...w]);for(const l of f){const T=y(l.metadata);n.push({id:String(l.id),...T===void 0?{}:{metadata:T}})}}return n},t=async(e,c,n)=>{const w=h(n.namespace),f=await e.nearest(w,c,n.topK??10),l=await a(f.map(L=>L.id),w),T=new Map(l.map(L=>[L.id,L])),N=[];for(const{distance:L,id:A}of f){const v=T.get(A);v!==void 0&&N.push({id:A,score:1-L,...n.returnMetadata==="none"||v.metadata===void 0?{}:{metadata:v.metadata}})}return{count:N.length,matches:N}};return{capabilities:u,deleteByIds:async(e,c)=>{if(await S(),e.length!==0)for(const n of O(e))await s(`DELETE FROM ${r} WHERE namespace = ? AND id IN (${I(n.length)})`,[h(c),...n]),await E?.remove(h(c),n)},getByIds:a,query:async e=>{await S();let c;if(e.embed&&e.input!==void 0)c=await e.embed(e.input);else throw new TypeError("@lunora/ai/rag: sqliteVectorStore query requires both `input` and `embed`");if(E!==void 0&&e.filter===void 0)return t(E,c,e);const n=await s(`SELECT id, vector, metadata FROM ${r} WHERE namespace = ? LIMIT ?`,[h(e.namespace),o+1]);if(n.length>o)throw new RangeError(`@lunora/ai/rag: sqliteVectorStore scanned ${String(n.length)} vectors in namespace "${h(e.namespace)}", over the ${String(o)} limit — search here is brute force and linear, so this namespace has outgrown it. Shard it further, or move this index to Vectorize or a pgvector backend`);const w=[];for(const l of n){const T=y(l.metadata);if(!x(T,e.filter))continue;const N=y(l.vector);N!==void 0&&w.push({id:String(l.id),score:F(c,N),...e.returnMetadata==="none"||T===void 0?{}:{metadata:T}})}const f=w.toSorted((l,T)=>T.score-l.score).slice(0,e.topK??10);return{count:f.length,matches:f}},upsert:b}};export{W as sqliteVectorStore};
|
|
@@ -140,6 +140,19 @@ declare const AI_GATEWAY_TOKEN_ENV = "LUNORA_AI_GATEWAY_TOKEN";
|
|
|
140
140
|
* call — telemetry configuration must not take inference down.
|
|
141
141
|
*/
|
|
142
142
|
declare const AI_GATEWAY_TAGS_ENV = "LUNORA_AI_GATEWAY_TAGS";
|
|
143
|
+
/**
|
|
144
|
+
* Env var carrying the base URL of a self-hosted OpenAI-compatible proxy
|
|
145
|
+
* (LiteLLM, OpenRouter, your own), e.g. `https://ai-proxy.internal/v1`.
|
|
146
|
+
*
|
|
147
|
+
* When set, `ctx.ai.model("<provider>/<model>")` sends the slug unchanged as the
|
|
148
|
+
* `model` of an OpenAI chat-completions request to this URL instead of routing
|
|
149
|
+
* it through Cloudflare AI Gateway — which needs no `AI` binding, so it is how
|
|
150
|
+
* `ctx.ai` works on hosts without Workers AI (celld). `@cf/…` ids still need
|
|
151
|
+
* the binding.
|
|
152
|
+
*/
|
|
153
|
+
declare const AI_PROXY_URL_ENV = "LUNORA_AI_PROXY_URL";
|
|
154
|
+
/** Env var carrying the bearer token sent to {@link AI_PROXY_URL_ENV}, when the proxy requires one. */
|
|
155
|
+
declare const AI_PROXY_TOKEN_ENV = "LUNORA_AI_PROXY_TOKEN";
|
|
143
156
|
/**
|
|
144
157
|
* Parse {@link AI_GATEWAY_TAGS_ENV} into tag fields. Non-string values are
|
|
145
158
|
* dropped rather than coerced — a number silently becoming `"1"` is a worse
|
|
@@ -203,6 +216,43 @@ interface AiGatewayOptions {
|
|
|
203
216
|
*/
|
|
204
217
|
metadata?: Record<string, string>;
|
|
205
218
|
}
|
|
219
|
+
/**
|
|
220
|
+
* Structural slice of the span handle `ctx.trace` hands its body — enough to
|
|
221
|
+
* attach a model call's usage once it is known. Declared here rather than
|
|
222
|
+
* imported so `@lunora/ai` takes no dependency on `@lunora/server`; the real
|
|
223
|
+
* handle is assignable to it.
|
|
224
|
+
* @experimental
|
|
225
|
+
*/
|
|
226
|
+
interface AiSpan {
|
|
227
|
+
setAttribute: (key: string, value: unknown) => void;
|
|
228
|
+
setAttributes: (fields: Record<string, unknown>) => void;
|
|
229
|
+
}
|
|
230
|
+
/**
|
|
231
|
+
* Structural slice of `ctx.trace` (the server `LunoraTracer`): runs `function_`
|
|
232
|
+
* inside a named span and hands it the span's {@link AiSpan}.
|
|
233
|
+
* @experimental
|
|
234
|
+
*/
|
|
235
|
+
type AiTracer = <T>(name: string, function_: (trace: AiTracer, span: AiSpan) => Promise<T> | T, attributes?: Record<string, unknown>) => Promise<T>;
|
|
236
|
+
/**
|
|
237
|
+
* Structural slice of `ctx.metrics` (the server `LunoraMetrics`) — only the
|
|
238
|
+
* counter, which is what usage accounting needs.
|
|
239
|
+
* @experimental
|
|
240
|
+
*/
|
|
241
|
+
interface AiMetrics {
|
|
242
|
+
count: (name: string, value?: number, attributes?: Record<string, unknown>) => void;
|
|
243
|
+
}
|
|
244
|
+
/**
|
|
245
|
+
* Where `ctx.ai` reports model usage. The generated `ctx.ai` passes the
|
|
246
|
+
* function's own `ctx.trace` / `ctx.metrics`, so every call made through
|
|
247
|
+
* `ctx.ai.model(...)` gets an `ai.generate` / `ai.stream` span and
|
|
248
|
+
* `gen_ai.usage.input_tokens` / `gen_ai.usage.output_tokens` /
|
|
249
|
+
* `gen_ai.usage.cost` counters attributed to that function.
|
|
250
|
+
* @experimental
|
|
251
|
+
*/
|
|
252
|
+
interface AiTelemetry {
|
|
253
|
+
metrics?: AiMetrics;
|
|
254
|
+
trace?: AiTracer;
|
|
255
|
+
}
|
|
206
256
|
/**
|
|
207
257
|
* `LunoraAiOptions` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
208
258
|
* @experimental
|
|
@@ -267,13 +317,21 @@ interface LunoraAiOptions {
|
|
|
267
317
|
* (e.g. `safePrompt`) before handing it to `@lunora/ai`.
|
|
268
318
|
*/
|
|
269
319
|
provider?: WorkersAiProviderLike;
|
|
320
|
+
/**
|
|
321
|
+
* Record every language-model call resolved by `model()` — a span plus
|
|
322
|
+
* token and cost counters (see {@link AiTelemetry}). Omitted, models are
|
|
323
|
+
* returned unwrapped.
|
|
324
|
+
*/
|
|
325
|
+
telemetry?: AiTelemetry;
|
|
270
326
|
}
|
|
271
327
|
/**
|
|
272
328
|
* A model to run against. The AI SDK's {@link LanguageModel} already admits a
|
|
273
329
|
* bare `string`, so this alias covers both arms of the provider-agnostic seam:
|
|
274
|
-
* a string id is
|
|
275
|
-
*
|
|
276
|
-
*
|
|
330
|
+
* a string id is resolved by `ctx.ai.model` — a Workers AI id (`@cf/…`), a
|
|
331
|
+
* `"<provider>/<model>"` slug (`anthropic/claude-sonnet-5`, `openai/gpt-5`, …)
|
|
332
|
+
* routed through Cloudflare AI Gateway, or a gateway dynamic route
|
|
333
|
+
* (`dynamic/<route>`); a built model object is bring-your-own (`@ai-sdk/openai`,
|
|
334
|
+
* `@ai-sdk/anthropic`, `@ai-sdk/google`, OpenRouter, …).
|
|
277
335
|
* @experimental
|
|
278
336
|
*/
|
|
279
337
|
type ModelInput = LanguageModel;
|
|
@@ -295,7 +353,13 @@ type EmbeddingModelInput = EmbeddingModel | string;
|
|
|
295
353
|
interface LunoraAi {
|
|
296
354
|
/** Resolve an {@link EmbeddingModel}: a string → Workers AI, an object → passthrough. */
|
|
297
355
|
embeddingModel: (model?: EmbeddingModelInput) => EmbeddingModel;
|
|
298
|
-
/**
|
|
356
|
+
/**
|
|
357
|
+
* Resolve a {@link LanguageModel}: a `@cf/…` id → Workers AI, a
|
|
358
|
+
* `"<provider>/<model>"` slug or `dynamic/<route>` → Cloudflare AI Gateway
|
|
359
|
+
* over the same binding (Unified Billing or the gateway's stored keys — no
|
|
360
|
+
* provider key in the app) or, with `LUNORA_AI_PROXY_URL` set, to that
|
|
361
|
+
* OpenAI-compatible proxy; an object → passthrough.
|
|
362
|
+
*/
|
|
299
363
|
model: (model?: ModelInput) => LanguageModel;
|
|
300
364
|
/**
|
|
301
365
|
* Raw Workers AI binding passthrough (void-style `ai.run`). Bypasses the AI
|
|
@@ -307,4 +371,4 @@ interface LunoraAi {
|
|
|
307
371
|
/** The underlying Workers AI provider — `ai.workersai("@cf/...")` for a raw model. */
|
|
308
372
|
workersai: WorkersAiProviderLike;
|
|
309
373
|
}
|
|
310
|
-
export { AI_DEFAULT_EMBEDDING_MODEL_ENV as A, EmbeddingModelInput as E, LunoraAiOptions as L, ModelInput as M, ResolvedAiGateway as R, WorkersAiProviderLike as W, LunoraAi as a, AI_DEFAULT_MODEL_ENV as b, AI_GATEWAY_ACCOUNT_ID_ENV as c, AI_GATEWAY_ID_ENV as d, AI_GATEWAY_METADATA_MAX_KEYS as e, AI_GATEWAY_TAGS_ENV as f, AI_GATEWAY_TOKEN_ENV as g,
|
|
374
|
+
export { AI_DEFAULT_EMBEDDING_MODEL_ENV as A, EmbeddingModelInput as E, LunoraAiOptions as L, ModelInput as M, ResolvedAiGateway as R, WorkersAiProviderLike as W, LunoraAi as a, AI_DEFAULT_MODEL_ENV as b, AI_GATEWAY_ACCOUNT_ID_ENV as c, AI_GATEWAY_ID_ENV as d, AI_GATEWAY_METADATA_MAX_KEYS as e, AI_GATEWAY_TAGS_ENV as f, AI_GATEWAY_TOKEN_ENV as g, AI_PROXY_TOKEN_ENV as h, AI_PROXY_URL_ENV as i, AiBindingLike as j, AiGatewayMetadata as k, AiGatewayOptions as l, AiMetrics as m, AiSpan as n, AiTelemetry as o, AiTracer as p, buildAiGatewayMetadataFields as q, readAiGatewayEnvTags as r, resolveAiGateway as s };
|
|
@@ -140,6 +140,19 @@ declare const AI_GATEWAY_TOKEN_ENV = "LUNORA_AI_GATEWAY_TOKEN";
|
|
|
140
140
|
* call — telemetry configuration must not take inference down.
|
|
141
141
|
*/
|
|
142
142
|
declare const AI_GATEWAY_TAGS_ENV = "LUNORA_AI_GATEWAY_TAGS";
|
|
143
|
+
/**
|
|
144
|
+
* Env var carrying the base URL of a self-hosted OpenAI-compatible proxy
|
|
145
|
+
* (LiteLLM, OpenRouter, your own), e.g. `https://ai-proxy.internal/v1`.
|
|
146
|
+
*
|
|
147
|
+
* When set, `ctx.ai.model("<provider>/<model>")` sends the slug unchanged as the
|
|
148
|
+
* `model` of an OpenAI chat-completions request to this URL instead of routing
|
|
149
|
+
* it through Cloudflare AI Gateway — which needs no `AI` binding, so it is how
|
|
150
|
+
* `ctx.ai` works on hosts without Workers AI (celld). `@cf/…` ids still need
|
|
151
|
+
* the binding.
|
|
152
|
+
*/
|
|
153
|
+
declare const AI_PROXY_URL_ENV = "LUNORA_AI_PROXY_URL";
|
|
154
|
+
/** Env var carrying the bearer token sent to {@link AI_PROXY_URL_ENV}, when the proxy requires one. */
|
|
155
|
+
declare const AI_PROXY_TOKEN_ENV = "LUNORA_AI_PROXY_TOKEN";
|
|
143
156
|
/**
|
|
144
157
|
* Parse {@link AI_GATEWAY_TAGS_ENV} into tag fields. Non-string values are
|
|
145
158
|
* dropped rather than coerced — a number silently becoming `"1"` is a worse
|
|
@@ -203,6 +216,43 @@ interface AiGatewayOptions {
|
|
|
203
216
|
*/
|
|
204
217
|
metadata?: Record<string, string>;
|
|
205
218
|
}
|
|
219
|
+
/**
|
|
220
|
+
* Structural slice of the span handle `ctx.trace` hands its body — enough to
|
|
221
|
+
* attach a model call's usage once it is known. Declared here rather than
|
|
222
|
+
* imported so `@lunora/ai` takes no dependency on `@lunora/server`; the real
|
|
223
|
+
* handle is assignable to it.
|
|
224
|
+
* @experimental
|
|
225
|
+
*/
|
|
226
|
+
interface AiSpan {
|
|
227
|
+
setAttribute: (key: string, value: unknown) => void;
|
|
228
|
+
setAttributes: (fields: Record<string, unknown>) => void;
|
|
229
|
+
}
|
|
230
|
+
/**
|
|
231
|
+
* Structural slice of `ctx.trace` (the server `LunoraTracer`): runs `function_`
|
|
232
|
+
* inside a named span and hands it the span's {@link AiSpan}.
|
|
233
|
+
* @experimental
|
|
234
|
+
*/
|
|
235
|
+
type AiTracer = <T>(name: string, function_: (trace: AiTracer, span: AiSpan) => Promise<T> | T, attributes?: Record<string, unknown>) => Promise<T>;
|
|
236
|
+
/**
|
|
237
|
+
* Structural slice of `ctx.metrics` (the server `LunoraMetrics`) — only the
|
|
238
|
+
* counter, which is what usage accounting needs.
|
|
239
|
+
* @experimental
|
|
240
|
+
*/
|
|
241
|
+
interface AiMetrics {
|
|
242
|
+
count: (name: string, value?: number, attributes?: Record<string, unknown>) => void;
|
|
243
|
+
}
|
|
244
|
+
/**
|
|
245
|
+
* Where `ctx.ai` reports model usage. The generated `ctx.ai` passes the
|
|
246
|
+
* function's own `ctx.trace` / `ctx.metrics`, so every call made through
|
|
247
|
+
* `ctx.ai.model(...)` gets an `ai.generate` / `ai.stream` span and
|
|
248
|
+
* `gen_ai.usage.input_tokens` / `gen_ai.usage.output_tokens` /
|
|
249
|
+
* `gen_ai.usage.cost` counters attributed to that function.
|
|
250
|
+
* @experimental
|
|
251
|
+
*/
|
|
252
|
+
interface AiTelemetry {
|
|
253
|
+
metrics?: AiMetrics;
|
|
254
|
+
trace?: AiTracer;
|
|
255
|
+
}
|
|
206
256
|
/**
|
|
207
257
|
* `LunoraAiOptions` is part of the experimental `@lunora/ai` API and may change without a major version bump.
|
|
208
258
|
* @experimental
|
|
@@ -267,13 +317,21 @@ interface LunoraAiOptions {
|
|
|
267
317
|
* (e.g. `safePrompt`) before handing it to `@lunora/ai`.
|
|
268
318
|
*/
|
|
269
319
|
provider?: WorkersAiProviderLike;
|
|
320
|
+
/**
|
|
321
|
+
* Record every language-model call resolved by `model()` — a span plus
|
|
322
|
+
* token and cost counters (see {@link AiTelemetry}). Omitted, models are
|
|
323
|
+
* returned unwrapped.
|
|
324
|
+
*/
|
|
325
|
+
telemetry?: AiTelemetry;
|
|
270
326
|
}
|
|
271
327
|
/**
|
|
272
328
|
* A model to run against. The AI SDK's {@link LanguageModel} already admits a
|
|
273
329
|
* bare `string`, so this alias covers both arms of the provider-agnostic seam:
|
|
274
|
-
* a string id is
|
|
275
|
-
*
|
|
276
|
-
*
|
|
330
|
+
* a string id is resolved by `ctx.ai.model` — a Workers AI id (`@cf/…`), a
|
|
331
|
+
* `"<provider>/<model>"` slug (`anthropic/claude-sonnet-5`, `openai/gpt-5`, …)
|
|
332
|
+
* routed through Cloudflare AI Gateway, or a gateway dynamic route
|
|
333
|
+
* (`dynamic/<route>`); a built model object is bring-your-own (`@ai-sdk/openai`,
|
|
334
|
+
* `@ai-sdk/anthropic`, `@ai-sdk/google`, OpenRouter, …).
|
|
277
335
|
* @experimental
|
|
278
336
|
*/
|
|
279
337
|
type ModelInput = LanguageModel;
|
|
@@ -295,7 +353,13 @@ type EmbeddingModelInput = EmbeddingModel | string;
|
|
|
295
353
|
interface LunoraAi {
|
|
296
354
|
/** Resolve an {@link EmbeddingModel}: a string → Workers AI, an object → passthrough. */
|
|
297
355
|
embeddingModel: (model?: EmbeddingModelInput) => EmbeddingModel;
|
|
298
|
-
/**
|
|
356
|
+
/**
|
|
357
|
+
* Resolve a {@link LanguageModel}: a `@cf/…` id → Workers AI, a
|
|
358
|
+
* `"<provider>/<model>"` slug or `dynamic/<route>` → Cloudflare AI Gateway
|
|
359
|
+
* over the same binding (Unified Billing or the gateway's stored keys — no
|
|
360
|
+
* provider key in the app) or, with `LUNORA_AI_PROXY_URL` set, to that
|
|
361
|
+
* OpenAI-compatible proxy; an object → passthrough.
|
|
362
|
+
*/
|
|
299
363
|
model: (model?: ModelInput) => LanguageModel;
|
|
300
364
|
/**
|
|
301
365
|
* Raw Workers AI binding passthrough (void-style `ai.run`). Bypasses the AI
|
|
@@ -307,4 +371,4 @@ interface LunoraAi {
|
|
|
307
371
|
/** The underlying Workers AI provider — `ai.workersai("@cf/...")` for a raw model. */
|
|
308
372
|
workersai: WorkersAiProviderLike;
|
|
309
373
|
}
|
|
310
|
-
export { AI_DEFAULT_EMBEDDING_MODEL_ENV as A, EmbeddingModelInput as E, LunoraAiOptions as L, ModelInput as M, ResolvedAiGateway as R, WorkersAiProviderLike as W, LunoraAi as a, AI_DEFAULT_MODEL_ENV as b, AI_GATEWAY_ACCOUNT_ID_ENV as c, AI_GATEWAY_ID_ENV as d, AI_GATEWAY_METADATA_MAX_KEYS as e, AI_GATEWAY_TAGS_ENV as f, AI_GATEWAY_TOKEN_ENV as g,
|
|
374
|
+
export { AI_DEFAULT_EMBEDDING_MODEL_ENV as A, EmbeddingModelInput as E, LunoraAiOptions as L, ModelInput as M, ResolvedAiGateway as R, WorkersAiProviderLike as W, LunoraAi as a, AI_DEFAULT_MODEL_ENV as b, AI_GATEWAY_ACCOUNT_ID_ENV as c, AI_GATEWAY_ID_ENV as d, AI_GATEWAY_METADATA_MAX_KEYS as e, AI_GATEWAY_TAGS_ENV as f, AI_GATEWAY_TOKEN_ENV as g, AI_PROXY_TOKEN_ENV as h, AI_PROXY_URL_ENV as i, AiBindingLike as j, AiGatewayMetadata as k, AiGatewayOptions as l, AiMetrics as m, AiSpan as n, AiTelemetry as o, AiTracer as p, buildAiGatewayMetadataFields as q, readAiGatewayEnvTags as r, resolveAiGateway as s };
|
|
@@ -0,0 +1 @@
|
|
|
1
|
+
import{estimateModelCost as d}from"./DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";const a=t=>typeof t=="number"&&Number.isFinite(t)&&t>=0?t:void 0,f=t=>{if(!(typeof t!="object"||t===null)){for(const o of Object.values(t))if(typeof o=="object"&&o!==null){const{cost:e}=o;if(typeof e=="number"&&Number.isFinite(e))return e}}},b=t=>{const o=t.modelId;return typeof o=="string"&&o.length>0?o:void 0},l=(t,o,e,u)=>{const s=a(o.usage?.inputTokens?.total),n=a(o.usage?.outputTokens?.total),c=f(o.providerMetadata),i=c??d(t,{inputTokens:s,outputTokens:n}),g=c===void 0?"estimated":"provider",r={"gen_ai.request.model":t??"unknown"};s!==void 0&&(e?.setAttribute("gen_ai.usage.input_tokens",s),u?.count("gen_ai.usage.input_tokens",s,r)),n!==void 0&&(e?.setAttribute("gen_ai.usage.output_tokens",n),u?.count("gen_ai.usage.output_tokens",n,r)),i!==void 0&&(e?.setAttributes({"gen_ai.usage.cost":i,"lunora.usage.cost.source":g}),u?.count("gen_ai.usage.cost",i,{...r,"lunora.usage.cost.source":g}))};export{b as m,l as r};
|
package/dist/rag/index.d.mts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { Tool } from 'ai';
|
|
2
|
-
import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-
|
|
2
|
+
import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-BNOWTjHf.mjs";
|
|
3
3
|
/**
|
|
4
4
|
* Built-in fixed-window chunker: split into `size`-char windows overlapping by
|
|
5
5
|
* `overlap` chars. Deliberately simple and deterministic — the zero-config
|
|
@@ -240,6 +240,13 @@ interface RagContext {
|
|
|
240
240
|
* attribute is absent (backward-compatible).
|
|
241
241
|
*/
|
|
242
242
|
conversationId?: string;
|
|
243
|
+
/**
|
|
244
|
+
* Optional `ctx.metrics` — an `ActionCtx`'s `ctx.metrics` satisfies it. When
|
|
245
|
+
* present, every embed counts its tokens and cost into the durable
|
|
246
|
+
* `gen_ai.usage.*` series, like a `ctx.ai.model(...)` call. `unknown` for the
|
|
247
|
+
* same decoupling reason as `trace`; `defineRag` narrows it.
|
|
248
|
+
*/
|
|
249
|
+
metrics?: unknown;
|
|
243
250
|
/**
|
|
244
251
|
* Optional `ctx.trace` span factory — an `ActionCtx`'s `ctx.trace` satisfies
|
|
245
252
|
* it structurally. When present, `defineRag` wraps each embedding-model
|
|
@@ -1196,6 +1203,21 @@ interface SqlLexicalStoreOptions {
|
|
|
1196
1203
|
declare const sqlLexicalStore: (options: SqlLexicalStoreOptions) => RagLexicalStore;
|
|
1197
1204
|
/** Options for {@link sqliteVectorStore}. */
|
|
1198
1205
|
interface SqliteVectorStoreOptions {
|
|
1206
|
+
/**
|
|
1207
|
+
* Keep a `sqlite-vec` `vec0` index (cosine distance) in `<table>_ann`, one
|
|
1208
|
+
* partition per namespace. Requires the extension in the executor's SQLite;
|
|
1209
|
+
* without it the first operation throws naming the missing module.
|
|
1210
|
+
*
|
|
1211
|
+
* The JSON table stays the source of truth: an index created over an
|
|
1212
|
+
* existing table is backfilled from it, and a match the table no longer
|
|
1213
|
+
* holds is dropped. Only an **unfiltered** query uses the index — `vec0`
|
|
1214
|
+
* applies a metadata filter after picking the `k` nearest, which would
|
|
1215
|
+
* return a short page, so a filtered query keeps the exact scan (and its
|
|
1216
|
+
* `maxScan` bound).
|
|
1217
|
+
*/
|
|
1218
|
+
ann?: {
|
|
1219
|
+
dimensions: number;
|
|
1220
|
+
};
|
|
1199
1221
|
/** Execute one statement. See {@link RagSqlExec}. */
|
|
1200
1222
|
exec: RagSqlExec;
|
|
1201
1223
|
/**
|
package/dist/rag/index.d.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import { Tool } from 'ai';
|
|
2
|
-
import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-
|
|
2
|
+
import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-BNOWTjHf.js";
|
|
3
3
|
/**
|
|
4
4
|
* Built-in fixed-window chunker: split into `size`-char windows overlapping by
|
|
5
5
|
* `overlap` chars. Deliberately simple and deterministic — the zero-config
|
|
@@ -240,6 +240,13 @@ interface RagContext {
|
|
|
240
240
|
* attribute is absent (backward-compatible).
|
|
241
241
|
*/
|
|
242
242
|
conversationId?: string;
|
|
243
|
+
/**
|
|
244
|
+
* Optional `ctx.metrics` — an `ActionCtx`'s `ctx.metrics` satisfies it. When
|
|
245
|
+
* present, every embed counts its tokens and cost into the durable
|
|
246
|
+
* `gen_ai.usage.*` series, like a `ctx.ai.model(...)` call. `unknown` for the
|
|
247
|
+
* same decoupling reason as `trace`; `defineRag` narrows it.
|
|
248
|
+
*/
|
|
249
|
+
metrics?: unknown;
|
|
243
250
|
/**
|
|
244
251
|
* Optional `ctx.trace` span factory — an `ActionCtx`'s `ctx.trace` satisfies
|
|
245
252
|
* it structurally. When present, `defineRag` wraps each embedding-model
|
|
@@ -1196,6 +1203,21 @@ interface SqlLexicalStoreOptions {
|
|
|
1196
1203
|
declare const sqlLexicalStore: (options: SqlLexicalStoreOptions) => RagLexicalStore;
|
|
1197
1204
|
/** Options for {@link sqliteVectorStore}. */
|
|
1198
1205
|
interface SqliteVectorStoreOptions {
|
|
1206
|
+
/**
|
|
1207
|
+
* Keep a `sqlite-vec` `vec0` index (cosine distance) in `<table>_ann`, one
|
|
1208
|
+
* partition per namespace. Requires the extension in the executor's SQLite;
|
|
1209
|
+
* without it the first operation throws naming the missing module.
|
|
1210
|
+
*
|
|
1211
|
+
* The JSON table stays the source of truth: an index created over an
|
|
1212
|
+
* existing table is backfilled from it, and a match the table no longer
|
|
1213
|
+
* holds is dropped. Only an **unfiltered** query uses the index — `vec0`
|
|
1214
|
+
* applies a metadata filter after picking the `k` nearest, which would
|
|
1215
|
+
* return a short page, so a filtered query keeps the exact scan (and its
|
|
1216
|
+
* `maxScan` bound).
|
|
1217
|
+
*/
|
|
1218
|
+
ann?: {
|
|
1219
|
+
dimensions: number;
|
|
1220
|
+
};
|
|
1199
1221
|
/** Execute one statement. See {@link RagSqlExec}. */
|
|
1200
1222
|
exec: RagSqlExec;
|
|
1201
1223
|
/**
|
package/dist/rag/index.mjs
CHANGED
|
@@ -1 +1 @@
|
|
|
1
|
-
import{default as o}from"../packem_shared/fixedWindowChunks-C461ahRE.mjs";import{markdownChunker as a,sentenceChunker as f,tokenChunker as n}from"../packem_shared/markdownChunker-iW8V4klY.mjs";import{default as x}from"../packem_shared/defineRag-
|
|
1
|
+
import{default as o}from"../packem_shared/fixedWindowChunks-C461ahRE.mjs";import{markdownChunker as a,sentenceChunker as f,tokenChunker as n}from"../packem_shared/markdownChunker-iW8V4klY.mjs";import{default as x}from"../packem_shared/defineRag-CMTzKfS7.mjs";import{contentHash as p,guessMimeTypeFromExtension as i}from"../packem_shared/contentHash-BIn6ECP8.mjs";import{hybridRank as d}from"../packem_shared/hybridRank-B4skyCLx.mjs";import{default as k}from"../packem_shared/bm25LexicalStore-DMUzAL0O.mjs";import{default as h}from"../packem_shared/matchesMetadataFilter-BbIOyA5g.mjs";import{batchReranker as g,scoreReranker as C}from"../packem_shared/batchReranker-Bc38FBLH.mjs";import{defineRagSource as E}from"../packem_shared/defineRagSource-Q3f3niU8.mjs";import{sqlLexicalStore as T}from"../packem_shared/sqlLexicalStore-4C_cIwef.mjs";import{sqliteVectorStore as y}from"../packem_shared/sqliteVectorStore-SjOFKoHo.mjs";import{ragSyncTriggers as q}from"../packem_shared/ragSyncTriggers-hgMcA4f2.mjs";import{VECTORIZE_CAPABILITIES as A,vectorizeStore as F}from"../packem_shared/VECTORIZE_CAPABILITIES-CUQDoxis.mjs";export{A as VECTORIZE_CAPABILITIES,g as batchReranker,k as bm25LexicalStore,p as contentHash,x as defineRag,E as defineRagSource,o as fixedWindowChunks,i as guessMimeTypeFromExtension,d as hybridRank,a as markdownChunker,h as matchesMetadataFilter,q as ragSyncTriggers,C as scoreReranker,f as sentenceChunker,T as sqlLexicalStore,y as sqliteVectorStore,n as tokenChunker,F as vectorizeStore};
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@lunora/ai",
|
|
3
|
-
"version": "1.0.0-alpha.
|
|
3
|
+
"version": "1.0.0-alpha.104",
|
|
4
4
|
"description": "Workers AI helper for Lunora: provider-agnostic AI SDK access from functions, Workers AI by default",
|
|
5
5
|
"keywords": [
|
|
6
6
|
"ai",
|
|
@@ -53,22 +53,12 @@
|
|
|
53
53
|
"access": "public"
|
|
54
54
|
},
|
|
55
55
|
"dependencies": {
|
|
56
|
+
"@ai-sdk/anthropic": "4.0.49",
|
|
57
|
+
"@ai-sdk/openai": "4.0.60",
|
|
56
58
|
"@lunora/errors": "1.0.0-alpha.47",
|
|
57
59
|
"ai": "7.0.93",
|
|
58
60
|
"workers-ai-provider": "4.0.0"
|
|
59
61
|
},
|
|
60
|
-
"peerDependencies": {
|
|
61
|
-
"@ai-sdk/anthropic": "^4.0.0",
|
|
62
|
-
"@ai-sdk/openai": "^4.0.0"
|
|
63
|
-
},
|
|
64
|
-
"peerDependenciesMeta": {
|
|
65
|
-
"@ai-sdk/anthropic": {
|
|
66
|
-
"optional": true
|
|
67
|
-
},
|
|
68
|
-
"@ai-sdk/openai": {
|
|
69
|
-
"optional": true
|
|
70
|
-
}
|
|
71
|
-
},
|
|
72
62
|
"engines": {
|
|
73
63
|
"node": "^22.15.0 || >=24.11.0"
|
|
74
64
|
}
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
const a=(t,o)=>{if(t===void 0)return;const e=t[o];return typeof e=="string"&&e.length>0?e:void 0};let A=!1,d=!1;const f=5,l=t=>{if(t===void 0)return;const o={};typeof t.functionPath=="string"&&t.functionPath.length>0&&(o.functionPath=t.functionPath),typeof t.traceId=="string"&&t.traceId.length>0&&(o.traceId=t.traceId);const e={};for(const[n,s]of Object.entries(t.tags??{}))typeof s=="string"&&s.length>0&&n.length>0&&!Object.hasOwn(o,n)&&(e[n]=s);Object.assign(e,o);const r=Object.keys(e);if(r.length===0)return;if(r.length<=f)return e;const i=r.slice(-f);return Object.fromEntries(i.map(n=>[n,e[n]]))},g=t=>{const o=l(t);return o===void 0?void 0:JSON.stringify(o)},I="LUNORA_AI_DEFAULT_MODEL",y="LUNORA_AI_DEFAULT_EMBEDDING_MODEL",E="LUNORA_AI_GATEWAY_ACCOUNT_ID",h="LUNORA_AI_GATEWAY_ID",_="LUNORA_AI_GATEWAY_TOKEN",u="LUNORA_AI_GATEWAY_TAGS",O=t=>{const o=a(t,u);if(o!==void 0)try{const e=JSON.parse(o);if(typeof e!="object"||e===null||Array.isArray(e))throw new TypeError("expected a JSON object");const r={};for(const[i,n]of Object.entries(e))typeof n=="string"&&n.length>0&&(r[i]=n);return Object.keys(r).length>0?r:void 0}catch{d||(d=!0,console.warn(`[lunora:ai] ${u} is not a flat JSON object of strings — AI Gateway tags from it are ignored.`));return}},T=(t,o,e="byo-provider")=>{const r=a(t,E),i=a(t,h);if(r===void 0||i===void 0)return;const n=a(t,_),s={};n!==void 0&&(s["cf-aig-authorization"]=`Bearer ${n}`,e==="workers-ai-binding"&&!A&&(A=!0,console.warn(`[lunora:ai] ${_} is set, but the Workers AI binding cannot send a gateway auth token — Cloudflare's native gateway option has no authorization field. The token is ignored on this path; use a bring-your-own AI SDK provider (which sends cf-aig-authorization), or make the AI Gateway unauthenticated for Workers AI.`)));const c=g(o);return c!==void 0&&(s["cf-aig-metadata"]=c),{accountId:r,baseURL:`https://gateway.ai.cloudflare.com/v1/${r}/${i}`,gatewayId:i,headers:s}};export{y as AI_DEFAULT_EMBEDDING_MODEL_ENV,I as AI_DEFAULT_MODEL_ENV,E as AI_GATEWAY_ACCOUNT_ID_ENV,h as AI_GATEWAY_ID_ENV,f as AI_GATEWAY_METADATA_MAX_KEYS,u as AI_GATEWAY_TAGS_ENV,_ as AI_GATEWAY_TOKEN_ENV,l as buildAiGatewayMetadataFields,O as readAiGatewayEnvTags,a as readEnv,T as resolveAiGateway};
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import{LunoraError as l}from"@lunora/errors";import{createWorkersAI as M}from"workers-ai-provider";import{readEnv as g,AI_DEFAULT_MODEL_ENV as w,AI_DEFAULT_EMBEDDING_MODEL_ENV as E,readAiGatewayEnvTags as y,buildAiGatewayMetadataFields as N,resolveAiGateway as h}from"./AI_DEFAULT_EMBEDDING_MODEL_ENV-B58f6yo6.mjs";const D=(d,o,n)=>{const v=o===void 0?void 0:y(o),a=v===void 0?n:{...n,tags:{...v,...n?.tags}},t=N(a);if(d!==void 0)return t!==void 0&&d.metadata===void 0?{...d,metadata:t}:d;if(o===void 0)return;const i=h(o,a,"workers-ai-binding");if(i!==void 0)return t===void 0?{id:i.gatewayId}:{id:i.gatewayId,metadata:t}},R=d=>{const{binding:o,defaultEmbeddingModel:n,defaultModel:v,env:a,gateway:t,metadata:i,provider:c}=d;if(!c&&!o)throw new l("INTERNAL","@lunora/ai: createAi requires a `binding` (env.AI) or a pre-built `provider`");const f=D(t,a,i),m=v??g(a,w),b=n??g(a,E),s=c??M({binding:o,gateway:f}),A=e=>{if(e===void 0){if(!m)throw new l("INTERNAL",`@lunora/ai: no model supplied and no default configured — pass a model id, or set ${w} in the Worker env (wrangler \`vars\` / \`.dev.vars\`)`);return s(m)}return typeof e=="string"?s(e):e},p=e=>{const r=s.textEmbeddingModel;if(typeof r!="function")throw new l("INTERNAL","@lunora/ai: the Workers AI provider does not expose `textEmbeddingModel`; pass an AI SDK EmbeddingModel (e.g. from @ai-sdk/openai) to embed()");return r.call(s,e)};return{embeddingModel:e=>{if(typeof e=="object")return e;const r=e??b;if(!r)throw new l("INTERNAL",`@lunora/ai: no embedding model supplied and no default configured — pass an embedding model id or an AI SDK EmbeddingModel, or set ${E} in the Worker env (wrangler \`vars\` / \`.dev.vars\`)`);return p(r)},model:A,run:async(e,r,u)=>{if(!o)throw new l("INTERNAL","@lunora/ai: ai.run requires the `binding` (env.AI) — it is unavailable when only a custom `provider` was supplied");const I=f!==void 0&&u?.gateway===void 0?{...u,gateway:f}:u;return o.run(e,r,I)},workersai:s}};export{R as default};
|
|
@@ -1,8 +0,0 @@
|
|
|
1
|
-
import{LunoraError as b,isLunoraError as _e}from"@lunora/errors";import{tool as Te,jsonSchema as Ae,embedMany as Me,embed as Ne}from"ai";import{s as De}from"./stable-key-B_BlboiY.mjs";import{estimateModelCost as Re}from"./DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";import Ce from"./fixedWindowChunks-C461ahRE.mjs";import{c as Be,I as Oe}from"./concurrent-C6nqBv41.mjs";import{contentHash as $e}from"./contentHash-BIn6ECP8.mjs";import{hybridRank as Ke}from"./hybridRank-B4skyCLx.mjs";import{VECTORIZE_CAPABILITIES as ae,vectorizeStore as Ue}from"./VECTORIZE_CAPABILITIES-CUQDoxis.mjs";const Qe=1e3,Ve=200,Fe=5,Le=4,ie=ae.maxMetadataBytes===!1?Number.POSITIVE_INFINITY:ae.maxMetadataBytes,je=2*1024,le="__ragChunk",me="__ragSource",C="__ragText",P="__ragHash",Y="__ragChunks",$="__ragImportance",he="__ragModel",ze=new Set([le,Y,P,$,he,me,C]),Pe=(e,d,h,w)=>{if(w===!1)return;const v=new TextEncoder().encode(JSON.stringify(e)).length;if(v<=w)return;const K=(typeof e[C]=="string"?new TextEncoder().encode(e[C]).length:0)*2>v?"lower `chunkSize` (it counts characters, not bytes — multibyte text costs up to 3 bytes each), or supply `textStore` to move chunk text out of metadata entirely":"attach less per-source `metadata`";throw new b("BAD_REQUEST",`@lunora/ai/rag: chunk ${String(d)} of "${h}" carries ${String(v)} bytes of metadata, over the store's ${String(w)}-byte per-vector ceiling — ${K}`)},Ye=(e,d,h)=>{if(h===!1)return;const w=new TextEncoder().encode(e).length;if(!(w<=h))throw new b("BAD_REQUEST",`@lunora/ai/rag: chunk id "${e}" for source "${d}" is ${String(w)} bytes, over the store's ${String(h)}-byte per-vector id ceiling — shorten the source id (hash long keys before indexing them) or shorten the \`namespace\`, which is prefixed onto every chunk id`)},He=/^[\w.-]{1,40}$/,pe=e=>e===void 0?"":`${encodeURIComponent(e)}#`,R=(e,d,h)=>`${pe(e)}${d}#${String(h)}`,de=(e,d)=>{const h=pe(d),w=h!==""&&e.startsWith(h)?e.slice(h.length):e,v=w.lastIndexOf("#"),A=v===-1?Number.NaN:Number(w.slice(v+1));return v===-1||!Number.isInteger(A)||A<0?{chunkIndex:0,sourceId:w}:{chunkIndex:A,sourceId:w.slice(0,v)}},qe=async e=>$e(new TextEncoder().encode(e)),We=e=>{try{return De([e.text,e.metadata,e.importance])}catch{return}},Ze=(e,d)=>{const h=[],w=[];for(const v of e)v.score>=d?h.push(v):w.push(v.id);return{kept:h,rejectedIds:w}},Xe=e=>e!==void 0&&Object.keys(e).length>0,j=e=>{if(!e)return;const d=Object.entries(e).filter(([h])=>!ze.has(h));return d.length>0?Object.fromEntries(d):void 0},ce=new Set,Je=e=>{ce.has(e)||(ce.add(e),console.warn(`[@lunora/ai/rag] index "${e}" is used without a namespace — in a multi-tenant/sharded
|
|
2
|
-
app this shares one tenant's chunks (text included) with every other tenant, since
|
|
3
|
-
Vectorize indexes are account-global. Pass \`namespace\` (the shard/tenant key) on both
|
|
4
|
-
index() and retrieve(). Single-tenant apps suppress this via { allowSharedNamespace: true }.`))},Ge=e=>e.map(d=>`[source:${d.sourceId}#${String(d.chunkIndex)}]
|
|
5
|
-
${d.text}`).join(`
|
|
6
|
-
|
|
7
|
-
`),ue=(e,d)=>{if(typeof e=="object")return e;if(d===void 0)throw new b("INTERNAL","@lunora/ai/rag: `embeddingModel` is a Workers AI model id (or omitted) but the bound context has no `ai` (env.AI). Pass an AI SDK EmbeddingModel object to embed without Workers AI, or bind a context whose `ctx.ai` is wired.");return d.embeddingModel(e)},z=e=>{const d=e.modelId;return typeof d=="string"&&d.length>0?d:void 0},et=e=>{if(!(typeof e!="object"||e===null)){for(const d of Object.values(e))if(typeof d=="object"&&d!==null){const{cost:h}=d;if(typeof h=="number"&&Number.isFinite(h))return h}}},tt=e=>{if(e===void 0)throw new b("INTERNAL","@lunora/ai/rag: the bound context has no `vectors` (env.VECTORIZE) and no `store` is configured — bind a context whose `ctx.vectors` is wired, or configure `store` (e.g. `sqliteVectorStore`) to back this index without Vectorize.");return e},mt=e=>{if(typeof e.index!="string"||e.index.length===0)throw new b("BAD_REQUEST","@lunora/ai/rag: `index` must be a non-empty Vectorize index name");const d=e.chunkSize??Qe,h=e.chunkOverlap??Ve;if(!Number.isInteger(d)||d<1)throw new b("BAD_REQUEST","@lunora/ai/rag: `chunkSize` must be a positive integer");if(!Number.isInteger(h)||h<0||h>=d)throw new b("BAD_REQUEST","@lunora/ai/rag: `chunkOverlap` must be a non-negative integer smaller than `chunkSize`");const w=ie-je;if(!e.chunk&&!e.textStore&&!e.store&&d>w)throw new b("BAD_REQUEST",`@lunora/ai/rag: \`chunkSize\` of ${String(d)} leaves no room under Vectorize's ${String(ie)}-byte metadata limit, which also carries this chunk's bookkeeping and any \`metadata\` you attach — keep it under ${String(w)}, or supply \`textStore\` to move chunk text out of metadata entirely`);const v=e.topK??Fe;if(!Number.isInteger(v)||v<1)throw new b("BAD_REQUEST","@lunora/ai/rag: `topK` must be a positive integer");if(e.maxEmbeddingDimensions!==void 0&&e.maxEmbeddingDimensions!==!1&&(!Number.isInteger(e.maxEmbeddingDimensions)||e.maxEmbeddingDimensions<1))throw new b("BAD_REQUEST","@lunora/ai/rag: `maxEmbeddingDimensions` must be a positive integer, or `false` to disable the check");if(e.embeddingModelVersion!==void 0&&!He.test(e.embeddingModelVersion))throw new b("BAD_REQUEST",'@lunora/ai/rag: `embeddingModelVersion` must match /^[A-Za-z0-9._-]{1,40}$/ (a short, stable tag like "bge-v1.5")');if(e.candidates!==void 0&&(!Number.isInteger(e.candidates)||e.candidates<1))throw new b("BAD_REQUEST","@lunora/ai/rag: `candidates` must be a positive integer");if(e.cacheEmbeddings!==void 0&&(!Number.isInteger(e.cacheEmbeddings)||e.cacheEmbeddings<0))throw new b("BAD_REQUEST","@lunora/ai/rag: `cacheEmbeddings` must be a non-negative integer");const A=e.cacheEmbeddings??0,K=e.rerank,fe=e.chunk??(p=>Ce(p,d,h)),{textStore:k}=e,B=e.embeddingModelVersion,Q=p=>B===void 0?p:p===void 0?B:`${B}::${p}`;return p=>{const E=e.store?e.store(p):Ue(tt(p.vectors),e.index),H=k?E.capabilities.maxTopK:E.capabilities.maxTopKWithMetadata,U=e.maxEmbeddingDimensions??E.capabilities.maxDimensions;let O;const q=typeof p.trace=="function"?p.trace:void 0;let W=U===!1;const Z=(t,r)=>{if(W||(W=!0,U===!1||t<=U))return;const s=z(r);throw new b("BAD_REQUEST",`@lunora/ai/rag: embedding model${s===void 0?"":` "${s}"`} produces ${String(t)}-dimension vectors, over the ${String(U)}-dimension ceiling of index "${e.index}" — either truncate them with the provider's \`dimensions\` option (Matryoshka models such as text-embedding-3-large support this), or set \`maxEmbeddingDimensions: false\` if this index is not Vectorize-backed`)},N=new Map,ge=(t,r)=>{if(A!==0)for(N.set(t,r);N.size>A;){const s=N.keys().next();if(s.done===!0)break;N.delete(s.value)}},X=async t=>{const r=N.get(t);if(r!==void 0)return r;O??=ue(e.embeddingModel,p.ai);const s=O,a=async n=>{const{embedding:m,providerMetadata:y,usage:l}=await Ne({model:s,value:t});if(Z(m.length,s),n!==void 0){const c=l.tokens;typeof c=="number"&&Number.isFinite(c)&&n.setAttribute("gen_ai.usage.input_tokens",c);const u=et(y),f=u??Re(z(s),{inputTokens:typeof c=="number"?c:void 0});f!==void 0&&(n.setAttribute("gen_ai.usage.cost",f),n.setAttribute("lunora.usage.cost.source",u===void 0?"estimated":"provider"))}return ge(t,m),m};if(q===void 0)return a();const o=z(s),i=typeof p.conversationId=="string"&&p.conversationId.length>0?p.conversationId:void 0;return q("ai.embed",(n,m)=>a(m),{"gen_ai.operation.name":"embeddings",...o===void 0?{}:{"gen_ai.request.model":o},...i===void 0?{}:{"gen_ai.conversation.id":i}})},be=async(t,r,s)=>{if(!e.transformQuery||r?.transformQuery===!1)return[t];const a=typeof p.conversationId=="string"&&p.conversationId.length>0?p.conversationId:void 0,o=await e.transformQuery(t,{conversationId:a,namespace:s}),i=(typeof o=="string"?[o]:[...o]).map(n=>n.trim()).filter(n=>n.length>0);return i.length>0?i:[t]},we=async t=>{const r=new Map,s=[...new Set(t.filter(a=>!N.has(a)))];if(s.length<2)return r;O??=ue(e.embeddingModel,p.ai);try{const{embeddings:a}=await Me({model:O,values:s});if(a.length!==s.length)return r;const[o]=a;o!==void 0&&Z(o.length,O);for(const[i,n]of s.entries())r.set(n,a[i])}catch(a){if(_e(a))throw a}return r},V=t=>{if(t===void 0){if(e.requireNamespace)throw new b("BAD_REQUEST",`@lunora/ai/rag: index "${e.index}" requires a namespace (requireNamespace is set) — pass the tenant/shard key on index()/retrieve()/remove()`);e.allowSharedNamespace||Je(e.index)}},J=async(t,r)=>{const[s]=await E.getByIds([R(r,t,0)],r),a=s?.metadata?.[P],o=s?.metadata?.[Y];return{chunks:typeof o=="number"&&Number.isInteger(o)&&o>0?o:void 0,hash:typeof a=="string"?a:void 0}},G=async(t,r,s,a)=>{const o=Array.from({length:s-r},(i,n)=>R(a,t,r+n));o.length!==0&&(await E.deleteByIds(o,a),await k?.remove?.(o,{namespace:a}),await e.lexicalStore?.remove?.(o,{namespace:a}))},ve=async t=>{if(V(t.namespace),t.importance!==void 0&&(typeof t.importance!="number"||t.importance<0||t.importance>1))throw new b("BAD_REQUEST","@lunora/ai/rag: `importance` must be a number in [0, 1]");const r=Q(t.namespace),s=We(t),a=await qe(s??t.text),o=await J(t.id,r);if(t.reindex!==!0&&s!==void 0&&o.hash===a&&o.chunks!==void 0)return{chunks:o.chunks,ids:Array.from({length:o.chunks},(c,u)=>R(r,t.id,u)),unchanged:!0};const i=fe(t.text),n=i.map((c,u)=>R(r,t.id,u)),m=n.at(-1);if(m!==void 0&&Ye(m,t.id,E.capabilities.maxIdBytes),i.length===0&&t.allowEmptySources===!1)throw new b("BAD_REQUEST",`@lunora/ai/rag: source "${t.id}" produced zero chunks — set allowEmptySources: true to allow this`);if(i.length>0){const c=i.map((u,f)=>({chunkIndex:f,id:n[f],sourceId:t.id,text:u,...t.metadata===void 0?{}:{metadata:t.metadata}}));k&&await k.put(c,{namespace:r}),e.lexicalStore&&await e.lexicalStore.index(c,{namespace:r})}const y=await we(i),l=async c=>y.get(c)??await X(c);return await Be(i,Oe,async(c,u)=>{const f=n[u],x={...t.metadata,[le]:u,[me]:t.id};k||(x[C]=c),t.importance!==void 0&&(x[$]=t.importance),u===0&&(x[P]=a,x[Y]=i.length,B!==void 0&&(x[he]=B)),Pe(x,u,t.id,E.capabilities.maxMetadataBytes),await E.upsert({embed:l,id:f,input:c,metadata:x,namespace:r}),t.onChunk?.({chunkIndex:u,id:f,text:c,total:i.length})}),o.chunks!==void 0&&o.chunks>i.length&&await G(t.id,i.length,o.chunks,r),{chunks:i.length,ids:n,unchanged:!1}},ye=async t=>{V(t.namespace);const r=Q(t.namespace),a=(await J(t.id,r)).chunks??1;await G(t.id,0,a,r)},ee=async(t,r)=>{const s=new Map;if(t.length===0)return s;if(k){const o=await k.getMany(t,{namespace:r});for(const[i,n]of t.entries()){const m=o[i];typeof m=="string"&&s.set(n,m)}return s}const a=await E.getByIds(t,r);for(const o of a){const i=o.metadata?.[C];typeof i=="string"&&s.set(o.id,i)}return s},Ee=async(t,r,s)=>{const a=r?.chunkContext?.before??0,o=r?.chunkContext?.after??0;if(a===0&&o===0)return t;if(!Number.isInteger(a)||a<0||!Number.isInteger(o)||o<0)throw new b("BAD_REQUEST","@lunora/ai/rag: `chunkContext.before`/`chunkContext.after` must be non-negative integers");const i=new Map(t.map(l=>[l.id,l.text])),n=new Set;for(const l of t)for(let c=-a;c<=o;c+=1){const u=l.chunkIndex+c,f=R(s,l.sourceId,u);c!==0&&u>=0&&!i.has(f)&&n.add(f)}const m=await ee([...n],s),y=(l,c)=>{const u=R(s,l,c);return i.get(u)??m.get(u)};return t.map(l=>{const c=[];for(let u=-a;u<=o;u+=1){const f=u===0?l.text:y(l.sourceId,l.chunkIndex+u);f!==void 0&&c.push(f)}return{...l,text:c.join(`
|
|
8
|
-
`)}})},xe=t=>{if(typeof t=="string"){const r=e.filters!==void 0&&Object.hasOwn(e.filters,t)?e.filters[t]:void 0;if(!r)throw new b("NOT_FOUND",`@lunora/ai/rag: unknown named filter "${t}" — must be one of the keys declared in RagConfig.filters`);return r.filter}return t},Ie=(t,r)=>t.matches.map(s=>{const a=s.metadata??{},o=de(s.id,r),i=a[C],n=a[$],m=typeof n=="number"&&n>=0&&n<=1?n:1;return{chunkIndex:o.chunkIndex,id:s.id,importance:m,metadata:j(a),score:s.score*m,sourceId:o.sourceId,text:typeof i=="string"?i:""}}),Se=async(t,r)=>{if(!k)return t;const s=t.map(n=>n.id),[a,o]=await Promise.all([ee(s,r),E.getByIds(s,r)]),i=new Map(o.map(n=>[n.id,n.metadata]));return t.flatMap(n=>{const m=a.get(n.id);if(m===void 0)return[];const y=i.get(n.id),l=y?.[$],c=typeof l=="number"&&l>=0&&l<=1?l:n.importance,f=(n.importance===0?0:n.score/n.importance)*c;return[{...n,importance:c,metadata:j(y)??n.metadata,score:f,text:m}]})},te=async(t,r,s)=>{const a=t.map(n=>n.id).filter(n=>!r.has(n)),o=a.length===0?[]:await E.getByIds(a,s),i=new Map(o.map(n=>[n.id,n.metadata]));return t.map(n=>{const m=de(n.id,s),y=i.get(n.id),l=y?.[$];return{chunkIndex:m.chunkIndex,id:n.id,importance:typeof l=="number"&&l>=0&&l<=1?l:1,metadata:j(y),score:n.score,sourceId:m.sourceId,text:n.text}})},ne=async(t,r)=>{V(r?.namespace);const s=Q(r?.namespace),a=xe(r?.filter),o=e.rlsFilter?await e.rlsFilter(p.auth):void 0,i=o?{...a,...o}:a,n=Math.min(r?.topK??v,H),m=await be(t,r,s),y=m[0],l=K!==void 0&&r?.rerank!==!1,u=l||e.lexicalStore!==void 0||e.graphStore!==void 0||m.length>1?Math.min(e.candidates??n*Le,H):n,f=r?.minScore,x=new Set,re=async g=>{const _=await E.query({embed:X,filter:i,input:g,namespace:s,returnMetadata:k?"indexed":"all",topK:u}),M=await Se(Ie(_,s),s);if(f===void 0)return M;const{kept:S,rejectedIds:T}=Ze(M,f);for(const ke of T)x.add(ke);return S},D=[{chunks:await re(y)}];for(const g of m.slice(1))D.push({chunks:await re(g)});if(e.lexicalStore){const g=await e.lexicalStore.search(y,{filter:i,namespace:s,topK:e.lexicalTopK??u}),_=new Set(D.flatMap(S=>S.chunks).map(S=>S.id)),M=g.filter(S=>!x.has(S.id)||_.has(S.id));D.push({chunks:await te(M,_,s)})}const F=D.flatMap(g=>[...g.chunks]);if(e.graphStore&&F.length>0&&(e.graphStore.enforcesFilter||!Xe(i))){const g=[...new Set(F.map(T=>T.sourceId))],_=await e.graphStore.related(g,{filter:i,namespace:s,topK:e.graphTopK??u}),M=new Set(F.map(T=>T.id)),S=_.filter(T=>!x.has(T.id)||M.has(T.id));D.push({chunks:await te(S,M,s),weight:"proximity"})}const L=D.filter(g=>g.chunks.length>0);let I=L.length>1?[...Ke(L)]:[...L[0]?.chunks??[]];I.sort((g,_)=>_.score-g.score),l&&(I=[...await K(y,I)]),I=I.slice(0,n),I=[...await Ee(I,r,s)];const se=[],oe=new Set;for(const g of I)oe.has(g.sourceId)||(oe.add(g.sourceId),se.push({id:g.sourceId,metadata:g.metadata,weight:g.importance}));return r?.onRetrieve?.({matches:I.length,query:t}),{chunks:I,context:Ge(I),sources:se}};return{asTool:t=>Te({description:t?.description??`Search the "${e.index}" knowledge base for passages relevant to a natural-language query.`,execute:async({query:r})=>ne(r,{namespace:t?.namespace,topK:t?.topK}),inputSchema:Ae({properties:{query:{description:"The natural-language search query.",type:"string"}},required:["query"],type:"object"})}),index:ve,remove:ye,retrieve:ne}}};export{mt as default};
|
|
@@ -1 +0,0 @@
|
|
|
1
|
-
import g from"./matchesMetadataFilter-BbIOyA5g.mjs";import{a as v,p as f,r as h,c as L,i as S}from"./sql-D5aqEMCY.mjs";const N="lunora_rag_vectors",p=5e4,w=100,I=null,d=s=>s??"",R=s=>{if(typeof s.exec!="function")throw new TypeError("@lunora/ai/rag: sqliteVectorStore requires an `exec` function");const n=v(s.table??N,"sqliteVectorStore `table`"),E=s.maxScan??p,{exec:c}=s,y={maxDimensions:s.maxDimensions??!1,maxIdBytes:!1,maxMetadataBytes:!1,maxTopK:w,maxTopKWithMetadata:w};let T;const m=async()=>{T??=(async()=>{await c(`CREATE TABLE IF NOT EXISTS ${n} (id TEXT NOT NULL, namespace TEXT NOT NULL DEFAULT '', vector TEXT NOT NULL, metadata TEXT, PRIMARY KEY (namespace, id))`,[]),await c(`CREATE INDEX IF NOT EXISTS ${n}_namespace ON ${n} (namespace)`,[])})().catch(e=>{throw T=void 0,e}),await T};return{capabilities:y,deleteByIds:async(e,t)=>{if(await m(),e.length!==0)for(const a of S(e))await c(`DELETE FROM ${n} WHERE namespace = ? AND id IN (${f(a.length)})`,[d(t),...a])},getByIds:async(e,t)=>{if(await m(),e.length===0)return[];const a=[];for(const i of S(e)){const l=await c(`SELECT id, metadata FROM ${n} WHERE namespace = ? AND id IN (${f(i.length)})`,[d(t),...i]);for(const r of l){const o=h(r.metadata);a.push({id:String(r.id),...o===void 0?{}:{metadata:o}})}}return a},query:async e=>{await m();let t;if(e.embed&&e.input!==void 0)t=await e.embed(e.input);else throw new TypeError("@lunora/ai/rag: sqliteVectorStore query requires both `input` and `embed`");const a=await c(`SELECT id, vector, metadata FROM ${n} WHERE namespace = ? LIMIT ?`,[d(e.namespace),E+1]);if(a.length>E)throw new RangeError(`@lunora/ai/rag: sqliteVectorStore scanned ${String(a.length)} vectors in namespace "${d(e.namespace)}", over the ${String(E)} limit — search here is brute force and linear, so this namespace has outgrown it. Shard it further, or move this index to Vectorize or a pgvector backend`);const i=[];for(const r of a){const o=h(r.metadata);if(!g(o,e.filter))continue;const u=h(r.vector);u!==void 0&&i.push({id:String(r.id),score:L(t,u),...e.returnMetadata==="none"||o===void 0?{}:{metadata:o}})}const l=i.toSorted((r,o)=>o.score-r.score).slice(0,e.topK??10);return{count:l.length,matches:l}},upsert:async e=>{if(await m(),!e.embed)throw new TypeError("@lunora/ai/rag: sqliteVectorStore requires an `embed` function on upsert");const t=await e.embed(e.input);await c(`INSERT INTO ${n} (id, namespace, vector, metadata) VALUES (${f(4)}) ON CONFLICT(namespace, id) DO UPDATE SET vector = excluded.vector, metadata = excluded.metadata`,[e.id,d(e.namespace),JSON.stringify([...t]),e.metadata===void 0?I:JSON.stringify(e.metadata)])}}};export{R as sqliteVectorStore};
|