@lunora/ai 1.0.0-alpha.103 → 1.0.0-alpha.105

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -54,11 +54,11 @@ yarn add @lunora/ai
54
54
  pnpm add @lunora/ai
55
55
  ```
56
56
 
57
- To use another provider, install it alongside (optional): `pnpm add @ai-sdk/openai`.
57
+ Other providers need no extra install: `ctx.ai.model("anthropic/claude-sonnet-5")` routes through Cloudflare AI Gateway (see below).
58
58
 
59
59
  ## Usage
60
60
 
61
- When a function uses AI, codegen wires a typed **`ctx.ai`** onto the action context (inference is an external call, so — like `ctx.fetch` — it lives on actions). Workers AI is the zero-config default; pass any AI SDK model to use another provider.
61
+ When a function uses AI, codegen wires a typed **`ctx.ai`** onto the action context (inference is an external call, so — like `ctx.fetch` — it lives on actions). Workers AI is the zero-config default; a `"<provider>/<model>"` id reaches any other provider through Cloudflare AI Gateway, and any AI SDK model object passes straight through.
62
62
 
63
63
  ```ts
64
64
  // lunora/summarize.ts — ctx.ai (codegen-wired), Workers AI by default
@@ -82,13 +82,14 @@ export const summarize = action.input({ text: v.string().max(20_000) }).action(a
82
82
  ```
83
83
 
84
84
  ```ts
85
- // Bring-your-own provider — same call surface, no lock-in
86
- import { streamText } from "@lunora/ai";
87
- import { openai } from "@ai-sdk/openai";
88
-
89
- const result = streamText({ model: openai("gpt-5"), messages });
85
+ // Any provider: change the string, nothing else. Routed through Cloudflare AI
86
+ // Gateway over the same `AI` binding — Unified Billing or keys stored on the
87
+ // gateway, so the app holds no provider API key.
88
+ const result = streamText({ model: ctx.ai.model("anthropic/claude-sonnet-5"), messages });
90
89
  ```
91
90
 
91
+ Every `ctx.ai.model(...)` call is traced and its tokens and cost are counted per function (`gen_ai.usage.*`), which Studio's **AI usage** page charts. `lunora ai gateway` creates a gateway for the app and sets `LUNORA_AI_GATEWAY_ID`; without it, calls use the account's `default` gateway.
92
+
92
93
  ```ts
93
94
  // RAG: embed via ctx.ai, store/search with @lunora/bindings/vectors (ctx.vectors)
94
95
  import { embed } from "@lunora/ai";
package/dist/index.d.mts CHANGED
@@ -1,5 +1,5 @@
1
- import { L as LunoraAiOptions, a as LunoraAi } from "./packem_shared/types.d-D3U2budn.mjs";
2
- export { A as AI_DEFAULT_EMBEDDING_MODEL_ENV, b as AI_DEFAULT_MODEL_ENV, c as AI_GATEWAY_ACCOUNT_ID_ENV, d as AI_GATEWAY_ID_ENV, e as AI_GATEWAY_METADATA_MAX_KEYS, f as AI_GATEWAY_TAGS_ENV, g as AI_GATEWAY_TOKEN_ENV, type h as AiBindingLike, type i as AiGatewayMetadata, type j as AiGatewayOptions, type E as EmbeddingModelInput, type M as ModelInput, type R as ResolvedAiGateway, type W as WorkersAiProviderLike, k as buildAiGatewayMetadataFields, r as readAiGatewayEnvTags, l as resolveAiGateway } from "./packem_shared/types.d-D3U2budn.mjs";
1
+ import { L as LunoraAiOptions, a as LunoraAi } from "./packem_shared/types.d-ChtZguP_.mjs";
2
+ export { A as AI_DEFAULT_EMBEDDING_MODEL_ENV, b as AI_DEFAULT_MODEL_ENV, c as AI_GATEWAY_ACCOUNT_ID_ENV, d as AI_GATEWAY_ID_ENV, e as AI_GATEWAY_METADATA_MAX_KEYS, f as AI_GATEWAY_TAGS_ENV, g as AI_GATEWAY_TOKEN_ENV, h as AI_PROXY_TOKEN_ENV, i as AI_PROXY_URL_ENV, type j as AiBindingLike, type k as AiGatewayMetadata, type l as AiGatewayOptions, type m as AiMetrics, type n as AiModelOptions, type o as AiRunOptions, type p as AiSpan, type q as AiTelemetry, type r as AiTracer, type E as EmbeddingModelInput, type M as ModelInput, type R as ResolvedAiGateway, type W as WorkersAiProviderLike, s as buildAiGatewayMetadataFields, t as readAiGatewayEnvTags, u as resolveAiGateway } from "./packem_shared/types.d-ChtZguP_.mjs";
3
3
  export { type EmbeddingModel, type LanguageModel, embed, embedMany, generateObject, generateText, hasToolCall, jsonSchema, streamObject, streamText, tool } from 'ai';
4
4
  export { createWorkersAI } from 'workers-ai-provider';
5
5
  /**
@@ -11,6 +11,11 @@ export { createWorkersAI } from 'workers-ai-provider';
11
11
  * (`@ai-sdk/openai`, `@ai-sdk/anthropic`, OpenRouter, …), so apps are never
12
12
  * locked to Workers AI. Pair `embed` with `@lunora/bindings/vectors` for RAG.
13
13
  *
14
+ * Without a binding, a string id resolves only as a `"<provider>/<model>"` slug
15
+ * through {@link AI_PROXY_URL_ENV}; everything else that needs the binding
16
+ * throws a directed error when called, never at construction — so the
17
+ * generated `ctx.ai` is always this facade.
18
+ *
14
19
  * Combine with the re-exported `generateText`/`streamText`/`generateObject`/
15
20
  * `embed`/`tool` from this package:
16
21
  *
package/dist/index.d.ts CHANGED
@@ -1,5 +1,5 @@
1
- import { L as LunoraAiOptions, a as LunoraAi } from "./packem_shared/types.d-D3U2budn.js";
2
- export { A as AI_DEFAULT_EMBEDDING_MODEL_ENV, b as AI_DEFAULT_MODEL_ENV, c as AI_GATEWAY_ACCOUNT_ID_ENV, d as AI_GATEWAY_ID_ENV, e as AI_GATEWAY_METADATA_MAX_KEYS, f as AI_GATEWAY_TAGS_ENV, g as AI_GATEWAY_TOKEN_ENV, type h as AiBindingLike, type i as AiGatewayMetadata, type j as AiGatewayOptions, type E as EmbeddingModelInput, type M as ModelInput, type R as ResolvedAiGateway, type W as WorkersAiProviderLike, k as buildAiGatewayMetadataFields, r as readAiGatewayEnvTags, l as resolveAiGateway } from "./packem_shared/types.d-D3U2budn.js";
1
+ import { L as LunoraAiOptions, a as LunoraAi } from "./packem_shared/types.d-ChtZguP_.js";
2
+ export { A as AI_DEFAULT_EMBEDDING_MODEL_ENV, b as AI_DEFAULT_MODEL_ENV, c as AI_GATEWAY_ACCOUNT_ID_ENV, d as AI_GATEWAY_ID_ENV, e as AI_GATEWAY_METADATA_MAX_KEYS, f as AI_GATEWAY_TAGS_ENV, g as AI_GATEWAY_TOKEN_ENV, h as AI_PROXY_TOKEN_ENV, i as AI_PROXY_URL_ENV, type j as AiBindingLike, type k as AiGatewayMetadata, type l as AiGatewayOptions, type m as AiMetrics, type n as AiModelOptions, type o as AiRunOptions, type p as AiSpan, type q as AiTelemetry, type r as AiTracer, type E as EmbeddingModelInput, type M as ModelInput, type R as ResolvedAiGateway, type W as WorkersAiProviderLike, s as buildAiGatewayMetadataFields, t as readAiGatewayEnvTags, u as resolveAiGateway } from "./packem_shared/types.d-ChtZguP_.js";
3
3
  export { type EmbeddingModel, type LanguageModel, embed, embedMany, generateObject, generateText, hasToolCall, jsonSchema, streamObject, streamText, tool } from 'ai';
4
4
  export { createWorkersAI } from 'workers-ai-provider';
5
5
  /**
@@ -11,6 +11,11 @@ export { createWorkersAI } from 'workers-ai-provider';
11
11
  * (`@ai-sdk/openai`, `@ai-sdk/anthropic`, OpenRouter, …), so apps are never
12
12
  * locked to Workers AI. Pair `embed` with `@lunora/bindings/vectors` for RAG.
13
13
  *
14
+ * Without a binding, a string id resolves only as a `"<provider>/<model>"` slug
15
+ * through {@link AI_PROXY_URL_ENV}; everything else that needs the binding
16
+ * throws a directed error when called, never at construction — so the
17
+ * generated `ctx.ai` is always this facade.
18
+ *
14
19
  * Combine with the re-exported `generateText`/`streamText`/`generateObject`/
15
20
  * `embed`/`tool` from this package:
16
21
  *
package/dist/index.mjs CHANGED
@@ -1 +1 @@
1
- import{default as _}from"./packem_shared/createAi-mXofcZid.mjs";import{AI_DEFAULT_EMBEDDING_MODEL_ENV as a,AI_DEFAULT_MODEL_ENV as E,AI_GATEWAY_ACCOUNT_ID_ENV as o,AI_GATEWAY_ID_ENV as r,AI_GATEWAY_METADATA_MAX_KEYS as T,AI_GATEWAY_TAGS_ENV as I,AI_GATEWAY_TOKEN_ENV as l,buildAiGatewayMetadataFields as m,readAiGatewayEnvTags as s,resolveAiGateway as D}from"./packem_shared/AI_DEFAULT_EMBEDDING_MODEL_ENV-B58f6yo6.mjs";import{DEFAULT_MODEL_PRICES as M,estimateModelCost as d,lookupModelPrice as N}from"./packem_shared/DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";import{embed as x,embedMany as O,generateObject as c,generateText as f,hasToolCall as p,jsonSchema as L,streamObject as V,streamText as W,tool as Y}from"ai";import{createWorkersAI as n}from"workers-ai-provider";export{a as AI_DEFAULT_EMBEDDING_MODEL_ENV,E as AI_DEFAULT_MODEL_ENV,o as AI_GATEWAY_ACCOUNT_ID_ENV,r as AI_GATEWAY_ID_ENV,T as AI_GATEWAY_METADATA_MAX_KEYS,I as AI_GATEWAY_TAGS_ENV,l as AI_GATEWAY_TOKEN_ENV,M as DEFAULT_MODEL_PRICES,m as buildAiGatewayMetadataFields,_ as createAi,n as createWorkersAI,x as embed,O as embedMany,d as estimateModelCost,c as generateObject,f as generateText,p as hasToolCall,L as jsonSchema,N as lookupModelPrice,s as readAiGatewayEnvTags,D as resolveAiGateway,V as streamObject,W as streamText,Y as tool};
1
+ import{default as _}from"./packem_shared/createAi-D76ClNVB.mjs";import{AI_DEFAULT_EMBEDDING_MODEL_ENV as t,AI_DEFAULT_MODEL_ENV as a,AI_GATEWAY_ACCOUNT_ID_ENV as o,AI_GATEWAY_ID_ENV as r,AI_GATEWAY_METADATA_MAX_KEYS as T,AI_GATEWAY_TAGS_ENV as I,AI_GATEWAY_TOKEN_ENV as N,AI_PROXY_TOKEN_ENV as l,AI_PROXY_URL_ENV as m,buildAiGatewayMetadataFields as s,readAiGatewayEnvTags as D,resolveAiGateway as G}from"./packem_shared/AI_DEFAULT_EMBEDDING_MODEL_ENV-IwcHZiKj.mjs";import{DEFAULT_MODEL_PRICES as O,estimateModelCost as d,lookupModelPrice as i}from"./packem_shared/DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";import{embed as Y,embedMany as x,generateObject as L,generateText as c,hasToolCall as f,jsonSchema as p,streamObject as W,streamText as b,tool as n}from"ai";import{createWorkersAI as U}from"workers-ai-provider";export{t as AI_DEFAULT_EMBEDDING_MODEL_ENV,a as AI_DEFAULT_MODEL_ENV,o as AI_GATEWAY_ACCOUNT_ID_ENV,r as AI_GATEWAY_ID_ENV,T as AI_GATEWAY_METADATA_MAX_KEYS,I as AI_GATEWAY_TAGS_ENV,N as AI_GATEWAY_TOKEN_ENV,l as AI_PROXY_TOKEN_ENV,m as AI_PROXY_URL_ENV,O as DEFAULT_MODEL_PRICES,s as buildAiGatewayMetadataFields,_ as createAi,U as createWorkersAI,Y as embed,x as embedMany,d as estimateModelCost,L as generateObject,c as generateText,f as hasToolCall,p as jsonSchema,i as lookupModelPrice,D as readAiGatewayEnvTags,G as resolveAiGateway,W as streamObject,b as streamText,n as tool};
@@ -0,0 +1 @@
1
+ const a=(t,n)=>{if(t===void 0)return;const e=t[n];return typeof e=="string"&&e.length>0?e:void 0};let _=!1,d=!1;const f=5,E=t=>{if(t===void 0)return;const n={};typeof t.functionPath=="string"&&t.functionPath.length>0&&(n.functionPath=t.functionPath),typeof t.traceId=="string"&&t.traceId.length>0&&(n.traceId=t.traceId);const e={};for(const[o,s]of Object.entries(t.tags??{}))typeof s=="string"&&s.length>0&&o.length>0&&!Object.hasOwn(n,o)&&(e[o]=s);Object.assign(e,n);const r=Object.keys(e);if(r.length===0)return;if(r.length<=f)return e;const i=r.slice(-f);return Object.fromEntries(i.map(o=>[o,e[o]]))},g=t=>{const n=E(t);return n===void 0?void 0:JSON.stringify(n)},h="LUNORA_AI_DEFAULT_MODEL",T="LUNORA_AI_DEFAULT_EMBEDDING_MODEL",l="LUNORA_AI_GATEWAY_ACCOUNT_ID",I="LUNORA_AI_GATEWAY_ID",c="LUNORA_AI_GATEWAY_TOKEN",u="LUNORA_AI_GATEWAY_TAGS",N="LUNORA_AI_PROXY_URL",y="LUNORA_AI_PROXY_TOKEN",v=t=>{const n=a(t,u);if(n!==void 0)try{const e=JSON.parse(n);if(typeof e!="object"||e===null||Array.isArray(e))throw new TypeError("expected a JSON object");const r={};for(const[i,o]of Object.entries(e))typeof o=="string"&&o.length>0&&(r[i]=o);return Object.keys(r).length>0?r:void 0}catch{d||(d=!0,console.warn(`[lunora:ai] ${u} is not a flat JSON object of strings — AI Gateway tags from it are ignored.`));return}},O=t=>{_||a(t,c)===void 0||(_=!0,console.warn(`[lunora:ai] ${c} is set, but the Workers AI binding cannot send a gateway auth token — Cloudflare's native gateway option has no authorization field. The token is ignored on this path; use a bring-your-own AI SDK provider (which sends cf-aig-authorization), or make the AI Gateway unauthenticated for Workers AI.`))},w=(t,n,e="byo-provider")=>{const r=a(t,l),i=a(t,I);if(r===void 0||i===void 0)return;const o=a(t,c),s={};o!==void 0&&(s["cf-aig-authorization"]=`Bearer ${o}`,e==="workers-ai-binding"&&O(t));const A=g(n);return A!==void 0&&(s["cf-aig-metadata"]=A),{accountId:r,baseURL:`https://gateway.ai.cloudflare.com/v1/${r}/${i}`,gatewayId:i,headers:s}};export{T as AI_DEFAULT_EMBEDDING_MODEL_ENV,h as AI_DEFAULT_MODEL_ENV,l as AI_GATEWAY_ACCOUNT_ID_ENV,I as AI_GATEWAY_ID_ENV,f as AI_GATEWAY_METADATA_MAX_KEYS,u as AI_GATEWAY_TAGS_ENV,c as AI_GATEWAY_TOKEN_ENV,y as AI_PROXY_TOKEN_ENV,N as AI_PROXY_URL_ENV,E as buildAiGatewayMetadataFields,v as readAiGatewayEnvTags,a as readEnv,w as resolveAiGateway,O as warnIgnoredBindingToken};
@@ -0,0 +1 @@
1
+ import{createOpenAI as S}from"@ai-sdk/openai";import{LunoraError as E}from"@lunora/errors";import{createWorkersAI as $}from"workers-ai-provider";import{anthropic as C}from"workers-ai-provider/anthropic";import{openai as W}from"workers-ai-provider/openai";import{buildAiGatewayMetadataFields as D,readEnv as A,AI_DEFAULT_MODEL_ENV as M,AI_DEFAULT_EMBEDDING_MODEL_ENV as I,AI_PROXY_URL_ENV as h,AI_PROXY_TOKEN_ENV as R,readAiGatewayEnvTags as j,AI_GATEWAY_ID_ENV as B,warnIgnoredBindingToken as V}from"./AI_DEFAULT_EMBEDDING_MODEL_ENV-IwcHZiKj.mjs";import{wrapLanguageModel as Y}from"ai";import{m as F,r as T}from"./usage-GPnTSqR8.mjs";import"./DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";const K={setAttribute:()=>{},setAttributes:()=>{}},O=async(e,t)=>t(O,K),q=(e,{metrics:t,trace:n=O})=>{const a={"gen_ai.operation.name":"chat","gen_ai.request.model":e};return{wrapGenerate:async({doGenerate:o})=>n("ai.generate",async(u,m)=>{const g=await o();return T(e,g,m,t),g},a),wrapStream:async({doStream:o})=>{const u=(w,s)=>{const c=w.stream.getReader();let l={};return{...w,stream:new ReadableStream({cancel:async f=>{s({outcome:l}),await c.cancel(f)},pull:async f=>{let v;try{v=await c.read()}catch(y){s({error:y,failed:!0,outcome:l}),f.error(y);return}if(v.done){s({outcome:l}),f.close();return}v.value.type==="finish"&&(l={providerMetadata:v.value.providerMetadata,usage:v.value.usage}),f.enqueue(v.value)}})}},m=Promise.withResolvers(),g=Promise.withResolvers();return n("ai.stream",async(w,s)=>{m.resolve(u(await o(),g.resolve));const c=await g.promise;if(T(e,c.outcome,s,t),c.failed===!0)throw c.error},a).catch(m.reject),m.promise}}},L=(e,t,n)=>{let a=e;return t!==void 0&&typeof e!="string"&&(a=Y({middleware:q(n??F(e)??"unknown",t),model:e})),a},X=[W,C],N=e=>!e.startsWith("@")&&e.includes("/"),b=e=>{throw new E("INTERNAL",`@lunora/ai: ${e} needs the \`AI\` binding (env.AI). Add an \`ai\` binding to wrangler.jsonc, or set ${h} to an OpenAI-compatible proxy for "<provider>/<model>" slugs.`)},P=()=>b("this model id");P.textEmbeddingModel=()=>b("this embedding model id");const H=3040,Z=/^(?:[A-Z]\w*:\s*)?3040\s*:/iu,z=e=>e?.code===H||e instanceof Error&&Z.test(e.message),J=new Set(["127.0.0.1","[::1]","localhost"]),Q=e=>{const t=A(e,h);if(t===void 0)return;const n=A(e,R),a=URL.canParse(t)?new URL(t):void 0;let o;if(a===void 0?o=`${h} is not a valid URL`:n!==void 0&&a.protocol!=="https:"&&!J.has(a.hostname)&&(o=`${h} (${a.origin}) is not HTTPS, so ${R} would travel in cleartext — use an https:// URL`),o!==void 0){const u=()=>{throw new E("INTERNAL",`@lunora/ai: ${o}`)};return{chat:u,embedding:u}}return S({apiKey:n??"",baseURL:t,name:"lunora-proxy"})},ee=(e,t)=>{const n=e===void 0?void 0:j(e);return n===void 0?t:{...t,tags:{...n,...t?.tags}}},te=(e,t,n)=>{const a=D(n);if(e!==void 0)return a!==void 0&&e.metadata===void 0?{...e,metadata:a}:e;if(t===void 0)return;const o=A(t,B);if(o!==void 0)return V(t),a===void 0?{id:o}:{id:o,metadata:a}},ve=e=>{const{binding:t,defaultEmbeddingModel:n,defaultModel:a,env:o,gateway:u,metadata:m,provider:g,telemetry:w}=e,s=Q(o),c=ee(o,m),l=te(u,o,c),f=u?.metadata===void 0?D(c):void 0,v=a??A(o,M),y=n??A(o,I),p=g??(t?$({binding:t,gateway:l,providers:X}):P),x=(r,i)=>N(r)?s!==void 0?s.chat(r):f===void 0?p(r):p(r,{metadata:f}):i?.rejectIfBusy===void 0?p(r):p(r,{rejectIfBusy:i.rejectIfBusy}),U=(r,i)=>{const d=r??v;if(d===void 0||d==="")throw new E("INTERNAL",`@lunora/ai: no model supplied and no default configured — pass a model id, or set ${M} in the Worker env (wrangler \`vars\` / \`.dev.vars\`)`);return typeof d=="string"?L(x(d,i),w,d):L(d,w)},k=r=>{if(s!==void 0&&N(r))return s.embedding(r);const i=p.textEmbeddingModel;if(typeof i!="function")throw new E("INTERNAL","@lunora/ai: the Workers AI provider does not expose `textEmbeddingModel`; pass an AI SDK EmbeddingModel (e.g. from @ai-sdk/openai) to embed()");return i.call(p,r)};return{embeddingModel:r=>{if(typeof r=="object")return r;const i=r??y;if(!i)throw new E("INTERNAL",`@lunora/ai: no embedding model supplied and no default configured — pass an embedding model id or an AI SDK EmbeddingModel, or set ${I} in the Worker env (wrangler \`vars\` / \`.dev.vars\`)`);return k(i)},model:U,run:async(r,i,d)=>{if(!t)return b("ai.run");const G=l!==void 0&&d?.gateway===void 0?{...d,gateway:l}:d;try{return await t.run(r,i,G)}catch(_){throw z(_)?new E("RATE_LIMITED",`@lunora/ai: Workers AI has no free capacity for ${r} (error 3040) — retry later`,{cause:_}):_}},workersai:p}};export{ve as default};
@@ -0,0 +1,8 @@
1
+ import{LunoraError as f,isLunoraError as Me}from"@lunora/errors";import{tool as Ae,jsonSchema as Ne,embedMany as De,embed as Re}from"ai";import{s as Ce}from"./stable-key-B_BlboiY.mjs";import{r as ie,m as Q}from"./usage-GPnTSqR8.mjs";import Be from"./fixedWindowChunks-C461ahRE.mjs";import{c as $e,I as Oe}from"./concurrent-C6nqBv41.mjs";import{contentHash as Ke}from"./contentHash-BIn6ECP8.mjs";import{hybridRank as Ue}from"./hybridRank-B4skyCLx.mjs";import{VECTORIZE_CAPABILITIES as de,vectorizeStore as Qe}from"./VECTORIZE_CAPABILITIES-CUQDoxis.mjs";const Ve=1e3,Fe=200,Le=5,ze=4,ce=de.maxMetadataBytes===!1?Number.POSITIVE_INFINITY:de.maxMetadataBytes,Pe=2*1024,he="__ragChunk",pe="__ragSource",B="__ragText",j="__ragHash",Y="__ragChunks",O="__ragImportance",ge="__ragModel",je=new Set([he,Y,j,O,ge,pe,B]),Ye=(e,l,h,w)=>{if(w===!1)return;const v=new TextEncoder().encode(JSON.stringify(e)).length;if(v<=w)return;const K=(typeof e[B]=="string"?new TextEncoder().encode(e[B]).length:0)*2>v?"lower `chunkSize` (it counts characters, not bytes — multibyte text costs up to 3 bytes each), or supply `textStore` to move chunk text out of metadata entirely":"attach less per-source `metadata`";throw new f("BAD_REQUEST",`@lunora/ai/rag: chunk ${String(l)} of "${h}" carries ${String(v)} bytes of metadata, over the store's ${String(w)}-byte per-vector ceiling — ${K}`)},He=(e,l,h)=>{if(h===!1)return;const w=new TextEncoder().encode(e).length;if(!(w<=h))throw new f("BAD_REQUEST",`@lunora/ai/rag: chunk id "${e}" for source "${l}" is ${String(w)} bytes, over the store's ${String(h)}-byte per-vector id ceiling — shorten the source id (hash long keys before indexing them) or shorten the \`namespace\`, which is prefixed onto every chunk id`)},qe=/^[\w.-]{1,40}$/,fe=e=>e===void 0?"":`${encodeURIComponent(e)}#`,C=(e,l,h)=>`${fe(e)}${l}#${String(h)}`,ue=(e,l)=>{const h=fe(l),w=h!==""&&e.startsWith(h)?e.slice(h.length):e,v=w.lastIndexOf("#"),M=v===-1?Number.NaN:Number(w.slice(v+1));return v===-1||!Number.isInteger(M)||M<0?{chunkIndex:0,sourceId:w}:{chunkIndex:M,sourceId:w.slice(0,v)}},We=async e=>Ke(new TextEncoder().encode(e)),Ze=e=>{try{return Ce([e.text,e.metadata,e.importance])}catch{return}},Xe=(e,l)=>{const h=[],w=[];for(const v of e)v.score>=l?h.push(v):w.push(v.id);return{kept:h,rejectedIds:w}},Je=e=>e!==void 0&&Object.keys(e).length>0,P=e=>{if(!e)return;const l=Object.entries(e).filter(([h])=>!je.has(h));return l.length>0?Object.fromEntries(l):void 0},le=new Set,Ge=e=>{le.has(e)||(le.add(e),console.warn(`[@lunora/ai/rag] index "${e}" is used without a namespace — in a multi-tenant/sharded
2
+ app this shares one tenant's chunks (text included) with every other tenant, since
3
+ Vectorize indexes are account-global. Pass \`namespace\` (the shard/tenant key) on both
4
+ index() and retrieve(). Single-tenant apps suppress this via { allowSharedNamespace: true }.`))},et=e=>e.map(l=>`[source:${l.sourceId}#${String(l.chunkIndex)}]
5
+ ${l.text}`).join(`
6
+
7
+ `),me=(e,l)=>{if(typeof e=="object")return e;if(l===void 0)throw new f("INTERNAL","@lunora/ai/rag: `embeddingModel` is a Workers AI model id (or omitted) but the bound context has no `ai` (env.AI). Pass an AI SDK EmbeddingModel object to embed without Workers AI, or bind a context whose `ctx.ai` is wired.");return l.embeddingModel(e)},tt=e=>{if(e===void 0)throw new f("INTERNAL","@lunora/ai/rag: the bound context has no `vectors` (env.VECTORIZE) and no `store` is configured — bind a context whose `ctx.vectors` is wired, or configure `store` (e.g. `sqliteVectorStore`) to back this index without Vectorize.");return e},mt=e=>{if(typeof e.index!="string"||e.index.length===0)throw new f("BAD_REQUEST","@lunora/ai/rag: `index` must be a non-empty Vectorize index name");const l=e.chunkSize??Ve,h=e.chunkOverlap??Fe;if(!Number.isInteger(l)||l<1)throw new f("BAD_REQUEST","@lunora/ai/rag: `chunkSize` must be a positive integer");if(!Number.isInteger(h)||h<0||h>=l)throw new f("BAD_REQUEST","@lunora/ai/rag: `chunkOverlap` must be a non-negative integer smaller than `chunkSize`");const w=ce-Pe;if(!e.chunk&&!e.textStore&&!e.store&&l>w)throw new f("BAD_REQUEST",`@lunora/ai/rag: \`chunkSize\` of ${String(l)} leaves no room under Vectorize's ${String(ce)}-byte metadata limit, which also carries this chunk's bookkeeping and any \`metadata\` you attach — keep it under ${String(w)}, or supply \`textStore\` to move chunk text out of metadata entirely`);const v=e.topK??Le;if(!Number.isInteger(v)||v<1)throw new f("BAD_REQUEST","@lunora/ai/rag: `topK` must be a positive integer");if(e.maxEmbeddingDimensions!==void 0&&e.maxEmbeddingDimensions!==!1&&(!Number.isInteger(e.maxEmbeddingDimensions)||e.maxEmbeddingDimensions<1))throw new f("BAD_REQUEST","@lunora/ai/rag: `maxEmbeddingDimensions` must be a positive integer, or `false` to disable the check");if(e.embeddingModelVersion!==void 0&&!qe.test(e.embeddingModelVersion))throw new f("BAD_REQUEST",'@lunora/ai/rag: `embeddingModelVersion` must match /^[A-Za-z0-9._-]{1,40}$/ (a short, stable tag like "bge-v1.5")');if(e.candidates!==void 0&&(!Number.isInteger(e.candidates)||e.candidates<1))throw new f("BAD_REQUEST","@lunora/ai/rag: `candidates` must be a positive integer");if(e.cacheEmbeddings!==void 0&&(!Number.isInteger(e.cacheEmbeddings)||e.cacheEmbeddings<0))throw new f("BAD_REQUEST","@lunora/ai/rag: `cacheEmbeddings` must be a non-negative integer");const M=e.cacheEmbeddings??0,K=e.rerank,we=e.chunk??(p=>Be(p,l,h)),{textStore:k}=e,$=e.embeddingModelVersion,V=p=>$===void 0?p:p===void 0?$:`${$}::${p}`;return p=>{const E=e.store?e.store(p):Qe(tt(p.vectors),e.index),H=k?E.capabilities.maxTopK:E.capabilities.maxTopKWithMetadata,U=e.maxEmbeddingDimensions??E.capabilities.maxDimensions;let N;const q=typeof p.trace=="function"?p.trace:void 0,W=typeof p.metrics?.count=="function"?p.metrics:void 0;let Z=U===!1;const X=(t,r)=>{if(Z||(Z=!0,U===!1||t<=U))return;const n=Q(r);throw new f("BAD_REQUEST",`@lunora/ai/rag: embedding model${n===void 0?"":` "${n}"`} produces ${String(t)}-dimension vectors, over the ${String(U)}-dimension ceiling of index "${e.index}" — either truncate them with the provider's \`dimensions\` option (Matryoshka models such as text-embedding-3-large support this), or set \`maxEmbeddingDimensions: false\` if this index is not Vectorize-backed`)},D=new Map,be=(t,r)=>{if(M!==0)for(D.set(t,r);D.size>M;){const n=D.keys().next();if(n.done===!0)break;D.delete(n.value)}},J=async t=>{const r=D.get(t);if(r!==void 0)return r;N??=me(e.embeddingModel,p.ai);const n=N,o=async a=>{const{embedding:c,providerMetadata:b,usage:d}=await Re({model:n,value:t});return X(c.length,n),ie(Q(n),{providerMetadata:b,usage:{inputTokens:{total:d.tokens}}},a,W),be(t,c),c};if(q===void 0)return o();const s=Q(n),i=typeof p.conversationId=="string"&&p.conversationId.length>0?p.conversationId:void 0;return q("ai.embed",(a,c)=>o(c),{"gen_ai.operation.name":"embeddings",...s===void 0?{}:{"gen_ai.request.model":s},...i===void 0?{}:{"gen_ai.conversation.id":i}})},ve=async(t,r,n)=>{if(!e.transformQuery||r?.transformQuery===!1)return[t];const o=typeof p.conversationId=="string"&&p.conversationId.length>0?p.conversationId:void 0,s=await e.transformQuery(t,{conversationId:o,namespace:n}),i=(typeof s=="string"?[s]:[...s]).map(a=>a.trim()).filter(a=>a.length>0);return i.length>0?i:[t]},ye=async t=>{const r=new Map,n=[...new Set(t.filter(o=>!D.has(o)))];if(n.length<2)return r;N??=me(e.embeddingModel,p.ai);try{const{embeddings:o,providerMetadata:s,usage:i}=await De({model:N,values:n});if(ie(Q(N),{providerMetadata:s,usage:{inputTokens:{total:i.tokens}}},void 0,W),o.length!==n.length)return r;const[a]=o;a!==void 0&&X(a.length,N);for(const[c,b]of n.entries())r.set(b,o[c])}catch(o){if(Me(o))throw o}return r},F=t=>{if(t===void 0){if(e.requireNamespace)throw new f("BAD_REQUEST",`@lunora/ai/rag: index "${e.index}" requires a namespace (requireNamespace is set) — pass the tenant/shard key on index()/retrieve()/remove()`);e.allowSharedNamespace||Ge(e.index)}},G=async(t,r)=>{const[n]=await E.getByIds([C(r,t,0)],r),o=n?.metadata?.[j],s=n?.metadata?.[Y];return{chunks:typeof s=="number"&&Number.isInteger(s)&&s>0?s:void 0,hash:typeof o=="string"?o:void 0}},ee=async(t,r,n,o)=>{const s=Array.from({length:n-r},(i,a)=>C(o,t,r+a));s.length!==0&&(await E.deleteByIds(s,o),await k?.remove?.(s,{namespace:o}),await e.lexicalStore?.remove?.(s,{namespace:o}))},Ee=async t=>{if(F(t.namespace),t.importance!==void 0&&(typeof t.importance!="number"||t.importance<0||t.importance>1))throw new f("BAD_REQUEST","@lunora/ai/rag: `importance` must be a number in [0, 1]");const r=V(t.namespace),n=Ze(t),o=await We(n??t.text),s=await G(t.id,r);if(t.reindex!==!0&&n!==void 0&&s.hash===o&&s.chunks!==void 0)return{chunks:s.chunks,ids:Array.from({length:s.chunks},(m,u)=>C(r,t.id,u)),unchanged:!0};const i=we(t.text),a=i.map((m,u)=>C(r,t.id,u)),c=a.at(-1);if(c!==void 0&&He(c,t.id,E.capabilities.maxIdBytes),i.length===0&&t.allowEmptySources===!1)throw new f("BAD_REQUEST",`@lunora/ai/rag: source "${t.id}" produced zero chunks — set allowEmptySources: true to allow this`);if(i.length>0){const m=i.map((u,y)=>({chunkIndex:y,id:a[y],sourceId:t.id,text:u,...t.metadata===void 0?{}:{metadata:t.metadata}}));k&&await k.put(m,{namespace:r}),e.lexicalStore&&await e.lexicalStore.index(m,{namespace:r})}const b=await ye(i),d=async m=>b.get(m)??await J(m);return await $e(i,Oe,async(m,u)=>{const y=a[u],x={...t.metadata,[he]:u,[pe]:t.id};k||(x[B]=m),t.importance!==void 0&&(x[O]=t.importance),u===0&&(x[j]=o,x[Y]=i.length,$!==void 0&&(x[ge]=$)),Ye(x,u,t.id,E.capabilities.maxMetadataBytes),await E.upsert({embed:d,id:y,input:m,metadata:x,namespace:r}),t.onChunk?.({chunkIndex:u,id:y,text:m,total:i.length})}),s.chunks!==void 0&&s.chunks>i.length&&await ee(t.id,i.length,s.chunks,r),{chunks:i.length,ids:a,unchanged:!1}},xe=async t=>{F(t.namespace);const r=V(t.namespace),o=(await G(t.id,r)).chunks??1;await ee(t.id,0,o,r)},te=async(t,r)=>{const n=new Map;if(t.length===0)return n;if(k){const s=await k.getMany(t,{namespace:r});for(const[i,a]of t.entries()){const c=s[i];typeof c=="string"&&n.set(a,c)}return n}const o=await E.getByIds(t,r);for(const s of o){const i=s.metadata?.[B];typeof i=="string"&&n.set(s.id,i)}return n},Ie=async(t,r,n)=>{const o=r?.chunkContext?.before??0,s=r?.chunkContext?.after??0;if(o===0&&s===0)return t;if(!Number.isInteger(o)||o<0||!Number.isInteger(s)||s<0)throw new f("BAD_REQUEST","@lunora/ai/rag: `chunkContext.before`/`chunkContext.after` must be non-negative integers");const i=new Map(t.map(d=>[d.id,d.text])),a=new Set;for(const d of t)for(let m=-o;m<=s;m+=1){const u=d.chunkIndex+m,y=C(n,d.sourceId,u);m!==0&&u>=0&&!i.has(y)&&a.add(y)}const c=await te([...a],n),b=(d,m)=>{const u=C(n,d,m);return i.get(u)??c.get(u)};return t.map(d=>{const m=[];for(let u=-o;u<=s;u+=1){const y=u===0?d.text:b(d.sourceId,d.chunkIndex+u);y!==void 0&&m.push(y)}return{...d,text:m.join(`
8
+ `)}})},Se=t=>{if(typeof t=="string"){const r=e.filters!==void 0&&Object.hasOwn(e.filters,t)?e.filters[t]:void 0;if(!r)throw new f("NOT_FOUND",`@lunora/ai/rag: unknown named filter "${t}" — must be one of the keys declared in RagConfig.filters`);return r.filter}return t},ke=(t,r)=>t.matches.map(n=>{const o=n.metadata??{},s=ue(n.id,r),i=o[B],a=o[O],c=typeof a=="number"&&a>=0&&a<=1?a:1;return{chunkIndex:s.chunkIndex,id:n.id,importance:c,metadata:P(o),score:n.score*c,sourceId:s.sourceId,text:typeof i=="string"?i:""}}),_e=async(t,r)=>{if(!k)return t;const n=t.map(a=>a.id),[o,s]=await Promise.all([te(n,r),E.getByIds(n,r)]),i=new Map(s.map(a=>[a.id,a.metadata]));return t.flatMap(a=>{const c=o.get(a.id);if(c===void 0)return[];const b=i.get(a.id),d=b?.[O],m=typeof d=="number"&&d>=0&&d<=1?d:a.importance,y=(a.importance===0?0:a.score/a.importance)*m;return[{...a,importance:m,metadata:P(b)??a.metadata,score:y,text:c}]})},re=async(t,r,n)=>{const o=t.map(a=>a.id).filter(a=>!r.has(a)),s=o.length===0?[]:await E.getByIds(o,n),i=new Map(s.map(a=>[a.id,a.metadata]));return t.map(a=>{const c=ue(a.id,n),b=i.get(a.id),d=b?.[O];return{chunkIndex:c.chunkIndex,id:a.id,importance:typeof d=="number"&&d>=0&&d<=1?d:1,metadata:P(b),score:a.score,sourceId:c.sourceId,text:a.text}})},ne=async(t,r)=>{F(r?.namespace);const n=V(r?.namespace),o=Se(r?.filter),s=e.rlsFilter?await e.rlsFilter(p.auth):void 0,i=s?{...o,...s}:o,a=Math.min(r?.topK??v,H),c=await ve(t,r,n),b=c[0],d=K!==void 0&&r?.rerank!==!1,u=d||e.lexicalStore!==void 0||e.graphStore!==void 0||c.length>1?Math.min(e.candidates??a*ze,H):a,y=r?.minScore,x=new Set,ae=async g=>{const _=await E.query({embed:J,filter:i,input:g,namespace:n,returnMetadata:k?"indexed":"all",topK:u}),A=await _e(ke(_,n),n);if(y===void 0)return A;const{kept:S,rejectedIds:T}=Xe(A,y);for(const Te of T)x.add(Te);return S},R=[{chunks:await ae(b)}];for(const g of c.slice(1))R.push({chunks:await ae(g)});if(e.lexicalStore){const g=await e.lexicalStore.search(b,{filter:i,namespace:n,topK:e.lexicalTopK??u}),_=new Set(R.flatMap(S=>S.chunks).map(S=>S.id)),A=g.filter(S=>!x.has(S.id)||_.has(S.id));R.push({chunks:await re(A,_,n)})}const L=R.flatMap(g=>[...g.chunks]);if(e.graphStore&&L.length>0&&(e.graphStore.enforcesFilter||!Je(i))){const g=[...new Set(L.map(T=>T.sourceId))],_=await e.graphStore.related(g,{filter:i,namespace:n,topK:e.graphTopK??u}),A=new Set(L.map(T=>T.id)),S=_.filter(T=>!x.has(T.id)||A.has(T.id));R.push({chunks:await re(S,A,n),weight:"proximity"})}const z=R.filter(g=>g.chunks.length>0);let I=z.length>1?[...Ue(z)]:[...z[0]?.chunks??[]];I.sort((g,_)=>_.score-g.score),d&&(I=[...await K(b,I)]),I=I.slice(0,a),I=[...await Ie(I,r,n)];const se=[],oe=new Set;for(const g of I)oe.has(g.sourceId)||(oe.add(g.sourceId),se.push({id:g.sourceId,metadata:g.metadata,weight:g.importance}));return r?.onRetrieve?.({matches:I.length,query:t}),{chunks:I,context:et(I),sources:se}};return{asTool:t=>Ae({description:t?.description??`Search the "${e.index}" knowledge base for passages relevant to a natural-language query.`,execute:async({query:r})=>ne(r,{namespace:t?.namespace,topK:t?.topK}),inputSchema:Ne({properties:{query:{description:"The natural-language search query.",type:"string"}},required:["query"],type:"object"})}),index:Ee,remove:xe,retrieve:ne}}};export{mt as default};
@@ -140,6 +140,19 @@ declare const AI_GATEWAY_TOKEN_ENV = "LUNORA_AI_GATEWAY_TOKEN";
140
140
  * call — telemetry configuration must not take inference down.
141
141
  */
142
142
  declare const AI_GATEWAY_TAGS_ENV = "LUNORA_AI_GATEWAY_TAGS";
143
+ /**
144
+ * Env var carrying the base URL of a self-hosted OpenAI-compatible proxy
145
+ * (LiteLLM, OpenRouter, your own), e.g. `https://ai-proxy.internal/v1`.
146
+ *
147
+ * When set, `ctx.ai.model("<provider>/<model>")` sends the slug unchanged as the
148
+ * `model` of an OpenAI chat-completions request to this URL instead of routing
149
+ * it through Cloudflare AI Gateway — which needs no `AI` binding, so it is how
150
+ * `ctx.ai` works on hosts without Workers AI (celld). `@cf/…` ids still need
151
+ * the binding.
152
+ */
153
+ declare const AI_PROXY_URL_ENV = "LUNORA_AI_PROXY_URL";
154
+ /** Env var carrying the bearer token sent to {@link AI_PROXY_URL_ENV}, when the proxy requires one. */
155
+ declare const AI_PROXY_TOKEN_ENV = "LUNORA_AI_PROXY_TOKEN";
143
156
  /**
144
157
  * Parse {@link AI_GATEWAY_TAGS_ENV} into tag fields. Non-string values are
145
158
  * dropped rather than coerced — a number silently becoming `"1"` is a worse
@@ -203,6 +216,73 @@ interface AiGatewayOptions {
203
216
  */
204
217
  metadata?: Record<string, string>;
205
218
  }
219
+ /**
220
+ * Options for the raw `ctx.ai.run(...)` passthrough — the Workers AI binding's
221
+ * third argument. Unlisted keys are forwarded to the binding unchanged.
222
+ * @experimental
223
+ */
224
+ interface AiRunOptions {
225
+ [key: string]: unknown;
226
+ /** Route this call through a Cloudflare AI Gateway. Defaults to the gateway `createAi` resolved. */
227
+ gateway?: AiGatewayOptions;
228
+ /**
229
+ * Fail at once instead of waiting in the Workers AI capacity queue when no
230
+ * capacity is free. The rejection surfaces as a `LunoraError` with code
231
+ * `RATE_LIMITED` (Workers AI error `3040`, HTTP 429).
232
+ */
233
+ rejectIfBusy?: boolean;
234
+ }
235
+ /**
236
+ * Per-call settings for `ctx.ai.model(...)`. Applied to Workers AI model ids
237
+ * (`@cf/…`) only; a gateway slug or a bring-your-own model ignores them.
238
+ * @experimental
239
+ */
240
+ interface AiModelOptions {
241
+ /**
242
+ * Fail at once instead of waiting in the Workers AI capacity queue. The
243
+ * provider reports the rejection as an AI SDK `APICallError` with
244
+ * `statusCode: 429`, which the AI SDK retries unless the call sets
245
+ * `maxRetries: 0`.
246
+ */
247
+ rejectIfBusy?: boolean;
248
+ }
249
+ /**
250
+ * Structural slice of the span handle `ctx.trace` hands its body — enough to
251
+ * attach a model call's usage once it is known. Declared here rather than
252
+ * imported so `@lunora/ai` takes no dependency on `@lunora/server`; the real
253
+ * handle is assignable to it.
254
+ * @experimental
255
+ */
256
+ interface AiSpan {
257
+ setAttribute: (key: string, value: unknown) => void;
258
+ setAttributes: (fields: Record<string, unknown>) => void;
259
+ }
260
+ /**
261
+ * Structural slice of `ctx.trace` (the server `LunoraTracer`): runs `function_`
262
+ * inside a named span and hands it the span's {@link AiSpan}.
263
+ * @experimental
264
+ */
265
+ type AiTracer = <T>(name: string, function_: (trace: AiTracer, span: AiSpan) => Promise<T> | T, attributes?: Record<string, unknown>) => Promise<T>;
266
+ /**
267
+ * Structural slice of `ctx.metrics` (the server `LunoraMetrics`) — only the
268
+ * counter, which is what usage accounting needs.
269
+ * @experimental
270
+ */
271
+ interface AiMetrics {
272
+ count: (name: string, value?: number, attributes?: Record<string, unknown>) => void;
273
+ }
274
+ /**
275
+ * Where `ctx.ai` reports model usage. The generated `ctx.ai` passes the
276
+ * function's own `ctx.trace` / `ctx.metrics`, so every call made through
277
+ * `ctx.ai.model(...)` gets an `ai.generate` / `ai.stream` span and
278
+ * `gen_ai.usage.input_tokens` / `gen_ai.usage.output_tokens` /
279
+ * `gen_ai.usage.cost` counters attributed to that function.
280
+ * @experimental
281
+ */
282
+ interface AiTelemetry {
283
+ metrics?: AiMetrics;
284
+ trace?: AiTracer;
285
+ }
206
286
  /**
207
287
  * `LunoraAiOptions` is part of the experimental `@lunora/ai` API and may change without a major version bump.
208
288
  * @experimental
@@ -267,13 +347,21 @@ interface LunoraAiOptions {
267
347
  * (e.g. `safePrompt`) before handing it to `@lunora/ai`.
268
348
  */
269
349
  provider?: WorkersAiProviderLike;
350
+ /**
351
+ * Record every language-model call resolved by `model()` — a span plus
352
+ * token and cost counters (see {@link AiTelemetry}). Omitted, models are
353
+ * returned unwrapped.
354
+ */
355
+ telemetry?: AiTelemetry;
270
356
  }
271
357
  /**
272
358
  * A model to run against. The AI SDK's {@link LanguageModel} already admits a
273
359
  * bare `string`, so this alias covers both arms of the provider-agnostic seam:
274
- * a string id is the Workers AI convenience path (resolved by `ctx.ai.model`),
275
- * a built model object is bring-your-own (`@ai-sdk/openai`, `@ai-sdk/anthropic`,
276
- * `@ai-sdk/google`, OpenRouter, …).
360
+ * a string id is resolved by `ctx.ai.model` — a Workers AI id (`@cf/…`), a
361
+ * `"<provider>/<model>"` slug (`anthropic/claude-sonnet-5`, `openai/gpt-5`, …)
362
+ * routed through Cloudflare AI Gateway, or a gateway dynamic route
363
+ * (`dynamic/<route>`); a built model object is bring-your-own (`@ai-sdk/openai`,
364
+ * `@ai-sdk/anthropic`, `@ai-sdk/google`, OpenRouter, …).
277
365
  * @experimental
278
366
  */
279
367
  type ModelInput = LanguageModel;
@@ -295,16 +383,22 @@ type EmbeddingModelInput = EmbeddingModel | string;
295
383
  interface LunoraAi {
296
384
  /** Resolve an {@link EmbeddingModel}: a string → Workers AI, an object → passthrough. */
297
385
  embeddingModel: (model?: EmbeddingModelInput) => EmbeddingModel;
298
- /** Resolve a {@link LanguageModel}: a string → Workers AI, an object → passthrough. */
299
- model: (model?: ModelInput) => LanguageModel;
386
+ /**
387
+ * Resolve a {@link LanguageModel}: a `@cf/…` id → Workers AI, a
388
+ * `"<provider>/<model>"` slug or `dynamic/<route>` → Cloudflare AI Gateway
389
+ * over the same binding (Unified Billing or the gateway's stored keys — no
390
+ * provider key in the app) or, with `LUNORA_AI_PROXY_URL` set, to that
391
+ * OpenAI-compatible proxy; an object → passthrough.
392
+ */
393
+ model: (model?: ModelInput, options?: AiModelOptions) => LanguageModel;
300
394
  /**
301
395
  * Raw Workers AI binding passthrough (void-style `ai.run`). Bypasses the AI
302
396
  * SDK entirely — useful for Workers-AI-only model families (image, ASR,
303
397
  * translation) not surfaced through the provider. Throws if no binding was
304
398
  * supplied.
305
399
  */
306
- run: (model: string, inputs: Record<string, unknown>, options?: Record<string, unknown>) => Promise<unknown>;
400
+ run: (model: string, inputs: Record<string, unknown>, options?: AiRunOptions) => Promise<unknown>;
307
401
  /** The underlying Workers AI provider — `ai.workersai("@cf/...")` for a raw model. */
308
402
  workersai: WorkersAiProviderLike;
309
403
  }
310
- export { AI_DEFAULT_EMBEDDING_MODEL_ENV as A, EmbeddingModelInput as E, LunoraAiOptions as L, ModelInput as M, ResolvedAiGateway as R, WorkersAiProviderLike as W, LunoraAi as a, AI_DEFAULT_MODEL_ENV as b, AI_GATEWAY_ACCOUNT_ID_ENV as c, AI_GATEWAY_ID_ENV as d, AI_GATEWAY_METADATA_MAX_KEYS as e, AI_GATEWAY_TAGS_ENV as f, AI_GATEWAY_TOKEN_ENV as g, AiBindingLike as h, AiGatewayMetadata as i, AiGatewayOptions as j, buildAiGatewayMetadataFields as k, resolveAiGateway as l, readAiGatewayEnvTags as r };
404
+ export { AI_DEFAULT_EMBEDDING_MODEL_ENV as A, EmbeddingModelInput as E, LunoraAiOptions as L, ModelInput as M, ResolvedAiGateway as R, WorkersAiProviderLike as W, LunoraAi as a, AI_DEFAULT_MODEL_ENV as b, AI_GATEWAY_ACCOUNT_ID_ENV as c, AI_GATEWAY_ID_ENV as d, AI_GATEWAY_METADATA_MAX_KEYS as e, AI_GATEWAY_TAGS_ENV as f, AI_GATEWAY_TOKEN_ENV as g, AI_PROXY_TOKEN_ENV as h, AI_PROXY_URL_ENV as i, AiBindingLike as j, AiGatewayMetadata as k, AiGatewayOptions as l, AiMetrics as m, AiModelOptions as n, AiRunOptions as o, AiSpan as p, AiTelemetry as q, AiTracer as r, buildAiGatewayMetadataFields as s, readAiGatewayEnvTags as t, resolveAiGateway as u };
@@ -140,6 +140,19 @@ declare const AI_GATEWAY_TOKEN_ENV = "LUNORA_AI_GATEWAY_TOKEN";
140
140
  * call — telemetry configuration must not take inference down.
141
141
  */
142
142
  declare const AI_GATEWAY_TAGS_ENV = "LUNORA_AI_GATEWAY_TAGS";
143
+ /**
144
+ * Env var carrying the base URL of a self-hosted OpenAI-compatible proxy
145
+ * (LiteLLM, OpenRouter, your own), e.g. `https://ai-proxy.internal/v1`.
146
+ *
147
+ * When set, `ctx.ai.model("<provider>/<model>")` sends the slug unchanged as the
148
+ * `model` of an OpenAI chat-completions request to this URL instead of routing
149
+ * it through Cloudflare AI Gateway — which needs no `AI` binding, so it is how
150
+ * `ctx.ai` works on hosts without Workers AI (celld). `@cf/…` ids still need
151
+ * the binding.
152
+ */
153
+ declare const AI_PROXY_URL_ENV = "LUNORA_AI_PROXY_URL";
154
+ /** Env var carrying the bearer token sent to {@link AI_PROXY_URL_ENV}, when the proxy requires one. */
155
+ declare const AI_PROXY_TOKEN_ENV = "LUNORA_AI_PROXY_TOKEN";
143
156
  /**
144
157
  * Parse {@link AI_GATEWAY_TAGS_ENV} into tag fields. Non-string values are
145
158
  * dropped rather than coerced — a number silently becoming `"1"` is a worse
@@ -203,6 +216,73 @@ interface AiGatewayOptions {
203
216
  */
204
217
  metadata?: Record<string, string>;
205
218
  }
219
+ /**
220
+ * Options for the raw `ctx.ai.run(...)` passthrough — the Workers AI binding's
221
+ * third argument. Unlisted keys are forwarded to the binding unchanged.
222
+ * @experimental
223
+ */
224
+ interface AiRunOptions {
225
+ [key: string]: unknown;
226
+ /** Route this call through a Cloudflare AI Gateway. Defaults to the gateway `createAi` resolved. */
227
+ gateway?: AiGatewayOptions;
228
+ /**
229
+ * Fail at once instead of waiting in the Workers AI capacity queue when no
230
+ * capacity is free. The rejection surfaces as a `LunoraError` with code
231
+ * `RATE_LIMITED` (Workers AI error `3040`, HTTP 429).
232
+ */
233
+ rejectIfBusy?: boolean;
234
+ }
235
+ /**
236
+ * Per-call settings for `ctx.ai.model(...)`. Applied to Workers AI model ids
237
+ * (`@cf/…`) only; a gateway slug or a bring-your-own model ignores them.
238
+ * @experimental
239
+ */
240
+ interface AiModelOptions {
241
+ /**
242
+ * Fail at once instead of waiting in the Workers AI capacity queue. The
243
+ * provider reports the rejection as an AI SDK `APICallError` with
244
+ * `statusCode: 429`, which the AI SDK retries unless the call sets
245
+ * `maxRetries: 0`.
246
+ */
247
+ rejectIfBusy?: boolean;
248
+ }
249
+ /**
250
+ * Structural slice of the span handle `ctx.trace` hands its body — enough to
251
+ * attach a model call's usage once it is known. Declared here rather than
252
+ * imported so `@lunora/ai` takes no dependency on `@lunora/server`; the real
253
+ * handle is assignable to it.
254
+ * @experimental
255
+ */
256
+ interface AiSpan {
257
+ setAttribute: (key: string, value: unknown) => void;
258
+ setAttributes: (fields: Record<string, unknown>) => void;
259
+ }
260
+ /**
261
+ * Structural slice of `ctx.trace` (the server `LunoraTracer`): runs `function_`
262
+ * inside a named span and hands it the span's {@link AiSpan}.
263
+ * @experimental
264
+ */
265
+ type AiTracer = <T>(name: string, function_: (trace: AiTracer, span: AiSpan) => Promise<T> | T, attributes?: Record<string, unknown>) => Promise<T>;
266
+ /**
267
+ * Structural slice of `ctx.metrics` (the server `LunoraMetrics`) — only the
268
+ * counter, which is what usage accounting needs.
269
+ * @experimental
270
+ */
271
+ interface AiMetrics {
272
+ count: (name: string, value?: number, attributes?: Record<string, unknown>) => void;
273
+ }
274
+ /**
275
+ * Where `ctx.ai` reports model usage. The generated `ctx.ai` passes the
276
+ * function's own `ctx.trace` / `ctx.metrics`, so every call made through
277
+ * `ctx.ai.model(...)` gets an `ai.generate` / `ai.stream` span and
278
+ * `gen_ai.usage.input_tokens` / `gen_ai.usage.output_tokens` /
279
+ * `gen_ai.usage.cost` counters attributed to that function.
280
+ * @experimental
281
+ */
282
+ interface AiTelemetry {
283
+ metrics?: AiMetrics;
284
+ trace?: AiTracer;
285
+ }
206
286
  /**
207
287
  * `LunoraAiOptions` is part of the experimental `@lunora/ai` API and may change without a major version bump.
208
288
  * @experimental
@@ -267,13 +347,21 @@ interface LunoraAiOptions {
267
347
  * (e.g. `safePrompt`) before handing it to `@lunora/ai`.
268
348
  */
269
349
  provider?: WorkersAiProviderLike;
350
+ /**
351
+ * Record every language-model call resolved by `model()` — a span plus
352
+ * token and cost counters (see {@link AiTelemetry}). Omitted, models are
353
+ * returned unwrapped.
354
+ */
355
+ telemetry?: AiTelemetry;
270
356
  }
271
357
  /**
272
358
  * A model to run against. The AI SDK's {@link LanguageModel} already admits a
273
359
  * bare `string`, so this alias covers both arms of the provider-agnostic seam:
274
- * a string id is the Workers AI convenience path (resolved by `ctx.ai.model`),
275
- * a built model object is bring-your-own (`@ai-sdk/openai`, `@ai-sdk/anthropic`,
276
- * `@ai-sdk/google`, OpenRouter, …).
360
+ * a string id is resolved by `ctx.ai.model` — a Workers AI id (`@cf/…`), a
361
+ * `"<provider>/<model>"` slug (`anthropic/claude-sonnet-5`, `openai/gpt-5`, …)
362
+ * routed through Cloudflare AI Gateway, or a gateway dynamic route
363
+ * (`dynamic/<route>`); a built model object is bring-your-own (`@ai-sdk/openai`,
364
+ * `@ai-sdk/anthropic`, `@ai-sdk/google`, OpenRouter, …).
277
365
  * @experimental
278
366
  */
279
367
  type ModelInput = LanguageModel;
@@ -295,16 +383,22 @@ type EmbeddingModelInput = EmbeddingModel | string;
295
383
  interface LunoraAi {
296
384
  /** Resolve an {@link EmbeddingModel}: a string → Workers AI, an object → passthrough. */
297
385
  embeddingModel: (model?: EmbeddingModelInput) => EmbeddingModel;
298
- /** Resolve a {@link LanguageModel}: a string → Workers AI, an object → passthrough. */
299
- model: (model?: ModelInput) => LanguageModel;
386
+ /**
387
+ * Resolve a {@link LanguageModel}: a `@cf/…` id → Workers AI, a
388
+ * `"<provider>/<model>"` slug or `dynamic/<route>` → Cloudflare AI Gateway
389
+ * over the same binding (Unified Billing or the gateway's stored keys — no
390
+ * provider key in the app) or, with `LUNORA_AI_PROXY_URL` set, to that
391
+ * OpenAI-compatible proxy; an object → passthrough.
392
+ */
393
+ model: (model?: ModelInput, options?: AiModelOptions) => LanguageModel;
300
394
  /**
301
395
  * Raw Workers AI binding passthrough (void-style `ai.run`). Bypasses the AI
302
396
  * SDK entirely — useful for Workers-AI-only model families (image, ASR,
303
397
  * translation) not surfaced through the provider. Throws if no binding was
304
398
  * supplied.
305
399
  */
306
- run: (model: string, inputs: Record<string, unknown>, options?: Record<string, unknown>) => Promise<unknown>;
400
+ run: (model: string, inputs: Record<string, unknown>, options?: AiRunOptions) => Promise<unknown>;
307
401
  /** The underlying Workers AI provider — `ai.workersai("@cf/...")` for a raw model. */
308
402
  workersai: WorkersAiProviderLike;
309
403
  }
310
- export { AI_DEFAULT_EMBEDDING_MODEL_ENV as A, EmbeddingModelInput as E, LunoraAiOptions as L, ModelInput as M, ResolvedAiGateway as R, WorkersAiProviderLike as W, LunoraAi as a, AI_DEFAULT_MODEL_ENV as b, AI_GATEWAY_ACCOUNT_ID_ENV as c, AI_GATEWAY_ID_ENV as d, AI_GATEWAY_METADATA_MAX_KEYS as e, AI_GATEWAY_TAGS_ENV as f, AI_GATEWAY_TOKEN_ENV as g, AiBindingLike as h, AiGatewayMetadata as i, AiGatewayOptions as j, buildAiGatewayMetadataFields as k, resolveAiGateway as l, readAiGatewayEnvTags as r };
404
+ export { AI_DEFAULT_EMBEDDING_MODEL_ENV as A, EmbeddingModelInput as E, LunoraAiOptions as L, ModelInput as M, ResolvedAiGateway as R, WorkersAiProviderLike as W, LunoraAi as a, AI_DEFAULT_MODEL_ENV as b, AI_GATEWAY_ACCOUNT_ID_ENV as c, AI_GATEWAY_ID_ENV as d, AI_GATEWAY_METADATA_MAX_KEYS as e, AI_GATEWAY_TAGS_ENV as f, AI_GATEWAY_TOKEN_ENV as g, AI_PROXY_TOKEN_ENV as h, AI_PROXY_URL_ENV as i, AiBindingLike as j, AiGatewayMetadata as k, AiGatewayOptions as l, AiMetrics as m, AiModelOptions as n, AiRunOptions as o, AiSpan as p, AiTelemetry as q, AiTracer as r, buildAiGatewayMetadataFields as s, readAiGatewayEnvTags as t, resolveAiGateway as u };
@@ -0,0 +1 @@
1
+ import{estimateModelCost as d}from"./DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";const a=t=>typeof t=="number"&&Number.isFinite(t)&&t>=0?t:void 0,f=t=>{if(!(typeof t!="object"||t===null)){for(const o of Object.values(t))if(typeof o=="object"&&o!==null){const{cost:e}=o;if(typeof e=="number"&&Number.isFinite(e))return e}}},b=t=>{const o=t.modelId;return typeof o=="string"&&o.length>0?o:void 0},l=(t,o,e,u)=>{const s=a(o.usage?.inputTokens?.total),n=a(o.usage?.outputTokens?.total),c=f(o.providerMetadata),i=c??d(t,{inputTokens:s,outputTokens:n}),g=c===void 0?"estimated":"provider",r={"gen_ai.request.model":t??"unknown"};s!==void 0&&(e?.setAttribute("gen_ai.usage.input_tokens",s),u?.count("gen_ai.usage.input_tokens",s,r)),n!==void 0&&(e?.setAttribute("gen_ai.usage.output_tokens",n),u?.count("gen_ai.usage.output_tokens",n,r)),i!==void 0&&(e?.setAttributes({"gen_ai.usage.cost":i,"lunora.usage.cost.source":g}),u?.count("gen_ai.usage.cost",i,{...r,"lunora.usage.cost.source":g}))};export{b as m,l as r};
@@ -1,5 +1,5 @@
1
1
  import { Tool } from 'ai';
2
- import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-D3U2budn.mjs";
2
+ import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-ChtZguP_.mjs";
3
3
  /**
4
4
  * Built-in fixed-window chunker: split into `size`-char windows overlapping by
5
5
  * `overlap` chars. Deliberately simple and deterministic — the zero-config
@@ -240,6 +240,13 @@ interface RagContext {
240
240
  * attribute is absent (backward-compatible).
241
241
  */
242
242
  conversationId?: string;
243
+ /**
244
+ * Optional `ctx.metrics` — an `ActionCtx`'s `ctx.metrics` satisfies it. When
245
+ * present, every embed counts its tokens and cost into the durable
246
+ * `gen_ai.usage.*` series, like a `ctx.ai.model(...)` call. `unknown` for the
247
+ * same decoupling reason as `trace`; `defineRag` narrows it.
248
+ */
249
+ metrics?: unknown;
243
250
  /**
244
251
  * Optional `ctx.trace` span factory — an `ActionCtx`'s `ctx.trace` satisfies
245
252
  * it structurally. When present, `defineRag` wraps each embedding-model
@@ -1,5 +1,5 @@
1
1
  import { Tool } from 'ai';
2
- import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-D3U2budn.js";
2
+ import { E as EmbeddingModelInput, a as LunoraAi } from "../packem_shared/types.d-ChtZguP_.js";
3
3
  /**
4
4
  * Built-in fixed-window chunker: split into `size`-char windows overlapping by
5
5
  * `overlap` chars. Deliberately simple and deterministic — the zero-config
@@ -240,6 +240,13 @@ interface RagContext {
240
240
  * attribute is absent (backward-compatible).
241
241
  */
242
242
  conversationId?: string;
243
+ /**
244
+ * Optional `ctx.metrics` — an `ActionCtx`'s `ctx.metrics` satisfies it. When
245
+ * present, every embed counts its tokens and cost into the durable
246
+ * `gen_ai.usage.*` series, like a `ctx.ai.model(...)` call. `unknown` for the
247
+ * same decoupling reason as `trace`; `defineRag` narrows it.
248
+ */
249
+ metrics?: unknown;
243
250
  /**
244
251
  * Optional `ctx.trace` span factory — an `ActionCtx`'s `ctx.trace` satisfies
245
252
  * it structurally. When present, `defineRag` wraps each embedding-model
@@ -1 +1 @@
1
- import{default as o}from"../packem_shared/fixedWindowChunks-C461ahRE.mjs";import{markdownChunker as a,sentenceChunker as f,tokenChunker as n}from"../packem_shared/markdownChunker-iW8V4klY.mjs";import{default as x}from"../packem_shared/defineRag-Cs7m4xJJ.mjs";import{contentHash as p,guessMimeTypeFromExtension as i}from"../packem_shared/contentHash-BIn6ECP8.mjs";import{hybridRank as d}from"../packem_shared/hybridRank-B4skyCLx.mjs";import{default as k}from"../packem_shared/bm25LexicalStore-DMUzAL0O.mjs";import{default as h}from"../packem_shared/matchesMetadataFilter-BbIOyA5g.mjs";import{batchReranker as g,scoreReranker as C}from"../packem_shared/batchReranker-Bc38FBLH.mjs";import{defineRagSource as E}from"../packem_shared/defineRagSource-Q3f3niU8.mjs";import{sqlLexicalStore as T}from"../packem_shared/sqlLexicalStore-4C_cIwef.mjs";import{sqliteVectorStore as y}from"../packem_shared/sqliteVectorStore-SjOFKoHo.mjs";import{ragSyncTriggers as q}from"../packem_shared/ragSyncTriggers-hgMcA4f2.mjs";import{VECTORIZE_CAPABILITIES as A,vectorizeStore as F}from"../packem_shared/VECTORIZE_CAPABILITIES-CUQDoxis.mjs";export{A as VECTORIZE_CAPABILITIES,g as batchReranker,k as bm25LexicalStore,p as contentHash,x as defineRag,E as defineRagSource,o as fixedWindowChunks,i as guessMimeTypeFromExtension,d as hybridRank,a as markdownChunker,h as matchesMetadataFilter,q as ragSyncTriggers,C as scoreReranker,f as sentenceChunker,T as sqlLexicalStore,y as sqliteVectorStore,n as tokenChunker,F as vectorizeStore};
1
+ import{default as o}from"../packem_shared/fixedWindowChunks-C461ahRE.mjs";import{markdownChunker as a,sentenceChunker as f,tokenChunker as n}from"../packem_shared/markdownChunker-iW8V4klY.mjs";import{default as x}from"../packem_shared/defineRag-CMTzKfS7.mjs";import{contentHash as p,guessMimeTypeFromExtension as i}from"../packem_shared/contentHash-BIn6ECP8.mjs";import{hybridRank as d}from"../packem_shared/hybridRank-B4skyCLx.mjs";import{default as k}from"../packem_shared/bm25LexicalStore-DMUzAL0O.mjs";import{default as h}from"../packem_shared/matchesMetadataFilter-BbIOyA5g.mjs";import{batchReranker as g,scoreReranker as C}from"../packem_shared/batchReranker-Bc38FBLH.mjs";import{defineRagSource as E}from"../packem_shared/defineRagSource-Q3f3niU8.mjs";import{sqlLexicalStore as T}from"../packem_shared/sqlLexicalStore-4C_cIwef.mjs";import{sqliteVectorStore as y}from"../packem_shared/sqliteVectorStore-SjOFKoHo.mjs";import{ragSyncTriggers as q}from"../packem_shared/ragSyncTriggers-hgMcA4f2.mjs";import{VECTORIZE_CAPABILITIES as A,vectorizeStore as F}from"../packem_shared/VECTORIZE_CAPABILITIES-CUQDoxis.mjs";export{A as VECTORIZE_CAPABILITIES,g as batchReranker,k as bm25LexicalStore,p as contentHash,x as defineRag,E as defineRagSource,o as fixedWindowChunks,i as guessMimeTypeFromExtension,d as hybridRank,a as markdownChunker,h as matchesMetadataFilter,q as ragSyncTriggers,C as scoreReranker,f as sentenceChunker,T as sqlLexicalStore,y as sqliteVectorStore,n as tokenChunker,F as vectorizeStore};
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "@lunora/ai",
3
- "version": "1.0.0-alpha.103",
3
+ "version": "1.0.0-alpha.105",
4
4
  "description": "Workers AI helper for Lunora: provider-agnostic AI SDK access from functions, Workers AI by default",
5
5
  "keywords": [
6
6
  "ai",
@@ -53,22 +53,12 @@
53
53
  "access": "public"
54
54
  },
55
55
  "dependencies": {
56
- "@lunora/errors": "1.0.0-alpha.47",
56
+ "@ai-sdk/anthropic": "4.0.49",
57
+ "@ai-sdk/openai": "4.0.60",
58
+ "@lunora/errors": "1.0.0-alpha.48",
57
59
  "ai": "7.0.93",
58
60
  "workers-ai-provider": "4.0.0"
59
61
  },
60
- "peerDependencies": {
61
- "@ai-sdk/anthropic": "^4.0.0",
62
- "@ai-sdk/openai": "^4.0.0"
63
- },
64
- "peerDependenciesMeta": {
65
- "@ai-sdk/anthropic": {
66
- "optional": true
67
- },
68
- "@ai-sdk/openai": {
69
- "optional": true
70
- }
71
- },
72
62
  "engines": {
73
63
  "node": "^22.15.0 || >=24.11.0"
74
64
  }
@@ -1 +0,0 @@
1
- const a=(t,o)=>{if(t===void 0)return;const e=t[o];return typeof e=="string"&&e.length>0?e:void 0};let A=!1,d=!1;const f=5,l=t=>{if(t===void 0)return;const o={};typeof t.functionPath=="string"&&t.functionPath.length>0&&(o.functionPath=t.functionPath),typeof t.traceId=="string"&&t.traceId.length>0&&(o.traceId=t.traceId);const e={};for(const[n,s]of Object.entries(t.tags??{}))typeof s=="string"&&s.length>0&&n.length>0&&!Object.hasOwn(o,n)&&(e[n]=s);Object.assign(e,o);const r=Object.keys(e);if(r.length===0)return;if(r.length<=f)return e;const i=r.slice(-f);return Object.fromEntries(i.map(n=>[n,e[n]]))},g=t=>{const o=l(t);return o===void 0?void 0:JSON.stringify(o)},I="LUNORA_AI_DEFAULT_MODEL",y="LUNORA_AI_DEFAULT_EMBEDDING_MODEL",E="LUNORA_AI_GATEWAY_ACCOUNT_ID",h="LUNORA_AI_GATEWAY_ID",_="LUNORA_AI_GATEWAY_TOKEN",u="LUNORA_AI_GATEWAY_TAGS",O=t=>{const o=a(t,u);if(o!==void 0)try{const e=JSON.parse(o);if(typeof e!="object"||e===null||Array.isArray(e))throw new TypeError("expected a JSON object");const r={};for(const[i,n]of Object.entries(e))typeof n=="string"&&n.length>0&&(r[i]=n);return Object.keys(r).length>0?r:void 0}catch{d||(d=!0,console.warn(`[lunora:ai] ${u} is not a flat JSON object of strings — AI Gateway tags from it are ignored.`));return}},T=(t,o,e="byo-provider")=>{const r=a(t,E),i=a(t,h);if(r===void 0||i===void 0)return;const n=a(t,_),s={};n!==void 0&&(s["cf-aig-authorization"]=`Bearer ${n}`,e==="workers-ai-binding"&&!A&&(A=!0,console.warn(`[lunora:ai] ${_} is set, but the Workers AI binding cannot send a gateway auth token — Cloudflare's native gateway option has no authorization field. The token is ignored on this path; use a bring-your-own AI SDK provider (which sends cf-aig-authorization), or make the AI Gateway unauthenticated for Workers AI.`)));const c=g(o);return c!==void 0&&(s["cf-aig-metadata"]=c),{accountId:r,baseURL:`https://gateway.ai.cloudflare.com/v1/${r}/${i}`,gatewayId:i,headers:s}};export{y as AI_DEFAULT_EMBEDDING_MODEL_ENV,I as AI_DEFAULT_MODEL_ENV,E as AI_GATEWAY_ACCOUNT_ID_ENV,h as AI_GATEWAY_ID_ENV,f as AI_GATEWAY_METADATA_MAX_KEYS,u as AI_GATEWAY_TAGS_ENV,_ as AI_GATEWAY_TOKEN_ENV,l as buildAiGatewayMetadataFields,O as readAiGatewayEnvTags,a as readEnv,T as resolveAiGateway};
@@ -1 +0,0 @@
1
- import{LunoraError as l}from"@lunora/errors";import{createWorkersAI as M}from"workers-ai-provider";import{readEnv as g,AI_DEFAULT_MODEL_ENV as w,AI_DEFAULT_EMBEDDING_MODEL_ENV as E,readAiGatewayEnvTags as y,buildAiGatewayMetadataFields as N,resolveAiGateway as h}from"./AI_DEFAULT_EMBEDDING_MODEL_ENV-B58f6yo6.mjs";const D=(d,o,n)=>{const v=o===void 0?void 0:y(o),a=v===void 0?n:{...n,tags:{...v,...n?.tags}},t=N(a);if(d!==void 0)return t!==void 0&&d.metadata===void 0?{...d,metadata:t}:d;if(o===void 0)return;const i=h(o,a,"workers-ai-binding");if(i!==void 0)return t===void 0?{id:i.gatewayId}:{id:i.gatewayId,metadata:t}},R=d=>{const{binding:o,defaultEmbeddingModel:n,defaultModel:v,env:a,gateway:t,metadata:i,provider:c}=d;if(!c&&!o)throw new l("INTERNAL","@lunora/ai: createAi requires a `binding` (env.AI) or a pre-built `provider`");const f=D(t,a,i),m=v??g(a,w),b=n??g(a,E),s=c??M({binding:o,gateway:f}),A=e=>{if(e===void 0){if(!m)throw new l("INTERNAL",`@lunora/ai: no model supplied and no default configured — pass a model id, or set ${w} in the Worker env (wrangler \`vars\` / \`.dev.vars\`)`);return s(m)}return typeof e=="string"?s(e):e},p=e=>{const r=s.textEmbeddingModel;if(typeof r!="function")throw new l("INTERNAL","@lunora/ai: the Workers AI provider does not expose `textEmbeddingModel`; pass an AI SDK EmbeddingModel (e.g. from @ai-sdk/openai) to embed()");return r.call(s,e)};return{embeddingModel:e=>{if(typeof e=="object")return e;const r=e??b;if(!r)throw new l("INTERNAL",`@lunora/ai: no embedding model supplied and no default configured — pass an embedding model id or an AI SDK EmbeddingModel, or set ${E} in the Worker env (wrangler \`vars\` / \`.dev.vars\`)`);return p(r)},model:A,run:async(e,r,u)=>{if(!o)throw new l("INTERNAL","@lunora/ai: ai.run requires the `binding` (env.AI) — it is unavailable when only a custom `provider` was supplied");const I=f!==void 0&&u?.gateway===void 0?{...u,gateway:f}:u;return o.run(e,r,I)},workersai:s}};export{R as default};
@@ -1,8 +0,0 @@
1
- import{LunoraError as b,isLunoraError as _e}from"@lunora/errors";import{tool as Te,jsonSchema as Ae,embedMany as Me,embed as Ne}from"ai";import{s as De}from"./stable-key-B_BlboiY.mjs";import{estimateModelCost as Re}from"./DEFAULT_MODEL_PRICES-CIJLs1Ah.mjs";import Ce from"./fixedWindowChunks-C461ahRE.mjs";import{c as Be,I as Oe}from"./concurrent-C6nqBv41.mjs";import{contentHash as $e}from"./contentHash-BIn6ECP8.mjs";import{hybridRank as Ke}from"./hybridRank-B4skyCLx.mjs";import{VECTORIZE_CAPABILITIES as ae,vectorizeStore as Ue}from"./VECTORIZE_CAPABILITIES-CUQDoxis.mjs";const Qe=1e3,Ve=200,Fe=5,Le=4,ie=ae.maxMetadataBytes===!1?Number.POSITIVE_INFINITY:ae.maxMetadataBytes,je=2*1024,le="__ragChunk",me="__ragSource",C="__ragText",P="__ragHash",Y="__ragChunks",$="__ragImportance",he="__ragModel",ze=new Set([le,Y,P,$,he,me,C]),Pe=(e,d,h,w)=>{if(w===!1)return;const v=new TextEncoder().encode(JSON.stringify(e)).length;if(v<=w)return;const K=(typeof e[C]=="string"?new TextEncoder().encode(e[C]).length:0)*2>v?"lower `chunkSize` (it counts characters, not bytes — multibyte text costs up to 3 bytes each), or supply `textStore` to move chunk text out of metadata entirely":"attach less per-source `metadata`";throw new b("BAD_REQUEST",`@lunora/ai/rag: chunk ${String(d)} of "${h}" carries ${String(v)} bytes of metadata, over the store's ${String(w)}-byte per-vector ceiling — ${K}`)},Ye=(e,d,h)=>{if(h===!1)return;const w=new TextEncoder().encode(e).length;if(!(w<=h))throw new b("BAD_REQUEST",`@lunora/ai/rag: chunk id "${e}" for source "${d}" is ${String(w)} bytes, over the store's ${String(h)}-byte per-vector id ceiling — shorten the source id (hash long keys before indexing them) or shorten the \`namespace\`, which is prefixed onto every chunk id`)},He=/^[\w.-]{1,40}$/,pe=e=>e===void 0?"":`${encodeURIComponent(e)}#`,R=(e,d,h)=>`${pe(e)}${d}#${String(h)}`,de=(e,d)=>{const h=pe(d),w=h!==""&&e.startsWith(h)?e.slice(h.length):e,v=w.lastIndexOf("#"),A=v===-1?Number.NaN:Number(w.slice(v+1));return v===-1||!Number.isInteger(A)||A<0?{chunkIndex:0,sourceId:w}:{chunkIndex:A,sourceId:w.slice(0,v)}},qe=async e=>$e(new TextEncoder().encode(e)),We=e=>{try{return De([e.text,e.metadata,e.importance])}catch{return}},Ze=(e,d)=>{const h=[],w=[];for(const v of e)v.score>=d?h.push(v):w.push(v.id);return{kept:h,rejectedIds:w}},Xe=e=>e!==void 0&&Object.keys(e).length>0,j=e=>{if(!e)return;const d=Object.entries(e).filter(([h])=>!ze.has(h));return d.length>0?Object.fromEntries(d):void 0},ce=new Set,Je=e=>{ce.has(e)||(ce.add(e),console.warn(`[@lunora/ai/rag] index "${e}" is used without a namespace — in a multi-tenant/sharded
2
- app this shares one tenant's chunks (text included) with every other tenant, since
3
- Vectorize indexes are account-global. Pass \`namespace\` (the shard/tenant key) on both
4
- index() and retrieve(). Single-tenant apps suppress this via { allowSharedNamespace: true }.`))},Ge=e=>e.map(d=>`[source:${d.sourceId}#${String(d.chunkIndex)}]
5
- ${d.text}`).join(`
6
-
7
- `),ue=(e,d)=>{if(typeof e=="object")return e;if(d===void 0)throw new b("INTERNAL","@lunora/ai/rag: `embeddingModel` is a Workers AI model id (or omitted) but the bound context has no `ai` (env.AI). Pass an AI SDK EmbeddingModel object to embed without Workers AI, or bind a context whose `ctx.ai` is wired.");return d.embeddingModel(e)},z=e=>{const d=e.modelId;return typeof d=="string"&&d.length>0?d:void 0},et=e=>{if(!(typeof e!="object"||e===null)){for(const d of Object.values(e))if(typeof d=="object"&&d!==null){const{cost:h}=d;if(typeof h=="number"&&Number.isFinite(h))return h}}},tt=e=>{if(e===void 0)throw new b("INTERNAL","@lunora/ai/rag: the bound context has no `vectors` (env.VECTORIZE) and no `store` is configured — bind a context whose `ctx.vectors` is wired, or configure `store` (e.g. `sqliteVectorStore`) to back this index without Vectorize.");return e},mt=e=>{if(typeof e.index!="string"||e.index.length===0)throw new b("BAD_REQUEST","@lunora/ai/rag: `index` must be a non-empty Vectorize index name");const d=e.chunkSize??Qe,h=e.chunkOverlap??Ve;if(!Number.isInteger(d)||d<1)throw new b("BAD_REQUEST","@lunora/ai/rag: `chunkSize` must be a positive integer");if(!Number.isInteger(h)||h<0||h>=d)throw new b("BAD_REQUEST","@lunora/ai/rag: `chunkOverlap` must be a non-negative integer smaller than `chunkSize`");const w=ie-je;if(!e.chunk&&!e.textStore&&!e.store&&d>w)throw new b("BAD_REQUEST",`@lunora/ai/rag: \`chunkSize\` of ${String(d)} leaves no room under Vectorize's ${String(ie)}-byte metadata limit, which also carries this chunk's bookkeeping and any \`metadata\` you attach — keep it under ${String(w)}, or supply \`textStore\` to move chunk text out of metadata entirely`);const v=e.topK??Fe;if(!Number.isInteger(v)||v<1)throw new b("BAD_REQUEST","@lunora/ai/rag: `topK` must be a positive integer");if(e.maxEmbeddingDimensions!==void 0&&e.maxEmbeddingDimensions!==!1&&(!Number.isInteger(e.maxEmbeddingDimensions)||e.maxEmbeddingDimensions<1))throw new b("BAD_REQUEST","@lunora/ai/rag: `maxEmbeddingDimensions` must be a positive integer, or `false` to disable the check");if(e.embeddingModelVersion!==void 0&&!He.test(e.embeddingModelVersion))throw new b("BAD_REQUEST",'@lunora/ai/rag: `embeddingModelVersion` must match /^[A-Za-z0-9._-]{1,40}$/ (a short, stable tag like "bge-v1.5")');if(e.candidates!==void 0&&(!Number.isInteger(e.candidates)||e.candidates<1))throw new b("BAD_REQUEST","@lunora/ai/rag: `candidates` must be a positive integer");if(e.cacheEmbeddings!==void 0&&(!Number.isInteger(e.cacheEmbeddings)||e.cacheEmbeddings<0))throw new b("BAD_REQUEST","@lunora/ai/rag: `cacheEmbeddings` must be a non-negative integer");const A=e.cacheEmbeddings??0,K=e.rerank,fe=e.chunk??(p=>Ce(p,d,h)),{textStore:k}=e,B=e.embeddingModelVersion,Q=p=>B===void 0?p:p===void 0?B:`${B}::${p}`;return p=>{const E=e.store?e.store(p):Ue(tt(p.vectors),e.index),H=k?E.capabilities.maxTopK:E.capabilities.maxTopKWithMetadata,U=e.maxEmbeddingDimensions??E.capabilities.maxDimensions;let O;const q=typeof p.trace=="function"?p.trace:void 0;let W=U===!1;const Z=(t,r)=>{if(W||(W=!0,U===!1||t<=U))return;const s=z(r);throw new b("BAD_REQUEST",`@lunora/ai/rag: embedding model${s===void 0?"":` "${s}"`} produces ${String(t)}-dimension vectors, over the ${String(U)}-dimension ceiling of index "${e.index}" — either truncate them with the provider's \`dimensions\` option (Matryoshka models such as text-embedding-3-large support this), or set \`maxEmbeddingDimensions: false\` if this index is not Vectorize-backed`)},N=new Map,ge=(t,r)=>{if(A!==0)for(N.set(t,r);N.size>A;){const s=N.keys().next();if(s.done===!0)break;N.delete(s.value)}},X=async t=>{const r=N.get(t);if(r!==void 0)return r;O??=ue(e.embeddingModel,p.ai);const s=O,a=async n=>{const{embedding:m,providerMetadata:y,usage:l}=await Ne({model:s,value:t});if(Z(m.length,s),n!==void 0){const c=l.tokens;typeof c=="number"&&Number.isFinite(c)&&n.setAttribute("gen_ai.usage.input_tokens",c);const u=et(y),f=u??Re(z(s),{inputTokens:typeof c=="number"?c:void 0});f!==void 0&&(n.setAttribute("gen_ai.usage.cost",f),n.setAttribute("lunora.usage.cost.source",u===void 0?"estimated":"provider"))}return ge(t,m),m};if(q===void 0)return a();const o=z(s),i=typeof p.conversationId=="string"&&p.conversationId.length>0?p.conversationId:void 0;return q("ai.embed",(n,m)=>a(m),{"gen_ai.operation.name":"embeddings",...o===void 0?{}:{"gen_ai.request.model":o},...i===void 0?{}:{"gen_ai.conversation.id":i}})},be=async(t,r,s)=>{if(!e.transformQuery||r?.transformQuery===!1)return[t];const a=typeof p.conversationId=="string"&&p.conversationId.length>0?p.conversationId:void 0,o=await e.transformQuery(t,{conversationId:a,namespace:s}),i=(typeof o=="string"?[o]:[...o]).map(n=>n.trim()).filter(n=>n.length>0);return i.length>0?i:[t]},we=async t=>{const r=new Map,s=[...new Set(t.filter(a=>!N.has(a)))];if(s.length<2)return r;O??=ue(e.embeddingModel,p.ai);try{const{embeddings:a}=await Me({model:O,values:s});if(a.length!==s.length)return r;const[o]=a;o!==void 0&&Z(o.length,O);for(const[i,n]of s.entries())r.set(n,a[i])}catch(a){if(_e(a))throw a}return r},V=t=>{if(t===void 0){if(e.requireNamespace)throw new b("BAD_REQUEST",`@lunora/ai/rag: index "${e.index}" requires a namespace (requireNamespace is set) — pass the tenant/shard key on index()/retrieve()/remove()`);e.allowSharedNamespace||Je(e.index)}},J=async(t,r)=>{const[s]=await E.getByIds([R(r,t,0)],r),a=s?.metadata?.[P],o=s?.metadata?.[Y];return{chunks:typeof o=="number"&&Number.isInteger(o)&&o>0?o:void 0,hash:typeof a=="string"?a:void 0}},G=async(t,r,s,a)=>{const o=Array.from({length:s-r},(i,n)=>R(a,t,r+n));o.length!==0&&(await E.deleteByIds(o,a),await k?.remove?.(o,{namespace:a}),await e.lexicalStore?.remove?.(o,{namespace:a}))},ve=async t=>{if(V(t.namespace),t.importance!==void 0&&(typeof t.importance!="number"||t.importance<0||t.importance>1))throw new b("BAD_REQUEST","@lunora/ai/rag: `importance` must be a number in [0, 1]");const r=Q(t.namespace),s=We(t),a=await qe(s??t.text),o=await J(t.id,r);if(t.reindex!==!0&&s!==void 0&&o.hash===a&&o.chunks!==void 0)return{chunks:o.chunks,ids:Array.from({length:o.chunks},(c,u)=>R(r,t.id,u)),unchanged:!0};const i=fe(t.text),n=i.map((c,u)=>R(r,t.id,u)),m=n.at(-1);if(m!==void 0&&Ye(m,t.id,E.capabilities.maxIdBytes),i.length===0&&t.allowEmptySources===!1)throw new b("BAD_REQUEST",`@lunora/ai/rag: source "${t.id}" produced zero chunks — set allowEmptySources: true to allow this`);if(i.length>0){const c=i.map((u,f)=>({chunkIndex:f,id:n[f],sourceId:t.id,text:u,...t.metadata===void 0?{}:{metadata:t.metadata}}));k&&await k.put(c,{namespace:r}),e.lexicalStore&&await e.lexicalStore.index(c,{namespace:r})}const y=await we(i),l=async c=>y.get(c)??await X(c);return await Be(i,Oe,async(c,u)=>{const f=n[u],x={...t.metadata,[le]:u,[me]:t.id};k||(x[C]=c),t.importance!==void 0&&(x[$]=t.importance),u===0&&(x[P]=a,x[Y]=i.length,B!==void 0&&(x[he]=B)),Pe(x,u,t.id,E.capabilities.maxMetadataBytes),await E.upsert({embed:l,id:f,input:c,metadata:x,namespace:r}),t.onChunk?.({chunkIndex:u,id:f,text:c,total:i.length})}),o.chunks!==void 0&&o.chunks>i.length&&await G(t.id,i.length,o.chunks,r),{chunks:i.length,ids:n,unchanged:!1}},ye=async t=>{V(t.namespace);const r=Q(t.namespace),a=(await J(t.id,r)).chunks??1;await G(t.id,0,a,r)},ee=async(t,r)=>{const s=new Map;if(t.length===0)return s;if(k){const o=await k.getMany(t,{namespace:r});for(const[i,n]of t.entries()){const m=o[i];typeof m=="string"&&s.set(n,m)}return s}const a=await E.getByIds(t,r);for(const o of a){const i=o.metadata?.[C];typeof i=="string"&&s.set(o.id,i)}return s},Ee=async(t,r,s)=>{const a=r?.chunkContext?.before??0,o=r?.chunkContext?.after??0;if(a===0&&o===0)return t;if(!Number.isInteger(a)||a<0||!Number.isInteger(o)||o<0)throw new b("BAD_REQUEST","@lunora/ai/rag: `chunkContext.before`/`chunkContext.after` must be non-negative integers");const i=new Map(t.map(l=>[l.id,l.text])),n=new Set;for(const l of t)for(let c=-a;c<=o;c+=1){const u=l.chunkIndex+c,f=R(s,l.sourceId,u);c!==0&&u>=0&&!i.has(f)&&n.add(f)}const m=await ee([...n],s),y=(l,c)=>{const u=R(s,l,c);return i.get(u)??m.get(u)};return t.map(l=>{const c=[];for(let u=-a;u<=o;u+=1){const f=u===0?l.text:y(l.sourceId,l.chunkIndex+u);f!==void 0&&c.push(f)}return{...l,text:c.join(`
8
- `)}})},xe=t=>{if(typeof t=="string"){const r=e.filters!==void 0&&Object.hasOwn(e.filters,t)?e.filters[t]:void 0;if(!r)throw new b("NOT_FOUND",`@lunora/ai/rag: unknown named filter "${t}" — must be one of the keys declared in RagConfig.filters`);return r.filter}return t},Ie=(t,r)=>t.matches.map(s=>{const a=s.metadata??{},o=de(s.id,r),i=a[C],n=a[$],m=typeof n=="number"&&n>=0&&n<=1?n:1;return{chunkIndex:o.chunkIndex,id:s.id,importance:m,metadata:j(a),score:s.score*m,sourceId:o.sourceId,text:typeof i=="string"?i:""}}),Se=async(t,r)=>{if(!k)return t;const s=t.map(n=>n.id),[a,o]=await Promise.all([ee(s,r),E.getByIds(s,r)]),i=new Map(o.map(n=>[n.id,n.metadata]));return t.flatMap(n=>{const m=a.get(n.id);if(m===void 0)return[];const y=i.get(n.id),l=y?.[$],c=typeof l=="number"&&l>=0&&l<=1?l:n.importance,f=(n.importance===0?0:n.score/n.importance)*c;return[{...n,importance:c,metadata:j(y)??n.metadata,score:f,text:m}]})},te=async(t,r,s)=>{const a=t.map(n=>n.id).filter(n=>!r.has(n)),o=a.length===0?[]:await E.getByIds(a,s),i=new Map(o.map(n=>[n.id,n.metadata]));return t.map(n=>{const m=de(n.id,s),y=i.get(n.id),l=y?.[$];return{chunkIndex:m.chunkIndex,id:n.id,importance:typeof l=="number"&&l>=0&&l<=1?l:1,metadata:j(y),score:n.score,sourceId:m.sourceId,text:n.text}})},ne=async(t,r)=>{V(r?.namespace);const s=Q(r?.namespace),a=xe(r?.filter),o=e.rlsFilter?await e.rlsFilter(p.auth):void 0,i=o?{...a,...o}:a,n=Math.min(r?.topK??v,H),m=await be(t,r,s),y=m[0],l=K!==void 0&&r?.rerank!==!1,u=l||e.lexicalStore!==void 0||e.graphStore!==void 0||m.length>1?Math.min(e.candidates??n*Le,H):n,f=r?.minScore,x=new Set,re=async g=>{const _=await E.query({embed:X,filter:i,input:g,namespace:s,returnMetadata:k?"indexed":"all",topK:u}),M=await Se(Ie(_,s),s);if(f===void 0)return M;const{kept:S,rejectedIds:T}=Ze(M,f);for(const ke of T)x.add(ke);return S},D=[{chunks:await re(y)}];for(const g of m.slice(1))D.push({chunks:await re(g)});if(e.lexicalStore){const g=await e.lexicalStore.search(y,{filter:i,namespace:s,topK:e.lexicalTopK??u}),_=new Set(D.flatMap(S=>S.chunks).map(S=>S.id)),M=g.filter(S=>!x.has(S.id)||_.has(S.id));D.push({chunks:await te(M,_,s)})}const F=D.flatMap(g=>[...g.chunks]);if(e.graphStore&&F.length>0&&(e.graphStore.enforcesFilter||!Xe(i))){const g=[...new Set(F.map(T=>T.sourceId))],_=await e.graphStore.related(g,{filter:i,namespace:s,topK:e.graphTopK??u}),M=new Set(F.map(T=>T.id)),S=_.filter(T=>!x.has(T.id)||M.has(T.id));D.push({chunks:await te(S,M,s),weight:"proximity"})}const L=D.filter(g=>g.chunks.length>0);let I=L.length>1?[...Ke(L)]:[...L[0]?.chunks??[]];I.sort((g,_)=>_.score-g.score),l&&(I=[...await K(y,I)]),I=I.slice(0,n),I=[...await Ee(I,r,s)];const se=[],oe=new Set;for(const g of I)oe.has(g.sourceId)||(oe.add(g.sourceId),se.push({id:g.sourceId,metadata:g.metadata,weight:g.importance}));return r?.onRetrieve?.({matches:I.length,query:t}),{chunks:I,context:Ge(I),sources:se}};return{asTool:t=>Te({description:t?.description??`Search the "${e.index}" knowledge base for passages relevant to a natural-language query.`,execute:async({query:r})=>ne(r,{namespace:t?.namespace,topK:t?.topK}),inputSchema:Ae({properties:{query:{description:"The natural-language search query.",type:"string"}},required:["query"],type:"object"})}),index:ve,remove:ye,retrieve:ne}}};export{mt as default};