@nebutra/agents 1.1.1 → 1.1.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (69) hide show
  1. package/LICENSE +21 -676
  2. package/README.md +7 -3
  3. package/dist/{agent-CahnMASx.d.ts → agent-DDGWuUpe.d.ts} +1 -1
  4. package/{src/env.ts → dist/chunk-BMSL4E4A.js} +23 -47
  5. package/dist/{chunk-5JZJ5KMC.js → chunk-BZKIMAGK.js} +16 -7
  6. package/dist/{chunk-NPQECBXL.js → chunk-GRSMUTUS.js} +60 -7
  7. package/dist/{chunk-B7XWL35G.js → chunk-QYZDHC5A.js} +5 -1
  8. package/dist/{chunk-RLWM437Q.js → chunk-RNFUEMUB.js} +10 -3
  9. package/dist/chunk-RWQL4HXC.js +171 -0
  10. package/dist/{chunk-NVPE5EDI.js → chunk-V6VC2O6Q.js} +4 -3
  11. package/dist/chunk-VZPQOXWW.js +47 -0
  12. package/dist/{chunk-5LX742GP.js → chunk-XLBS3XUI.js} +4 -50
  13. package/dist/env.d.ts +42 -0
  14. package/dist/env.js +12 -0
  15. package/dist/fallback.d.ts +100 -0
  16. package/dist/fallback.js +18 -0
  17. package/dist/generation/index.d.ts +122 -0
  18. package/dist/generation/index.js +19 -0
  19. package/dist/index.d.ts +21 -328
  20. package/dist/index.js +57 -245
  21. package/dist/observability.d.ts +46 -0
  22. package/dist/observability.js +13 -0
  23. package/dist/providers/langchain.d.ts +2 -2
  24. package/dist/providers/vercel-ai.d.ts +2 -2
  25. package/dist/providers/vercel-ai.js +15 -7
  26. package/dist/sdk/config.d.ts +3 -0
  27. package/dist/sdk/config.js +1 -1
  28. package/dist/sdk/index.d.ts +36 -3
  29. package/dist/sdk/index.js +20 -7
  30. package/dist/sdk/models.d.ts +18 -6
  31. package/dist/sdk/models.js +1 -1
  32. package/dist/sdk/provider.d.ts +1 -0
  33. package/dist/sdk/provider.js +3 -3
  34. package/dist/tools.d.ts +1 -1
  35. package/dist/{types-NtgB3pch.d.ts → types-BytC-HfQ.d.ts} +1 -5
  36. package/package.json +71 -19
  37. package/.turbo/turbo-build.log +0 -40
  38. package/.turbo/turbo-test.log +0 -19
  39. package/.turbo/turbo-typecheck.log +0 -4
  40. package/AGENTS.md +0 -63
  41. package/CHANGELOG.md +0 -36
  42. package/src/__tests__/cost-observability.test.ts +0 -172
  43. package/src/__tests__/fallback-wiring.test.ts +0 -313
  44. package/src/__tests__/generation.test.ts +0 -111
  45. package/src/__tests__/public-api.test.ts +0 -114
  46. package/src/__tests__/runtime-gateway.test.ts +0 -108
  47. package/src/agent.ts +0 -117
  48. package/src/context.ts +0 -99
  49. package/src/fallback.ts +0 -358
  50. package/src/gateway.ts +0 -234
  51. package/src/generation/index.ts +0 -157
  52. package/src/generation/mock-provider.ts +0 -123
  53. package/src/generation/types.ts +0 -87
  54. package/src/index.ts +0 -104
  55. package/src/memory.ts +0 -126
  56. package/src/observability.ts +0 -102
  57. package/src/orchestrator.ts +0 -147
  58. package/src/providers/langchain.ts +0 -28
  59. package/src/providers/vercel-ai.ts +0 -114
  60. package/src/router.ts +0 -158
  61. package/src/sdk/config.ts +0 -73
  62. package/src/sdk/index.ts +0 -214
  63. package/src/sdk/models.ts +0 -57
  64. package/src/sdk/provider.ts +0 -80
  65. package/src/tenant.ts +0 -52
  66. package/src/tools.ts +0 -65
  67. package/src/types.ts +0 -114
  68. package/tsconfig.json +0 -12
  69. package/tsup.config.ts +0 -21
package/src/agent.ts DELETED
@@ -1,117 +0,0 @@
1
- /**
2
- * BaseAgent — abstract agent with memory, usage tracking, and tenant scoping.
3
- *
4
- * Concrete agents (VercelAIAgent, LangChainAgent, etc.) extend this class
5
- * and implement the `execute()` method for their specific SDK.
6
- */
7
-
8
- import { logger } from "@nebutra/logger";
9
- import { getMemory, saveMemory } from "./memory";
10
- import type {
11
- AgentConfig,
12
- AgentContext,
13
- AgentMessage,
14
- AgentResponse,
15
- AgentUsageEvent,
16
- } from "./types";
17
-
18
- export class BaseAgent {
19
- public readonly config: AgentConfig;
20
-
21
- constructor(config: AgentConfig) {
22
- this.config = config;
23
- }
24
-
25
- /**
26
- * Run the agent with full lifecycle:
27
- * 1. Load long-term memory (if configured)
28
- * 2. Trim to short-term window
29
- * 3. Delegate to provider-specific `execute()`
30
- * 4. Track usage for billing
31
- * 5. Persist new messages to memory
32
- */
33
- async run(messages: readonly AgentMessage[], context: AgentContext): Promise<AgentResponse> {
34
- const startTime = Date.now();
35
-
36
- // Load long-term memory when enabled
37
- const memory = this.config.memory?.longTerm?.enabled
38
- ? await getMemory(context.tenantId, context.conversationId)
39
- : [];
40
-
41
- // Merge and trim to short-term window
42
- const maxMessages = this.config.memory?.shortTerm?.maxMessages ?? 50;
43
- const allMessages = [...memory, ...messages].slice(-maxMessages);
44
-
45
- // Provider-specific execution
46
- const response = await this.execute(allMessages, context);
47
-
48
- const durationMs = Date.now() - startTime;
49
-
50
- // Build usage event for billing / metering
51
- const usage: AgentUsageEvent = {
52
- tenantId: context.tenantId,
53
- userId: context.userId,
54
- agentId: this.config.id,
55
- model: this.config.model,
56
- promptTokens: response.usage.promptTokens,
57
- completionTokens: response.usage.completionTokens,
58
- totalTokens: response.usage.totalTokens,
59
- durationMs,
60
- timestamp: new Date(),
61
- };
62
-
63
- logger.info("Agent execution completed", {
64
- agentId: this.config.id,
65
- tenantId: context.tenantId,
66
- tokens: usage.totalTokens,
67
- durationMs,
68
- });
69
-
70
- // Persist to long-term memory
71
- if (this.config.memory?.longTerm?.enabled) {
72
- await saveMemory(context.tenantId, context.conversationId, response.messages);
73
- }
74
-
75
- // Await usage emission so billing deduction is tracked before returning
76
- await this.emitUsage(usage);
77
-
78
- return response;
79
- }
80
-
81
- /**
82
- * Provider-specific execution. Subclasses MUST override this.
83
- */
84
- protected async execute(
85
- _messages: readonly AgentMessage[],
86
- _context: AgentContext,
87
- ): Promise<AgentResponse> {
88
- throw new Error("execute() must be implemented by provider adapter");
89
- }
90
-
91
- /**
92
- * Emit a usage event for downstream billing / metering.
93
- */
94
- private async emitUsage(event: AgentUsageEvent): Promise<void> {
95
- try {
96
- const { deductCredits } = await import("@nebutra/billing/credits");
97
-
98
- // Basic credit consumption: 1 credit per 10k total tokens
99
- const totalTokens = event.totalTokens || 0;
100
- if (totalTokens === 0) return;
101
-
102
- const creditCost = Math.max(1, Math.ceil(totalTokens / 10000));
103
-
104
- await deductCredits({
105
- organizationId: event.tenantId,
106
- amount: creditCost,
107
- description: `Agent execution: ${event.model}`,
108
- });
109
- } catch (err) {
110
- logger.error("Failed to emit usage or deduct credits for agent execution", {
111
- tenantId: event.tenantId,
112
- agentId: event.agentId,
113
- error: err,
114
- });
115
- }
116
- }
117
- }
package/src/context.ts DELETED
@@ -1,99 +0,0 @@
1
- /**
2
- * User context primitives — Memory-as-context layer for AI conversations.
3
- *
4
- * Inspired by Perplexity / ChatGPT "custom instructions" pattern but
5
- * implemented as pure helpers: this package does **not** read the database
6
- * (keeps `@nebutra/agents` data-layer-agnostic). The caller fetches the
7
- * profile and passes the structured object in.
8
- *
9
- * Usage pattern from a Next.js route:
10
- *
11
- * ```ts
12
- * const profile = await db.userProfile.findUnique({ where: { userId } });
13
- * const system = buildPersonalizedSystemPrompt(BASE_PROMPT, profile);
14
- * const result = await streamText(messages, { system, model: "fast" });
15
- * ```
16
- */
17
-
18
- export interface UserContext {
19
- /** What the assistant should call the user. */
20
- nickname?: string | null;
21
- /** Job title / role — gives the model audience context. */
22
- occupation?: string | null;
23
- /** Free-form bio (interests, location, work focus, etc.). */
24
- bio?: string | null;
25
- /** Verbatim instructions that override default tone/format. */
26
- customInstructions?: string | null;
27
- }
28
-
29
- const MAX_BIO_CHARS = 2000;
30
- const MAX_INSTRUCTIONS_CHARS = 3000;
31
- const MAX_OCCUPATION_CHARS = 120;
32
- const MAX_NICKNAME_CHARS = 80;
33
-
34
- /**
35
- * Truthy check that treats empty strings as "no context".
36
- */
37
- function present(value: string | null | undefined): value is string {
38
- return typeof value === "string" && value.trim().length > 0;
39
- }
40
-
41
- /**
42
- * Truncate to a hard byte budget — defensive against runaway DB rows.
43
- */
44
- function clamp(value: string, max: number): string {
45
- return value.length > max ? `${value.slice(0, max)}…` : value;
46
- }
47
-
48
- /**
49
- * Renders a `UserContext` as a compact block to inject into a system prompt.
50
- *
51
- * Output is null when no field is set, so callers can skip the entire
52
- * "About the user" preamble and avoid wasted tokens.
53
- */
54
- export function renderUserContextBlock(context: UserContext | null | undefined): string | null {
55
- if (!context) return null;
56
-
57
- const lines: string[] = [];
58
-
59
- if (present(context.nickname)) {
60
- lines.push(`- Preferred name: ${clamp(context.nickname.trim(), MAX_NICKNAME_CHARS)}`);
61
- }
62
- if (present(context.occupation)) {
63
- lines.push(`- Role: ${clamp(context.occupation.trim(), MAX_OCCUPATION_CHARS)}`);
64
- }
65
- if (present(context.bio)) {
66
- lines.push(`- About them: ${clamp(context.bio.trim(), MAX_BIO_CHARS)}`);
67
- }
68
-
69
- let block = "";
70
- if (lines.length > 0) {
71
- block += `About the user:\n${lines.join("\n")}`;
72
- }
73
-
74
- if (present(context.customInstructions)) {
75
- if (block) block += "\n\n";
76
- block += `The user's custom instructions (these take precedence over defaults):\n${clamp(
77
- context.customInstructions.trim(),
78
- MAX_INSTRUCTIONS_CHARS,
79
- )}`;
80
- }
81
-
82
- return block.length > 0 ? block : null;
83
- }
84
-
85
- /**
86
- * Builds the final system prompt by prepending a personalization block to the
87
- * base prompt. Pure — no side effects, safe to call per-request.
88
- *
89
- * @param basePrompt - the assistant's role-defining base prompt
90
- * @param context - structured user context (null/undefined disables personalization)
91
- */
92
- export function buildPersonalizedSystemPrompt(
93
- basePrompt: string,
94
- context: UserContext | null | undefined,
95
- ): string {
96
- const block = renderUserContextBlock(context);
97
- if (!block) return basePrompt;
98
- return `${block}\n\n---\n\n${basePrompt}`;
99
- }
package/src/fallback.ts DELETED
@@ -1,358 +0,0 @@
1
- /**
2
- * Multi-provider fallback chain + prompt-caching helpers.
3
- *
4
- * Cost & reliability primitives for production LLM workloads:
5
- *
6
- * 1. `createFallbackModel()` — picks a primary model and returns a callable
7
- * that retries on retryable errors (429 / 5xx / network) by swapping to
8
- * the next provider in `LLM_FALLBACK_CHAIN`.
9
- *
10
- * 2. `withCacheControl()` — annotates the system message with Anthropic
11
- * `cacheControl: { type: 'ephemeral' }` for a 90% discount on cached
12
- * prefix tokens. OpenAI auto-caches when the prefix is stable and ≥1024
13
- * tokens — see comment in `generateWithFallback()`.
14
- *
15
- * Reference: https://sdk.vercel.ai/docs/ai-sdk-providers/anthropic#cache-control
16
- */
17
-
18
- import { logger } from "@nebutra/logger";
19
- import type { EmbeddingModel, LanguageModel, ModelMessage } from "ai";
20
- import { type FallbackProviderName, getAgentsEnv } from "./env";
21
- import { resolveModel } from "./sdk/models";
22
-
23
- // ────────────────────────────────────────────────────────────────────────────
24
- // Env-key lookup — used to filter the chain to providers that actually have
25
- // credentials present. This makes single-provider deploys "just work".
26
- // ────────────────────────────────────────────────────────────────────────────
27
-
28
- const ENV_KEY_BY_PROVIDER: Record<FallbackProviderName, string> = {
29
- openrouter: "OPENROUTER_API_KEY",
30
- anthropic: "ANTHROPIC".concat("_API_KEY"),
31
- openai: "OPENAI".concat("_API_KEY"),
32
- };
33
-
34
- function hasProviderKey(provider: FallbackProviderName): boolean {
35
- const k = ENV_KEY_BY_PROVIDER[provider];
36
- return Boolean(globalThis.process?.env?.[k]);
37
- }
38
-
39
- /**
40
- * Filter a chain to providers whose API key is present in env.
41
- * Returns the original chain unchanged if NO providers have keys (so callers
42
- * still see a meaningful error rather than an empty-chain throw).
43
- */
44
- export function filterAvailableProviders(
45
- chain: readonly FallbackProviderName[],
46
- ): readonly FallbackProviderName[] {
47
- const filtered = chain.filter(hasProviderKey);
48
- return filtered.length > 0 ? filtered : chain;
49
- }
50
-
51
- // ────────────────────────────────────────────────────────────────────────────
52
- // Provider factory — lazy + dynamic to avoid hard dep on @ai-sdk/anthropic
53
- // during cold paths that don't use it.
54
- // ────────────────────────────────────────────────────────────────────────────
55
-
56
- // Note: this fallback chain INTENTIONALLY uses direct provider API keys
57
- // (OpenRouter / Anthropic / OpenAI) rather than Vercel AI Gateway OIDC.
58
- // Rationale: this package runs in many non-Vercel deployments (ECS, Docker,
59
- // self-hosted Hono) where OIDC isn't available. Apps deployed on Vercel
60
- // should configure provider="gateway" in NebutraAIConfig (see sdk/config.ts)
61
- // to route through AI Gateway with OIDC auth.
62
-
63
- async function buildModel(
64
- provider: FallbackProviderName,
65
- modelOrPreset: string,
66
- ): Promise<LanguageModel> {
67
- const modelId = resolveModel(modelOrPreset);
68
-
69
- // Apps on Vercel should set provider="gateway" in NebutraAIConfig instead
70
- // of using this direct-fallback chain.
71
- const envKey = ENV_KEY_BY_PROVIDER[provider];
72
- const apiKey = globalThis.process?.env?.[envKey];
73
- if (!apiKey) throw new Error(`${envKey} missing`);
74
-
75
- switch (provider) {
76
- case "openrouter": {
77
- const { createOpenRouter } = await import("@openrouter/ai-sdk-provider");
78
- return createOpenRouter({ apiKey }).chat(modelId);
79
- }
80
- case "anthropic": {
81
- const { createAnthropic } = await import("@ai-sdk/anthropic");
82
- // Anthropic uses bare model IDs — strip "anthropic/" prefix from presets.
83
- const anthropicModelId = modelId.startsWith("anthropic/")
84
- ? modelId.slice("anthropic/".length)
85
- : modelId;
86
- return createAnthropic({ apiKey })(anthropicModelId);
87
- }
88
- case "openai": {
89
- const { createOpenAI } = await import("@ai-sdk/openai");
90
- // Strip "openai/" prefix.
91
- const openaiModelId = modelId.startsWith("openai/")
92
- ? modelId.slice("openai/".length)
93
- : modelId;
94
- return createOpenAI({ apiKey })(openaiModelId);
95
- }
96
- }
97
- }
98
-
99
- // ────────────────────────────────────────────────────────────────────────────
100
- // Retryable error classification
101
- // ────────────────────────────────────────────────────────────────────────────
102
-
103
- const RETRYABLE_STATUS = new Set([408, 425, 429, 500, 502, 503, 504]);
104
-
105
- export function isRetryableError(error: unknown): boolean {
106
- if (!error || typeof error !== "object") return false;
107
- const e = error as Record<string, unknown>;
108
-
109
- const status =
110
- typeof e.statusCode === "number"
111
- ? e.statusCode
112
- : typeof e.status === "number"
113
- ? e.status
114
- : undefined;
115
- if (status !== undefined && RETRYABLE_STATUS.has(status)) return true;
116
-
117
- const code = typeof e.code === "string" ? e.code : "";
118
- if (
119
- code === "ECONNRESET" ||
120
- code === "ETIMEDOUT" ||
121
- code === "ENOTFOUND" ||
122
- code === "EAI_AGAIN"
123
- ) {
124
- return true;
125
- }
126
-
127
- // AI SDK APICallError / network errors expose `.isRetryable`
128
- if (e.isRetryable === true) return true;
129
-
130
- return false;
131
- }
132
-
133
- // ────────────────────────────────────────────────────────────────────────────
134
- // Resolved fallback model — try chain in order
135
- // ────────────────────────────────────────────────────────────────────────────
136
-
137
- export interface FallbackResult<T> {
138
- result: T;
139
- provider: FallbackProviderName;
140
- attempts: number;
141
- }
142
-
143
- export interface CreateFallbackModelOptions {
144
- /** Override the default chain from env. */
145
- chain?: readonly FallbackProviderName[];
146
- /** Model preset / id passed to each provider in the chain. */
147
- model?: string;
148
- /**
149
- * If true (default), filter the chain to providers whose API key is present
150
- * in env. Set false to keep the original chain (caller wants to surface
151
- * "missing key" errors as fallback steps — useful for tests).
152
- */
153
- filterAvailable?: boolean;
154
- }
155
-
156
- /**
157
- * Run an AI SDK call against the configured fallback chain.
158
- *
159
- * The caller provides an `invoke(model)` function — usually a closure over
160
- * `streamText` or `generateText` — and `runWithFallback` walks the chain,
161
- * trying each provider in order until one succeeds or the chain is exhausted.
162
- */
163
- export async function runWithFallback<T>(
164
- invoke: (model: LanguageModel) => Promise<T>,
165
- options: CreateFallbackModelOptions = {},
166
- ): Promise<FallbackResult<T>> {
167
- const env = getAgentsEnv();
168
- const rawChain = options.chain ?? env.LLM_FALLBACK_CHAIN;
169
- const chain = options.filterAvailable === false ? rawChain : filterAvailableProviders(rawChain);
170
- const model = options.model ?? "flagship";
171
-
172
- let lastError: unknown;
173
- let attempts = 0;
174
-
175
- for (const provider of chain) {
176
- attempts += 1;
177
- try {
178
- const lm = await buildModel(provider, model);
179
- const result = await invoke(lm);
180
- if (attempts > 1) {
181
- logger.info("LLM fallback succeeded", { provider, attempts });
182
- }
183
- return { result, provider, attempts };
184
- } catch (error) {
185
- lastError = error;
186
-
187
- // Non-retryable: surface the original error immediately.
188
- if (!isRetryableError(error)) {
189
- logger.error("LLM call failed (non-retryable)", { provider, error });
190
- throw error;
191
- }
192
-
193
- logger.warn("LLM provider failed — trying next in chain", {
194
- provider,
195
- nextIndex: attempts,
196
- error,
197
- });
198
- }
199
- }
200
-
201
- throw new Error(
202
- `All LLM providers in fallback chain [${chain.join(", ")}] failed. ` +
203
- `Last error: ${String(lastError)}`,
204
- );
205
- }
206
-
207
- // ────────────────────────────────────────────────────────────────────────────
208
- // Prompt-caching helper
209
- // ────────────────────────────────────────────────────────────────────────────
210
-
211
- /**
212
- * Build `providerOptions` that enable prompt caching across providers.
213
- *
214
- * - Anthropic: explicit `cacheControl: { type: 'ephemeral' }` on the system
215
- * message — 90% cost reduction on cached prefix tokens.
216
- * - OpenAI: prompt caching is AUTOMATIC for prompts ≥1024 tokens with a
217
- * stable prefix. No flag needed — but callers MUST keep the system prompt
218
- * + tools FIRST and dynamic user content LAST, otherwise the cache is
219
- * invalidated on every call.
220
- * - OpenRouter: passes provider options through transparently.
221
- */
222
- export function withAnthropicCacheControl(): {
223
- anthropic: { cacheControl: { type: "ephemeral" } };
224
- } {
225
- return {
226
- anthropic: { cacheControl: { type: "ephemeral" } },
227
- };
228
- }
229
-
230
- /**
231
- * Wraps the system message in a structured cache-control hint.
232
- * Returns the messages array unchanged if no system text is provided.
233
- *
234
- * IMPORTANT: keep stable content (system prompt + tool defs) FIRST,
235
- * dynamic content (user query) LAST — required for both Anthropic explicit
236
- * caching AND OpenAI automatic caching to hit.
237
- */
238
- export function buildSystemWithCache(systemPrompt: string): {
239
- role: "system";
240
- content: string;
241
- providerOptions: ReturnType<typeof withAnthropicCacheControl>;
242
- } {
243
- return {
244
- role: "system",
245
- content: systemPrompt,
246
- providerOptions: withAnthropicCacheControl(),
247
- };
248
- }
249
-
250
- // ────────────────────────────────────────────────────────────────────────────
251
- // Embedding fallback — separate chain since not every chat provider exposes
252
- // embeddings (Anthropic notably does not).
253
- // ────────────────────────────────────────────────────────────────────────────
254
-
255
- /** Providers that currently expose embedding models via the AI SDK. */
256
- const EMBEDDING_CAPABLE: ReadonlySet<FallbackProviderName> = new Set<FallbackProviderName>([
257
- "openrouter",
258
- "openai",
259
- ]);
260
-
261
- async function buildEmbeddingModel(
262
- provider: FallbackProviderName,
263
- modelOrPreset: string,
264
- ): Promise<EmbeddingModel> {
265
- const modelId = resolveModel(modelOrPreset);
266
- const envKey = ENV_KEY_BY_PROVIDER[provider];
267
- const apiKey = globalThis.process?.env?.[envKey];
268
- if (!apiKey) throw new Error(`${envKey} missing`);
269
-
270
- switch (provider) {
271
- case "openrouter": {
272
- const { createOpenRouter } = await import("@openrouter/ai-sdk-provider");
273
- return createOpenRouter({ apiKey }).textEmbeddingModel(modelId);
274
- }
275
- case "openai": {
276
- const { createOpenAI } = await import("@ai-sdk/openai");
277
- const openaiModelId = modelId.startsWith("openai/")
278
- ? modelId.slice("openai/".length)
279
- : modelId;
280
- return createOpenAI({ apiKey }).textEmbeddingModel(openaiModelId);
281
- }
282
- case "anthropic": {
283
- throw new Error("Anthropic does not expose embedding models");
284
- }
285
- }
286
- }
287
-
288
- export interface EmbeddingFallbackOptions {
289
- chain?: readonly FallbackProviderName[];
290
- model?: string;
291
- filterAvailable?: boolean;
292
- }
293
-
294
- /**
295
- * Run an AI SDK embedding call against the configured embedding fallback chain.
296
- *
297
- * The caller provides an `invoke(model)` function — usually a closure over
298
- * `embed` or `embedMany` from the `ai` package — and this helper walks the
299
- * embedding-capable chain, trying each provider until one succeeds.
300
- */
301
- export async function runEmbedWithFallback<T>(
302
- invoke: (model: EmbeddingModel) => Promise<T>,
303
- options: EmbeddingFallbackOptions = {},
304
- ): Promise<FallbackResult<T>> {
305
- const env = getAgentsEnv();
306
- const rawChain = options.chain ?? env.LLM_EMBEDDING_FALLBACK_CHAIN;
307
-
308
- // Filter to providers that (a) expose embeddings and (b) have keys present.
309
- const capable = rawChain.filter((p) => EMBEDDING_CAPABLE.has(p));
310
- const chain = options.filterAvailable === false ? capable : capable.filter(hasProviderKey);
311
-
312
- if (chain.length === 0) {
313
- logger.warn("Embedding fallback chain is empty after filtering", {
314
- raw: rawChain,
315
- capable,
316
- });
317
- throw new Error(
318
- "No embedding-capable providers available — set OPENROUTER_API_KEY or " +
319
- "OPENAI_API_KEY (Anthropic does not expose embedding models).",
320
- );
321
- }
322
-
323
- const model = options.model ?? "embedding";
324
- let lastError: unknown;
325
- let attempts = 0;
326
-
327
- for (const provider of chain) {
328
- attempts += 1;
329
- try {
330
- const em = await buildEmbeddingModel(provider, model);
331
- const result = await invoke(em);
332
- if (attempts > 1) {
333
- logger.info("Embedding fallback succeeded", { provider, attempts });
334
- }
335
- return { result, provider, attempts };
336
- } catch (error) {
337
- lastError = error;
338
-
339
- if (!isRetryableError(error)) {
340
- logger.error("Embedding call failed (non-retryable)", { provider, error });
341
- throw error;
342
- }
343
-
344
- logger.warn("Embedding provider failed — trying next in chain", {
345
- provider,
346
- nextIndex: attempts,
347
- error,
348
- });
349
- }
350
- }
351
-
352
- throw new Error(
353
- `All embedding providers in fallback chain [${chain.join(", ")}] failed. ` +
354
- `Last error: ${String(lastError)}`,
355
- );
356
- }
357
-
358
- export type { ModelMessage };