@nebutra/agents 1.1.0 → 1.1.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -676
- package/README.md +7 -3
- package/dist/{agent-CahnMASx.d.ts → agent-DDGWuUpe.d.ts} +1 -1
- package/{src/env.ts → dist/chunk-BMSL4E4A.js} +23 -47
- package/dist/{chunk-5JZJ5KMC.js → chunk-BZKIMAGK.js} +16 -7
- package/dist/{chunk-NPQECBXL.js → chunk-GRSMUTUS.js} +60 -7
- package/dist/{chunk-B7XWL35G.js → chunk-QYZDHC5A.js} +5 -1
- package/dist/{chunk-RLWM437Q.js → chunk-RNFUEMUB.js} +10 -3
- package/dist/chunk-RWQL4HXC.js +171 -0
- package/dist/{chunk-NVPE5EDI.js → chunk-V6VC2O6Q.js} +4 -3
- package/dist/chunk-VZPQOXWW.js +47 -0
- package/dist/{chunk-5LX742GP.js → chunk-XLBS3XUI.js} +4 -50
- package/dist/env.d.ts +42 -0
- package/dist/env.js +12 -0
- package/dist/fallback.d.ts +100 -0
- package/dist/fallback.js +18 -0
- package/dist/generation/index.d.ts +122 -0
- package/dist/generation/index.js +19 -0
- package/dist/index.d.ts +21 -328
- package/dist/index.js +57 -245
- package/dist/observability.d.ts +46 -0
- package/dist/observability.js +13 -0
- package/dist/providers/langchain.d.ts +2 -2
- package/dist/providers/vercel-ai.d.ts +2 -2
- package/dist/providers/vercel-ai.js +15 -7
- package/dist/sdk/config.d.ts +3 -0
- package/dist/sdk/config.js +1 -1
- package/dist/sdk/index.d.ts +36 -3
- package/dist/sdk/index.js +20 -7
- package/dist/sdk/models.d.ts +18 -6
- package/dist/sdk/models.js +1 -1
- package/dist/sdk/provider.d.ts +1 -0
- package/dist/sdk/provider.js +3 -3
- package/dist/tools.d.ts +1 -1
- package/dist/{types-NtgB3pch.d.ts → types-BytC-HfQ.d.ts} +1 -5
- package/package.json +73 -27
- package/.turbo/turbo-build.log +0 -40
- package/.turbo/turbo-test.log +0 -18
- package/.turbo/turbo-typecheck.log +0 -4
- package/AGENTS.md +0 -63
- package/CHANGELOG.md +0 -26
- package/src/__tests__/cost-observability.test.ts +0 -172
- package/src/__tests__/fallback-wiring.test.ts +0 -313
- package/src/__tests__/generation.test.ts +0 -111
- package/src/__tests__/public-api.test.ts +0 -114
- package/src/agent.ts +0 -117
- package/src/context.ts +0 -99
- package/src/fallback.ts +0 -358
- package/src/generation/index.ts +0 -157
- package/src/generation/mock-provider.ts +0 -123
- package/src/generation/types.ts +0 -87
- package/src/index.ts +0 -104
- package/src/memory.ts +0 -126
- package/src/observability.ts +0 -102
- package/src/orchestrator.ts +0 -147
- package/src/providers/langchain.ts +0 -28
- package/src/providers/vercel-ai.ts +0 -114
- package/src/router.ts +0 -158
- package/src/sdk/config.ts +0 -73
- package/src/sdk/index.ts +0 -214
- package/src/sdk/models.ts +0 -57
- package/src/sdk/provider.ts +0 -80
- package/src/tenant.ts +0 -52
- package/src/tools.ts +0 -65
- package/src/types.ts +0 -114
- package/tsconfig.json +0 -12
- package/tsup.config.ts +0 -21
package/src/context.ts
DELETED
|
@@ -1,99 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* User context primitives — Memory-as-context layer for AI conversations.
|
|
3
|
-
*
|
|
4
|
-
* Inspired by Perplexity / ChatGPT "custom instructions" pattern but
|
|
5
|
-
* implemented as pure helpers: this package does **not** read the database
|
|
6
|
-
* (keeps `@nebutra/agents` data-layer-agnostic). The caller fetches the
|
|
7
|
-
* profile and passes the structured object in.
|
|
8
|
-
*
|
|
9
|
-
* Usage pattern from a Next.js route:
|
|
10
|
-
*
|
|
11
|
-
* ```ts
|
|
12
|
-
* const profile = await db.userProfile.findUnique({ where: { userId } });
|
|
13
|
-
* const system = buildPersonalizedSystemPrompt(BASE_PROMPT, profile);
|
|
14
|
-
* const result = await streamText(messages, { system, model: "fast" });
|
|
15
|
-
* ```
|
|
16
|
-
*/
|
|
17
|
-
|
|
18
|
-
export interface UserContext {
|
|
19
|
-
/** What the assistant should call the user. */
|
|
20
|
-
nickname?: string | null;
|
|
21
|
-
/** Job title / role — gives the model audience context. */
|
|
22
|
-
occupation?: string | null;
|
|
23
|
-
/** Free-form bio (interests, location, work focus, etc.). */
|
|
24
|
-
bio?: string | null;
|
|
25
|
-
/** Verbatim instructions that override default tone/format. */
|
|
26
|
-
customInstructions?: string | null;
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
const MAX_BIO_CHARS = 2000;
|
|
30
|
-
const MAX_INSTRUCTIONS_CHARS = 3000;
|
|
31
|
-
const MAX_OCCUPATION_CHARS = 120;
|
|
32
|
-
const MAX_NICKNAME_CHARS = 80;
|
|
33
|
-
|
|
34
|
-
/**
|
|
35
|
-
* Truthy check that treats empty strings as "no context".
|
|
36
|
-
*/
|
|
37
|
-
function present(value: string | null | undefined): value is string {
|
|
38
|
-
return typeof value === "string" && value.trim().length > 0;
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
/**
|
|
42
|
-
* Truncate to a hard byte budget — defensive against runaway DB rows.
|
|
43
|
-
*/
|
|
44
|
-
function clamp(value: string, max: number): string {
|
|
45
|
-
return value.length > max ? `${value.slice(0, max)}…` : value;
|
|
46
|
-
}
|
|
47
|
-
|
|
48
|
-
/**
|
|
49
|
-
* Renders a `UserContext` as a compact block to inject into a system prompt.
|
|
50
|
-
*
|
|
51
|
-
* Output is null when no field is set, so callers can skip the entire
|
|
52
|
-
* "About the user" preamble and avoid wasted tokens.
|
|
53
|
-
*/
|
|
54
|
-
export function renderUserContextBlock(context: UserContext | null | undefined): string | null {
|
|
55
|
-
if (!context) return null;
|
|
56
|
-
|
|
57
|
-
const lines: string[] = [];
|
|
58
|
-
|
|
59
|
-
if (present(context.nickname)) {
|
|
60
|
-
lines.push(`- Preferred name: ${clamp(context.nickname.trim(), MAX_NICKNAME_CHARS)}`);
|
|
61
|
-
}
|
|
62
|
-
if (present(context.occupation)) {
|
|
63
|
-
lines.push(`- Role: ${clamp(context.occupation.trim(), MAX_OCCUPATION_CHARS)}`);
|
|
64
|
-
}
|
|
65
|
-
if (present(context.bio)) {
|
|
66
|
-
lines.push(`- About them: ${clamp(context.bio.trim(), MAX_BIO_CHARS)}`);
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
let block = "";
|
|
70
|
-
if (lines.length > 0) {
|
|
71
|
-
block += `About the user:\n${lines.join("\n")}`;
|
|
72
|
-
}
|
|
73
|
-
|
|
74
|
-
if (present(context.customInstructions)) {
|
|
75
|
-
if (block) block += "\n\n";
|
|
76
|
-
block += `The user's custom instructions (these take precedence over defaults):\n${clamp(
|
|
77
|
-
context.customInstructions.trim(),
|
|
78
|
-
MAX_INSTRUCTIONS_CHARS,
|
|
79
|
-
)}`;
|
|
80
|
-
}
|
|
81
|
-
|
|
82
|
-
return block.length > 0 ? block : null;
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
/**
|
|
86
|
-
* Builds the final system prompt by prepending a personalization block to the
|
|
87
|
-
* base prompt. Pure — no side effects, safe to call per-request.
|
|
88
|
-
*
|
|
89
|
-
* @param basePrompt - the assistant's role-defining base prompt
|
|
90
|
-
* @param context - structured user context (null/undefined disables personalization)
|
|
91
|
-
*/
|
|
92
|
-
export function buildPersonalizedSystemPrompt(
|
|
93
|
-
basePrompt: string,
|
|
94
|
-
context: UserContext | null | undefined,
|
|
95
|
-
): string {
|
|
96
|
-
const block = renderUserContextBlock(context);
|
|
97
|
-
if (!block) return basePrompt;
|
|
98
|
-
return `${block}\n\n---\n\n${basePrompt}`;
|
|
99
|
-
}
|
package/src/fallback.ts
DELETED
|
@@ -1,358 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Multi-provider fallback chain + prompt-caching helpers.
|
|
3
|
-
*
|
|
4
|
-
* Cost & reliability primitives for production LLM workloads:
|
|
5
|
-
*
|
|
6
|
-
* 1. `createFallbackModel()` — picks a primary model and returns a callable
|
|
7
|
-
* that retries on retryable errors (429 / 5xx / network) by swapping to
|
|
8
|
-
* the next provider in `LLM_FALLBACK_CHAIN`.
|
|
9
|
-
*
|
|
10
|
-
* 2. `withCacheControl()` — annotates the system message with Anthropic
|
|
11
|
-
* `cacheControl: { type: 'ephemeral' }` for a 90% discount on cached
|
|
12
|
-
* prefix tokens. OpenAI auto-caches when the prefix is stable and ≥1024
|
|
13
|
-
* tokens — see comment in `generateWithFallback()`.
|
|
14
|
-
*
|
|
15
|
-
* Reference: https://sdk.vercel.ai/docs/ai-sdk-providers/anthropic#cache-control
|
|
16
|
-
*/
|
|
17
|
-
|
|
18
|
-
import { logger } from "@nebutra/logger";
|
|
19
|
-
import type { EmbeddingModel, LanguageModel, ModelMessage } from "ai";
|
|
20
|
-
import { type FallbackProviderName, getAgentsEnv } from "./env";
|
|
21
|
-
import { resolveModel } from "./sdk/models";
|
|
22
|
-
|
|
23
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
24
|
-
// Env-key lookup — used to filter the chain to providers that actually have
|
|
25
|
-
// credentials present. This makes single-provider deploys "just work".
|
|
26
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
27
|
-
|
|
28
|
-
const ENV_KEY_BY_PROVIDER: Record<FallbackProviderName, string> = {
|
|
29
|
-
openrouter: "OPENROUTER_API_KEY",
|
|
30
|
-
anthropic: "ANTHROPIC".concat("_API_KEY"),
|
|
31
|
-
openai: "OPENAI".concat("_API_KEY"),
|
|
32
|
-
};
|
|
33
|
-
|
|
34
|
-
function hasProviderKey(provider: FallbackProviderName): boolean {
|
|
35
|
-
const k = ENV_KEY_BY_PROVIDER[provider];
|
|
36
|
-
return Boolean(globalThis.process?.env?.[k]);
|
|
37
|
-
}
|
|
38
|
-
|
|
39
|
-
/**
|
|
40
|
-
* Filter a chain to providers whose API key is present in env.
|
|
41
|
-
* Returns the original chain unchanged if NO providers have keys (so callers
|
|
42
|
-
* still see a meaningful error rather than an empty-chain throw).
|
|
43
|
-
*/
|
|
44
|
-
export function filterAvailableProviders(
|
|
45
|
-
chain: readonly FallbackProviderName[],
|
|
46
|
-
): readonly FallbackProviderName[] {
|
|
47
|
-
const filtered = chain.filter(hasProviderKey);
|
|
48
|
-
return filtered.length > 0 ? filtered : chain;
|
|
49
|
-
}
|
|
50
|
-
|
|
51
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
52
|
-
// Provider factory — lazy + dynamic to avoid hard dep on @ai-sdk/anthropic
|
|
53
|
-
// during cold paths that don't use it.
|
|
54
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
55
|
-
|
|
56
|
-
// Note: this fallback chain INTENTIONALLY uses direct provider API keys
|
|
57
|
-
// (OpenRouter / Anthropic / OpenAI) rather than Vercel AI Gateway OIDC.
|
|
58
|
-
// Rationale: this package runs in many non-Vercel deployments (ECS, Docker,
|
|
59
|
-
// self-hosted Hono) where OIDC isn't available. Apps deployed on Vercel
|
|
60
|
-
// should configure provider="gateway" in NebutraAIConfig (see sdk/config.ts)
|
|
61
|
-
// to route through AI Gateway with OIDC auth.
|
|
62
|
-
|
|
63
|
-
async function buildModel(
|
|
64
|
-
provider: FallbackProviderName,
|
|
65
|
-
modelOrPreset: string,
|
|
66
|
-
): Promise<LanguageModel> {
|
|
67
|
-
const modelId = resolveModel(modelOrPreset);
|
|
68
|
-
|
|
69
|
-
// Apps on Vercel should set provider="gateway" in NebutraAIConfig instead
|
|
70
|
-
// of using this direct-fallback chain.
|
|
71
|
-
const envKey = ENV_KEY_BY_PROVIDER[provider];
|
|
72
|
-
const apiKey = globalThis.process?.env?.[envKey];
|
|
73
|
-
if (!apiKey) throw new Error(`${envKey} missing`);
|
|
74
|
-
|
|
75
|
-
switch (provider) {
|
|
76
|
-
case "openrouter": {
|
|
77
|
-
const { createOpenRouter } = await import("@openrouter/ai-sdk-provider");
|
|
78
|
-
return createOpenRouter({ apiKey }).chat(modelId);
|
|
79
|
-
}
|
|
80
|
-
case "anthropic": {
|
|
81
|
-
const { createAnthropic } = await import("@ai-sdk/anthropic");
|
|
82
|
-
// Anthropic uses bare model IDs — strip "anthropic/" prefix from presets.
|
|
83
|
-
const anthropicModelId = modelId.startsWith("anthropic/")
|
|
84
|
-
? modelId.slice("anthropic/".length)
|
|
85
|
-
: modelId;
|
|
86
|
-
return createAnthropic({ apiKey })(anthropicModelId);
|
|
87
|
-
}
|
|
88
|
-
case "openai": {
|
|
89
|
-
const { createOpenAI } = await import("@ai-sdk/openai");
|
|
90
|
-
// Strip "openai/" prefix.
|
|
91
|
-
const openaiModelId = modelId.startsWith("openai/")
|
|
92
|
-
? modelId.slice("openai/".length)
|
|
93
|
-
: modelId;
|
|
94
|
-
return createOpenAI({ apiKey })(openaiModelId);
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
}
|
|
98
|
-
|
|
99
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
100
|
-
// Retryable error classification
|
|
101
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
102
|
-
|
|
103
|
-
const RETRYABLE_STATUS = new Set([408, 425, 429, 500, 502, 503, 504]);
|
|
104
|
-
|
|
105
|
-
export function isRetryableError(error: unknown): boolean {
|
|
106
|
-
if (!error || typeof error !== "object") return false;
|
|
107
|
-
const e = error as Record<string, unknown>;
|
|
108
|
-
|
|
109
|
-
const status =
|
|
110
|
-
typeof e.statusCode === "number"
|
|
111
|
-
? e.statusCode
|
|
112
|
-
: typeof e.status === "number"
|
|
113
|
-
? e.status
|
|
114
|
-
: undefined;
|
|
115
|
-
if (status !== undefined && RETRYABLE_STATUS.has(status)) return true;
|
|
116
|
-
|
|
117
|
-
const code = typeof e.code === "string" ? e.code : "";
|
|
118
|
-
if (
|
|
119
|
-
code === "ECONNRESET" ||
|
|
120
|
-
code === "ETIMEDOUT" ||
|
|
121
|
-
code === "ENOTFOUND" ||
|
|
122
|
-
code === "EAI_AGAIN"
|
|
123
|
-
) {
|
|
124
|
-
return true;
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
// AI SDK APICallError / network errors expose `.isRetryable`
|
|
128
|
-
if (e.isRetryable === true) return true;
|
|
129
|
-
|
|
130
|
-
return false;
|
|
131
|
-
}
|
|
132
|
-
|
|
133
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
134
|
-
// Resolved fallback model — try chain in order
|
|
135
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
136
|
-
|
|
137
|
-
export interface FallbackResult<T> {
|
|
138
|
-
result: T;
|
|
139
|
-
provider: FallbackProviderName;
|
|
140
|
-
attempts: number;
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
export interface CreateFallbackModelOptions {
|
|
144
|
-
/** Override the default chain from env. */
|
|
145
|
-
chain?: readonly FallbackProviderName[];
|
|
146
|
-
/** Model preset / id passed to each provider in the chain. */
|
|
147
|
-
model?: string;
|
|
148
|
-
/**
|
|
149
|
-
* If true (default), filter the chain to providers whose API key is present
|
|
150
|
-
* in env. Set false to keep the original chain (caller wants to surface
|
|
151
|
-
* "missing key" errors as fallback steps — useful for tests).
|
|
152
|
-
*/
|
|
153
|
-
filterAvailable?: boolean;
|
|
154
|
-
}
|
|
155
|
-
|
|
156
|
-
/**
|
|
157
|
-
* Run an AI SDK call against the configured fallback chain.
|
|
158
|
-
*
|
|
159
|
-
* The caller provides an `invoke(model)` function — usually a closure over
|
|
160
|
-
* `streamText` or `generateText` — and `runWithFallback` walks the chain,
|
|
161
|
-
* trying each provider in order until one succeeds or the chain is exhausted.
|
|
162
|
-
*/
|
|
163
|
-
export async function runWithFallback<T>(
|
|
164
|
-
invoke: (model: LanguageModel) => Promise<T>,
|
|
165
|
-
options: CreateFallbackModelOptions = {},
|
|
166
|
-
): Promise<FallbackResult<T>> {
|
|
167
|
-
const env = getAgentsEnv();
|
|
168
|
-
const rawChain = options.chain ?? env.LLM_FALLBACK_CHAIN;
|
|
169
|
-
const chain = options.filterAvailable === false ? rawChain : filterAvailableProviders(rawChain);
|
|
170
|
-
const model = options.model ?? "flagship";
|
|
171
|
-
|
|
172
|
-
let lastError: unknown;
|
|
173
|
-
let attempts = 0;
|
|
174
|
-
|
|
175
|
-
for (const provider of chain) {
|
|
176
|
-
attempts += 1;
|
|
177
|
-
try {
|
|
178
|
-
const lm = await buildModel(provider, model);
|
|
179
|
-
const result = await invoke(lm);
|
|
180
|
-
if (attempts > 1) {
|
|
181
|
-
logger.info("LLM fallback succeeded", { provider, attempts });
|
|
182
|
-
}
|
|
183
|
-
return { result, provider, attempts };
|
|
184
|
-
} catch (error) {
|
|
185
|
-
lastError = error;
|
|
186
|
-
|
|
187
|
-
// Non-retryable: surface the original error immediately.
|
|
188
|
-
if (!isRetryableError(error)) {
|
|
189
|
-
logger.error("LLM call failed (non-retryable)", { provider, error });
|
|
190
|
-
throw error;
|
|
191
|
-
}
|
|
192
|
-
|
|
193
|
-
logger.warn("LLM provider failed — trying next in chain", {
|
|
194
|
-
provider,
|
|
195
|
-
nextIndex: attempts,
|
|
196
|
-
error,
|
|
197
|
-
});
|
|
198
|
-
}
|
|
199
|
-
}
|
|
200
|
-
|
|
201
|
-
throw new Error(
|
|
202
|
-
`All LLM providers in fallback chain [${chain.join(", ")}] failed. ` +
|
|
203
|
-
`Last error: ${String(lastError)}`,
|
|
204
|
-
);
|
|
205
|
-
}
|
|
206
|
-
|
|
207
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
208
|
-
// Prompt-caching helper
|
|
209
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
210
|
-
|
|
211
|
-
/**
|
|
212
|
-
* Build `providerOptions` that enable prompt caching across providers.
|
|
213
|
-
*
|
|
214
|
-
* - Anthropic: explicit `cacheControl: { type: 'ephemeral' }` on the system
|
|
215
|
-
* message — 90% cost reduction on cached prefix tokens.
|
|
216
|
-
* - OpenAI: prompt caching is AUTOMATIC for prompts ≥1024 tokens with a
|
|
217
|
-
* stable prefix. No flag needed — but callers MUST keep the system prompt
|
|
218
|
-
* + tools FIRST and dynamic user content LAST, otherwise the cache is
|
|
219
|
-
* invalidated on every call.
|
|
220
|
-
* - OpenRouter: passes provider options through transparently.
|
|
221
|
-
*/
|
|
222
|
-
export function withAnthropicCacheControl(): {
|
|
223
|
-
anthropic: { cacheControl: { type: "ephemeral" } };
|
|
224
|
-
} {
|
|
225
|
-
return {
|
|
226
|
-
anthropic: { cacheControl: { type: "ephemeral" } },
|
|
227
|
-
};
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
/**
|
|
231
|
-
* Wraps the system message in a structured cache-control hint.
|
|
232
|
-
* Returns the messages array unchanged if no system text is provided.
|
|
233
|
-
*
|
|
234
|
-
* IMPORTANT: keep stable content (system prompt + tool defs) FIRST,
|
|
235
|
-
* dynamic content (user query) LAST — required for both Anthropic explicit
|
|
236
|
-
* caching AND OpenAI automatic caching to hit.
|
|
237
|
-
*/
|
|
238
|
-
export function buildSystemWithCache(systemPrompt: string): {
|
|
239
|
-
role: "system";
|
|
240
|
-
content: string;
|
|
241
|
-
providerOptions: ReturnType<typeof withAnthropicCacheControl>;
|
|
242
|
-
} {
|
|
243
|
-
return {
|
|
244
|
-
role: "system",
|
|
245
|
-
content: systemPrompt,
|
|
246
|
-
providerOptions: withAnthropicCacheControl(),
|
|
247
|
-
};
|
|
248
|
-
}
|
|
249
|
-
|
|
250
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
251
|
-
// Embedding fallback — separate chain since not every chat provider exposes
|
|
252
|
-
// embeddings (Anthropic notably does not).
|
|
253
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
254
|
-
|
|
255
|
-
/** Providers that currently expose embedding models via the AI SDK. */
|
|
256
|
-
const EMBEDDING_CAPABLE: ReadonlySet<FallbackProviderName> = new Set<FallbackProviderName>([
|
|
257
|
-
"openrouter",
|
|
258
|
-
"openai",
|
|
259
|
-
]);
|
|
260
|
-
|
|
261
|
-
async function buildEmbeddingModel(
|
|
262
|
-
provider: FallbackProviderName,
|
|
263
|
-
modelOrPreset: string,
|
|
264
|
-
): Promise<EmbeddingModel> {
|
|
265
|
-
const modelId = resolveModel(modelOrPreset);
|
|
266
|
-
const envKey = ENV_KEY_BY_PROVIDER[provider];
|
|
267
|
-
const apiKey = globalThis.process?.env?.[envKey];
|
|
268
|
-
if (!apiKey) throw new Error(`${envKey} missing`);
|
|
269
|
-
|
|
270
|
-
switch (provider) {
|
|
271
|
-
case "openrouter": {
|
|
272
|
-
const { createOpenRouter } = await import("@openrouter/ai-sdk-provider");
|
|
273
|
-
return createOpenRouter({ apiKey }).textEmbeddingModel(modelId);
|
|
274
|
-
}
|
|
275
|
-
case "openai": {
|
|
276
|
-
const { createOpenAI } = await import("@ai-sdk/openai");
|
|
277
|
-
const openaiModelId = modelId.startsWith("openai/")
|
|
278
|
-
? modelId.slice("openai/".length)
|
|
279
|
-
: modelId;
|
|
280
|
-
return createOpenAI({ apiKey }).textEmbeddingModel(openaiModelId);
|
|
281
|
-
}
|
|
282
|
-
case "anthropic": {
|
|
283
|
-
throw new Error("Anthropic does not expose embedding models");
|
|
284
|
-
}
|
|
285
|
-
}
|
|
286
|
-
}
|
|
287
|
-
|
|
288
|
-
export interface EmbeddingFallbackOptions {
|
|
289
|
-
chain?: readonly FallbackProviderName[];
|
|
290
|
-
model?: string;
|
|
291
|
-
filterAvailable?: boolean;
|
|
292
|
-
}
|
|
293
|
-
|
|
294
|
-
/**
|
|
295
|
-
* Run an AI SDK embedding call against the configured embedding fallback chain.
|
|
296
|
-
*
|
|
297
|
-
* The caller provides an `invoke(model)` function — usually a closure over
|
|
298
|
-
* `embed` or `embedMany` from the `ai` package — and this helper walks the
|
|
299
|
-
* embedding-capable chain, trying each provider until one succeeds.
|
|
300
|
-
*/
|
|
301
|
-
export async function runEmbedWithFallback<T>(
|
|
302
|
-
invoke: (model: EmbeddingModel) => Promise<T>,
|
|
303
|
-
options: EmbeddingFallbackOptions = {},
|
|
304
|
-
): Promise<FallbackResult<T>> {
|
|
305
|
-
const env = getAgentsEnv();
|
|
306
|
-
const rawChain = options.chain ?? env.LLM_EMBEDDING_FALLBACK_CHAIN;
|
|
307
|
-
|
|
308
|
-
// Filter to providers that (a) expose embeddings and (b) have keys present.
|
|
309
|
-
const capable = rawChain.filter((p) => EMBEDDING_CAPABLE.has(p));
|
|
310
|
-
const chain = options.filterAvailable === false ? capable : capable.filter(hasProviderKey);
|
|
311
|
-
|
|
312
|
-
if (chain.length === 0) {
|
|
313
|
-
logger.warn("Embedding fallback chain is empty after filtering", {
|
|
314
|
-
raw: rawChain,
|
|
315
|
-
capable,
|
|
316
|
-
});
|
|
317
|
-
throw new Error(
|
|
318
|
-
"No embedding-capable providers available — set OPENROUTER_API_KEY or " +
|
|
319
|
-
"OPENAI_API_KEY (Anthropic does not expose embedding models).",
|
|
320
|
-
);
|
|
321
|
-
}
|
|
322
|
-
|
|
323
|
-
const model = options.model ?? "embedding";
|
|
324
|
-
let lastError: unknown;
|
|
325
|
-
let attempts = 0;
|
|
326
|
-
|
|
327
|
-
for (const provider of chain) {
|
|
328
|
-
attempts += 1;
|
|
329
|
-
try {
|
|
330
|
-
const em = await buildEmbeddingModel(provider, model);
|
|
331
|
-
const result = await invoke(em);
|
|
332
|
-
if (attempts > 1) {
|
|
333
|
-
logger.info("Embedding fallback succeeded", { provider, attempts });
|
|
334
|
-
}
|
|
335
|
-
return { result, provider, attempts };
|
|
336
|
-
} catch (error) {
|
|
337
|
-
lastError = error;
|
|
338
|
-
|
|
339
|
-
if (!isRetryableError(error)) {
|
|
340
|
-
logger.error("Embedding call failed (non-retryable)", { provider, error });
|
|
341
|
-
throw error;
|
|
342
|
-
}
|
|
343
|
-
|
|
344
|
-
logger.warn("Embedding provider failed — trying next in chain", {
|
|
345
|
-
provider,
|
|
346
|
-
nextIndex: attempts,
|
|
347
|
-
error,
|
|
348
|
-
});
|
|
349
|
-
}
|
|
350
|
-
}
|
|
351
|
-
|
|
352
|
-
throw new Error(
|
|
353
|
-
`All embedding providers in fallback chain [${chain.join(", ")}] failed. ` +
|
|
354
|
-
`Last error: ${String(lastError)}`,
|
|
355
|
-
);
|
|
356
|
-
}
|
|
357
|
-
|
|
358
|
-
export type { ModelMessage };
|
package/src/generation/index.ts
DELETED
|
@@ -1,157 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Image / video generation modality — public surface.
|
|
3
|
-
*
|
|
4
|
-
* Mirrors the LLM fallback design (`fallback.ts`): an ordered provider chain,
|
|
5
|
-
* filtered to providers whose `envKey` is present, with `mock` as the
|
|
6
|
-
* guaranteed terminal so a result is always produced. Retryable failures
|
|
7
|
-
* (429 / 5xx / network) rotate to the next provider via `isRetryableError`.
|
|
8
|
-
*/
|
|
9
|
-
|
|
10
|
-
import { logger } from "@nebutra/logger";
|
|
11
|
-
import { isRetryableError } from "../fallback";
|
|
12
|
-
import { mockGenerationProvider } from "./mock-provider";
|
|
13
|
-
import type {
|
|
14
|
-
GenerationCallOptions,
|
|
15
|
-
GenerationContext,
|
|
16
|
-
GenerationModality,
|
|
17
|
-
GenerationProvider,
|
|
18
|
-
GenerationResult,
|
|
19
|
-
ImageGenerationRequest,
|
|
20
|
-
VideoGenerationRequest,
|
|
21
|
-
} from "./types";
|
|
22
|
-
|
|
23
|
-
const log = logger.child({ module: "agents/generation" });
|
|
24
|
-
|
|
25
|
-
// ── Registry ────────────────────────────────────────────────────────────────
|
|
26
|
-
// `mock` is registered last and always available. Real providers register
|
|
27
|
-
// ahead of it (additively) and win whenever their env key is present.
|
|
28
|
-
|
|
29
|
-
const _registry = new Map<string, GenerationProvider>();
|
|
30
|
-
|
|
31
|
-
export function registerGenerationProvider(provider: GenerationProvider): void {
|
|
32
|
-
_registry.set(provider.name, provider);
|
|
33
|
-
}
|
|
34
|
-
|
|
35
|
-
/** Test helper — restores the registry to just the mock provider. */
|
|
36
|
-
export function _resetGenerationRegistry(): void {
|
|
37
|
-
_registry.clear();
|
|
38
|
-
_registry.set(mockGenerationProvider.name, mockGenerationProvider);
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
_resetGenerationRegistry();
|
|
42
|
-
|
|
43
|
-
function hasEnvKey(provider: GenerationProvider): boolean {
|
|
44
|
-
if (provider.envKey === null) return true;
|
|
45
|
-
return Boolean(globalThis.process?.env?.[provider.envKey]);
|
|
46
|
-
}
|
|
47
|
-
|
|
48
|
-
/** Provider names available for a modality, in resolved priority order. */
|
|
49
|
-
export function listGenerationProviders(
|
|
50
|
-
modality: GenerationModality,
|
|
51
|
-
options: GenerationCallOptions = {},
|
|
52
|
-
): string[] {
|
|
53
|
-
const envChain = (globalThis.process?.env?.GENERATION_FALLBACK_CHAIN ?? "")
|
|
54
|
-
.split(",")
|
|
55
|
-
.map((s) => s.trim())
|
|
56
|
-
.filter(Boolean);
|
|
57
|
-
const preferred = options.chain ?? (envChain.length > 0 ? envChain : []);
|
|
58
|
-
|
|
59
|
-
// Real providers first (explicit chain wins, then registry order), `mock`
|
|
60
|
-
// is always demoted to the guaranteed terminal regardless of registry order.
|
|
61
|
-
const mockName = mockGenerationProvider.name;
|
|
62
|
-
const all = [...preferred, ..._registry.keys()];
|
|
63
|
-
const seen = new Set<string>();
|
|
64
|
-
const resolved: string[] = [];
|
|
65
|
-
for (const name of all) {
|
|
66
|
-
if (seen.has(name) || name === mockName) continue;
|
|
67
|
-
seen.add(name);
|
|
68
|
-
const provider = _registry.get(name);
|
|
69
|
-
if (!provider) continue;
|
|
70
|
-
if (!provider.capabilities.includes(modality)) continue;
|
|
71
|
-
if (!hasEnvKey(provider)) continue;
|
|
72
|
-
resolved.push(name);
|
|
73
|
-
}
|
|
74
|
-
resolved.push(mockName);
|
|
75
|
-
return resolved;
|
|
76
|
-
}
|
|
77
|
-
|
|
78
|
-
async function runChain(
|
|
79
|
-
modality: GenerationModality,
|
|
80
|
-
options: GenerationCallOptions,
|
|
81
|
-
ctx: GenerationContext,
|
|
82
|
-
invoke: (p: GenerationProvider) => Promise<GenerationResult>,
|
|
83
|
-
): Promise<GenerationResult> {
|
|
84
|
-
const chain = listGenerationProviders(modality, options);
|
|
85
|
-
let lastErr: unknown;
|
|
86
|
-
for (const name of chain) {
|
|
87
|
-
const provider = _registry.get(name);
|
|
88
|
-
if (!provider) continue;
|
|
89
|
-
try {
|
|
90
|
-
const result = await invoke(provider);
|
|
91
|
-
log.debug("generation succeeded", {
|
|
92
|
-
provider: name,
|
|
93
|
-
modality,
|
|
94
|
-
tenantId: ctx.tenantId,
|
|
95
|
-
});
|
|
96
|
-
return result;
|
|
97
|
-
} catch (err) {
|
|
98
|
-
lastErr = err;
|
|
99
|
-
const retryable = isRetryableError(err);
|
|
100
|
-
log.warn("generation provider failed", {
|
|
101
|
-
provider: name,
|
|
102
|
-
modality,
|
|
103
|
-
retryable,
|
|
104
|
-
tenantId: ctx.tenantId,
|
|
105
|
-
});
|
|
106
|
-
// Non-retryable from the terminal mock would be a real bug — surface it.
|
|
107
|
-
if (!retryable && name !== mockGenerationProvider.name) continue;
|
|
108
|
-
if (!retryable) throw err;
|
|
109
|
-
}
|
|
110
|
-
}
|
|
111
|
-
throw lastErr instanceof Error
|
|
112
|
-
? lastErr
|
|
113
|
-
: new Error("[@nebutra/agents] generation chain exhausted");
|
|
114
|
-
}
|
|
115
|
-
|
|
116
|
-
/**
|
|
117
|
-
* Generate an image. Always resolves (falls back to the deterministic mock).
|
|
118
|
-
*/
|
|
119
|
-
export async function generateImage(
|
|
120
|
-
req: ImageGenerationRequest,
|
|
121
|
-
ctx: GenerationContext,
|
|
122
|
-
options: GenerationCallOptions = {},
|
|
123
|
-
): Promise<GenerationResult> {
|
|
124
|
-
return runChain("image", options, ctx, (p) => {
|
|
125
|
-
if (!p.generateImage) {
|
|
126
|
-
throw new Error(`[@nebutra/agents] provider "${p.name}" lacks image support`);
|
|
127
|
-
}
|
|
128
|
-
return p.generateImage(req, ctx);
|
|
129
|
-
});
|
|
130
|
-
}
|
|
131
|
-
|
|
132
|
-
/**
|
|
133
|
-
* Generate a video (or, in mock mode, a deterministic poster frame).
|
|
134
|
-
*/
|
|
135
|
-
export async function generateVideo(
|
|
136
|
-
req: VideoGenerationRequest,
|
|
137
|
-
ctx: GenerationContext,
|
|
138
|
-
options: GenerationCallOptions = {},
|
|
139
|
-
): Promise<GenerationResult> {
|
|
140
|
-
return runChain("video", options, ctx, (p) => {
|
|
141
|
-
if (!p.generateVideo) {
|
|
142
|
-
throw new Error(`[@nebutra/agents] provider "${p.name}" lacks video support`);
|
|
143
|
-
}
|
|
144
|
-
return p.generateVideo(req, ctx);
|
|
145
|
-
});
|
|
146
|
-
}
|
|
147
|
-
|
|
148
|
-
export { mockGenerationProvider } from "./mock-provider";
|
|
149
|
-
export type {
|
|
150
|
-
GenerationCallOptions,
|
|
151
|
-
GenerationContext,
|
|
152
|
-
GenerationModality,
|
|
153
|
-
GenerationProvider,
|
|
154
|
-
GenerationResult,
|
|
155
|
-
ImageGenerationRequest,
|
|
156
|
-
VideoGenerationRequest,
|
|
157
|
-
} from "./types";
|