@nebutra/agents 1.1.1 → 2.0.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -676
- package/README.md +7 -3
- package/dist/{agent-CahnMASx.d.ts → agent-DDGWuUpe.d.ts} +1 -1
- package/dist/{chunk-B7XWL35G.js → chunk-7VL333VJ.js} +5 -1
- package/dist/{chunk-5LX742GP.js → chunk-C4CC5UEF.js} +25 -53
- package/{src/env.ts → dist/chunk-FCXWOXII.js} +23 -47
- package/dist/chunk-HVLFZW6E.js +71 -0
- package/dist/chunk-HZQXXUKB.js +171 -0
- package/dist/{chunk-RLWM437Q.js → chunk-KC6SOI5Z.js} +19 -4
- package/dist/{chunk-NPQECBXL.js → chunk-S5N743EP.js} +60 -7
- package/dist/chunk-VZPQOXWW.js +47 -0
- package/dist/{chunk-NVPE5EDI.js → chunk-YBCIJKC7.js} +20 -3
- package/dist/env.d.ts +45 -0
- package/dist/env.js +12 -0
- package/dist/fallback.d.ts +100 -0
- package/dist/fallback.js +18 -0
- package/dist/generation/index.d.ts +122 -0
- package/dist/generation/index.js +19 -0
- package/dist/index.d.ts +22 -329
- package/dist/index.js +59 -245
- package/dist/observability.d.ts +46 -0
- package/dist/observability.js +13 -0
- package/dist/providers/langchain.d.ts +2 -2
- package/dist/providers/vercel-ai.d.ts +2 -2
- package/dist/providers/vercel-ai.js +15 -7
- package/dist/sdk/config.d.ts +6 -0
- package/dist/sdk/config.js +2 -1
- package/dist/sdk/index.d.ts +37 -4
- package/dist/sdk/index.js +23 -8
- package/dist/sdk/models.d.ts +20 -27
- package/dist/sdk/models.js +5 -3
- package/dist/sdk/provider.d.ts +1 -0
- package/dist/sdk/provider.js +3 -3
- package/dist/tools.d.ts +1 -1
- package/dist/{types-NtgB3pch.d.ts → types-BytC-HfQ.d.ts} +1 -5
- package/package.json +73 -19
- package/.turbo/turbo-build.log +0 -40
- package/.turbo/turbo-test.log +0 -19
- package/.turbo/turbo-typecheck.log +0 -4
- package/AGENTS.md +0 -63
- package/CHANGELOG.md +0 -36
- package/dist/chunk-5JZJ5KMC.js +0 -37
- package/src/__tests__/cost-observability.test.ts +0 -172
- package/src/__tests__/fallback-wiring.test.ts +0 -313
- package/src/__tests__/generation.test.ts +0 -111
- package/src/__tests__/public-api.test.ts +0 -114
- package/src/__tests__/runtime-gateway.test.ts +0 -108
- package/src/agent.ts +0 -117
- package/src/context.ts +0 -99
- package/src/fallback.ts +0 -358
- package/src/gateway.ts +0 -234
- package/src/generation/index.ts +0 -157
- package/src/generation/mock-provider.ts +0 -123
- package/src/generation/types.ts +0 -87
- package/src/index.ts +0 -104
- package/src/memory.ts +0 -126
- package/src/observability.ts +0 -102
- package/src/orchestrator.ts +0 -147
- package/src/providers/langchain.ts +0 -28
- package/src/providers/vercel-ai.ts +0 -114
- package/src/router.ts +0 -158
- package/src/sdk/config.ts +0 -73
- package/src/sdk/index.ts +0 -214
- package/src/sdk/models.ts +0 -57
- package/src/sdk/provider.ts +0 -80
- package/src/tenant.ts +0 -52
- package/src/tools.ts +0 -65
- package/src/types.ts +0 -114
- package/tsconfig.json +0 -12
- package/tsup.config.ts +0 -21
package/src/agent.ts
DELETED
|
@@ -1,117 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* BaseAgent — abstract agent with memory, usage tracking, and tenant scoping.
|
|
3
|
-
*
|
|
4
|
-
* Concrete agents (VercelAIAgent, LangChainAgent, etc.) extend this class
|
|
5
|
-
* and implement the `execute()` method for their specific SDK.
|
|
6
|
-
*/
|
|
7
|
-
|
|
8
|
-
import { logger } from "@nebutra/logger";
|
|
9
|
-
import { getMemory, saveMemory } from "./memory";
|
|
10
|
-
import type {
|
|
11
|
-
AgentConfig,
|
|
12
|
-
AgentContext,
|
|
13
|
-
AgentMessage,
|
|
14
|
-
AgentResponse,
|
|
15
|
-
AgentUsageEvent,
|
|
16
|
-
} from "./types";
|
|
17
|
-
|
|
18
|
-
export class BaseAgent {
|
|
19
|
-
public readonly config: AgentConfig;
|
|
20
|
-
|
|
21
|
-
constructor(config: AgentConfig) {
|
|
22
|
-
this.config = config;
|
|
23
|
-
}
|
|
24
|
-
|
|
25
|
-
/**
|
|
26
|
-
* Run the agent with full lifecycle:
|
|
27
|
-
* 1. Load long-term memory (if configured)
|
|
28
|
-
* 2. Trim to short-term window
|
|
29
|
-
* 3. Delegate to provider-specific `execute()`
|
|
30
|
-
* 4. Track usage for billing
|
|
31
|
-
* 5. Persist new messages to memory
|
|
32
|
-
*/
|
|
33
|
-
async run(messages: readonly AgentMessage[], context: AgentContext): Promise<AgentResponse> {
|
|
34
|
-
const startTime = Date.now();
|
|
35
|
-
|
|
36
|
-
// Load long-term memory when enabled
|
|
37
|
-
const memory = this.config.memory?.longTerm?.enabled
|
|
38
|
-
? await getMemory(context.tenantId, context.conversationId)
|
|
39
|
-
: [];
|
|
40
|
-
|
|
41
|
-
// Merge and trim to short-term window
|
|
42
|
-
const maxMessages = this.config.memory?.shortTerm?.maxMessages ?? 50;
|
|
43
|
-
const allMessages = [...memory, ...messages].slice(-maxMessages);
|
|
44
|
-
|
|
45
|
-
// Provider-specific execution
|
|
46
|
-
const response = await this.execute(allMessages, context);
|
|
47
|
-
|
|
48
|
-
const durationMs = Date.now() - startTime;
|
|
49
|
-
|
|
50
|
-
// Build usage event for billing / metering
|
|
51
|
-
const usage: AgentUsageEvent = {
|
|
52
|
-
tenantId: context.tenantId,
|
|
53
|
-
userId: context.userId,
|
|
54
|
-
agentId: this.config.id,
|
|
55
|
-
model: this.config.model,
|
|
56
|
-
promptTokens: response.usage.promptTokens,
|
|
57
|
-
completionTokens: response.usage.completionTokens,
|
|
58
|
-
totalTokens: response.usage.totalTokens,
|
|
59
|
-
durationMs,
|
|
60
|
-
timestamp: new Date(),
|
|
61
|
-
};
|
|
62
|
-
|
|
63
|
-
logger.info("Agent execution completed", {
|
|
64
|
-
agentId: this.config.id,
|
|
65
|
-
tenantId: context.tenantId,
|
|
66
|
-
tokens: usage.totalTokens,
|
|
67
|
-
durationMs,
|
|
68
|
-
});
|
|
69
|
-
|
|
70
|
-
// Persist to long-term memory
|
|
71
|
-
if (this.config.memory?.longTerm?.enabled) {
|
|
72
|
-
await saveMemory(context.tenantId, context.conversationId, response.messages);
|
|
73
|
-
}
|
|
74
|
-
|
|
75
|
-
// Await usage emission so billing deduction is tracked before returning
|
|
76
|
-
await this.emitUsage(usage);
|
|
77
|
-
|
|
78
|
-
return response;
|
|
79
|
-
}
|
|
80
|
-
|
|
81
|
-
/**
|
|
82
|
-
* Provider-specific execution. Subclasses MUST override this.
|
|
83
|
-
*/
|
|
84
|
-
protected async execute(
|
|
85
|
-
_messages: readonly AgentMessage[],
|
|
86
|
-
_context: AgentContext,
|
|
87
|
-
): Promise<AgentResponse> {
|
|
88
|
-
throw new Error("execute() must be implemented by provider adapter");
|
|
89
|
-
}
|
|
90
|
-
|
|
91
|
-
/**
|
|
92
|
-
* Emit a usage event for downstream billing / metering.
|
|
93
|
-
*/
|
|
94
|
-
private async emitUsage(event: AgentUsageEvent): Promise<void> {
|
|
95
|
-
try {
|
|
96
|
-
const { deductCredits } = await import("@nebutra/billing/credits");
|
|
97
|
-
|
|
98
|
-
// Basic credit consumption: 1 credit per 10k total tokens
|
|
99
|
-
const totalTokens = event.totalTokens || 0;
|
|
100
|
-
if (totalTokens === 0) return;
|
|
101
|
-
|
|
102
|
-
const creditCost = Math.max(1, Math.ceil(totalTokens / 10000));
|
|
103
|
-
|
|
104
|
-
await deductCredits({
|
|
105
|
-
organizationId: event.tenantId,
|
|
106
|
-
amount: creditCost,
|
|
107
|
-
description: `Agent execution: ${event.model}`,
|
|
108
|
-
});
|
|
109
|
-
} catch (err) {
|
|
110
|
-
logger.error("Failed to emit usage or deduct credits for agent execution", {
|
|
111
|
-
tenantId: event.tenantId,
|
|
112
|
-
agentId: event.agentId,
|
|
113
|
-
error: err,
|
|
114
|
-
});
|
|
115
|
-
}
|
|
116
|
-
}
|
|
117
|
-
}
|
package/src/context.ts
DELETED
|
@@ -1,99 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* User context primitives — Memory-as-context layer for AI conversations.
|
|
3
|
-
*
|
|
4
|
-
* Inspired by Perplexity / ChatGPT "custom instructions" pattern but
|
|
5
|
-
* implemented as pure helpers: this package does **not** read the database
|
|
6
|
-
* (keeps `@nebutra/agents` data-layer-agnostic). The caller fetches the
|
|
7
|
-
* profile and passes the structured object in.
|
|
8
|
-
*
|
|
9
|
-
* Usage pattern from a Next.js route:
|
|
10
|
-
*
|
|
11
|
-
* ```ts
|
|
12
|
-
* const profile = await db.userProfile.findUnique({ where: { userId } });
|
|
13
|
-
* const system = buildPersonalizedSystemPrompt(BASE_PROMPT, profile);
|
|
14
|
-
* const result = await streamText(messages, { system, model: "fast" });
|
|
15
|
-
* ```
|
|
16
|
-
*/
|
|
17
|
-
|
|
18
|
-
export interface UserContext {
|
|
19
|
-
/** What the assistant should call the user. */
|
|
20
|
-
nickname?: string | null;
|
|
21
|
-
/** Job title / role — gives the model audience context. */
|
|
22
|
-
occupation?: string | null;
|
|
23
|
-
/** Free-form bio (interests, location, work focus, etc.). */
|
|
24
|
-
bio?: string | null;
|
|
25
|
-
/** Verbatim instructions that override default tone/format. */
|
|
26
|
-
customInstructions?: string | null;
|
|
27
|
-
}
|
|
28
|
-
|
|
29
|
-
const MAX_BIO_CHARS = 2000;
|
|
30
|
-
const MAX_INSTRUCTIONS_CHARS = 3000;
|
|
31
|
-
const MAX_OCCUPATION_CHARS = 120;
|
|
32
|
-
const MAX_NICKNAME_CHARS = 80;
|
|
33
|
-
|
|
34
|
-
/**
|
|
35
|
-
* Truthy check that treats empty strings as "no context".
|
|
36
|
-
*/
|
|
37
|
-
function present(value: string | null | undefined): value is string {
|
|
38
|
-
return typeof value === "string" && value.trim().length > 0;
|
|
39
|
-
}
|
|
40
|
-
|
|
41
|
-
/**
|
|
42
|
-
* Truncate to a hard byte budget — defensive against runaway DB rows.
|
|
43
|
-
*/
|
|
44
|
-
function clamp(value: string, max: number): string {
|
|
45
|
-
return value.length > max ? `${value.slice(0, max)}…` : value;
|
|
46
|
-
}
|
|
47
|
-
|
|
48
|
-
/**
|
|
49
|
-
* Renders a `UserContext` as a compact block to inject into a system prompt.
|
|
50
|
-
*
|
|
51
|
-
* Output is null when no field is set, so callers can skip the entire
|
|
52
|
-
* "About the user" preamble and avoid wasted tokens.
|
|
53
|
-
*/
|
|
54
|
-
export function renderUserContextBlock(context: UserContext | null | undefined): string | null {
|
|
55
|
-
if (!context) return null;
|
|
56
|
-
|
|
57
|
-
const lines: string[] = [];
|
|
58
|
-
|
|
59
|
-
if (present(context.nickname)) {
|
|
60
|
-
lines.push(`- Preferred name: ${clamp(context.nickname.trim(), MAX_NICKNAME_CHARS)}`);
|
|
61
|
-
}
|
|
62
|
-
if (present(context.occupation)) {
|
|
63
|
-
lines.push(`- Role: ${clamp(context.occupation.trim(), MAX_OCCUPATION_CHARS)}`);
|
|
64
|
-
}
|
|
65
|
-
if (present(context.bio)) {
|
|
66
|
-
lines.push(`- About them: ${clamp(context.bio.trim(), MAX_BIO_CHARS)}`);
|
|
67
|
-
}
|
|
68
|
-
|
|
69
|
-
let block = "";
|
|
70
|
-
if (lines.length > 0) {
|
|
71
|
-
block += `About the user:\n${lines.join("\n")}`;
|
|
72
|
-
}
|
|
73
|
-
|
|
74
|
-
if (present(context.customInstructions)) {
|
|
75
|
-
if (block) block += "\n\n";
|
|
76
|
-
block += `The user's custom instructions (these take precedence over defaults):\n${clamp(
|
|
77
|
-
context.customInstructions.trim(),
|
|
78
|
-
MAX_INSTRUCTIONS_CHARS,
|
|
79
|
-
)}`;
|
|
80
|
-
}
|
|
81
|
-
|
|
82
|
-
return block.length > 0 ? block : null;
|
|
83
|
-
}
|
|
84
|
-
|
|
85
|
-
/**
|
|
86
|
-
* Builds the final system prompt by prepending a personalization block to the
|
|
87
|
-
* base prompt. Pure — no side effects, safe to call per-request.
|
|
88
|
-
*
|
|
89
|
-
* @param basePrompt - the assistant's role-defining base prompt
|
|
90
|
-
* @param context - structured user context (null/undefined disables personalization)
|
|
91
|
-
*/
|
|
92
|
-
export function buildPersonalizedSystemPrompt(
|
|
93
|
-
basePrompt: string,
|
|
94
|
-
context: UserContext | null | undefined,
|
|
95
|
-
): string {
|
|
96
|
-
const block = renderUserContextBlock(context);
|
|
97
|
-
if (!block) return basePrompt;
|
|
98
|
-
return `${block}\n\n---\n\n${basePrompt}`;
|
|
99
|
-
}
|
package/src/fallback.ts
DELETED
|
@@ -1,358 +0,0 @@
|
|
|
1
|
-
/**
|
|
2
|
-
* Multi-provider fallback chain + prompt-caching helpers.
|
|
3
|
-
*
|
|
4
|
-
* Cost & reliability primitives for production LLM workloads:
|
|
5
|
-
*
|
|
6
|
-
* 1. `createFallbackModel()` — picks a primary model and returns a callable
|
|
7
|
-
* that retries on retryable errors (429 / 5xx / network) by swapping to
|
|
8
|
-
* the next provider in `LLM_FALLBACK_CHAIN`.
|
|
9
|
-
*
|
|
10
|
-
* 2. `withCacheControl()` — annotates the system message with Anthropic
|
|
11
|
-
* `cacheControl: { type: 'ephemeral' }` for a 90% discount on cached
|
|
12
|
-
* prefix tokens. OpenAI auto-caches when the prefix is stable and ≥1024
|
|
13
|
-
* tokens — see comment in `generateWithFallback()`.
|
|
14
|
-
*
|
|
15
|
-
* Reference: https://sdk.vercel.ai/docs/ai-sdk-providers/anthropic#cache-control
|
|
16
|
-
*/
|
|
17
|
-
|
|
18
|
-
import { logger } from "@nebutra/logger";
|
|
19
|
-
import type { EmbeddingModel, LanguageModel, ModelMessage } from "ai";
|
|
20
|
-
import { type FallbackProviderName, getAgentsEnv } from "./env";
|
|
21
|
-
import { resolveModel } from "./sdk/models";
|
|
22
|
-
|
|
23
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
24
|
-
// Env-key lookup — used to filter the chain to providers that actually have
|
|
25
|
-
// credentials present. This makes single-provider deploys "just work".
|
|
26
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
27
|
-
|
|
28
|
-
const ENV_KEY_BY_PROVIDER: Record<FallbackProviderName, string> = {
|
|
29
|
-
openrouter: "OPENROUTER_API_KEY",
|
|
30
|
-
anthropic: "ANTHROPIC".concat("_API_KEY"),
|
|
31
|
-
openai: "OPENAI".concat("_API_KEY"),
|
|
32
|
-
};
|
|
33
|
-
|
|
34
|
-
function hasProviderKey(provider: FallbackProviderName): boolean {
|
|
35
|
-
const k = ENV_KEY_BY_PROVIDER[provider];
|
|
36
|
-
return Boolean(globalThis.process?.env?.[k]);
|
|
37
|
-
}
|
|
38
|
-
|
|
39
|
-
/**
|
|
40
|
-
* Filter a chain to providers whose API key is present in env.
|
|
41
|
-
* Returns the original chain unchanged if NO providers have keys (so callers
|
|
42
|
-
* still see a meaningful error rather than an empty-chain throw).
|
|
43
|
-
*/
|
|
44
|
-
export function filterAvailableProviders(
|
|
45
|
-
chain: readonly FallbackProviderName[],
|
|
46
|
-
): readonly FallbackProviderName[] {
|
|
47
|
-
const filtered = chain.filter(hasProviderKey);
|
|
48
|
-
return filtered.length > 0 ? filtered : chain;
|
|
49
|
-
}
|
|
50
|
-
|
|
51
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
52
|
-
// Provider factory — lazy + dynamic to avoid hard dep on @ai-sdk/anthropic
|
|
53
|
-
// during cold paths that don't use it.
|
|
54
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
55
|
-
|
|
56
|
-
// Note: this fallback chain INTENTIONALLY uses direct provider API keys
|
|
57
|
-
// (OpenRouter / Anthropic / OpenAI) rather than Vercel AI Gateway OIDC.
|
|
58
|
-
// Rationale: this package runs in many non-Vercel deployments (ECS, Docker,
|
|
59
|
-
// self-hosted Hono) where OIDC isn't available. Apps deployed on Vercel
|
|
60
|
-
// should configure provider="gateway" in NebutraAIConfig (see sdk/config.ts)
|
|
61
|
-
// to route through AI Gateway with OIDC auth.
|
|
62
|
-
|
|
63
|
-
async function buildModel(
|
|
64
|
-
provider: FallbackProviderName,
|
|
65
|
-
modelOrPreset: string,
|
|
66
|
-
): Promise<LanguageModel> {
|
|
67
|
-
const modelId = resolveModel(modelOrPreset);
|
|
68
|
-
|
|
69
|
-
// Apps on Vercel should set provider="gateway" in NebutraAIConfig instead
|
|
70
|
-
// of using this direct-fallback chain.
|
|
71
|
-
const envKey = ENV_KEY_BY_PROVIDER[provider];
|
|
72
|
-
const apiKey = globalThis.process?.env?.[envKey];
|
|
73
|
-
if (!apiKey) throw new Error(`${envKey} missing`);
|
|
74
|
-
|
|
75
|
-
switch (provider) {
|
|
76
|
-
case "openrouter": {
|
|
77
|
-
const { createOpenRouter } = await import("@openrouter/ai-sdk-provider");
|
|
78
|
-
return createOpenRouter({ apiKey }).chat(modelId);
|
|
79
|
-
}
|
|
80
|
-
case "anthropic": {
|
|
81
|
-
const { createAnthropic } = await import("@ai-sdk/anthropic");
|
|
82
|
-
// Anthropic uses bare model IDs — strip "anthropic/" prefix from presets.
|
|
83
|
-
const anthropicModelId = modelId.startsWith("anthropic/")
|
|
84
|
-
? modelId.slice("anthropic/".length)
|
|
85
|
-
: modelId;
|
|
86
|
-
return createAnthropic({ apiKey })(anthropicModelId);
|
|
87
|
-
}
|
|
88
|
-
case "openai": {
|
|
89
|
-
const { createOpenAI } = await import("@ai-sdk/openai");
|
|
90
|
-
// Strip "openai/" prefix.
|
|
91
|
-
const openaiModelId = modelId.startsWith("openai/")
|
|
92
|
-
? modelId.slice("openai/".length)
|
|
93
|
-
: modelId;
|
|
94
|
-
return createOpenAI({ apiKey })(openaiModelId);
|
|
95
|
-
}
|
|
96
|
-
}
|
|
97
|
-
}
|
|
98
|
-
|
|
99
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
100
|
-
// Retryable error classification
|
|
101
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
102
|
-
|
|
103
|
-
const RETRYABLE_STATUS = new Set([408, 425, 429, 500, 502, 503, 504]);
|
|
104
|
-
|
|
105
|
-
export function isRetryableError(error: unknown): boolean {
|
|
106
|
-
if (!error || typeof error !== "object") return false;
|
|
107
|
-
const e = error as Record<string, unknown>;
|
|
108
|
-
|
|
109
|
-
const status =
|
|
110
|
-
typeof e.statusCode === "number"
|
|
111
|
-
? e.statusCode
|
|
112
|
-
: typeof e.status === "number"
|
|
113
|
-
? e.status
|
|
114
|
-
: undefined;
|
|
115
|
-
if (status !== undefined && RETRYABLE_STATUS.has(status)) return true;
|
|
116
|
-
|
|
117
|
-
const code = typeof e.code === "string" ? e.code : "";
|
|
118
|
-
if (
|
|
119
|
-
code === "ECONNRESET" ||
|
|
120
|
-
code === "ETIMEDOUT" ||
|
|
121
|
-
code === "ENOTFOUND" ||
|
|
122
|
-
code === "EAI_AGAIN"
|
|
123
|
-
) {
|
|
124
|
-
return true;
|
|
125
|
-
}
|
|
126
|
-
|
|
127
|
-
// AI SDK APICallError / network errors expose `.isRetryable`
|
|
128
|
-
if (e.isRetryable === true) return true;
|
|
129
|
-
|
|
130
|
-
return false;
|
|
131
|
-
}
|
|
132
|
-
|
|
133
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
134
|
-
// Resolved fallback model — try chain in order
|
|
135
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
136
|
-
|
|
137
|
-
export interface FallbackResult<T> {
|
|
138
|
-
result: T;
|
|
139
|
-
provider: FallbackProviderName;
|
|
140
|
-
attempts: number;
|
|
141
|
-
}
|
|
142
|
-
|
|
143
|
-
export interface CreateFallbackModelOptions {
|
|
144
|
-
/** Override the default chain from env. */
|
|
145
|
-
chain?: readonly FallbackProviderName[];
|
|
146
|
-
/** Model preset / id passed to each provider in the chain. */
|
|
147
|
-
model?: string;
|
|
148
|
-
/**
|
|
149
|
-
* If true (default), filter the chain to providers whose API key is present
|
|
150
|
-
* in env. Set false to keep the original chain (caller wants to surface
|
|
151
|
-
* "missing key" errors as fallback steps — useful for tests).
|
|
152
|
-
*/
|
|
153
|
-
filterAvailable?: boolean;
|
|
154
|
-
}
|
|
155
|
-
|
|
156
|
-
/**
|
|
157
|
-
* Run an AI SDK call against the configured fallback chain.
|
|
158
|
-
*
|
|
159
|
-
* The caller provides an `invoke(model)` function — usually a closure over
|
|
160
|
-
* `streamText` or `generateText` — and `runWithFallback` walks the chain,
|
|
161
|
-
* trying each provider in order until one succeeds or the chain is exhausted.
|
|
162
|
-
*/
|
|
163
|
-
export async function runWithFallback<T>(
|
|
164
|
-
invoke: (model: LanguageModel) => Promise<T>,
|
|
165
|
-
options: CreateFallbackModelOptions = {},
|
|
166
|
-
): Promise<FallbackResult<T>> {
|
|
167
|
-
const env = getAgentsEnv();
|
|
168
|
-
const rawChain = options.chain ?? env.LLM_FALLBACK_CHAIN;
|
|
169
|
-
const chain = options.filterAvailable === false ? rawChain : filterAvailableProviders(rawChain);
|
|
170
|
-
const model = options.model ?? "flagship";
|
|
171
|
-
|
|
172
|
-
let lastError: unknown;
|
|
173
|
-
let attempts = 0;
|
|
174
|
-
|
|
175
|
-
for (const provider of chain) {
|
|
176
|
-
attempts += 1;
|
|
177
|
-
try {
|
|
178
|
-
const lm = await buildModel(provider, model);
|
|
179
|
-
const result = await invoke(lm);
|
|
180
|
-
if (attempts > 1) {
|
|
181
|
-
logger.info("LLM fallback succeeded", { provider, attempts });
|
|
182
|
-
}
|
|
183
|
-
return { result, provider, attempts };
|
|
184
|
-
} catch (error) {
|
|
185
|
-
lastError = error;
|
|
186
|
-
|
|
187
|
-
// Non-retryable: surface the original error immediately.
|
|
188
|
-
if (!isRetryableError(error)) {
|
|
189
|
-
logger.error("LLM call failed (non-retryable)", { provider, error });
|
|
190
|
-
throw error;
|
|
191
|
-
}
|
|
192
|
-
|
|
193
|
-
logger.warn("LLM provider failed — trying next in chain", {
|
|
194
|
-
provider,
|
|
195
|
-
nextIndex: attempts,
|
|
196
|
-
error,
|
|
197
|
-
});
|
|
198
|
-
}
|
|
199
|
-
}
|
|
200
|
-
|
|
201
|
-
throw new Error(
|
|
202
|
-
`All LLM providers in fallback chain [${chain.join(", ")}] failed. ` +
|
|
203
|
-
`Last error: ${String(lastError)}`,
|
|
204
|
-
);
|
|
205
|
-
}
|
|
206
|
-
|
|
207
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
208
|
-
// Prompt-caching helper
|
|
209
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
210
|
-
|
|
211
|
-
/**
|
|
212
|
-
* Build `providerOptions` that enable prompt caching across providers.
|
|
213
|
-
*
|
|
214
|
-
* - Anthropic: explicit `cacheControl: { type: 'ephemeral' }` on the system
|
|
215
|
-
* message — 90% cost reduction on cached prefix tokens.
|
|
216
|
-
* - OpenAI: prompt caching is AUTOMATIC for prompts ≥1024 tokens with a
|
|
217
|
-
* stable prefix. No flag needed — but callers MUST keep the system prompt
|
|
218
|
-
* + tools FIRST and dynamic user content LAST, otherwise the cache is
|
|
219
|
-
* invalidated on every call.
|
|
220
|
-
* - OpenRouter: passes provider options through transparently.
|
|
221
|
-
*/
|
|
222
|
-
export function withAnthropicCacheControl(): {
|
|
223
|
-
anthropic: { cacheControl: { type: "ephemeral" } };
|
|
224
|
-
} {
|
|
225
|
-
return {
|
|
226
|
-
anthropic: { cacheControl: { type: "ephemeral" } },
|
|
227
|
-
};
|
|
228
|
-
}
|
|
229
|
-
|
|
230
|
-
/**
|
|
231
|
-
* Wraps the system message in a structured cache-control hint.
|
|
232
|
-
* Returns the messages array unchanged if no system text is provided.
|
|
233
|
-
*
|
|
234
|
-
* IMPORTANT: keep stable content (system prompt + tool defs) FIRST,
|
|
235
|
-
* dynamic content (user query) LAST — required for both Anthropic explicit
|
|
236
|
-
* caching AND OpenAI automatic caching to hit.
|
|
237
|
-
*/
|
|
238
|
-
export function buildSystemWithCache(systemPrompt: string): {
|
|
239
|
-
role: "system";
|
|
240
|
-
content: string;
|
|
241
|
-
providerOptions: ReturnType<typeof withAnthropicCacheControl>;
|
|
242
|
-
} {
|
|
243
|
-
return {
|
|
244
|
-
role: "system",
|
|
245
|
-
content: systemPrompt,
|
|
246
|
-
providerOptions: withAnthropicCacheControl(),
|
|
247
|
-
};
|
|
248
|
-
}
|
|
249
|
-
|
|
250
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
251
|
-
// Embedding fallback — separate chain since not every chat provider exposes
|
|
252
|
-
// embeddings (Anthropic notably does not).
|
|
253
|
-
// ────────────────────────────────────────────────────────────────────────────
|
|
254
|
-
|
|
255
|
-
/** Providers that currently expose embedding models via the AI SDK. */
|
|
256
|
-
const EMBEDDING_CAPABLE: ReadonlySet<FallbackProviderName> = new Set<FallbackProviderName>([
|
|
257
|
-
"openrouter",
|
|
258
|
-
"openai",
|
|
259
|
-
]);
|
|
260
|
-
|
|
261
|
-
async function buildEmbeddingModel(
|
|
262
|
-
provider: FallbackProviderName,
|
|
263
|
-
modelOrPreset: string,
|
|
264
|
-
): Promise<EmbeddingModel> {
|
|
265
|
-
const modelId = resolveModel(modelOrPreset);
|
|
266
|
-
const envKey = ENV_KEY_BY_PROVIDER[provider];
|
|
267
|
-
const apiKey = globalThis.process?.env?.[envKey];
|
|
268
|
-
if (!apiKey) throw new Error(`${envKey} missing`);
|
|
269
|
-
|
|
270
|
-
switch (provider) {
|
|
271
|
-
case "openrouter": {
|
|
272
|
-
const { createOpenRouter } = await import("@openrouter/ai-sdk-provider");
|
|
273
|
-
return createOpenRouter({ apiKey }).textEmbeddingModel(modelId);
|
|
274
|
-
}
|
|
275
|
-
case "openai": {
|
|
276
|
-
const { createOpenAI } = await import("@ai-sdk/openai");
|
|
277
|
-
const openaiModelId = modelId.startsWith("openai/")
|
|
278
|
-
? modelId.slice("openai/".length)
|
|
279
|
-
: modelId;
|
|
280
|
-
return createOpenAI({ apiKey }).textEmbeddingModel(openaiModelId);
|
|
281
|
-
}
|
|
282
|
-
case "anthropic": {
|
|
283
|
-
throw new Error("Anthropic does not expose embedding models");
|
|
284
|
-
}
|
|
285
|
-
}
|
|
286
|
-
}
|
|
287
|
-
|
|
288
|
-
export interface EmbeddingFallbackOptions {
|
|
289
|
-
chain?: readonly FallbackProviderName[];
|
|
290
|
-
model?: string;
|
|
291
|
-
filterAvailable?: boolean;
|
|
292
|
-
}
|
|
293
|
-
|
|
294
|
-
/**
|
|
295
|
-
* Run an AI SDK embedding call against the configured embedding fallback chain.
|
|
296
|
-
*
|
|
297
|
-
* The caller provides an `invoke(model)` function — usually a closure over
|
|
298
|
-
* `embed` or `embedMany` from the `ai` package — and this helper walks the
|
|
299
|
-
* embedding-capable chain, trying each provider until one succeeds.
|
|
300
|
-
*/
|
|
301
|
-
export async function runEmbedWithFallback<T>(
|
|
302
|
-
invoke: (model: EmbeddingModel) => Promise<T>,
|
|
303
|
-
options: EmbeddingFallbackOptions = {},
|
|
304
|
-
): Promise<FallbackResult<T>> {
|
|
305
|
-
const env = getAgentsEnv();
|
|
306
|
-
const rawChain = options.chain ?? env.LLM_EMBEDDING_FALLBACK_CHAIN;
|
|
307
|
-
|
|
308
|
-
// Filter to providers that (a) expose embeddings and (b) have keys present.
|
|
309
|
-
const capable = rawChain.filter((p) => EMBEDDING_CAPABLE.has(p));
|
|
310
|
-
const chain = options.filterAvailable === false ? capable : capable.filter(hasProviderKey);
|
|
311
|
-
|
|
312
|
-
if (chain.length === 0) {
|
|
313
|
-
logger.warn("Embedding fallback chain is empty after filtering", {
|
|
314
|
-
raw: rawChain,
|
|
315
|
-
capable,
|
|
316
|
-
});
|
|
317
|
-
throw new Error(
|
|
318
|
-
"No embedding-capable providers available — set OPENROUTER_API_KEY or " +
|
|
319
|
-
"OPENAI_API_KEY (Anthropic does not expose embedding models).",
|
|
320
|
-
);
|
|
321
|
-
}
|
|
322
|
-
|
|
323
|
-
const model = options.model ?? "embedding";
|
|
324
|
-
let lastError: unknown;
|
|
325
|
-
let attempts = 0;
|
|
326
|
-
|
|
327
|
-
for (const provider of chain) {
|
|
328
|
-
attempts += 1;
|
|
329
|
-
try {
|
|
330
|
-
const em = await buildEmbeddingModel(provider, model);
|
|
331
|
-
const result = await invoke(em);
|
|
332
|
-
if (attempts > 1) {
|
|
333
|
-
logger.info("Embedding fallback succeeded", { provider, attempts });
|
|
334
|
-
}
|
|
335
|
-
return { result, provider, attempts };
|
|
336
|
-
} catch (error) {
|
|
337
|
-
lastError = error;
|
|
338
|
-
|
|
339
|
-
if (!isRetryableError(error)) {
|
|
340
|
-
logger.error("Embedding call failed (non-retryable)", { provider, error });
|
|
341
|
-
throw error;
|
|
342
|
-
}
|
|
343
|
-
|
|
344
|
-
logger.warn("Embedding provider failed — trying next in chain", {
|
|
345
|
-
provider,
|
|
346
|
-
nextIndex: attempts,
|
|
347
|
-
error,
|
|
348
|
-
});
|
|
349
|
-
}
|
|
350
|
-
}
|
|
351
|
-
|
|
352
|
-
throw new Error(
|
|
353
|
-
`All embedding providers in fallback chain [${chain.join(", ")}] failed. ` +
|
|
354
|
-
`Last error: ${String(lastError)}`,
|
|
355
|
-
);
|
|
356
|
-
}
|
|
357
|
-
|
|
358
|
-
export type { ModelMessage };
|