@nebutra/agents 1.1.0 → 1.1.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/.turbo/turbo-build.log +6 -6
- package/.turbo/turbo-test.log +11 -10
- package/.turbo/turbo-typecheck.log +1 -1
- package/CHANGELOG.md +11 -1
- package/package.json +6 -12
- package/src/__tests__/cost-observability.test.ts +1 -1
- package/src/__tests__/runtime-gateway.test.ts +108 -0
- package/src/gateway.ts +234 -0
package/.turbo/turbo-build.log
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
|
|
2
|
-
> @nebutra/agents@1.1.
|
|
2
|
+
> @nebutra/agents@1.1.1 build /home/runner/work/Nebutra-Sailor/Nebutra-Sailor/packages/ai/agents
|
|
3
3
|
> tsup
|
|
4
4
|
|
|
5
5
|
[34mCLI[39m Building entry: src/index.ts, src/tools.ts, src/providers/langchain.ts, src/providers/vercel-ai.ts, src/sdk/config.ts, src/sdk/index.ts, src/sdk/models.ts, src/sdk/provider.ts
|
|
@@ -12,22 +12,22 @@
|
|
|
12
12
|
[32mESM[39m [1mdist/index.js [22m[32m16.08 KB[39m
|
|
13
13
|
[32mESM[39m [1mdist/providers/langchain.js [22m[32m536.00 B[39m
|
|
14
14
|
[32mESM[39m [1mdist/sdk/config.js [22m[32m166.00 B[39m
|
|
15
|
-
[32mESM[39m [1mdist/tools.js [22m[32m203.00 B[39m
|
|
16
15
|
[32mESM[39m [1mdist/chunk-UVL2UVVM.js [22m[32m1.26 KB[39m
|
|
16
|
+
[32mESM[39m [1mdist/tools.js [22m[32m203.00 B[39m
|
|
17
17
|
[32mESM[39m [1mdist/providers/vercel-ai.js [22m[32m2.43 KB[39m
|
|
18
18
|
[32mESM[39m [1mdist/chunk-B7XWL35G.js [22m[32m1.51 KB[39m
|
|
19
19
|
[32mESM[39m [1mdist/chunk-RDOFKRI6.js [22m[32m4.71 KB[39m
|
|
20
20
|
[32mESM[39m [1mdist/sdk/index.js [22m[32m534.00 B[39m
|
|
21
|
-
[32mESM[39m [1mdist/chunk-5LX742GP.js [22m[32m7.78 KB[39m
|
|
22
21
|
[32mESM[39m [1mdist/chunk-NPQECBXL.js [22m[32m2.37 KB[39m
|
|
22
|
+
[32mESM[39m [1mdist/chunk-5LX742GP.js [22m[32m7.78 KB[39m
|
|
23
|
+
[32mESM[39m [1mdist/sdk/provider.js [22m[32m190.00 B[39m
|
|
23
24
|
[32mESM[39m [1mdist/sdk/models.js [22m[32m102.00 B[39m
|
|
24
25
|
[32mESM[39m [1mdist/chunk-RLWM437Q.js [22m[32m1.63 KB[39m
|
|
25
26
|
[32mESM[39m [1mdist/chunk-NVPE5EDI.js [22m[32m1.49 KB[39m
|
|
26
27
|
[32mESM[39m [1mdist/chunk-5JZJ5KMC.js [22m[32m1.22 KB[39m
|
|
27
|
-
[32mESM[39m
|
|
28
|
-
[32mESM[39m ⚡️ Build success in 33ms
|
|
28
|
+
[32mESM[39m ⚡️ Build success in 48ms
|
|
29
29
|
[34mDTS[39m Build start
|
|
30
|
-
[32mDTS[39m ⚡️ Build success in
|
|
30
|
+
[32mDTS[39m ⚡️ Build success in 5575ms
|
|
31
31
|
[32mDTS[39m [1mdist/index.d.ts [22m[32m19.71 KB[39m
|
|
32
32
|
[32mDTS[39m [1mdist/tools.d.ts [22m[32m724.00 B[39m
|
|
33
33
|
[32mDTS[39m [1mdist/providers/langchain.d.ts [22m[32m546.00 B[39m
|
package/.turbo/turbo-test.log
CHANGED
|
@@ -1,18 +1,19 @@
|
|
|
1
1
|
|
|
2
|
-
> @nebutra/agents@1.1.
|
|
2
|
+
> @nebutra/agents@1.1.1 test /home/runner/work/Nebutra-Sailor/Nebutra-Sailor/packages/ai/agents
|
|
3
3
|
> vitest run
|
|
4
4
|
|
|
5
5
|
|
|
6
6
|
[1m[30m[46m RUN [49m[39m[22m [36mv4.1.4 [39m[90m/home/runner/work/Nebutra-Sailor/Nebutra-Sailor/packages/ai/agents[39m
|
|
7
7
|
|
|
8
|
-
[32m✓[39m src/__tests__/fallback-wiring.test.ts [2m([22m[2m9 tests[22m[2m)[22m[32m
|
|
9
|
-
[32m✓[39m src/__tests__/cost-observability.test.ts [2m([22m[2m18 tests[22m[2m)[22m[33m
|
|
10
|
-
[33m[2m✓[22m[39m falls through retryable errors and returns the next provider's result [33m
|
|
11
|
-
[32m✓[39m src/__tests__/public-api.test.ts [2m([22m[2m16 tests[22m[2m)[22m[32m
|
|
12
|
-
[32m✓[39m src/__tests__/
|
|
8
|
+
[32m✓[39m src/__tests__/fallback-wiring.test.ts [2m([22m[2m9 tests[22m[2m)[22m[32m 235[2mms[22m[39m
|
|
9
|
+
[32m✓[39m src/__tests__/cost-observability.test.ts [2m([22m[2m18 tests[22m[2m)[22m[33m 510[2mms[22m[39m
|
|
10
|
+
[33m[2m✓[22m[39m falls through retryable errors and returns the next provider's result [33m 335[2mms[22m[39m
|
|
11
|
+
[32m✓[39m src/__tests__/public-api.test.ts [2m([22m[2m16 tests[22m[2m)[22m[32m 72[2mms[22m[39m
|
|
12
|
+
[32m✓[39m src/__tests__/runtime-gateway.test.ts [2m([22m[2m3 tests[22m[2m)[22m[32m 80[2mms[22m[39m
|
|
13
|
+
[32m✓[39m src/__tests__/generation.test.ts [2m([22m[2m7 tests[22m[2m)[22m[32m 61[2mms[22m[39m
|
|
13
14
|
|
|
14
|
-
[2m Test Files [22m [1m[
|
|
15
|
-
[2m Tests [22m [1m[
|
|
16
|
-
[2m Start at [22m
|
|
17
|
-
[2m Duration [22m 4.
|
|
15
|
+
[2m Test Files [22m [1m[32m5 passed[39m[22m[90m (5)[39m
|
|
16
|
+
[2m Tests [22m [1m[32m53 passed[39m[22m[90m (53)[39m
|
|
17
|
+
[2m Start at [22m 08:15:21
|
|
18
|
+
[2m Duration [22m 4.98s[2m (transform 2.66s, setup 0ms, import 7.42s, tests 958ms, environment 1ms)[22m
|
|
18
19
|
|
package/CHANGELOG.md
CHANGED
|
@@ -1,11 +1,21 @@
|
|
|
1
1
|
# @nebutra/agents
|
|
2
2
|
|
|
3
|
+
## 1.1.1
|
|
4
|
+
|
|
5
|
+
### Patch Changes
|
|
6
|
+
|
|
7
|
+
- Publish registry package metadata under the MIT license.
|
|
8
|
+
|
|
9
|
+
- Updated dependencies []:
|
|
10
|
+
- @nebutra/billing@0.1.2
|
|
11
|
+
- @nebutra/cache@0.0.2
|
|
12
|
+
- @nebutra/logger@0.1.1
|
|
13
|
+
|
|
3
14
|
## 1.1.0
|
|
4
15
|
|
|
5
16
|
### Minor Changes
|
|
6
17
|
|
|
7
18
|
- [`092a1ce`](https://github.com/Nebutra/Nebutra-Sailor/commit/092a1ce810965e2d81767642e4bad05b80df81f4) Thanks [@TsekaLuk](https://github.com/TsekaLuk)! - Add the Atelier agentic creative-canvas capability.
|
|
8
|
-
|
|
9
19
|
- `@nebutra/agents`: new image/video **generation modality**
|
|
10
20
|
(`@nebutra/agents/generation`) on the same env-key-gated provider layer as
|
|
11
21
|
the LLM fallback chain, with a deterministic always-available mock provider.
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "@nebutra/agents",
|
|
3
|
-
"version": "1.1.
|
|
3
|
+
"version": "1.1.1",
|
|
4
4
|
"description": "Nebutra AI runtime: multi-agent orchestration + Vercel AI SDK helpers (absorbed @nebutra/ai-sdk in 1.0.0)",
|
|
5
5
|
"private": false,
|
|
6
6
|
"license": "MIT",
|
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
"nebutra": {
|
|
9
9
|
"featureId": "agents",
|
|
10
10
|
"category": "ai",
|
|
11
|
+
"surface": "model-runtime",
|
|
11
12
|
"summary": "Nebutra AI runtime: multi-agent orchestration + Vercel AI SDK helpers"
|
|
12
13
|
},
|
|
13
14
|
"main": "./src/index.ts",
|
|
@@ -30,20 +31,13 @@
|
|
|
30
31
|
"@ai-sdk/anthropic": "^3.0.76",
|
|
31
32
|
"@ai-sdk/openai": "^3.0.41",
|
|
32
33
|
"@openrouter/ai-sdk-provider": "^2.3.3",
|
|
34
|
+
"ai": "^6.0.0",
|
|
33
35
|
"langfuse": "^3.38.20",
|
|
34
36
|
"langfuse-vercel": "^3.38.20",
|
|
35
37
|
"zod": "^4.3.6",
|
|
36
|
-
"@nebutra/billing": "0.1.
|
|
37
|
-
"@nebutra/cache": "0.0.
|
|
38
|
-
"@nebutra/logger": "0.1.
|
|
39
|
-
},
|
|
40
|
-
"peerDependencies": {
|
|
41
|
-
"ai": "^6.0.0"
|
|
42
|
-
},
|
|
43
|
-
"peerDependenciesMeta": {
|
|
44
|
-
"ai": {
|
|
45
|
-
"optional": true
|
|
46
|
-
}
|
|
38
|
+
"@nebutra/billing": "0.1.2",
|
|
39
|
+
"@nebutra/cache": "0.0.2",
|
|
40
|
+
"@nebutra/logger": "0.1.1"
|
|
47
41
|
},
|
|
48
42
|
"devDependencies": {
|
|
49
43
|
"@types/node": "^22.19.15",
|
|
@@ -31,7 +31,7 @@ describe("withAnthropicCacheControl()", () => {
|
|
|
31
31
|
});
|
|
32
32
|
|
|
33
33
|
describe("isRetryableError()", () => {
|
|
34
|
-
it.each([429, 500, 502, 503, 504, 408])("marks status %i as retryable", (statusCode) => {
|
|
34
|
+
it.each([429, 500, 502, 503, 504, 408])("marks status %i as retryable", (statusCode: number) => {
|
|
35
35
|
expect(isRetryableError({ statusCode })).toBe(true);
|
|
36
36
|
});
|
|
37
37
|
|
|
@@ -0,0 +1,108 @@
|
|
|
1
|
+
import { describe, expect, it } from "vitest";
|
|
2
|
+
import { AgentRuntimeGateway, type AgentRuntimeGatewayProvider } from "../gateway";
|
|
3
|
+
|
|
4
|
+
function provider(
|
|
5
|
+
id: string,
|
|
6
|
+
text: string,
|
|
7
|
+
options: {
|
|
8
|
+
capabilities?: readonly string[];
|
|
9
|
+
fail?: "retryable" | "fatal";
|
|
10
|
+
calls?: { count: number };
|
|
11
|
+
} = {},
|
|
12
|
+
): AgentRuntimeGatewayProvider {
|
|
13
|
+
return {
|
|
14
|
+
id,
|
|
15
|
+
model: `${id}-model`,
|
|
16
|
+
capabilities: new Set(options.capabilities ?? ["reasoning", "tools"]),
|
|
17
|
+
async complete() {
|
|
18
|
+
if (options.calls) options.calls.count += 1;
|
|
19
|
+
if (options.fail === "retryable") {
|
|
20
|
+
throw Object.assign(new Error("rate limited"), { statusCode: 429 });
|
|
21
|
+
}
|
|
22
|
+
if (options.fail === "fatal") {
|
|
23
|
+
throw Object.assign(new Error("unauthorized"), { statusCode: 401 });
|
|
24
|
+
}
|
|
25
|
+
return {
|
|
26
|
+
id: `${id}-call`,
|
|
27
|
+
model: `${id}-model`,
|
|
28
|
+
provider: id,
|
|
29
|
+
text,
|
|
30
|
+
usage: { inputTokens: 2, outputTokens: 3, totalTokens: 5 },
|
|
31
|
+
};
|
|
32
|
+
},
|
|
33
|
+
};
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
describe("AgentRuntimeGateway", () => {
|
|
37
|
+
it("routes by capability and records tenant-aware usage decisions", async () => {
|
|
38
|
+
const gateway = new AgentRuntimeGateway({
|
|
39
|
+
providers: [
|
|
40
|
+
provider("vision-only", "no", { capabilities: ["vision"] }),
|
|
41
|
+
provider("reasoning-tools", "hello", { capabilities: ["reasoning", "tools"] }),
|
|
42
|
+
],
|
|
43
|
+
});
|
|
44
|
+
|
|
45
|
+
const response = await gateway.complete({
|
|
46
|
+
capability: "reasoning+tools",
|
|
47
|
+
tenantId: "org_1",
|
|
48
|
+
userId: "user_1",
|
|
49
|
+
requestId: "req_1",
|
|
50
|
+
messages: [{ role: "user", content: "hi" }],
|
|
51
|
+
});
|
|
52
|
+
|
|
53
|
+
expect(response.text).toBe("hello");
|
|
54
|
+
expect(gateway.usageReport()).toMatchObject({ calls: 1, totalTokens: 5 });
|
|
55
|
+
expect(gateway.debugLog()[0]).toMatchObject({
|
|
56
|
+
requestId: "req_1",
|
|
57
|
+
tenantId: "org_1",
|
|
58
|
+
userId: "user_1",
|
|
59
|
+
decision: {
|
|
60
|
+
provider: "reasoning-tools",
|
|
61
|
+
fallbackIndex: 0,
|
|
62
|
+
},
|
|
63
|
+
ok: true,
|
|
64
|
+
});
|
|
65
|
+
});
|
|
66
|
+
|
|
67
|
+
it("falls back only for retryable provider errors", async () => {
|
|
68
|
+
const retryable = new AgentRuntimeGateway({
|
|
69
|
+
providers: [provider("primary", "no", { fail: "retryable" }), provider("fallback", "yes")],
|
|
70
|
+
});
|
|
71
|
+
|
|
72
|
+
await expect(
|
|
73
|
+
retryable.complete({
|
|
74
|
+
capability: "reasoning",
|
|
75
|
+
messages: [{ role: "user", content: "hi" }],
|
|
76
|
+
}),
|
|
77
|
+
).resolves.toMatchObject({ provider: "fallback", text: "yes" });
|
|
78
|
+
|
|
79
|
+
const fatal = new AgentRuntimeGateway({
|
|
80
|
+
providers: [provider("primary", "no", { fail: "fatal" }), provider("fallback", "yes")],
|
|
81
|
+
});
|
|
82
|
+
|
|
83
|
+
await expect(
|
|
84
|
+
fatal.complete({
|
|
85
|
+
capability: "reasoning",
|
|
86
|
+
messages: [{ role: "user", content: "hi" }],
|
|
87
|
+
}),
|
|
88
|
+
).rejects.toThrow(/unauthorized/);
|
|
89
|
+
expect(fatal.debugLog()).toHaveLength(1);
|
|
90
|
+
});
|
|
91
|
+
|
|
92
|
+
it("uses a stable prompt cache without invoking the provider twice", async () => {
|
|
93
|
+
const calls = { count: 0 };
|
|
94
|
+
const gateway = new AgentRuntimeGateway({
|
|
95
|
+
providers: [provider("primary", "cached", { calls })],
|
|
96
|
+
});
|
|
97
|
+
const request = {
|
|
98
|
+
capability: "reasoning",
|
|
99
|
+
messages: [{ role: "user" as const, content: "same prompt" }],
|
|
100
|
+
};
|
|
101
|
+
|
|
102
|
+
await gateway.complete(request);
|
|
103
|
+
await gateway.complete(request);
|
|
104
|
+
|
|
105
|
+
expect(calls.count).toBe(1);
|
|
106
|
+
expect(gateway.cacheStats()).toMatchObject({ hits: 1, misses: 1, size: 1 });
|
|
107
|
+
});
|
|
108
|
+
});
|
package/src/gateway.ts
ADDED
|
@@ -0,0 +1,234 @@
|
|
|
1
|
+
import { isRetryableError } from "./fallback";
|
|
2
|
+
|
|
3
|
+
export type AgentRuntimeGatewayMessageRole = "system" | "user" | "assistant" | "tool";
|
|
4
|
+
|
|
5
|
+
export interface AgentRuntimeGatewayMessage {
|
|
6
|
+
readonly role: AgentRuntimeGatewayMessageRole;
|
|
7
|
+
readonly content: string;
|
|
8
|
+
}
|
|
9
|
+
|
|
10
|
+
export interface AgentRuntimeGatewayUsage {
|
|
11
|
+
readonly inputTokens?: number;
|
|
12
|
+
readonly outputTokens?: number;
|
|
13
|
+
readonly totalTokens?: number;
|
|
14
|
+
}
|
|
15
|
+
|
|
16
|
+
export interface AgentRuntimeGatewayCompletion {
|
|
17
|
+
readonly id: string;
|
|
18
|
+
readonly provider: string;
|
|
19
|
+
readonly model: string;
|
|
20
|
+
readonly text: string;
|
|
21
|
+
readonly usage?: AgentRuntimeGatewayUsage;
|
|
22
|
+
readonly raw?: unknown;
|
|
23
|
+
}
|
|
24
|
+
|
|
25
|
+
export interface AgentRuntimeGatewayProvider {
|
|
26
|
+
readonly id: string;
|
|
27
|
+
readonly model: string;
|
|
28
|
+
readonly capabilities: ReadonlySet<string>;
|
|
29
|
+
complete(
|
|
30
|
+
messages: readonly AgentRuntimeGatewayMessage[],
|
|
31
|
+
options?: AgentRuntimeGatewayCompleteOptions,
|
|
32
|
+
): Promise<AgentRuntimeGatewayCompletion>;
|
|
33
|
+
}
|
|
34
|
+
|
|
35
|
+
export interface AgentRuntimeGatewayCompleteOptions {
|
|
36
|
+
readonly temperature?: number;
|
|
37
|
+
readonly maxTokens?: number;
|
|
38
|
+
readonly signal?: AbortSignal;
|
|
39
|
+
}
|
|
40
|
+
|
|
41
|
+
export interface AgentRuntimeGatewayRequest extends AgentRuntimeGatewayCompleteOptions {
|
|
42
|
+
readonly capability: string;
|
|
43
|
+
readonly messages: readonly AgentRuntimeGatewayMessage[];
|
|
44
|
+
readonly tenantId?: string;
|
|
45
|
+
readonly userId?: string;
|
|
46
|
+
readonly requestId?: string;
|
|
47
|
+
readonly cacheKey?: string;
|
|
48
|
+
readonly maxFallbacks?: number;
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
export interface AgentRuntimeGatewayDecision {
|
|
52
|
+
readonly provider: string;
|
|
53
|
+
readonly reason: string;
|
|
54
|
+
readonly fallbackIndex: number;
|
|
55
|
+
}
|
|
56
|
+
|
|
57
|
+
export interface AgentRuntimeGatewayDebugEntry {
|
|
58
|
+
readonly requestId: string;
|
|
59
|
+
readonly tenantId?: string;
|
|
60
|
+
readonly userId?: string;
|
|
61
|
+
readonly decision: AgentRuntimeGatewayDecision;
|
|
62
|
+
readonly ok: boolean;
|
|
63
|
+
readonly error?: string;
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
export interface AgentRuntimeGatewayUsageReport {
|
|
67
|
+
readonly calls: number;
|
|
68
|
+
readonly inputTokens: number;
|
|
69
|
+
readonly outputTokens: number;
|
|
70
|
+
readonly totalTokens: number;
|
|
71
|
+
readonly estimatedUsd: number;
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
export interface AgentRuntimeGatewayOptions {
|
|
75
|
+
readonly providers: readonly AgentRuntimeGatewayProvider[];
|
|
76
|
+
readonly estimateUsd?: (usage: Required<AgentRuntimeGatewayUsage>) => number;
|
|
77
|
+
readonly requestId?: () => string;
|
|
78
|
+
}
|
|
79
|
+
|
|
80
|
+
interface CacheEntry {
|
|
81
|
+
readonly response: AgentRuntimeGatewayCompletion;
|
|
82
|
+
}
|
|
83
|
+
|
|
84
|
+
function capabilityParts(capability: string): string[] {
|
|
85
|
+
return capability
|
|
86
|
+
.split(/[+,\s]+/)
|
|
87
|
+
.map((part) => part.trim())
|
|
88
|
+
.filter(Boolean);
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
function cacheKey(request: AgentRuntimeGatewayRequest): string {
|
|
92
|
+
return (
|
|
93
|
+
request.cacheKey ??
|
|
94
|
+
JSON.stringify({
|
|
95
|
+
capability: request.capability,
|
|
96
|
+
prefix: request.messages.slice(0, Math.max(1, request.messages.length - 1)),
|
|
97
|
+
last: request.messages.at(-1),
|
|
98
|
+
})
|
|
99
|
+
);
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
function normalizeUsage(
|
|
103
|
+
usage: AgentRuntimeGatewayUsage | undefined,
|
|
104
|
+
): Required<AgentRuntimeGatewayUsage> {
|
|
105
|
+
const inputTokens = usage?.inputTokens ?? 0;
|
|
106
|
+
const outputTokens = usage?.outputTokens ?? 0;
|
|
107
|
+
return {
|
|
108
|
+
inputTokens,
|
|
109
|
+
outputTokens,
|
|
110
|
+
totalTokens: usage?.totalTokens ?? inputTokens + outputTokens,
|
|
111
|
+
};
|
|
112
|
+
}
|
|
113
|
+
|
|
114
|
+
function defaultRequestId(): string {
|
|
115
|
+
return `agents-gw-${Date.now()}-${Math.random().toString(16).slice(2)}`;
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
function defaultCostEstimate(usage: Required<AgentRuntimeGatewayUsage>): number {
|
|
119
|
+
return usage.totalTokens * 0.000_001;
|
|
120
|
+
}
|
|
121
|
+
|
|
122
|
+
export class AgentRuntimeGateway {
|
|
123
|
+
readonly #providers: readonly AgentRuntimeGatewayProvider[];
|
|
124
|
+
readonly #cache = new Map<string, CacheEntry>();
|
|
125
|
+
readonly #debug: AgentRuntimeGatewayDebugEntry[] = [];
|
|
126
|
+
readonly #estimateUsd: (usage: Required<AgentRuntimeGatewayUsage>) => number;
|
|
127
|
+
readonly #requestId: () => string;
|
|
128
|
+
#hits = 0;
|
|
129
|
+
#misses = 0;
|
|
130
|
+
#usage: AgentRuntimeGatewayUsageReport = {
|
|
131
|
+
calls: 0,
|
|
132
|
+
inputTokens: 0,
|
|
133
|
+
outputTokens: 0,
|
|
134
|
+
totalTokens: 0,
|
|
135
|
+
estimatedUsd: 0,
|
|
136
|
+
};
|
|
137
|
+
|
|
138
|
+
constructor(options: AgentRuntimeGatewayOptions) {
|
|
139
|
+
this.#providers = options.providers;
|
|
140
|
+
this.#estimateUsd = options.estimateUsd ?? defaultCostEstimate;
|
|
141
|
+
this.#requestId = options.requestId ?? defaultRequestId;
|
|
142
|
+
}
|
|
143
|
+
|
|
144
|
+
async complete(request: AgentRuntimeGatewayRequest): Promise<AgentRuntimeGatewayCompletion> {
|
|
145
|
+
const resolvedRequestId = request.requestId ?? this.#requestId();
|
|
146
|
+
const resolvedCacheKey = cacheKey(request);
|
|
147
|
+
const cached = this.#cache.get(resolvedCacheKey);
|
|
148
|
+
if (cached) {
|
|
149
|
+
this.#hits += 1;
|
|
150
|
+
return cached.response;
|
|
151
|
+
}
|
|
152
|
+
this.#misses += 1;
|
|
153
|
+
|
|
154
|
+
const providers = this.route(request);
|
|
155
|
+
let lastError: unknown;
|
|
156
|
+
const max = Math.min(request.maxFallbacks ?? providers.length, providers.length);
|
|
157
|
+
|
|
158
|
+
for (let index = 0; index < max; index += 1) {
|
|
159
|
+
const provider = providers[index];
|
|
160
|
+
if (!provider) continue;
|
|
161
|
+
|
|
162
|
+
const decision: AgentRuntimeGatewayDecision = {
|
|
163
|
+
provider: provider.id,
|
|
164
|
+
fallbackIndex: index,
|
|
165
|
+
reason: `matched capability "${request.capability}"`,
|
|
166
|
+
};
|
|
167
|
+
|
|
168
|
+
try {
|
|
169
|
+
const response = await provider.complete(request.messages, {
|
|
170
|
+
...(request.temperature !== undefined && { temperature: request.temperature }),
|
|
171
|
+
...(request.maxTokens !== undefined && { maxTokens: request.maxTokens }),
|
|
172
|
+
...(request.signal !== undefined && { signal: request.signal }),
|
|
173
|
+
});
|
|
174
|
+
this.#recordUsage(response);
|
|
175
|
+
this.#cache.set(resolvedCacheKey, { response });
|
|
176
|
+
this.#debug.push({
|
|
177
|
+
requestId: resolvedRequestId,
|
|
178
|
+
...(request.tenantId !== undefined && { tenantId: request.tenantId }),
|
|
179
|
+
...(request.userId !== undefined && { userId: request.userId }),
|
|
180
|
+
decision,
|
|
181
|
+
ok: true,
|
|
182
|
+
});
|
|
183
|
+
return response;
|
|
184
|
+
} catch (error) {
|
|
185
|
+
lastError = error;
|
|
186
|
+
this.#debug.push({
|
|
187
|
+
requestId: resolvedRequestId,
|
|
188
|
+
...(request.tenantId !== undefined && { tenantId: request.tenantId }),
|
|
189
|
+
...(request.userId !== undefined && { userId: request.userId }),
|
|
190
|
+
decision,
|
|
191
|
+
ok: false,
|
|
192
|
+
error: error instanceof Error ? error.message : String(error),
|
|
193
|
+
});
|
|
194
|
+
|
|
195
|
+
if (!isRetryableError(error)) throw error;
|
|
196
|
+
}
|
|
197
|
+
}
|
|
198
|
+
|
|
199
|
+
throw new Error(
|
|
200
|
+
`All AgentRuntimeGateway providers failed for capability "${request.capability}". Last error: ${String(lastError)}`,
|
|
201
|
+
);
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
route(request: Pick<AgentRuntimeGatewayRequest, "capability">): AgentRuntimeGatewayProvider[] {
|
|
205
|
+
const parts = capabilityParts(request.capability);
|
|
206
|
+
const matched = this.#providers.filter((provider) =>
|
|
207
|
+
parts.every((part) => provider.capabilities.has(part)),
|
|
208
|
+
);
|
|
209
|
+
return matched.length > 0 ? matched : [...this.#providers];
|
|
210
|
+
}
|
|
211
|
+
|
|
212
|
+
cacheStats(): { hits: number; misses: number; size: number } {
|
|
213
|
+
return { hits: this.#hits, misses: this.#misses, size: this.#cache.size };
|
|
214
|
+
}
|
|
215
|
+
|
|
216
|
+
usageReport(): AgentRuntimeGatewayUsageReport {
|
|
217
|
+
return { ...this.#usage };
|
|
218
|
+
}
|
|
219
|
+
|
|
220
|
+
debugLog(): readonly AgentRuntimeGatewayDebugEntry[] {
|
|
221
|
+
return [...this.#debug];
|
|
222
|
+
}
|
|
223
|
+
|
|
224
|
+
#recordUsage(response: AgentRuntimeGatewayCompletion): void {
|
|
225
|
+
const usage = normalizeUsage(response.usage);
|
|
226
|
+
this.#usage = {
|
|
227
|
+
calls: this.#usage.calls + 1,
|
|
228
|
+
inputTokens: this.#usage.inputTokens + usage.inputTokens,
|
|
229
|
+
outputTokens: this.#usage.outputTokens + usage.outputTokens,
|
|
230
|
+
totalTokens: this.#usage.totalTokens + usage.totalTokens,
|
|
231
|
+
estimatedUsd: Number((this.#usage.estimatedUsd + this.#estimateUsd(usage)).toFixed(6)),
|
|
232
|
+
};
|
|
233
|
+
}
|
|
234
|
+
}
|