@sriinnu/kosha-discovery 1.2.0 → 1.5.1
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +125 -52
- package/dist/aliases.d.ts +6 -1
- package/dist/aliases.d.ts.map +1 -1
- package/dist/aliases.js +38 -12
- package/dist/aliases.js.map +1 -1
- package/dist/cache.d.ts.map +1 -1
- package/dist/cache.js +40 -3
- package/dist/cache.js.map +1 -1
- package/dist/claude-generation.d.ts +59 -0
- package/dist/claude-generation.d.ts.map +1 -0
- package/dist/claude-generation.js +117 -0
- package/dist/claude-generation.js.map +1 -0
- package/dist/cli-cmd-doctor.d.ts +17 -0
- package/dist/cli-cmd-doctor.d.ts.map +1 -0
- package/dist/cli-cmd-doctor.js +199 -0
- package/dist/cli-cmd-doctor.js.map +1 -0
- package/dist/cli-cmd-model.js +2 -2
- package/dist/cli-cmd-model.js.map +1 -1
- package/dist/cli-cmd-spend.d.ts +10 -0
- package/dist/cli-cmd-spend.d.ts.map +1 -0
- package/dist/cli-cmd-spend.js +130 -0
- package/dist/cli-cmd-spend.js.map +1 -0
- package/dist/cli-commands.d.ts +4 -1
- package/dist/cli-commands.d.ts.map +1 -1
- package/dist/cli-commands.js +10 -2
- package/dist/cli-commands.js.map +1 -1
- package/dist/cli-format.js.map +1 -1
- package/dist/cli-help.d.ts.map +1 -1
- package/dist/cli-help.js +13 -2
- package/dist/cli-help.js.map +1 -1
- package/dist/cli.js +10 -0
- package/dist/cli.js.map +1 -1
- package/dist/cost.d.ts +152 -0
- package/dist/cost.d.ts.map +1 -0
- package/dist/cost.js +377 -0
- package/dist/cost.js.map +1 -0
- package/dist/credentials/resolver.d.ts +10 -0
- package/dist/credentials/resolver.d.ts.map +1 -1
- package/dist/credentials/resolver.js +64 -31
- package/dist/credentials/resolver.js.map +1 -1
- package/dist/discovery/anthropic.d.ts +19 -1
- package/dist/discovery/anthropic.d.ts.map +1 -1
- package/dist/discovery/anthropic.js +112 -8
- package/dist/discovery/anthropic.js.map +1 -1
- package/dist/discovery/base.d.ts +7 -0
- package/dist/discovery/base.d.ts.map +1 -1
- package/dist/discovery/base.js +32 -4
- package/dist/discovery/base.js.map +1 -1
- package/dist/discovery/bedrock.d.ts.map +1 -1
- package/dist/discovery/cerebras.d.ts.map +1 -1
- package/dist/discovery/cohere.d.ts.map +1 -1
- package/dist/discovery/deepinfra.d.ts.map +1 -1
- package/dist/discovery/deepseek.d.ts.map +1 -1
- package/dist/discovery/fireworks.d.ts.map +1 -1
- package/dist/discovery/glm.d.ts.map +1 -1
- package/dist/discovery/google.d.ts +3 -1
- package/dist/discovery/google.d.ts.map +1 -1
- package/dist/discovery/google.js +13 -6
- package/dist/discovery/google.js.map +1 -1
- package/dist/discovery/groq.d.ts.map +1 -1
- package/dist/discovery/index.d.ts +2 -0
- package/dist/discovery/index.d.ts.map +1 -1
- package/dist/discovery/index.js +6 -0
- package/dist/discovery/index.js.map +1 -1
- package/dist/discovery/llama-cpp.d.ts.map +1 -1
- package/dist/discovery/lmstudio.d.ts +26 -0
- package/dist/discovery/lmstudio.d.ts.map +1 -0
- package/dist/discovery/lmstudio.js +93 -0
- package/dist/discovery/lmstudio.js.map +1 -0
- package/dist/discovery/minimax.d.ts.map +1 -1
- package/dist/discovery/mistral.d.ts.map +1 -1
- package/dist/discovery/moonshot.d.ts.map +1 -1
- package/dist/discovery/nvidia.d.ts.map +1 -1
- package/dist/discovery/ollama.d.ts.map +1 -1
- package/dist/discovery/openai-compatible.d.ts.map +1 -1
- package/dist/discovery/openai.d.ts.map +1 -1
- package/dist/discovery/openrouter.d.ts +29 -5
- package/dist/discovery/openrouter.d.ts.map +1 -1
- package/dist/discovery/openrouter.js +56 -18
- package/dist/discovery/openrouter.js.map +1 -1
- package/dist/discovery/perplexity.d.ts.map +1 -1
- package/dist/discovery/promo-overrides.js.map +1 -1
- package/dist/discovery/static-direct.d.ts +9 -1
- package/dist/discovery/static-direct.d.ts.map +1 -1
- package/dist/discovery/static-direct.js +68 -7
- package/dist/discovery/static-direct.js.map +1 -1
- package/dist/discovery/together.d.ts.map +1 -1
- package/dist/discovery/vercel.d.ts.map +1 -1
- package/dist/discovery/vertex.d.ts.map +1 -1
- package/dist/discovery/vllm.d.ts +26 -0
- package/dist/discovery/vllm.d.ts.map +1 -0
- package/dist/discovery/vllm.js +93 -0
- package/dist/discovery/vllm.js.map +1 -0
- package/dist/discovery/zai.d.ts.map +1 -1
- package/dist/discovery-contract.d.ts +7 -0
- package/dist/discovery-contract.d.ts.map +1 -1
- package/dist/discovery-contract.js.map +1 -1
- package/dist/discovery-routes.d.ts +9 -0
- package/dist/discovery-routes.d.ts.map +1 -1
- package/dist/discovery-routes.js +106 -18
- package/dist/discovery-routes.js.map +1 -1
- package/dist/enrichment/litellm.d.ts +1 -1
- package/dist/enrichment/litellm.d.ts.map +1 -1
- package/dist/enrichment/litellm.js +3 -3
- package/dist/entry.d.ts +14 -0
- package/dist/entry.d.ts.map +1 -0
- package/dist/entry.js +26 -0
- package/dist/entry.js.map +1 -0
- package/dist/index.d.ts +10 -0
- package/dist/index.d.ts.map +1 -1
- package/dist/index.js +5 -0
- package/dist/index.js.map +1 -1
- package/dist/mcp-server.d.ts +207 -2
- package/dist/mcp-server.d.ts.map +1 -1
- package/dist/mcp-server.js +203 -51
- package/dist/mcp-server.js.map +1 -1
- package/dist/model-features.d.ts.map +1 -1
- package/dist/model-features.js +11 -2
- package/dist/model-features.js.map +1 -1
- package/dist/normalize.d.ts +7 -7
- package/dist/normalize.js +8 -8
- package/dist/provider-catalog.d.ts.map +1 -1
- package/dist/provider-catalog.js +26 -0
- package/dist/provider-catalog.js.map +1 -1
- package/dist/proxy.d.ts +49 -7
- package/dist/proxy.d.ts.map +1 -1
- package/dist/proxy.js +754 -89
- package/dist/proxy.js.map +1 -1
- package/dist/registry-query.js +2 -2
- package/dist/registry-query.js.map +1 -1
- package/dist/registry-routing.d.ts +65 -0
- package/dist/registry-routing.d.ts.map +1 -0
- package/dist/registry-routing.js +165 -0
- package/dist/registry-routing.js.map +1 -0
- package/dist/registry-runtime.d.ts +2 -0
- package/dist/registry-runtime.d.ts.map +1 -1
- package/dist/registry-runtime.js +177 -23
- package/dist/registry-runtime.js.map +1 -1
- package/dist/registry-selection.js +6 -0
- package/dist/registry-selection.js.map +1 -1
- package/dist/registry.d.ts +69 -0
- package/dist/registry.d.ts.map +1 -1
- package/dist/registry.js +133 -1
- package/dist/registry.js.map +1 -1
- package/dist/resilience.d.ts +15 -0
- package/dist/resilience.d.ts.map +1 -1
- package/dist/resilience.js +23 -1
- package/dist/resilience.js.map +1 -1
- package/dist/server.d.ts +44 -2
- package/dist/server.d.ts.map +1 -1
- package/dist/server.js +328 -33
- package/dist/server.js.map +1 -1
- package/dist/tally.d.ts +80 -0
- package/dist/tally.d.ts.map +1 -0
- package/dist/tally.js +176 -0
- package/dist/tally.js.map +1 -0
- package/dist/types.d.ts +22 -1
- package/dist/types.d.ts.map +1 -1
- package/dist/wire-anthropic.d.ts +264 -0
- package/dist/wire-anthropic.d.ts.map +1 -0
- package/dist/wire-anthropic.js +960 -0
- package/dist/wire-anthropic.js.map +1 -0
- package/package.json +14 -10
- package/logo.png +0 -0
package/dist/proxy.js
CHANGED
|
@@ -15,31 +15,162 @@
|
|
|
15
15
|
* <N>k minimum context window in tokens
|
|
16
16
|
* provider:<id> pin to a specific serving-layer provider
|
|
17
17
|
*
|
|
18
|
-
* Supported transports: openai, openai-compatible-http, ollama.
|
|
19
|
-
* Anthropic
|
|
20
|
-
*
|
|
18
|
+
* Supported transports: openai, openai-compatible-http, ollama, anthropic.
|
|
19
|
+
* Anthropic is proxied through the OpenAI ↔ Anthropic wire translator
|
|
20
|
+
* (`wire-anthropic.ts`): streaming, tools / tool calls, image_url parts,
|
|
21
|
+
* response_format, and reasoning_effort are carried; audio input and
|
|
22
|
+
* non-function tools fail over to a native OpenAI-compatible route. Google,
|
|
23
|
+
* Bedrock, and Vertex speak cloud-SDK wire formats and are not yet proxied.
|
|
24
|
+
*
|
|
25
|
+
* Spend accounting: every successful forward writes a ledger row with the
|
|
26
|
+
* pre-flight estimate AND, when the upstream returned a `usage` block (JSON
|
|
27
|
+
* or SSE), the reconciled actual cost. Budget enforcement prefers the actual.
|
|
21
28
|
*
|
|
22
29
|
* Response headers added by the proxy:
|
|
23
|
-
* x-kosha-model
|
|
24
|
-
* x-kosha-provider
|
|
25
|
-
* x-kosha-requested
|
|
30
|
+
* x-kosha-model — resolved model ID
|
|
31
|
+
* x-kosha-provider — resolved provider
|
|
32
|
+
* x-kosha-requested — original model string from the caller
|
|
33
|
+
* x-kosha-attempt-chain — provider:status for each attempt
|
|
34
|
+
* x-kosha-estimated-cost-usd — pre-flight estimate
|
|
35
|
+
* x-kosha-actual-cost-usd — reconciled from upstream usage (non-streaming)
|
|
36
|
+
* x-kosha-usage-source — upstream | estimate (non-streaming)
|
|
37
|
+
* x-kosha-wire-notes — Anthropic translator notes (dropped / degraded fields)
|
|
26
38
|
* @module
|
|
27
39
|
*/
|
|
28
|
-
import {
|
|
40
|
+
import { randomUUID } from "node:crypto";
|
|
41
|
+
import { getProviderDescriptor, listProviderDescriptors, providerExecutionCredentialRequired } from "./provider-catalog.js";
|
|
29
42
|
import { fallbackRegistryCredential, getRegistryCredentialResolver } from "./registry-runtime.js";
|
|
43
|
+
import { parseRouteStrategy } from "./registry-routing.js";
|
|
44
|
+
import { actualCostFromUsage, appendLedgerEntry, estimateRequestCost, readMonthlyBudgetUsd, readSpendForMonth, readTenantBudgetUsd, } from "./cost.js";
|
|
45
|
+
import { UnsupportedWireContentError, coerceOpenAIChatRequest, translateAnthropicStreamToOpenAI, translateAnthropicToOpenAI, translateOpenAIToAnthropicWithNotes, } from "./wire-anthropic.js";
|
|
46
|
+
const proxyCounters = {
|
|
47
|
+
requestsTotal: 0,
|
|
48
|
+
errorsTotal: 0,
|
|
49
|
+
byProvider: {},
|
|
50
|
+
};
|
|
51
|
+
/** Record one forwarded outcome: total + per-provider requests/errors/latency. */
|
|
52
|
+
function bumpProxyMetric(provider, ok, latencyMs) {
|
|
53
|
+
proxyCounters.requestsTotal += 1;
|
|
54
|
+
if (!ok)
|
|
55
|
+
proxyCounters.errorsTotal += 1;
|
|
56
|
+
let c = proxyCounters.byProvider[provider];
|
|
57
|
+
if (!c) {
|
|
58
|
+
c = { requests: 0, errors: 0, latencySumMs: 0 };
|
|
59
|
+
proxyCounters.byProvider[provider] = c;
|
|
60
|
+
}
|
|
61
|
+
c.requests += 1;
|
|
62
|
+
if (!ok)
|
|
63
|
+
c.errors += 1;
|
|
64
|
+
c.latencySumMs += latencyMs;
|
|
65
|
+
}
|
|
66
|
+
/**
|
|
67
|
+
* Snapshot the proxy hot-path counters. In-memory only (resets on restart);
|
|
68
|
+
* complements the breaker / provider-observation data the registry exposes.
|
|
69
|
+
*/
|
|
70
|
+
export function snapshotProxyMetrics() {
|
|
71
|
+
const byProvider = {};
|
|
72
|
+
for (const [provider, c] of Object.entries(proxyCounters.byProvider)) {
|
|
73
|
+
byProvider[provider] = {
|
|
74
|
+
requests: c.requests,
|
|
75
|
+
errors: c.errors,
|
|
76
|
+
avgLatencyMs: c.requests > 0 ? Math.round(c.latencySumMs / c.requests) : 0,
|
|
77
|
+
};
|
|
78
|
+
}
|
|
79
|
+
return { requestsTotal: proxyCounters.requestsTotal, errorsTotal: proxyCounters.errorsTotal, byProvider };
|
|
80
|
+
}
|
|
81
|
+
// ---------------------------------------------------------------------------
|
|
82
|
+
// Upstream host allowlist (SSRF guard)
|
|
83
|
+
// ---------------------------------------------------------------------------
|
|
84
|
+
/**
|
|
85
|
+
* Hostnames the proxy is allowed to forward to. Built once at module load
|
|
86
|
+
* from the in-process provider catalog (NOT from any file/cache value).
|
|
87
|
+
* Even though buildUpstreamUrl already only takes the host from the in-process
|
|
88
|
+
* catalog for non-local providers, validating the parsed URL's hostname
|
|
89
|
+
* against this list here lets static analyzers (CodeQL `js/request-forgery`)
|
|
90
|
+
* recognize this code as a sanitizer for the outbound fetch.
|
|
91
|
+
*/
|
|
92
|
+
const ALLOWED_UPSTREAM_HOSTS = (() => {
|
|
93
|
+
const hosts = new Set();
|
|
94
|
+
for (const descriptor of listProviderDescriptors()) {
|
|
95
|
+
try {
|
|
96
|
+
hosts.add(new URL(descriptor.defaultBaseUrl).hostname);
|
|
97
|
+
}
|
|
98
|
+
catch {
|
|
99
|
+
// A malformed catalog entry should not block loading the module.
|
|
100
|
+
}
|
|
101
|
+
}
|
|
102
|
+
return Object.freeze(Array.from(hosts));
|
|
103
|
+
})();
|
|
104
|
+
/**
|
|
105
|
+
* Return the upstream URL if-and-only-if its hostname is one we trust.
|
|
106
|
+
* Returns null for an unparseable URL, an exotic protocol, or a hostname
|
|
107
|
+
* outside the catalog allowlist / loopback families. The caller refuses the
|
|
108
|
+
* request in that case rather than dialing a tainted host. Written with
|
|
109
|
+
* explicit `===` checks against literal strings so static taint analysis
|
|
110
|
+
* recognizes this function as a sanitizer.
|
|
111
|
+
*/
|
|
112
|
+
function safeUpstreamUrl(rawUrl, isLocalProvider) {
|
|
113
|
+
let parsed;
|
|
114
|
+
try {
|
|
115
|
+
parsed = new URL(rawUrl);
|
|
116
|
+
}
|
|
117
|
+
catch {
|
|
118
|
+
return null;
|
|
119
|
+
}
|
|
120
|
+
// Block exotic schemes that fetch would otherwise accept (data:, file:, …).
|
|
121
|
+
if (parsed.protocol !== "https:" && parsed.protocol !== "http:")
|
|
122
|
+
return null;
|
|
123
|
+
const host = parsed.hostname;
|
|
124
|
+
if (isLocalProvider) {
|
|
125
|
+
// Loopback-only for user-configurable local runtimes. Explicit literal
|
|
126
|
+
// comparisons let CodeQL recognize this as a sanitizer.
|
|
127
|
+
if (host === "localhost")
|
|
128
|
+
return parsed;
|
|
129
|
+
if (host === "127.0.0.1")
|
|
130
|
+
return parsed;
|
|
131
|
+
if (host === "::1")
|
|
132
|
+
return parsed;
|
|
133
|
+
if (host === "0.0.0.0")
|
|
134
|
+
return parsed;
|
|
135
|
+
return null;
|
|
136
|
+
}
|
|
137
|
+
// Non-local: hostname must match one drawn from the in-process catalog.
|
|
138
|
+
for (const allowed of ALLOWED_UPSTREAM_HOSTS) {
|
|
139
|
+
if (host === allowed)
|
|
140
|
+
return parsed;
|
|
141
|
+
}
|
|
142
|
+
return null;
|
|
143
|
+
}
|
|
30
144
|
// native-http providers that speak the OpenAI wire format natively.
|
|
31
145
|
// All openai-compatible-http providers are implicitly forwardable.
|
|
32
146
|
const NATIVE_OPENAI_WIRE = new Set(["openai", "ollama"]);
|
|
147
|
+
// native-http providers whose wire format we translate into OpenAI-compatible
|
|
148
|
+
// at the proxy boundary, so the SDK contract stays OpenAI throughout.
|
|
149
|
+
const TRANSLATABLE_WIRE = new Set(["anthropic"]);
|
|
150
|
+
const KOSHA_PREFIX = "kosha:";
|
|
151
|
+
/**
|
|
152
|
+
* Parse a `kosha:<strategy>[filters]` selector. Returns null when the model
|
|
153
|
+
* string is not a kosha selector or names an unknown strategy (the caller
|
|
154
|
+
* then falls through to direct model/alias resolution).
|
|
155
|
+
*
|
|
156
|
+
* Examples: `kosha:cheapest`, `kosha:fastest[tool_use,128k]`,
|
|
157
|
+
* `kosha:reliable[provider:groq]`, `kosha:balanced[vision]`.
|
|
158
|
+
*/
|
|
33
159
|
function parseKoshaHint(model) {
|
|
34
|
-
if (!model.startsWith(
|
|
160
|
+
if (!model.startsWith(KOSHA_PREFIX))
|
|
35
161
|
return null;
|
|
36
|
-
const
|
|
37
|
-
|
|
38
|
-
|
|
39
|
-
const
|
|
40
|
-
|
|
162
|
+
const rest = model.slice(KOSHA_PREFIX.length);
|
|
163
|
+
const bracket = /\[([^\]]*)\]/.exec(rest);
|
|
164
|
+
const head = (bracket ? rest.slice(0, bracket.index) : rest).trim();
|
|
165
|
+
const strategy = parseRouteStrategy(head);
|
|
166
|
+
if (!strategy)
|
|
167
|
+
return null;
|
|
168
|
+
const hint = { strategy };
|
|
169
|
+
if (!bracket)
|
|
170
|
+
return hint;
|
|
171
|
+
for (const part of bracket[1].split(",").map((s) => s.trim()).filter(Boolean)) {
|
|
41
172
|
if (part.startsWith("provider:")) {
|
|
42
|
-
hint.provider = part.slice(
|
|
173
|
+
hint.provider = part.slice("provider:".length);
|
|
43
174
|
}
|
|
44
175
|
else if (/^\d+k$/i.test(part)) {
|
|
45
176
|
hint.minContext = Number.parseInt(part, 10) * 1_000;
|
|
@@ -54,13 +185,18 @@ function parseKoshaHint(model) {
|
|
|
54
185
|
// Model resolution
|
|
55
186
|
// ---------------------------------------------------------------------------
|
|
56
187
|
function isForwardable(model) {
|
|
188
|
+
// Retired models are no longer served upstream — never offer them as a
|
|
189
|
+
// proxy route, even when their transport would otherwise be forwardable.
|
|
190
|
+
if (model.status === "retired")
|
|
191
|
+
return false;
|
|
57
192
|
const descriptor = getProviderDescriptor(model.provider);
|
|
58
193
|
if (!descriptor)
|
|
59
194
|
return false;
|
|
60
195
|
if (descriptor.transport === "openai-compatible-http")
|
|
61
196
|
return true;
|
|
62
|
-
if (descriptor.transport === "native-http")
|
|
63
|
-
return NATIVE_OPENAI_WIRE.has(model.provider);
|
|
197
|
+
if (descriptor.transport === "native-http") {
|
|
198
|
+
return NATIVE_OPENAI_WIRE.has(model.provider) || TRANSLATABLE_WIRE.has(model.provider);
|
|
199
|
+
}
|
|
64
200
|
return false; // cloud-sdk (bedrock, vertex) needs per-SDK translation
|
|
65
201
|
}
|
|
66
202
|
function isExecutableRoute(registry, model) {
|
|
@@ -71,56 +207,205 @@ function isExecutableRoute(registry, model) {
|
|
|
71
207
|
return true;
|
|
72
208
|
return registry.provider(model.provider)?.authenticated === true;
|
|
73
209
|
}
|
|
74
|
-
|
|
210
|
+
/**
|
|
211
|
+
* Resolve an ordered list of forwardable candidate routes for `requested`.
|
|
212
|
+
*
|
|
213
|
+
* The order is the failover sequence: routes we hold credentials for come
|
|
214
|
+
* first (ranked by the selector's strategy for kosha: hints), then any other
|
|
215
|
+
* forwardable route. Duplicates (same provider + model id) are removed. An
|
|
216
|
+
* empty list means nothing forwardable matched.
|
|
217
|
+
*/
|
|
218
|
+
function resolveProxyCandidates(registry, requested) {
|
|
75
219
|
const hint = parseKoshaHint(requested);
|
|
76
220
|
if (hint !== null) {
|
|
77
|
-
const
|
|
221
|
+
const ranked = registry.rankedRoutes({
|
|
78
222
|
mode: "chat",
|
|
79
223
|
capability: hint.capability,
|
|
80
224
|
provider: hint.provider,
|
|
81
225
|
limit: 20,
|
|
82
|
-
});
|
|
83
|
-
const
|
|
84
|
-
?
|
|
85
|
-
:
|
|
86
|
-
|
|
87
|
-
|
|
88
|
-
|
|
89
|
-
|
|
226
|
+
}, hint.strategy);
|
|
227
|
+
const filtered = hint.minContext
|
|
228
|
+
? ranked.filter((r) => r.model.contextWindow >= hint.minContext)
|
|
229
|
+
: ranked;
|
|
230
|
+
const exec = filtered.filter((r) => isExecutableRoute(registry, r.model)).map((r) => r.model);
|
|
231
|
+
const fwd = filtered
|
|
232
|
+
.filter((r) => isForwardable(r.model) && !isExecutableRoute(registry, r.model))
|
|
233
|
+
.map((r) => r.model);
|
|
234
|
+
return dedupeModels([...exec, ...fwd]);
|
|
90
235
|
}
|
|
91
236
|
// For a specific model ID or alias, prefer the canonical card but fall back
|
|
92
237
|
// to any forwardable route if the primary provider isn't proxiable (e.g.
|
|
93
|
-
// "claude-sonnet-
|
|
238
|
+
// "claude-sonnet-5" resolves to Anthropic by default, but if the caller
|
|
94
239
|
// only has an OpenRouter key, we should route through OpenRouter instead).
|
|
95
240
|
const primary = registry.model(requested);
|
|
96
|
-
if (primary && isExecutableRoute(registry, primary))
|
|
97
|
-
return primary;
|
|
98
241
|
const routes = registry.modelRoutes(requested);
|
|
99
|
-
|
|
100
|
-
|
|
101
|
-
|
|
102
|
-
|
|
103
|
-
|
|
242
|
+
const ordered = [];
|
|
243
|
+
if (primary && isExecutableRoute(registry, primary))
|
|
244
|
+
ordered.push(primary);
|
|
245
|
+
for (const route of routes)
|
|
246
|
+
if (isExecutableRoute(registry, route))
|
|
247
|
+
ordered.push(route);
|
|
248
|
+
if (primary && isForwardable(primary))
|
|
249
|
+
ordered.push(primary);
|
|
250
|
+
for (const route of routes)
|
|
251
|
+
if (isForwardable(route))
|
|
252
|
+
ordered.push(route);
|
|
253
|
+
return dedupeModels(ordered);
|
|
254
|
+
}
|
|
255
|
+
/** Allowed shape of a tenant tag: short, filesystem/log-safe identifier. */
|
|
256
|
+
const TENANT_TAG_PATTERN = /^[A-Za-z0-9_.-]{1,64}$/;
|
|
257
|
+
/**
|
|
258
|
+
* Extract a tenant tag from the request. Two carriers are accepted:
|
|
259
|
+
*
|
|
260
|
+
* - `x-kosha-tenant: <name>` — preferred, and the only option when
|
|
261
|
+
* `KOSHA_PROXY_TOKEN` is set (the Authorization header then carries the
|
|
262
|
+
* operator token instead).
|
|
263
|
+
* - `Authorization: Bearer kosha-tenant-<name>` — legacy carrier, kept for
|
|
264
|
+
* callers that can only set a bearer token.
|
|
265
|
+
*
|
|
266
|
+
* The tag is a bucketing label only (per-tenant ledger rows + budget), never
|
|
267
|
+
* authentication — the bearer is consumed and replaced with the resolved
|
|
268
|
+
* upstream credential before forwarding. Returns null for any non-conforming
|
|
269
|
+
* value so downstream code doesn't have to guard for empties.
|
|
270
|
+
*/
|
|
271
|
+
export function parseTenantTag(authHeader, tenantHeader) {
|
|
272
|
+
const explicit = tenantHeader?.trim();
|
|
273
|
+
if (explicit && TENANT_TAG_PATTERN.test(explicit))
|
|
274
|
+
return explicit;
|
|
275
|
+
if (!authHeader)
|
|
276
|
+
return null;
|
|
277
|
+
const match = /^\s*Bearer\s+kosha-tenant-([A-Za-z0-9_.-]{1,64})\s*$/i.exec(authHeader);
|
|
278
|
+
return match ? match[1] : null;
|
|
279
|
+
}
|
|
280
|
+
/**
|
|
281
|
+
* Pull a `{ message, type }` error shape out of an Anthropic upstream error
|
|
282
|
+
* body. Anthropic errors are `{ type, error: { type, message } }`; a flat
|
|
283
|
+
* `{ message, type }` is also tolerated. Defaults keep the OpenAI error
|
|
284
|
+
* envelope well-formed when the body doesn't match either shape.
|
|
285
|
+
*/
|
|
286
|
+
function extractAnthropicError(raw) {
|
|
287
|
+
if (raw && typeof raw === "object") {
|
|
288
|
+
const obj = raw;
|
|
289
|
+
const err = obj.error && typeof obj.error === "object" ? obj.error : obj;
|
|
290
|
+
return {
|
|
291
|
+
message: typeof err.message === "string" ? err.message : "anthropic upstream error",
|
|
292
|
+
type: typeof err.type === "string" ? err.type : "upstream_error",
|
|
293
|
+
};
|
|
294
|
+
}
|
|
295
|
+
return { message: "anthropic upstream error", type: "upstream_error" };
|
|
296
|
+
}
|
|
297
|
+
/** Stable de-dup of model cards by provider + id, preserving first occurrence. */
|
|
298
|
+
function dedupeModels(models) {
|
|
299
|
+
const seen = new Set();
|
|
300
|
+
const out = [];
|
|
301
|
+
for (const model of models) {
|
|
302
|
+
const key = `${model.provider}:${model.id}`;
|
|
303
|
+
if (seen.has(key))
|
|
304
|
+
continue;
|
|
305
|
+
seen.add(key);
|
|
306
|
+
out.push(model);
|
|
307
|
+
}
|
|
308
|
+
return out;
|
|
309
|
+
}
|
|
310
|
+
function resolveProxyModel(registry, requested) {
|
|
311
|
+
return resolveProxyCandidates(registry, requested)[0] ?? null;
|
|
104
312
|
}
|
|
105
313
|
// ---------------------------------------------------------------------------
|
|
106
314
|
// URL builder
|
|
107
315
|
// ---------------------------------------------------------------------------
|
|
108
316
|
function buildUpstreamUrl(model, registry) {
|
|
109
|
-
const info = registry.provider(model.provider);
|
|
110
317
|
const descriptor = getProviderDescriptor(model.provider);
|
|
111
|
-
|
|
112
|
-
|
|
113
|
-
|
|
318
|
+
// For non-local providers we ONLY use the in-process catalog's
|
|
319
|
+
// defaultBaseUrl. The registry's `baseUrl` is loaded from disk and is
|
|
320
|
+
// untrusted for routing decisions; using it in the URL would let a
|
|
321
|
+
// poisoned cache redirect outbound traffic (SSRF / request forgery).
|
|
322
|
+
// Local providers are user-configurable, so we still let the registry
|
|
323
|
+
// override the default host — but the safeUpstreamUrl() loopback check
|
|
324
|
+
// in the request handler refuses anything outside the loopback families.
|
|
325
|
+
let base;
|
|
326
|
+
if (descriptor?.isLocal) {
|
|
327
|
+
const info = registry.provider(model.provider);
|
|
328
|
+
base = (info?.baseUrl ?? descriptor.defaultBaseUrl).replace(/\/$/, "");
|
|
329
|
+
}
|
|
330
|
+
else {
|
|
331
|
+
base = (descriptor?.defaultBaseUrl ?? "").replace(/\/$/, "");
|
|
332
|
+
}
|
|
114
333
|
// Some native/local roots expose the OpenAI-compatible layer under /v1.
|
|
115
334
|
if ((model.provider === "openai" || model.provider === "ollama" || model.provider === "llama.cpp") && !base.endsWith("/v1")) {
|
|
116
335
|
return `${base}/v1/chat/completions`;
|
|
117
336
|
}
|
|
118
337
|
return `${base}/chat/completions`;
|
|
119
338
|
}
|
|
339
|
+
/**
|
|
340
|
+
* Make caller- or upstream-derived text safe to reflect into a response
|
|
341
|
+
* header: printable ASCII only, bounded length. HTTP header values must be
|
|
342
|
+
* ISO-8859-1 and undici's `Headers.set` throws on anything above U+00FF —
|
|
343
|
+
* which, if it happened after the upstream call, would turn a paid request
|
|
344
|
+
* into an unrecorded 500.
|
|
345
|
+
*/
|
|
346
|
+
function headerSafe(value, max = 400) {
|
|
347
|
+
return value.replace(/[^\x20-\x7e]/g, "?").slice(0, max);
|
|
348
|
+
}
|
|
349
|
+
/**
|
|
350
|
+
* Pass an OpenAI-compatible SSE stream through byte-for-byte while watching
|
|
351
|
+
* for a `usage` object in the events (providers emit it on the final chunk
|
|
352
|
+
* when the caller set `stream_options.include_usage`). Resolves with the last
|
|
353
|
+
* usage seen, or null when the stream never carried one — on every terminal
|
|
354
|
+
* path, including the client cancelling or the upstream erroring.
|
|
355
|
+
*/
|
|
356
|
+
function observeOpenAIStreamUsage(upstream) {
|
|
357
|
+
const decoder = new TextDecoder();
|
|
358
|
+
let buffer = "";
|
|
359
|
+
let lastUsage = null;
|
|
360
|
+
let resolve;
|
|
361
|
+
const usage = new Promise((r) => {
|
|
362
|
+
resolve = r;
|
|
363
|
+
});
|
|
364
|
+
const scan = (flushAll) => {
|
|
365
|
+
const parts = buffer.split(/\r?\n\r?\n/);
|
|
366
|
+
buffer = flushAll ? "" : (parts.pop() ?? "");
|
|
367
|
+
// Never let a stream without blank-line separators grow the buffer unbounded.
|
|
368
|
+
if (buffer.length > 65_536)
|
|
369
|
+
buffer = buffer.slice(-65_536);
|
|
370
|
+
for (const part of parts) {
|
|
371
|
+
for (const line of part.split(/\r?\n/)) {
|
|
372
|
+
if (!line.startsWith("data:"))
|
|
373
|
+
continue;
|
|
374
|
+
const data = line.slice(5).trim();
|
|
375
|
+
if (!data || data === "[DONE]")
|
|
376
|
+
continue;
|
|
377
|
+
try {
|
|
378
|
+
const evt = JSON.parse(data);
|
|
379
|
+
if (evt && typeof evt === "object" && evt.usage && typeof evt.usage === "object")
|
|
380
|
+
lastUsage = evt.usage;
|
|
381
|
+
}
|
|
382
|
+
catch {
|
|
383
|
+
// partial / non-JSON line — ignore
|
|
384
|
+
}
|
|
385
|
+
}
|
|
386
|
+
}
|
|
387
|
+
};
|
|
388
|
+
const stream = upstream.pipeThrough(new TransformStream({
|
|
389
|
+
transform(chunk, controller) {
|
|
390
|
+
controller.enqueue(chunk);
|
|
391
|
+
buffer += decoder.decode(chunk, { stream: true });
|
|
392
|
+
scan(false);
|
|
393
|
+
},
|
|
394
|
+
flush() {
|
|
395
|
+
buffer += decoder.decode();
|
|
396
|
+
scan(true);
|
|
397
|
+
resolve(lastUsage);
|
|
398
|
+
},
|
|
399
|
+
cancel() {
|
|
400
|
+
resolve(lastUsage);
|
|
401
|
+
},
|
|
402
|
+
}));
|
|
403
|
+
return { stream, usage };
|
|
404
|
+
}
|
|
120
405
|
// ---------------------------------------------------------------------------
|
|
121
406
|
// Route registration
|
|
122
407
|
// ---------------------------------------------------------------------------
|
|
123
|
-
export function registerProxyRoutes(app, registry) {
|
|
408
|
+
export function registerProxyRoutes(app, registry, shutdownSignal) {
|
|
124
409
|
// OpenAI-compatible model list — returns all models the proxy can forward.
|
|
125
410
|
// Required for SDK compatibility: OpenAI clients call this before chat requests.
|
|
126
411
|
app.get("/proxy/v1/models", (ctx) => {
|
|
@@ -148,66 +433,446 @@ export function registerProxyRoutes(app, registry) {
|
|
|
148
433
|
if (!requested) {
|
|
149
434
|
return ctx.json({ error: "missing required field: model" }, 400);
|
|
150
435
|
}
|
|
151
|
-
// ── Resolve
|
|
152
|
-
const
|
|
153
|
-
if (
|
|
436
|
+
// ── Resolve candidate routes (failover order) ──────────────────
|
|
437
|
+
const candidates = resolveProxyCandidates(registry, requested);
|
|
438
|
+
if (candidates.length === 0) {
|
|
439
|
+
// Distinguish "found but not proxiable" (422) from "not found" (404).
|
|
440
|
+
const primary = registry.model(requested);
|
|
441
|
+
if (primary && !isForwardable(primary)) {
|
|
442
|
+
const descriptor = getProviderDescriptor(primary.provider);
|
|
443
|
+
return ctx.json({
|
|
444
|
+
error: `provider '${primary.provider}' uses '${descriptor?.transport ?? "unknown"}' transport — proxy not yet supported`,
|
|
445
|
+
resolvedModel: primary.id,
|
|
446
|
+
resolvedProvider: primary.provider,
|
|
447
|
+
}, 422);
|
|
448
|
+
}
|
|
154
449
|
return ctx.json({ error: `no model found for '${requested}'` }, 404);
|
|
155
450
|
}
|
|
156
|
-
|
|
157
|
-
|
|
158
|
-
|
|
159
|
-
|
|
160
|
-
|
|
161
|
-
|
|
162
|
-
|
|
451
|
+
// `requested` is caller-controlled. Reduce it to printable ASCII and
|
|
452
|
+
// bound the length before reflecting it into a response header: the
|
|
453
|
+
// Headers API throws on CR/LF and on any non-Latin-1 character, which
|
|
454
|
+
// would turn a malformed model string into an unhandled 500.
|
|
455
|
+
const safeRequested = headerSafe(requested, 200);
|
|
456
|
+
// Optional per-tenant tag, drawn from a kosha-tenant-<name> bearer token.
|
|
457
|
+
// We don't trust the value as authentication — it just buckets ledger
|
|
458
|
+
// rows and per-tenant budgets. The proxy still resolves real upstream
|
|
459
|
+
// credentials from env / CLI files as usual.
|
|
460
|
+
const tenant = parseTenantTag(ctx.req.header("authorization"), ctx.req.header("x-kosha-tenant"));
|
|
461
|
+
// ── Budget gate ────────────────────────────────────────────────
|
|
462
|
+
// When a budget is configured we MUST fail closed: if the ledger is
|
|
463
|
+
// unreadable (permissions, FS error, partial mount) we cannot prove the
|
|
464
|
+
// caller is under budget, so we refuse. Treating the read failure as
|
|
465
|
+
// $0 spent would let an attacker bypass the cap by corrupting the
|
|
466
|
+
// ledger file.
|
|
467
|
+
// The global cap is always checked against TOTAL spend — never a
|
|
468
|
+
// tenant's slice — so a caller cannot escape it by inventing a fresh
|
|
469
|
+
// tenant tag per request. A per-tenant cap is an additional gate.
|
|
470
|
+
const budget = readMonthlyBudgetUsd();
|
|
471
|
+
const tenantBudget = tenant ? readTenantBudgetUsd() : null;
|
|
472
|
+
if (budget !== null || tenantBudget !== null) {
|
|
473
|
+
let spent = 0;
|
|
474
|
+
let tenantSpent = 0;
|
|
475
|
+
try {
|
|
476
|
+
if (budget !== null)
|
|
477
|
+
spent = await readSpendForMonth(Date.now());
|
|
478
|
+
if (tenantBudget !== null)
|
|
479
|
+
tenantSpent = await readSpendForMonth(Date.now(), tenant);
|
|
480
|
+
}
|
|
481
|
+
catch (err) {
|
|
482
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
483
|
+
return ctx.json({
|
|
484
|
+
error: "budget enforcement unavailable — ledger could not be read",
|
|
485
|
+
detail: message,
|
|
486
|
+
budgetUsd: budget ?? tenantBudget,
|
|
487
|
+
}, 503);
|
|
488
|
+
}
|
|
489
|
+
if (budget !== null && spent >= budget) {
|
|
490
|
+
return ctx.json({ error: "monthly budget exceeded", spentUsd: spent, budgetUsd: budget }, 429, {
|
|
491
|
+
"x-kosha-budget-remaining-usd": Math.max(0, budget - spent).toFixed(4),
|
|
492
|
+
"x-kosha-budget-usd": budget.toFixed(2),
|
|
493
|
+
});
|
|
494
|
+
}
|
|
495
|
+
if (tenantBudget !== null && tenantSpent >= tenantBudget) {
|
|
496
|
+
return ctx.json({ error: "tenant monthly budget exceeded", tenant, spentUsd: tenantSpent, budgetUsd: tenantBudget }, 429, {
|
|
497
|
+
"x-kosha-budget-remaining-usd": Math.max(0, tenantBudget - tenantSpent).toFixed(4),
|
|
498
|
+
"x-kosha-budget-usd": tenantBudget.toFixed(2),
|
|
499
|
+
});
|
|
500
|
+
}
|
|
163
501
|
}
|
|
164
|
-
// ── Resolve credential ─────────────────────────────────────────
|
|
165
502
|
// Use the full 5-tier CredentialResolver (CLI files, ADC, OAuth, env vars)
|
|
166
503
|
// so the proxy honours the same credential sources as discovery.
|
|
167
504
|
const resolver = await getRegistryCredentialResolver();
|
|
168
|
-
|
|
169
|
-
|
|
170
|
-
|
|
171
|
-
|
|
172
|
-
|
|
173
|
-
|
|
174
|
-
|
|
175
|
-
|
|
176
|
-
|
|
505
|
+
// ── Forward, failing over on transport / 5xx errors ────────────
|
|
506
|
+
// We try ranked candidates in order, bounding the number of actual
|
|
507
|
+
// upstream fetches so a wave of dead providers can't stall the caller.
|
|
508
|
+
// A 4xx is the caller's own fault (bad request, bad key) and is returned
|
|
509
|
+
// as-is; a 5xx or network error rolls over to the next candidate.
|
|
510
|
+
const MAX_FETCHES = 3;
|
|
511
|
+
// Hard ceiling on a single upstream call so a hung provider can't stall
|
|
512
|
+
// the caller or hold a connection open past shutdown.
|
|
513
|
+
const UPSTREAM_TIMEOUT_MS = 30_000;
|
|
514
|
+
const attemptChain = [];
|
|
515
|
+
let fetchAttempts = 0;
|
|
516
|
+
let lastNoCred = null;
|
|
517
|
+
let sawCredentialedRoute = false;
|
|
518
|
+
let lastUnsupportedWire = null;
|
|
519
|
+
const wantsStream = body.stream === true;
|
|
520
|
+
const includeUsage = wantsStream && body.stream_options?.include_usage === true;
|
|
521
|
+
for (const model of candidates) {
|
|
522
|
+
const credential = resolver
|
|
523
|
+
? await resolver.resolve(model.provider)
|
|
524
|
+
: fallbackRegistryCredential(model.provider);
|
|
525
|
+
const bearerToken = credential.apiKey ?? credential.accessToken;
|
|
526
|
+
const descriptor = getProviderDescriptor(model.provider);
|
|
527
|
+
if (descriptor && providerExecutionCredentialRequired(descriptor) && !bearerToken) {
|
|
528
|
+
const envHint = descriptor.credentialEnvVars.length
|
|
529
|
+
? descriptor.credentialEnvVars.join(" or ")
|
|
530
|
+
: descriptor.primaryCredentialEnvVar;
|
|
531
|
+
lastNoCred = { provider: model.provider, envHint };
|
|
532
|
+
attemptChain.push(`${model.provider}:no-credential`);
|
|
533
|
+
continue; // can't call this provider — try the next route
|
|
534
|
+
}
|
|
535
|
+
sawCredentialedRoute = true;
|
|
536
|
+
if (fetchAttempts >= MAX_FETCHES)
|
|
537
|
+
break;
|
|
538
|
+
const upstreamUrl = buildUpstreamUrl(model, registry);
|
|
539
|
+
// The proxy never forwards to a host outside the in-process catalog
|
|
540
|
+
// allowlist. The registry's `provider.baseUrl` is loaded from disk
|
|
541
|
+
// cache (or could come from a user config file), so a poisoned
|
|
542
|
+
// value must NOT redirect a request to an attacker-chosen host —
|
|
543
|
+
// this is the SSRF / `js/request-forgery` guard.
|
|
544
|
+
const safeUrl = safeUpstreamUrl(upstreamUrl, descriptor?.isLocal === true);
|
|
545
|
+
if (!safeUrl) {
|
|
546
|
+
attemptChain.push(`${model.provider}:untrusted-host`);
|
|
547
|
+
continue;
|
|
548
|
+
}
|
|
549
|
+
const usesAnthropicWire = model.provider === "anthropic";
|
|
550
|
+
const upstreamHeaders = { "content-type": "application/json" };
|
|
551
|
+
let upstreamBody;
|
|
552
|
+
let upstreamUrlString;
|
|
553
|
+
let wireNotes = [];
|
|
554
|
+
if (usesAnthropicWire) {
|
|
555
|
+
// Anthropic /v1/messages: auth via x-api-key + anthropic-version,
|
|
556
|
+
// body translated from OpenAI chat-completions on the way in,
|
|
557
|
+
// response (JSON or SSE) translated back on the way out. The
|
|
558
|
+
// translator is fail-safe: content it cannot carry throws
|
|
559
|
+
// UnsupportedWireContentError rather than being silently mangled;
|
|
560
|
+
// we then fail over to a native-OpenAI route for the same model
|
|
561
|
+
// (OpenRouter, etc.) and only 422 if none exists.
|
|
562
|
+
upstreamHeaders["x-api-key"] = bearerToken ?? "";
|
|
563
|
+
upstreamHeaders["anthropic-version"] = "2023-06-01";
|
|
564
|
+
let translation;
|
|
565
|
+
try {
|
|
566
|
+
translation = translateOpenAIToAnthropicWithNotes({ ...coerceOpenAIChatRequest(body), model: model.id });
|
|
567
|
+
}
|
|
568
|
+
catch (err) {
|
|
569
|
+
if (err instanceof UnsupportedWireContentError) {
|
|
570
|
+
attemptChain.push(`${model.provider}:unsupported-wire-content`);
|
|
571
|
+
lastUnsupportedWire = err.message;
|
|
572
|
+
continue;
|
|
573
|
+
}
|
|
574
|
+
throw err;
|
|
575
|
+
}
|
|
576
|
+
wireNotes = translation.notes;
|
|
577
|
+
if (model.maxOutputTokens > 0 && translation.request.max_tokens > model.maxOutputTokens) {
|
|
578
|
+
wireNotes.push(`clamped max_tokens ${translation.request.max_tokens} to ${model.maxOutputTokens}: model output cap`);
|
|
579
|
+
translation.request.max_tokens = model.maxOutputTokens;
|
|
580
|
+
}
|
|
581
|
+
upstreamBody = JSON.stringify(translation.request);
|
|
582
|
+
// Rebuild the URL via the WHATWG URL API rather than a string
|
|
583
|
+
// `replace`, so the pathname swap can't be tripped by a fragment
|
|
584
|
+
// or query-string that happens to contain `/chat/completions`.
|
|
585
|
+
// `safeUrl` is already allowlist-validated, so reusing its
|
|
586
|
+
// origin keeps the SSRF guarantee intact.
|
|
587
|
+
const anthropicUrl = new URL(safeUrl.toString());
|
|
588
|
+
anthropicUrl.pathname = "/v1/messages";
|
|
589
|
+
upstreamUrlString = anthropicUrl.toString();
|
|
590
|
+
}
|
|
591
|
+
else {
|
|
592
|
+
if (bearerToken)
|
|
593
|
+
upstreamHeaders.authorization = `Bearer ${bearerToken}`;
|
|
594
|
+
upstreamBody = JSON.stringify({ ...body, model: model.id });
|
|
595
|
+
upstreamUrlString = safeUrl.toString();
|
|
596
|
+
}
|
|
597
|
+
fetchAttempts += 1;
|
|
598
|
+
const attemptStart = Date.now();
|
|
599
|
+
// Bound the upstream call so a hung provider can't stall the caller,
|
|
600
|
+
// and abort in-flight requests when kosha is shutting down so we
|
|
601
|
+
// drain instead of dropping streams mid-flight. The timer covers
|
|
602
|
+
// headers and non-streaming bodies only: a streamed body legitimately
|
|
603
|
+
// runs for minutes on 128K-output models, so once we start relaying
|
|
604
|
+
// a stream the timer is released and only the shutdown signal (and
|
|
605
|
+
// the client hanging up) can end it.
|
|
606
|
+
const upstreamAbort = new AbortController();
|
|
607
|
+
const upstreamTimer = setTimeout(() => upstreamAbort.abort(new DOMException(`upstream did not respond within ${UPSTREAM_TIMEOUT_MS}ms`, "TimeoutError")), UPSTREAM_TIMEOUT_MS);
|
|
608
|
+
const releaseTimer = () => clearTimeout(upstreamTimer);
|
|
609
|
+
let upstream;
|
|
610
|
+
try {
|
|
611
|
+
const signal = shutdownSignal ? AbortSignal.any([upstreamAbort.signal, shutdownSignal]) : upstreamAbort.signal;
|
|
612
|
+
upstream = await fetch(upstreamUrlString, {
|
|
613
|
+
method: "POST",
|
|
614
|
+
headers: upstreamHeaders,
|
|
615
|
+
body: upstreamBody,
|
|
616
|
+
signal,
|
|
617
|
+
});
|
|
618
|
+
}
|
|
619
|
+
catch (err) {
|
|
620
|
+
releaseTimer();
|
|
621
|
+
attemptChain.push(`${model.provider}:error`);
|
|
622
|
+
const errLatency = Date.now() - attemptStart;
|
|
623
|
+
const errorType = err instanceof Error && err.name === "TimeoutError" ? "timeout" : "transport";
|
|
624
|
+
registry.recordProxyOutcome(model.provider, { ok: false, latencyMs: errLatency, errorType });
|
|
625
|
+
bumpProxyMetric(model.provider, false, errLatency);
|
|
626
|
+
if (fetchAttempts < MAX_FETCHES)
|
|
627
|
+
continue; // fail over
|
|
628
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
629
|
+
return ctx.json({ error: `upstream unreachable: ${message}`, resolvedProvider: model.provider, attemptChain }, 502);
|
|
630
|
+
}
|
|
631
|
+
// A retryable upstream failure (5xx) rolls over to the next route,
|
|
632
|
+
// unless we've spent our fetch budget — then we surface it.
|
|
633
|
+
if (upstream.status >= 500 && fetchAttempts < MAX_FETCHES) {
|
|
634
|
+
releaseTimer();
|
|
635
|
+
attemptChain.push(`${model.provider}:${upstream.status}`);
|
|
636
|
+
await upstream.body?.cancel().catch(() => { });
|
|
637
|
+
registry.recordProxyOutcome(model.provider, {
|
|
638
|
+
ok: false,
|
|
639
|
+
status: upstream.status,
|
|
640
|
+
latencyMs: Date.now() - attemptStart,
|
|
641
|
+
errorType: "transport",
|
|
642
|
+
});
|
|
643
|
+
bumpProxyMetric(model.provider, false, Date.now() - attemptStart);
|
|
644
|
+
continue;
|
|
645
|
+
}
|
|
646
|
+
attemptChain.push(`${model.provider}:${upstream.status}`);
|
|
647
|
+
// Feed the real inference outcome into the health tracker so
|
|
648
|
+
// kosha:reliable/fastest/balanced rank on actual proxy traffic, not
|
|
649
|
+
// just discovery-ping latency. A 4xx is the caller's fault and not a
|
|
650
|
+
// provider-health signal, so we neither credit nor penalize it.
|
|
651
|
+
const attemptLatencyMs = Date.now() - attemptStart;
|
|
652
|
+
if (upstream.ok) {
|
|
653
|
+
registry.recordProxyOutcome(model.provider, { ok: true, status: upstream.status, latencyMs: attemptLatencyMs });
|
|
654
|
+
bumpProxyMetric(model.provider, true, attemptLatencyMs);
|
|
655
|
+
}
|
|
656
|
+
else if (upstream.status >= 500) {
|
|
657
|
+
registry.recordProxyOutcome(model.provider, {
|
|
658
|
+
ok: false,
|
|
659
|
+
status: upstream.status,
|
|
660
|
+
latencyMs: attemptLatencyMs,
|
|
661
|
+
errorType: "transport",
|
|
662
|
+
});
|
|
663
|
+
bumpProxyMetric(model.provider, false, attemptLatencyMs);
|
|
664
|
+
}
|
|
665
|
+
// Pre-flight cost estimate: (input tokens × inputRate + expected
|
|
666
|
+
// output × outputRate), reflected on the response so the caller can
|
|
667
|
+
// audit before paying. When the upstream returns a usage block we
|
|
668
|
+
// also reconcile the actual cost and record both on the ledger.
|
|
669
|
+
const estimate = estimateRequestCost(model, body);
|
|
670
|
+
const responseHeaders = new Headers({
|
|
671
|
+
"x-kosha-model": model.id,
|
|
672
|
+
"x-kosha-provider": model.provider,
|
|
673
|
+
"x-kosha-requested": safeRequested,
|
|
674
|
+
"x-kosha-attempt-chain": headerSafe(attemptChain.join(",")),
|
|
675
|
+
});
|
|
676
|
+
if (wireNotes.length > 0)
|
|
677
|
+
responseHeaders.set("x-kosha-wire-notes", headerSafe(wireNotes.join("; ")));
|
|
678
|
+
if (estimate)
|
|
679
|
+
responseHeaders.set("x-kosha-estimated-cost-usd", estimate.estimatedUsd.toFixed(6));
|
|
680
|
+
const upstreamStatus = upstream.status;
|
|
681
|
+
const upstreamOk = upstream.ok;
|
|
682
|
+
const requestId = randomUUID();
|
|
683
|
+
// Only charge the ledger on a successful upstream response. A 4xx
|
|
684
|
+
// is the caller's fault (malformed request, bad key) and costs
|
|
685
|
+
// nothing upstream — charging it would let malformed traffic
|
|
686
|
+
// exhaust a tenant's monthly budget. Append failures are
|
|
687
|
+
// observability losses, never a reason to fail the request.
|
|
688
|
+
//
|
|
689
|
+
// The request row is written the moment we know the upstream said
|
|
690
|
+
// yes — before any body is read or relayed — so a client that
|
|
691
|
+
// disconnects mid-stream, a stream that errors, or a later header
|
|
692
|
+
// failure can never leave a paid request unrecorded. Streaming
|
|
693
|
+
// responses learn their real usage only at the end; that lands as a
|
|
694
|
+
// separate `adjustment` row carrying the delta.
|
|
695
|
+
const recordSpend = (actual) => {
|
|
696
|
+
if (!estimate || !upstreamOk)
|
|
697
|
+
return;
|
|
698
|
+
appendLedgerEntry({
|
|
699
|
+
ts: Date.now(),
|
|
700
|
+
provider: model.provider,
|
|
701
|
+
modelId: model.id,
|
|
702
|
+
requested: safeRequested,
|
|
703
|
+
tenant,
|
|
704
|
+
estimatedUsd: estimate.estimatedUsd,
|
|
705
|
+
estimatedInputTokens: estimate.inputTokens,
|
|
706
|
+
estimatedOutputTokens: estimate.expectedOutputTokens,
|
|
707
|
+
upstreamStatus,
|
|
708
|
+
kind: "request",
|
|
709
|
+
requestId,
|
|
710
|
+
...(actual
|
|
711
|
+
? {
|
|
712
|
+
actualUsd: actual.usd,
|
|
713
|
+
actualInputTokens: actual.inputTokens,
|
|
714
|
+
actualOutputTokens: actual.outputTokens,
|
|
715
|
+
cacheReadTokens: actual.cacheReadTokens,
|
|
716
|
+
cacheWriteTokens: actual.cacheWriteTokens,
|
|
717
|
+
usageSource: "upstream",
|
|
718
|
+
}
|
|
719
|
+
: { usageSource: "estimate" }),
|
|
720
|
+
}).catch(() => { });
|
|
721
|
+
};
|
|
722
|
+
const recordAdjustment = (actual) => {
|
|
723
|
+
if (!estimate || !upstreamOk || !actual)
|
|
724
|
+
return;
|
|
725
|
+
appendLedgerEntry({
|
|
726
|
+
ts: Date.now(),
|
|
727
|
+
provider: model.provider,
|
|
728
|
+
modelId: model.id,
|
|
729
|
+
requested: safeRequested,
|
|
730
|
+
tenant,
|
|
731
|
+
estimatedUsd: 0,
|
|
732
|
+
estimatedInputTokens: 0,
|
|
733
|
+
estimatedOutputTokens: 0,
|
|
734
|
+
upstreamStatus,
|
|
735
|
+
kind: "adjustment",
|
|
736
|
+
requestId,
|
|
737
|
+
adjustmentUsd: actual.usd - estimate.estimatedUsd,
|
|
738
|
+
adjustmentInputTokens: actual.inputTokens - estimate.inputTokens,
|
|
739
|
+
adjustmentOutputTokens: actual.outputTokens - estimate.expectedOutputTokens,
|
|
740
|
+
actualUsd: actual.usd,
|
|
741
|
+
actualInputTokens: actual.inputTokens,
|
|
742
|
+
actualOutputTokens: actual.outputTokens,
|
|
743
|
+
cacheReadTokens: actual.cacheReadTokens,
|
|
744
|
+
cacheWriteTokens: actual.cacheWriteTokens,
|
|
745
|
+
usageSource: "upstream",
|
|
746
|
+
}).catch(() => { });
|
|
747
|
+
};
|
|
748
|
+
const reflectActual = (actual) => {
|
|
749
|
+
if (actual) {
|
|
750
|
+
responseHeaders.set("x-kosha-actual-cost-usd", actual.usd.toFixed(6));
|
|
751
|
+
responseHeaders.set("x-kosha-usage-source", "upstream");
|
|
752
|
+
}
|
|
753
|
+
else {
|
|
754
|
+
responseHeaders.set("x-kosha-usage-source", "estimate");
|
|
755
|
+
}
|
|
756
|
+
};
|
|
757
|
+
if (usesAnthropicWire) {
|
|
758
|
+
// Error bodies are re-shaped into the OpenAI error envelope
|
|
759
|
+
// ({ error: { message, type, code } }) so a client using the
|
|
760
|
+
// OpenAI SDK reads a structured error instead of an Anthropic
|
|
761
|
+
// blob it can't parse. Anthropic returns JSON errors even for
|
|
762
|
+
// stream requests, so this branch runs before the SSE one.
|
|
763
|
+
responseHeaders.set("content-type", "application/json");
|
|
764
|
+
if (!upstream.ok) {
|
|
765
|
+
let raw;
|
|
766
|
+
try {
|
|
767
|
+
raw = await upstream.json();
|
|
768
|
+
}
|
|
769
|
+
catch {
|
|
770
|
+
raw = null;
|
|
771
|
+
}
|
|
772
|
+
releaseTimer();
|
|
773
|
+
const aErr = extractAnthropicError(raw);
|
|
774
|
+
return new Response(JSON.stringify({ error: { ...aErr, code: String(upstream.status) } }), { status: upstream.status, headers: responseHeaders });
|
|
775
|
+
}
|
|
776
|
+
if (wantsStream) {
|
|
777
|
+
if (!upstream.body) {
|
|
778
|
+
releaseTimer();
|
|
779
|
+
recordSpend(null);
|
|
780
|
+
return new Response(JSON.stringify({ error: { message: "anthropic upstream returned no stream body", type: "upstream_error" } }), { status: 502, headers: responseHeaders });
|
|
781
|
+
}
|
|
782
|
+
// Estimate now; reconcile when (if) the stream reports usage.
|
|
783
|
+
recordSpend(null);
|
|
784
|
+
const { stream, usage } = translateAnthropicStreamToOpenAI(upstream.body, model.id, { includeUsage });
|
|
785
|
+
usage.then((u) => recordAdjustment(actualCostFromUsage(model, u, "anthropic"))).catch(() => { });
|
|
786
|
+
releaseTimer();
|
|
787
|
+
responseHeaders.set("content-type", "text/event-stream; charset=utf-8");
|
|
788
|
+
responseHeaders.set("cache-control", "no-cache");
|
|
789
|
+
return new Response(stream, { status: 200, headers: responseHeaders });
|
|
790
|
+
}
|
|
791
|
+
let raw;
|
|
792
|
+
try {
|
|
793
|
+
raw = await upstream.json();
|
|
794
|
+
}
|
|
795
|
+
catch {
|
|
796
|
+
releaseTimer();
|
|
797
|
+
recordSpend(null);
|
|
798
|
+
return new Response(JSON.stringify({ error: { message: "anthropic upstream returned unparseable JSON", type: "upstream_error" } }), { status: 502, headers: responseHeaders });
|
|
799
|
+
}
|
|
800
|
+
releaseTimer();
|
|
801
|
+
const anthropicResponse = raw;
|
|
802
|
+
const actual = actualCostFromUsage(model, anthropicResponse.usage, "anthropic");
|
|
803
|
+
reflectActual(actual);
|
|
804
|
+
recordSpend(actual);
|
|
805
|
+
const translated = translateAnthropicToOpenAI(anthropicResponse, model.id);
|
|
806
|
+
return new Response(JSON.stringify(translated), { status: upstream.status, headers: responseHeaders });
|
|
807
|
+
}
|
|
808
|
+
// ── Native OpenAI-compatible passthrough ───────────────────────
|
|
809
|
+
const contentType = upstream.headers.get("content-type") ?? "";
|
|
810
|
+
if (contentType)
|
|
811
|
+
responseHeaders.set("content-type", headerSafe(contentType, 200));
|
|
812
|
+
if (!upstream.ok || !upstream.body) {
|
|
813
|
+
releaseTimer();
|
|
814
|
+
if (upstream.ok)
|
|
815
|
+
recordSpend(null);
|
|
816
|
+
return new Response(upstream.body, { status: upstream.status, headers: responseHeaders });
|
|
817
|
+
}
|
|
818
|
+
if (contentType.includes("text/event-stream")) {
|
|
819
|
+
// Bytes pass through untouched; we only watch for a trailing
|
|
820
|
+
// usage chunk (present when the caller asked for include_usage).
|
|
821
|
+
// Estimate now; reconcile when the stream ends with usage.
|
|
822
|
+
recordSpend(null);
|
|
823
|
+
const { stream, usage } = observeOpenAIStreamUsage(upstream.body);
|
|
824
|
+
usage.then((u) => recordAdjustment(actualCostFromUsage(model, u, "openai"))).catch(() => { });
|
|
825
|
+
releaseTimer();
|
|
826
|
+
return new Response(stream, { status: upstream.status, headers: responseHeaders });
|
|
827
|
+
}
|
|
828
|
+
if (contentType.includes("application/json")) {
|
|
829
|
+
let text;
|
|
830
|
+
try {
|
|
831
|
+
text = await upstream.text();
|
|
832
|
+
}
|
|
833
|
+
catch (err) {
|
|
834
|
+
releaseTimer();
|
|
835
|
+
recordSpend(null);
|
|
836
|
+
const message = err instanceof Error ? err.message : String(err);
|
|
837
|
+
return ctx.json({ error: `upstream body could not be read: ${message}`, resolvedProvider: model.provider, attemptChain }, 502);
|
|
838
|
+
}
|
|
839
|
+
releaseTimer();
|
|
840
|
+
let usageRaw = null;
|
|
841
|
+
try {
|
|
842
|
+
usageRaw = JSON.parse(text).usage ?? null;
|
|
843
|
+
}
|
|
844
|
+
catch {
|
|
845
|
+
// Not JSON after all — forward as-is with the estimate only.
|
|
846
|
+
}
|
|
847
|
+
const actual = actualCostFromUsage(model, usageRaw, "openai");
|
|
848
|
+
reflectActual(actual);
|
|
849
|
+
recordSpend(actual);
|
|
850
|
+
return new Response(text, { status: upstream.status, headers: responseHeaders });
|
|
851
|
+
}
|
|
852
|
+
releaseTimer();
|
|
853
|
+
reflectActual(null);
|
|
854
|
+
recordSpend(null);
|
|
855
|
+
return new Response(upstream.body, { status: upstream.status, headers: responseHeaders });
|
|
856
|
+
}
|
|
857
|
+
// Exhausted candidates without a returnable response.
|
|
858
|
+
if (!sawCredentialedRoute && lastNoCred) {
|
|
177
859
|
return ctx.json({
|
|
178
|
-
error: `no credential found for '${
|
|
179
|
-
hint: envHint ? `set ${envHint}` : "configure credentials for this provider",
|
|
180
|
-
|
|
181
|
-
|
|
860
|
+
error: `no credential found for '${lastNoCred.provider}'`,
|
|
861
|
+
hint: lastNoCred.envHint ? `set ${lastNoCred.envHint}` : "configure credentials for this provider",
|
|
862
|
+
resolvedProvider: lastNoCred.provider,
|
|
863
|
+
attemptChain,
|
|
182
864
|
}, 401);
|
|
183
865
|
}
|
|
184
|
-
//
|
|
185
|
-
|
|
186
|
-
|
|
187
|
-
if (
|
|
188
|
-
|
|
189
|
-
|
|
190
|
-
|
|
191
|
-
|
|
192
|
-
method: "POST",
|
|
193
|
-
headers: upstreamHeaders,
|
|
194
|
-
body: JSON.stringify({ ...body, model: model.id }),
|
|
195
|
-
});
|
|
196
|
-
}
|
|
197
|
-
catch (err) {
|
|
198
|
-
const message = err instanceof Error ? err.message : String(err);
|
|
199
|
-
return ctx.json({ error: `upstream unreachable: ${message}`, resolvedProvider: model.provider }, 502);
|
|
866
|
+
// 422 only when NO upstream was ever contacted: the translator skipped
|
|
867
|
+
// every candidate. If a native route was actually tried and failed,
|
|
868
|
+
// that failure (502) is the honest answer.
|
|
869
|
+
if (lastUnsupportedWire && fetchAttempts === 0) {
|
|
870
|
+
return ctx.json({
|
|
871
|
+
error: `request uses content the Anthropic wire translator cannot carry and no native OpenAI-compatible route was available: ${lastUnsupportedWire}`,
|
|
872
|
+
attemptChain,
|
|
873
|
+
}, 422);
|
|
200
874
|
}
|
|
201
|
-
|
|
202
|
-
const responseHeaders = new Headers({
|
|
203
|
-
"x-kosha-model": model.id,
|
|
204
|
-
"x-kosha-provider": model.provider,
|
|
205
|
-
"x-kosha-requested": requested,
|
|
206
|
-
});
|
|
207
|
-
const ct = upstream.headers.get("content-type");
|
|
208
|
-
if (ct)
|
|
209
|
-
responseHeaders.set("content-type", ct);
|
|
210
|
-
return new Response(upstream.body, { status: upstream.status, headers: responseHeaders });
|
|
875
|
+
return ctx.json({ error: "all upstream providers failed", attemptChain }, 502);
|
|
211
876
|
});
|
|
212
877
|
}
|
|
213
878
|
//# sourceMappingURL=proxy.js.map
|