opencode-ufr 0.2.0
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/LICENSE +21 -0
- package/README.md +184 -0
- package/bin/ufr.ts +4 -0
- package/models.json +206 -0
- package/package.json +16 -0
- package/src/cli/catalog.ts +45 -0
- package/src/cli/connect.ts +146 -0
- package/src/cli/daemon-client.ts +19 -0
- package/src/cli/index.ts +93 -0
- package/src/cli/io.ts +98 -0
- package/src/cli/keys.ts +95 -0
- package/src/cli/stats.ts +37 -0
- package/src/cli/status.ts +48 -0
- package/src/cli/stop.ts +8 -0
- package/src/cli/ufr-check.ts +43 -0
- package/src/daemon/breaker.ts +146 -0
- package/src/daemon/catalog-source.ts +85 -0
- package/src/daemon/catalog.ts +127 -0
- package/src/daemon/daemon.ts +397 -0
- package/src/daemon/fallbacks.ts +41 -0
- package/src/daemon/keypool.ts +109 -0
- package/src/daemon/log.ts +15 -0
- package/src/daemon/main.ts +18 -0
- package/src/daemon/model.ts +15 -0
- package/src/daemon/router.ts +392 -0
- package/src/daemon/server.ts +54 -0
- package/src/daemon/stats.ts +105 -0
- package/src/daemon/transport.ts +21 -0
- package/src/daemon/upstream.ts +68 -0
- package/src/daemon/vpn/fortinet.ts +466 -0
- package/src/daemon/vpn/ip.ts +176 -0
- package/src/daemon/vpn/manager.ts +258 -0
- package/src/daemon/vpn/ppp.ts +450 -0
- package/src/daemon/vpn/proxy.ts +102 -0
- package/src/daemon/vpn/stack.ts +140 -0
- package/src/daemon/vpn/tcp.ts +356 -0
- package/src/daemon/window.ts +126 -0
- package/src/plugin/connect.ts +113 -0
- package/src/plugin/ensure-daemon.ts +101 -0
- package/src/plugin/index.ts +93 -0
- package/src/plugin/model-info.ts +35 -0
- package/src/shared/config.ts +170 -0
- package/src/shared/connect.ts +100 -0
- package/src/shared/errors.ts +6 -0
- package/src/shared/fs.ts +31 -0
- package/src/shared/models-file.ts +68 -0
- package/src/shared/paths.ts +62 -0
- package/src/shared/secrets.ts +62 -0
- package/src/shared/sleep.ts +15 -0
- package/src/shared/time.ts +6 -0
- package/src/shared/version.ts +3 -0
- package/src/shared/vpn.ts +8 -0
|
@@ -0,0 +1,18 @@
|
|
|
1
|
+
import { resolvePaths } from "../shared/paths"
|
|
2
|
+
import { KeyringStore } from "../shared/secrets"
|
|
3
|
+
import { AlreadyRunningError, startDaemon } from "./daemon"
|
|
4
|
+
import { createLogger } from "./log"
|
|
5
|
+
|
|
6
|
+
const paths = resolvePaths()
|
|
7
|
+
const log = createLogger(paths.logFile)
|
|
8
|
+
try {
|
|
9
|
+
const d = await startDaemon({ paths, secrets: new KeyringStore(), log, onStopped: () => process.exit(0) })
|
|
10
|
+
for (const sig of ["SIGINT", "SIGTERM"] as const) process.on(sig, () => void d.stop())
|
|
11
|
+
} catch (e) {
|
|
12
|
+
if (e instanceof AlreadyRunningError) {
|
|
13
|
+
log("another daemon is already running — exiting")
|
|
14
|
+
process.exit(0)
|
|
15
|
+
}
|
|
16
|
+
log(`fatal: ${(e as Error).message}`)
|
|
17
|
+
process.exit(1)
|
|
18
|
+
}
|
|
@@ -0,0 +1,15 @@
|
|
|
1
|
+
/** USD per 1M tokens. UFR bills prefix-cached input in full (measured 2026-09-28). */
|
|
2
|
+
export type ModelPrice = { input: number; output: number; cacheRead: number; cacheWrite: number }
|
|
3
|
+
|
|
4
|
+
export type Model = {
|
|
5
|
+
id: string
|
|
6
|
+
name: string
|
|
7
|
+
tier: "free" | "paid"
|
|
8
|
+
vision: boolean
|
|
9
|
+
tools: boolean
|
|
10
|
+
context: number
|
|
11
|
+
maxOutput: number
|
|
12
|
+
price: ModelPrice | null
|
|
13
|
+
hasEntry: boolean
|
|
14
|
+
hidden: boolean
|
|
15
|
+
}
|
|
@@ -0,0 +1,392 @@
|
|
|
1
|
+
import type { Config } from "../shared/config"
|
|
2
|
+
import { errorResponse } from "../shared/errors"
|
|
3
|
+
import type { BreakerRegistry } from "./breaker"
|
|
4
|
+
import { type Catalog, resolveModel } from "./catalog"
|
|
5
|
+
import type { KeyPool } from "./keypool"
|
|
6
|
+
import { type Stats, costUsd } from "./stats"
|
|
7
|
+
import type { Transport } from "./transport"
|
|
8
|
+
import { type UpstreamResult, callUpstream } from "./upstream"
|
|
9
|
+
import type { SlidingWindow } from "./window"
|
|
10
|
+
|
|
11
|
+
export type RouterDeps = {
|
|
12
|
+
config: Config
|
|
13
|
+
catalog: () => Catalog
|
|
14
|
+
keys: KeyPool
|
|
15
|
+
pool: SlidingWindow
|
|
16
|
+
breakers: BreakerRegistry
|
|
17
|
+
stats: Stats
|
|
18
|
+
transport: Transport
|
|
19
|
+
now: () => number
|
|
20
|
+
sleep: (ms: number, signal?: AbortSignal) => Promise<void>
|
|
21
|
+
onUpstream?: (ok: boolean, message: string) => void
|
|
22
|
+
}
|
|
23
|
+
|
|
24
|
+
type Usage = { prompt: number; completion: number }
|
|
25
|
+
type LoopOk = { ok: true; response: Response; model: string; keyAlias: string; attempts: number; upstreamAbort?: AbortController }
|
|
26
|
+
type LoopFail = { ok: false; response: Response; model: string; keyAlias: string | null; attempts: number; errorType: string }
|
|
27
|
+
|
|
28
|
+
export function usageOf(json: unknown): Usage | null {
|
|
29
|
+
const u = (json as { usage?: { prompt_tokens?: number; completion_tokens?: number } } | null)?.usage
|
|
30
|
+
return u ? { prompt: u.prompt_tokens ?? 0, completion: u.completion_tokens ?? 0 } : null
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
function addUsage(a: Usage | null, b: Usage | null): Usage | null {
|
|
34
|
+
if (!a) return b
|
|
35
|
+
if (!b) return a
|
|
36
|
+
return { prompt: a.prompt + b.prompt, completion: a.completion + b.completion }
|
|
37
|
+
}
|
|
38
|
+
|
|
39
|
+
/** Sums two per-call costs; null only when neither call's model had a known price. */
|
|
40
|
+
function addCost(a: number | null, b: number | null): number | null {
|
|
41
|
+
if (a === null && b === null) return null
|
|
42
|
+
return (a ?? 0) + (b ?? 0)
|
|
43
|
+
}
|
|
44
|
+
|
|
45
|
+
type Choice = { finish_reason?: string; message?: { content?: string | null; reasoning_content?: string | null; reasoning?: string | null } }
|
|
46
|
+
|
|
47
|
+
/** glm-5.2 can spend its whole budget thinking and return no text (observed 2026-09-08). */
|
|
48
|
+
export function isReasoningStarved(json: unknown): boolean {
|
|
49
|
+
const c = (json as { choices?: Choice[] } | null)?.choices?.[0]
|
|
50
|
+
if (!c || c.finish_reason !== "length") return false
|
|
51
|
+
const content = (c.message?.content ?? "").trim()
|
|
52
|
+
const reasoning = (c.message?.reasoning_content ?? c.message?.reasoning ?? "").trim()
|
|
53
|
+
return content === "" && reasoning !== ""
|
|
54
|
+
}
|
|
55
|
+
|
|
56
|
+
export class Router {
|
|
57
|
+
private active = 0
|
|
58
|
+
|
|
59
|
+
constructor(private readonly d: RouterDeps) {}
|
|
60
|
+
|
|
61
|
+
/** Client requests being answered right now (streams count until they end). */
|
|
62
|
+
get inFlight(): number {
|
|
63
|
+
return this.active
|
|
64
|
+
}
|
|
65
|
+
|
|
66
|
+
async handleChat(body: Record<string, unknown>, signal?: AbortSignal): Promise<Response> {
|
|
67
|
+
this.active++
|
|
68
|
+
try {
|
|
69
|
+
return await this.handle(body, signal)
|
|
70
|
+
} finally {
|
|
71
|
+
this.active--
|
|
72
|
+
}
|
|
73
|
+
}
|
|
74
|
+
|
|
75
|
+
private async handle(body: Record<string, unknown>, signal?: AbortSignal): Promise<Response> {
|
|
76
|
+
const t0 = this.d.now()
|
|
77
|
+
const requested = typeof body.model === "string" ? body.model.trim() : ""
|
|
78
|
+
if (!requested) return errorResponse(400, "invalid_request", "body.model is required")
|
|
79
|
+
const cat = this.d.catalog()
|
|
80
|
+
const group = resolveModel(cat, requested)
|
|
81
|
+
const chain = [group, ...(cat.chains.get(group) ?? [])]
|
|
82
|
+
|
|
83
|
+
const target = this.d.breakers.firstAvailable(chain)
|
|
84
|
+
if (!target) {
|
|
85
|
+
return this.fail(t0, group, null, 0, false, "upstream_circuit_open",
|
|
86
|
+
errorResponse(429, "upstream_circuit_open",
|
|
87
|
+
`UFR is refusing ${group} and every fallback right now; the gateway waits instead of hammering it`,
|
|
88
|
+
this.d.breakers.minRetryAfterMs(chain)))
|
|
89
|
+
}
|
|
90
|
+
|
|
91
|
+
const slot = this.d.pool.reserve()
|
|
92
|
+
if (slot.verdict === "reject") {
|
|
93
|
+
// target's half-open probe (if any) was consumed by firstAvailable above but
|
|
94
|
+
// never gets an outcome — resolve it so the ladder doesn't escalate on a probe timeout.
|
|
95
|
+
this.d.breakers.get(target).onOtherFailure()
|
|
96
|
+
return this.fail(t0, group, null, 0, false, "upstream_pool_cap",
|
|
97
|
+
errorResponse(429, "upstream_pool_cap", `hourly cap of ${this.d.config.limits.poolPerHour} requests reached`, slot.waitMs))
|
|
98
|
+
}
|
|
99
|
+
if (slot.waitMs > 0) {
|
|
100
|
+
try {
|
|
101
|
+
await this.d.sleep(slot.waitMs, signal)
|
|
102
|
+
} catch {
|
|
103
|
+
this.d.pool.release(slot.at)
|
|
104
|
+
this.d.breakers.get(target).onOtherFailure()
|
|
105
|
+
return this.fail(t0, group, null, 0, false, "client_closed",
|
|
106
|
+
errorResponse(499, "client_closed", "client went away while waiting for a slot"))
|
|
107
|
+
}
|
|
108
|
+
}
|
|
109
|
+
|
|
110
|
+
const stream = body.stream === true
|
|
111
|
+
const max = this.d.config.limits.maxUpstreamAttempts
|
|
112
|
+
const res = await this.loop(body, group, target, stream, max, signal)
|
|
113
|
+
if (!res.ok) return this.fail(slot.at, res.model, res.keyAlias, res.attempts, true, res.errorType, res.response)
|
|
114
|
+
if (stream) return this.streamOut(slot.at, res)
|
|
115
|
+
return this.jsonOut(slot.at, body, group, res, max, signal)
|
|
116
|
+
}
|
|
117
|
+
|
|
118
|
+
/** Upstream calls for one client request: other key, next model, context hub. */
|
|
119
|
+
private async loop(
|
|
120
|
+
body: Record<string, unknown>,
|
|
121
|
+
group: string,
|
|
122
|
+
start: string,
|
|
123
|
+
stream: boolean,
|
|
124
|
+
maxAttempts: number,
|
|
125
|
+
signal?: AbortSignal,
|
|
126
|
+
): Promise<LoopOk | LoopFail> {
|
|
127
|
+
const { config, keys, breakers } = this.d
|
|
128
|
+
const cat = this.d.catalog()
|
|
129
|
+
let model = start
|
|
130
|
+
let attempts = 0
|
|
131
|
+
let lastKey: string | null = null
|
|
132
|
+
let rateLimitedHere = false
|
|
133
|
+
let contextHopped = false
|
|
134
|
+
const tried = new Set<string>()
|
|
135
|
+
const counted = new Set<string>()
|
|
136
|
+
const giveUp = (m: string) => {
|
|
137
|
+
if (counted.has(m)) return
|
|
138
|
+
counted.add(m)
|
|
139
|
+
breakers.get(m).onRateLimited() // once per client request per group
|
|
140
|
+
}
|
|
141
|
+
const moveOn = (): boolean => {
|
|
142
|
+
const next = this.nextModel(model, group)
|
|
143
|
+
if (!next) return false
|
|
144
|
+
model = next
|
|
145
|
+
tried.clear()
|
|
146
|
+
rateLimitedHere = false
|
|
147
|
+
return true
|
|
148
|
+
}
|
|
149
|
+
const fail = (status: number, type: string, message: string, retryAfterMs?: number): LoopFail => ({
|
|
150
|
+
ok: false,
|
|
151
|
+
response: errorResponse(status, type, message, retryAfterMs),
|
|
152
|
+
model,
|
|
153
|
+
keyAlias: lastKey,
|
|
154
|
+
attempts,
|
|
155
|
+
errorType: type,
|
|
156
|
+
})
|
|
157
|
+
|
|
158
|
+
while (attempts < maxAttempts) {
|
|
159
|
+
if (signal?.aborted) {
|
|
160
|
+
breakers.get(model).onOtherFailure() // model's probe (if any) got no outcome
|
|
161
|
+
return fail(499, "client_closed", "client went away")
|
|
162
|
+
}
|
|
163
|
+
const k = keys.acquire(tried)
|
|
164
|
+
if (k.kind === "none") {
|
|
165
|
+
if (k.reason === "no_keys") {
|
|
166
|
+
breakers.get(model).onOtherFailure() // model's probe (if any) got no outcome
|
|
167
|
+
return fail(401, "no_keys", "no valid UFR API key configured — run `ufr keys add <alias>`")
|
|
168
|
+
}
|
|
169
|
+
if (k.reason === "exhausted") {
|
|
170
|
+
breakers.get(model).onOtherFailure() // model's probe (if any) got no outcome
|
|
171
|
+
return fail(429, "key_pool_exhausted",
|
|
172
|
+
`every UFR key is at its limit of ${config.limits.keyRpm} requests per ${config.limits.keyWindowS} s`, k.retryAfterMs)
|
|
173
|
+
}
|
|
174
|
+
// all_tried: every usable key already failed on this model
|
|
175
|
+
if (rateLimitedHere) giveUp(model)
|
|
176
|
+
if (!moveOn()) break
|
|
177
|
+
continue
|
|
178
|
+
}
|
|
179
|
+
if (k.waitMs > 0) {
|
|
180
|
+
try {
|
|
181
|
+
await this.d.sleep(k.waitMs, signal)
|
|
182
|
+
} catch {
|
|
183
|
+
keys.release(k.alias, k.at)
|
|
184
|
+
breakers.get(model).onOtherFailure() // model's probe (if any) got no outcome
|
|
185
|
+
return fail(499, "client_closed", "client went away while waiting for a key")
|
|
186
|
+
}
|
|
187
|
+
}
|
|
188
|
+
attempts++
|
|
189
|
+
tried.add(k.alias)
|
|
190
|
+
lastKey = k.alias
|
|
191
|
+
const upstreamBody: Record<string, unknown> = { ...body, model }
|
|
192
|
+
if (stream) {
|
|
193
|
+
upstreamBody.stream_options = { ...((body.stream_options as Record<string, unknown> | undefined) ?? {}), include_usage: true }
|
|
194
|
+
}
|
|
195
|
+
// Bun's fetch does not propagate a body reader's cancel() into aborting the
|
|
196
|
+
// underlying request — give streaming attempts their own controller so
|
|
197
|
+
// streamOut can actually cancel UFR when the client goes away.
|
|
198
|
+
const upstreamAbort = stream ? new AbortController() : null
|
|
199
|
+
const attemptSignal = upstreamAbort ? (signal ? AbortSignal.any([signal, upstreamAbort.signal]) : upstreamAbort.signal) : signal
|
|
200
|
+
let r: UpstreamResult
|
|
201
|
+
try {
|
|
202
|
+
r = await callUpstream({
|
|
203
|
+
transport: this.d.transport,
|
|
204
|
+
baseUrl: config.upstream.baseUrl,
|
|
205
|
+
key: k.secret,
|
|
206
|
+
body: upstreamBody,
|
|
207
|
+
timeoutMs: config.upstream.requestTimeoutS * 1000,
|
|
208
|
+
signal: attemptSignal,
|
|
209
|
+
stream,
|
|
210
|
+
})
|
|
211
|
+
} catch (e) {
|
|
212
|
+
breakers.get(model).onOtherFailure() // model's probe (if any) got no outcome
|
|
213
|
+
if (signal?.aborted) return fail(499, "client_closed", "client went away")
|
|
214
|
+
// Not the client: something else escaped callUpstream (e.g. the connection
|
|
215
|
+
// dropped while the error body was read) — an upstream failure.
|
|
216
|
+
return fail(503, "transport_unreachable", `UFR call failed (${e instanceof Error ? e.message : String(e)})`)
|
|
217
|
+
}
|
|
218
|
+
switch (r.kind) {
|
|
219
|
+
case "ok":
|
|
220
|
+
breakers.get(model).onSuccess()
|
|
221
|
+
this.d.onUpstream?.(true, "")
|
|
222
|
+
return { ok: true, response: r.response, model, keyAlias: k.alias, attempts, upstreamAbort: upstreamAbort ?? undefined }
|
|
223
|
+
case "rate_limited":
|
|
224
|
+
keys.onRateLimited(k.alias, model)
|
|
225
|
+
rateLimitedHere = true
|
|
226
|
+
if (tried.size >= Math.min(2, keys.size)) {
|
|
227
|
+
giveUp(model)
|
|
228
|
+
// At the attempt cap, don't advance to (and consume the probe of) a model
|
|
229
|
+
// that will never actually be called — let the while-condition end the loop
|
|
230
|
+
// and report the failure under the model that was really tried.
|
|
231
|
+
if (attempts < maxAttempts && !moveOn()) {
|
|
232
|
+
return fail(429, "upstream_rate_limited", `UFR rate-limited ${group} on every key and fallback tried`)
|
|
233
|
+
}
|
|
234
|
+
}
|
|
235
|
+
continue
|
|
236
|
+
case "auth_invalid":
|
|
237
|
+
keys.onInvalid(k.alias)
|
|
238
|
+
continue
|
|
239
|
+
case "context_overflow": {
|
|
240
|
+
// UFR answered this model with its own 400 — not a wall, so this resolves
|
|
241
|
+
// its probe (if any) as evidence rather than leaving it to time out.
|
|
242
|
+
breakers.get(model).onOtherFailure()
|
|
243
|
+
const hub = cat.contextChains.get(model)?.[0]
|
|
244
|
+
// The hub has its own breaker; at the attempt cap it would never be called,
|
|
245
|
+
// so don't take (and wedge) its half-open probe.
|
|
246
|
+
if (hub && !contextHopped && attempts < maxAttempts && breakers.get(hub).allow().allowed) {
|
|
247
|
+
contextHopped = true
|
|
248
|
+
model = hub
|
|
249
|
+
tried.clear()
|
|
250
|
+
rateLimitedHere = false
|
|
251
|
+
continue
|
|
252
|
+
}
|
|
253
|
+
return { ok: false, response: new Response(r.body, { status: r.status, headers: { "content-type": "application/json" } }),
|
|
254
|
+
model, keyAlias: k.alias, attempts, errorType: "context_overflow" }
|
|
255
|
+
}
|
|
256
|
+
case "unreachable":
|
|
257
|
+
breakers.get(model).onOtherFailure()
|
|
258
|
+
this.d.onUpstream?.(false, r.message)
|
|
259
|
+
return fail(503, "transport_unreachable", r.message)
|
|
260
|
+
case "error":
|
|
261
|
+
breakers.get(model).onOtherFailure()
|
|
262
|
+
return { ok: false, response: new Response(r.body, { status: r.status, headers: { "content-type": r.contentType } }),
|
|
263
|
+
model, keyAlias: k.alias, attempts, errorType: `upstream_${r.status}` }
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
if (rateLimitedHere) giveUp(model)
|
|
267
|
+
return fail(429, "upstream_rate_limited", `UFR rate-limited ${group}; gave up after ${attempts} upstream calls`)
|
|
268
|
+
}
|
|
269
|
+
|
|
270
|
+
private nextModel(current: string, group: string): string | null {
|
|
271
|
+
const cat = this.d.catalog()
|
|
272
|
+
const chain = [group, ...(cat.chains.get(group) ?? [])]
|
|
273
|
+
const i = chain.indexOf(current)
|
|
274
|
+
if (i < 0) return null
|
|
275
|
+
for (const m of chain.slice(i + 1)) if (this.d.breakers.get(m).allow().allowed) return m
|
|
276
|
+
return null
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
private async jsonOut(ts: number, body: Record<string, unknown>, group: string, res: LoopOk, max: number, signal?: AbortSignal): Promise<Response> {
|
|
280
|
+
let json: unknown = await res.response.json()
|
|
281
|
+
let usage = usageOf(json)
|
|
282
|
+
let cost = usage ? costUsd(this.priceOf(res.model), usage.prompt, usage.completion) : null
|
|
283
|
+
let attempts = res.attempts
|
|
284
|
+
let model = res.model
|
|
285
|
+
let keyAlias = res.keyAlias
|
|
286
|
+
if (isReasoningStarved(json) && attempts < max) {
|
|
287
|
+
const retry = await this.loop({ ...body, reasoning_effort: "none" }, group, res.model, false, max - attempts, signal)
|
|
288
|
+
attempts += retry.attempts
|
|
289
|
+
if (retry.ok) {
|
|
290
|
+
const again: unknown = await retry.response.json()
|
|
291
|
+
const u2 = usageOf(again)
|
|
292
|
+
// Price each call at the model it was actually answered by — a retry that
|
|
293
|
+
// falls back must not price the first call at the fallback's rate.
|
|
294
|
+
cost = addCost(cost, u2 ? costUsd(this.priceOf(retry.model), u2.prompt, u2.completion) : null)
|
|
295
|
+
usage = addUsage(usage, u2)
|
|
296
|
+
if (!isReasoningStarved(again)) {
|
|
297
|
+
json = again
|
|
298
|
+
model = retry.model
|
|
299
|
+
keyAlias = retry.keyAlias
|
|
300
|
+
}
|
|
301
|
+
}
|
|
302
|
+
}
|
|
303
|
+
this.record(ts, model, keyAlias, 200, usage, attempts, null, true, cost)
|
|
304
|
+
return Response.json(json)
|
|
305
|
+
}
|
|
306
|
+
|
|
307
|
+
private priceOf(model: string) {
|
|
308
|
+
return this.d.catalog().models.get(model)?.price ?? null
|
|
309
|
+
}
|
|
310
|
+
|
|
311
|
+
/** Byte-for-byte pass-through; records the usage chunk; cancelling cancels UFR. */
|
|
312
|
+
private streamOut(ts: number, res: LoopOk): Response {
|
|
313
|
+
const reader = res.response.body!.getReader()
|
|
314
|
+
const decoder = new TextDecoder()
|
|
315
|
+
let buf = ""
|
|
316
|
+
let usage: Usage | null = null
|
|
317
|
+
let finished = false
|
|
318
|
+
this.active++ // handleChat's own count ends when it returns; the stream keeps one until it is done
|
|
319
|
+
const finish = (errorType: string | null) => {
|
|
320
|
+
if (finished) return
|
|
321
|
+
finished = true
|
|
322
|
+
this.active--
|
|
323
|
+
this.record(ts, res.model, res.keyAlias, 200, usage, res.attempts, errorType, true)
|
|
324
|
+
}
|
|
325
|
+
const scan = (chunk: Uint8Array) => {
|
|
326
|
+
buf += decoder.decode(chunk, { stream: true })
|
|
327
|
+
let nl: number
|
|
328
|
+
while ((nl = buf.indexOf("\n")) >= 0) {
|
|
329
|
+
const line = buf.slice(0, nl).trim()
|
|
330
|
+
buf = buf.slice(nl + 1)
|
|
331
|
+
if (!line.startsWith("data:")) continue
|
|
332
|
+
const data = line.slice(5).trim()
|
|
333
|
+
if (!data || data === "[DONE]") continue
|
|
334
|
+
try {
|
|
335
|
+
const u = usageOf(JSON.parse(data))
|
|
336
|
+
if (u) usage = u
|
|
337
|
+
} catch {
|
|
338
|
+
// not JSON: pass it through untouched
|
|
339
|
+
}
|
|
340
|
+
}
|
|
341
|
+
}
|
|
342
|
+
const out = new ReadableStream<Uint8Array>({
|
|
343
|
+
pull: async (ctl) => {
|
|
344
|
+
try {
|
|
345
|
+
const { value, done } = await reader.read()
|
|
346
|
+
if (done) {
|
|
347
|
+
finish(null)
|
|
348
|
+
ctl.close()
|
|
349
|
+
return
|
|
350
|
+
}
|
|
351
|
+
scan(value)
|
|
352
|
+
ctl.enqueue(value)
|
|
353
|
+
} catch (e) {
|
|
354
|
+
finish("stream_error")
|
|
355
|
+
ctl.error(e)
|
|
356
|
+
}
|
|
357
|
+
},
|
|
358
|
+
cancel: async (reason) => {
|
|
359
|
+
finish("client_closed")
|
|
360
|
+
res.upstreamAbort?.abort() // reader.cancel() alone does not abort the upstream fetch on Bun
|
|
361
|
+
await reader.cancel(reason).catch(() => {})
|
|
362
|
+
},
|
|
363
|
+
})
|
|
364
|
+
return new Response(out, { status: 200, headers: { "content-type": "text/event-stream; charset=utf-8", "cache-control": "no-cache" } })
|
|
365
|
+
}
|
|
366
|
+
|
|
367
|
+
/** costOverride, when passed, is used as-is (e.g. a reasoning retry's own per-call sum);
|
|
368
|
+
* otherwise cost is derived from `model`'s price, as for every other single-call outcome. */
|
|
369
|
+
private record(ts: number, model: string, keyAlias: string | null, status: number, usage: Usage | null,
|
|
370
|
+
attempts: number, errorType: string | null, poolAdmitted: boolean, costOverride?: number | null): void {
|
|
371
|
+
const cost = costOverride !== undefined ? costOverride : (usage ? costUsd(this.priceOf(model), usage.prompt, usage.completion) : null)
|
|
372
|
+
this.d.stats.record({
|
|
373
|
+
ts,
|
|
374
|
+
model,
|
|
375
|
+
keyAlias,
|
|
376
|
+
status,
|
|
377
|
+
promptTokens: usage?.prompt ?? 0,
|
|
378
|
+
completionTokens: usage?.completion ?? 0,
|
|
379
|
+
costUsd: cost,
|
|
380
|
+
latencyMs: Math.max(0, this.d.now() - ts),
|
|
381
|
+
attempts,
|
|
382
|
+
errorType,
|
|
383
|
+
poolAdmitted,
|
|
384
|
+
})
|
|
385
|
+
}
|
|
386
|
+
|
|
387
|
+
private fail(ts: number, model: string, keyAlias: string | null, attempts: number, poolAdmitted: boolean,
|
|
388
|
+
errorType: string, response: Response): Response {
|
|
389
|
+
this.record(ts, model, keyAlias, response.status, null, attempts, errorType, poolAdmitted)
|
|
390
|
+
return response
|
|
391
|
+
}
|
|
392
|
+
}
|
|
@@ -0,0 +1,54 @@
|
|
|
1
|
+
import { errorResponse } from "../shared/errors"
|
|
2
|
+
import type { Router } from "./router"
|
|
3
|
+
|
|
4
|
+
export type ServerDeps = {
|
|
5
|
+
port: number
|
|
6
|
+
token: string
|
|
7
|
+
version: string
|
|
8
|
+
router: Router
|
|
9
|
+
models: () => unknown[]
|
|
10
|
+
status: () => unknown
|
|
11
|
+
onActivity: () => void
|
|
12
|
+
onShutdown: () => void
|
|
13
|
+
}
|
|
14
|
+
|
|
15
|
+
export function startServer(d: ServerDeps): { port: number; stop: () => void } {
|
|
16
|
+
const server = Bun.serve({
|
|
17
|
+
hostname: "127.0.0.1",
|
|
18
|
+
port: d.port,
|
|
19
|
+
idleTimeout: 30,
|
|
20
|
+
fetch: async (req, srv) => {
|
|
21
|
+
const url = new URL(req.url)
|
|
22
|
+
if (req.method === "GET" && url.pathname === "/health") {
|
|
23
|
+
return Response.json({ ok: true, version: d.version, pid: process.pid })
|
|
24
|
+
}
|
|
25
|
+
if (req.headers.get("authorization") !== `Bearer ${d.token}`) {
|
|
26
|
+
return errorResponse(401, "unauthorized", "missing or wrong local gateway token")
|
|
27
|
+
}
|
|
28
|
+
d.onActivity()
|
|
29
|
+
const route = `${req.method} ${url.pathname}`
|
|
30
|
+
if (route === "GET /v1/models") return Response.json({ object: "list", data: d.models() })
|
|
31
|
+
if (route === "POST /v1/chat/completions") {
|
|
32
|
+
srv.timeout(req, 0) // glm-5.2 can think for minutes before the first byte
|
|
33
|
+
let body: unknown
|
|
34
|
+
try {
|
|
35
|
+
body = await req.json()
|
|
36
|
+
} catch {
|
|
37
|
+
return errorResponse(400, "invalid_request", "body must be JSON")
|
|
38
|
+
}
|
|
39
|
+
if (typeof body !== "object" || body === null || Array.isArray(body)) {
|
|
40
|
+
return errorResponse(400, "invalid_request", "body must be a JSON object")
|
|
41
|
+
}
|
|
42
|
+
return d.router.handleChat(body as Record<string, unknown>, req.signal)
|
|
43
|
+
}
|
|
44
|
+
if (route === "POST /v1/_client/heartbeat") return Response.json({ ok: true })
|
|
45
|
+
if (route === "GET /v1/_status") return Response.json(d.status())
|
|
46
|
+
if (route === "POST /v1/_shutdown") {
|
|
47
|
+
setTimeout(d.onShutdown, 20)
|
|
48
|
+
return Response.json({ ok: true })
|
|
49
|
+
}
|
|
50
|
+
return errorResponse(404, "not_found", `no route ${route}`)
|
|
51
|
+
},
|
|
52
|
+
})
|
|
53
|
+
return { port: server.port as number, stop: () => server.stop(true) }
|
|
54
|
+
}
|
|
@@ -0,0 +1,105 @@
|
|
|
1
|
+
import { Database } from "bun:sqlite"
|
|
2
|
+
import { mkdirSync } from "node:fs"
|
|
3
|
+
import { dirname } from "node:path"
|
|
4
|
+
import type { ModelPrice } from "./model"
|
|
5
|
+
|
|
6
|
+
export type RequestRow = {
|
|
7
|
+
ts: number
|
|
8
|
+
model: string
|
|
9
|
+
keyAlias: string | null
|
|
10
|
+
status: number
|
|
11
|
+
promptTokens: number
|
|
12
|
+
completionTokens: number
|
|
13
|
+
costUsd: number | null
|
|
14
|
+
latencyMs: number
|
|
15
|
+
attempts: number
|
|
16
|
+
errorType: string | null
|
|
17
|
+
poolAdmitted: boolean
|
|
18
|
+
}
|
|
19
|
+
|
|
20
|
+
export type SummaryRow = {
|
|
21
|
+
name: string
|
|
22
|
+
requests: number
|
|
23
|
+
errors: number
|
|
24
|
+
promptTokens: number
|
|
25
|
+
completionTokens: number
|
|
26
|
+
costUsd: number
|
|
27
|
+
unpriced: number
|
|
28
|
+
}
|
|
29
|
+
|
|
30
|
+
/** Prompt tokens are all billed at the input price — UFR has no cache discount. */
|
|
31
|
+
export function costUsd(price: ModelPrice | null, promptTokens: number, completionTokens: number): number | null {
|
|
32
|
+
if (!price) return null
|
|
33
|
+
return (promptTokens * price.input + completionTokens * price.output) / 1_000_000
|
|
34
|
+
}
|
|
35
|
+
|
|
36
|
+
export class Stats {
|
|
37
|
+
private readonly db: Database
|
|
38
|
+
|
|
39
|
+
constructor(path: string) {
|
|
40
|
+
if (path !== ":memory:") mkdirSync(dirname(path), { recursive: true })
|
|
41
|
+
this.db = new Database(path, { create: true })
|
|
42
|
+
if (path !== ":memory:") this.db.exec("PRAGMA journal_mode = WAL")
|
|
43
|
+
this.db.exec(`CREATE TABLE IF NOT EXISTS requests (
|
|
44
|
+
ts INTEGER NOT NULL, model TEXT NOT NULL, key_alias TEXT, status INTEGER NOT NULL,
|
|
45
|
+
prompt_tokens INTEGER NOT NULL, completion_tokens INTEGER NOT NULL, cost_usd REAL,
|
|
46
|
+
latency_ms INTEGER NOT NULL, attempts INTEGER NOT NULL, error_type TEXT, pool_admitted INTEGER NOT NULL)`)
|
|
47
|
+
this.db.exec("CREATE INDEX IF NOT EXISTS requests_ts ON requests(ts)")
|
|
48
|
+
this.db.exec("CREATE TABLE IF NOT EXISTS kv (k TEXT PRIMARY KEY, v TEXT NOT NULL)")
|
|
49
|
+
}
|
|
50
|
+
|
|
51
|
+
record(r: RequestRow): void {
|
|
52
|
+
this.db
|
|
53
|
+
.query(
|
|
54
|
+
`INSERT INTO requests (ts, model, key_alias, status, prompt_tokens, completion_tokens, cost_usd,
|
|
55
|
+
latency_ms, attempts, error_type, pool_admitted) VALUES (?, ?, ?, ?, ?, ?, ?, ?, ?, ?, ?)`,
|
|
56
|
+
)
|
|
57
|
+
.run(r.ts, r.model, r.keyAlias, r.status, r.promptTokens, r.completionTokens, r.costUsd,
|
|
58
|
+
Math.round(r.latencyMs), r.attempts, r.errorType, r.poolAdmitted ? 1 : 0)
|
|
59
|
+
}
|
|
60
|
+
|
|
61
|
+
/** Admission times for rebuilding the pool window after a restart. */
|
|
62
|
+
poolAdmissionsSince(sinceTs: number): number[] {
|
|
63
|
+
const rows = this.db.query("SELECT ts FROM requests WHERE pool_admitted = 1 AND ts > ? ORDER BY ts").all(sinceTs) as { ts: number }[]
|
|
64
|
+
return rows.map((r) => r.ts)
|
|
65
|
+
}
|
|
66
|
+
|
|
67
|
+
spendByKeySince(sinceTs: number): Record<string, number> {
|
|
68
|
+
const rows = this.db
|
|
69
|
+
.query("SELECT key_alias AS k, COALESCE(SUM(cost_usd), 0) AS c FROM requests WHERE ts >= ? AND key_alias IS NOT NULL GROUP BY key_alias")
|
|
70
|
+
.all(sinceTs) as { k: string; c: number }[]
|
|
71
|
+
return Object.fromEntries(rows.map((r) => [r.k, r.c]))
|
|
72
|
+
}
|
|
73
|
+
|
|
74
|
+
summary(sinceTs: number): { byModel: SummaryRow[]; byKey: SummaryRow[] } {
|
|
75
|
+
const q = (col: string) =>
|
|
76
|
+
this.db
|
|
77
|
+
.query(
|
|
78
|
+
`SELECT ${col} AS name, COUNT(*) AS requests,
|
|
79
|
+
SUM(CASE WHEN status >= 400 THEN 1 ELSE 0 END) AS errors,
|
|
80
|
+
SUM(prompt_tokens) AS promptTokens, SUM(completion_tokens) AS completionTokens,
|
|
81
|
+
COALESCE(SUM(cost_usd), 0) AS costUsd,
|
|
82
|
+
SUM(CASE WHEN cost_usd IS NULL AND status < 400 THEN 1 ELSE 0 END) AS unpriced
|
|
83
|
+
FROM requests WHERE ts >= ? GROUP BY ${col} ORDER BY costUsd DESC, requests DESC`,
|
|
84
|
+
)
|
|
85
|
+
.all(sinceTs) as SummaryRow[]
|
|
86
|
+
return { byModel: q("model"), byKey: q("COALESCE(key_alias, '-')") }
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
prune(beforeTs: number): number {
|
|
90
|
+
return this.db.query("DELETE FROM requests WHERE ts < ?").run(beforeTs).changes
|
|
91
|
+
}
|
|
92
|
+
|
|
93
|
+
setKv(k: string, v: string): void {
|
|
94
|
+
this.db.query("INSERT INTO kv (k, v) VALUES (?, ?) ON CONFLICT(k) DO UPDATE SET v = excluded.v").run(k, v)
|
|
95
|
+
}
|
|
96
|
+
|
|
97
|
+
getKv(k: string): string | null {
|
|
98
|
+
const r = this.db.query("SELECT v FROM kv WHERE k = ?").get(k) as { v: string } | null
|
|
99
|
+
return r?.v ?? null
|
|
100
|
+
}
|
|
101
|
+
|
|
102
|
+
close(): void {
|
|
103
|
+
this.db.close()
|
|
104
|
+
}
|
|
105
|
+
}
|
|
@@ -0,0 +1,21 @@
|
|
|
1
|
+
import type { Config } from "../shared/config"
|
|
2
|
+
import type { VpnManager } from "./vpn/manager"
|
|
3
|
+
|
|
4
|
+
/**
|
|
5
|
+
* How the daemon reaches UFR. "direct" only uses the OS network (user is on
|
|
6
|
+
* the campus network or their own VPN). "auto" routes through the built-in
|
|
7
|
+
* Fortinet tunnel (userspace TCP stack — no TUN, no routes, no admin rights)
|
|
8
|
+
* whenever UFR is not reachable directly.
|
|
9
|
+
*/
|
|
10
|
+
export type Transport = { name: string; fetch(url: string, init?: RequestInit): Promise<Response> }
|
|
11
|
+
|
|
12
|
+
export const directTransport: Transport = { name: "direct", fetch: (url, init) => fetch(url, init) }
|
|
13
|
+
|
|
14
|
+
export function createTransport(cfg: Config["transport"], o: { vpn?: VpnManager | null } = {}): Transport {
|
|
15
|
+
switch (cfg.type) {
|
|
16
|
+
case "direct":
|
|
17
|
+
return directTransport
|
|
18
|
+
case "auto":
|
|
19
|
+
return o.vpn?.transport() ?? directTransport
|
|
20
|
+
}
|
|
21
|
+
}
|
|
@@ -0,0 +1,68 @@
|
|
|
1
|
+
import { VPN_MESSAGE, isVpnPage } from "../shared/vpn"
|
|
2
|
+
import type { Transport } from "./transport"
|
|
3
|
+
|
|
4
|
+
export type UpstreamResult =
|
|
5
|
+
| { kind: "ok"; response: Response }
|
|
6
|
+
| { kind: "rate_limited"; status: number; body: string }
|
|
7
|
+
| { kind: "context_overflow"; status: number; body: string }
|
|
8
|
+
| { kind: "auth_invalid"; status: number; body: string }
|
|
9
|
+
| { kind: "unreachable"; message: string }
|
|
10
|
+
| { kind: "error"; status: number; body: string; contentType: string }
|
|
11
|
+
|
|
12
|
+
const CONTEXT_RE = /maximum context length|max input tokens|context length|context window|too many tokens|prompt is too long/i
|
|
13
|
+
|
|
14
|
+
/** One call to UFR. Throws only if the caller's own signal aborted. */
|
|
15
|
+
export async function callUpstream(o: {
|
|
16
|
+
transport: Transport
|
|
17
|
+
baseUrl: string
|
|
18
|
+
key: string
|
|
19
|
+
body: unknown
|
|
20
|
+
timeoutMs: number
|
|
21
|
+
signal?: AbortSignal
|
|
22
|
+
stream: boolean
|
|
23
|
+
}): Promise<UpstreamResult> {
|
|
24
|
+
const timeout = AbortSignal.timeout(o.timeoutMs)
|
|
25
|
+
const signal = o.signal ? AbortSignal.any([o.signal, timeout]) : timeout
|
|
26
|
+
let res: Response
|
|
27
|
+
try {
|
|
28
|
+
res = await o.transport.fetch(`${o.baseUrl}/chat/completions`, {
|
|
29
|
+
method: "POST",
|
|
30
|
+
headers: {
|
|
31
|
+
Authorization: `Bearer ${o.key}`,
|
|
32
|
+
"Content-Type": "application/json",
|
|
33
|
+
Accept: o.stream ? "text/event-stream" : "application/json",
|
|
34
|
+
},
|
|
35
|
+
body: JSON.stringify(o.body),
|
|
36
|
+
signal,
|
|
37
|
+
redirect: "manual",
|
|
38
|
+
})
|
|
39
|
+
} catch (e) {
|
|
40
|
+
if (o.signal?.aborted) throw e
|
|
41
|
+
if (timeout.aborted) {
|
|
42
|
+
const message = `UFR did not answer within ${Math.round(o.timeoutMs / 1000)} s`
|
|
43
|
+
return {
|
|
44
|
+
kind: "error",
|
|
45
|
+
status: 504,
|
|
46
|
+
body: JSON.stringify({ error: { message, type: "upstream_timeout", code: 504 } }),
|
|
47
|
+
contentType: "application/json",
|
|
48
|
+
}
|
|
49
|
+
}
|
|
50
|
+
return { kind: "unreachable", message: `cannot reach UFR (${(e as Error).message}) — are you connected to the uni VPN?` }
|
|
51
|
+
}
|
|
52
|
+
const type = res.headers.get("content-type") ?? ""
|
|
53
|
+
if (res.status >= 300 && res.status < 400) {
|
|
54
|
+
// Not proof of the VPN wall (ruling R12): only UFR's HTML page below is.
|
|
55
|
+
await res.body?.cancel()
|
|
56
|
+
return { kind: "unreachable", message: `UFR answered with a redirect (HTTP ${res.status}) instead of JSON` }
|
|
57
|
+
}
|
|
58
|
+
if (type.includes("text/html")) {
|
|
59
|
+
const html = await res.text()
|
|
60
|
+
return { kind: "unreachable", message: isVpnPage(html) ? VPN_MESSAGE : `UFR answered with an HTML page (HTTP ${res.status}) instead of JSON` }
|
|
61
|
+
}
|
|
62
|
+
if (res.ok) return { kind: "ok", response: res }
|
|
63
|
+
const body = await res.text()
|
|
64
|
+
if (res.status === 429) return { kind: "rate_limited", status: 429, body }
|
|
65
|
+
if (res.status === 401 || res.status === 403) return { kind: "auth_invalid", status: res.status, body }
|
|
66
|
+
if ((res.status === 400 || res.status === 413) && CONTEXT_RE.test(body)) return { kind: "context_overflow", status: res.status, body }
|
|
67
|
+
return { kind: "error", status: res.status, body, contentType: type || "application/json" }
|
|
68
|
+
}
|