opencode-ufr 0.2.8 → 0.2.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +65 -62
- package/models.json +11 -5
- package/package.json +14 -6
- package/skills/pdf2md/SKILL.md +32 -0
- package/src/client/gateway.ts +101 -0
- package/src/client/pdf2md.ts +747 -0
- package/src/daemon/catalog-source.ts +9 -0
- package/src/daemon/daemon.ts +70 -18
- package/src/daemon/keypool.ts +12 -7
- package/src/daemon/main.ts +7 -1
- package/src/daemon/probe-all.ts +212 -0
- package/src/daemon/probe.ts +1 -1
- package/src/daemon/router.ts +215 -17
- package/src/daemon/server.ts +33 -1
- package/src/daemon/stats.ts +13 -1
- package/src/daemon/upstream.ts +14 -3
- package/src/daemon/vpn/fortinet.ts +42 -29
- package/src/daemon/vpn/manager.ts +50 -16
- package/src/daemon/vpn/proxy.ts +3 -1
- package/src/plugin/connect.ts +4 -4
- package/src/plugin/ensure-daemon.ts +48 -19
- package/src/plugin/index.ts +121 -46
- package/src/plugin/tui.tsx +77 -0
- package/src/shared/config.ts +53 -4
- package/src/shared/connect.ts +3 -4
- package/src/shared/daemon-client.ts +29 -0
- package/src/shared/errors.ts +1 -1
- package/src/shared/fs.ts +8 -3
- package/src/shared/keys.ts +65 -0
- package/src/shared/models-file.ts +1 -1
- package/src/shared/secrets.ts +2 -6
- package/bin/ufr.ts +0 -4
- package/src/cli/catalog.ts +0 -44
- package/src/cli/connect.ts +0 -146
- package/src/cli/daemon-client.ts +0 -19
- package/src/cli/index.ts +0 -97
- package/src/cli/io.ts +0 -98
- package/src/cli/keys.ts +0 -130
- package/src/cli/stats.ts +0 -37
- package/src/cli/status.ts +0 -48
- package/src/cli/stop.ts +0 -8
- package/src/cli/ufr-check.ts +0 -43
package/src/daemon/router.ts
CHANGED
|
@@ -115,6 +115,155 @@ export class Router {
|
|
|
115
115
|
return this.jsonOut(slot.at, body, group, res, max, signal)
|
|
116
116
|
}
|
|
117
117
|
|
|
118
|
+
/**
|
|
119
|
+
* One raw upstream call for local scripts (context probes, pdf2md): the same
|
|
120
|
+
* central key rotation and soft rate limiting as chat — the caller waits for
|
|
121
|
+
* a key slot instead of being told to back off — but no fallback chain, no
|
|
122
|
+
* context-hub hop and no reasoning retry. The caller gets exactly the model
|
|
123
|
+
* it asked for, with UFR's status and body passed through verbatim: a context
|
|
124
|
+
* probe needs the 400 body that names the limit, and a transcription needs
|
|
125
|
+
* the model it chose, not a fallback. Non-streaming by design.
|
|
126
|
+
*/
|
|
127
|
+
async handleRelay(body: Record<string, unknown>, signal?: AbortSignal): Promise<Response> {
|
|
128
|
+
this.active++
|
|
129
|
+
try {
|
|
130
|
+
return await this.relayOnce(body, signal)
|
|
131
|
+
} finally {
|
|
132
|
+
this.active--
|
|
133
|
+
}
|
|
134
|
+
}
|
|
135
|
+
|
|
136
|
+
private async relayOnce(body: Record<string, unknown>, signal?: AbortSignal): Promise<Response> {
|
|
137
|
+
this.active++ // a long-lived relay must be visible in the status inFlight count
|
|
138
|
+
try {
|
|
139
|
+
return await this.relay(body, signal)
|
|
140
|
+
} finally {
|
|
141
|
+
this.active--
|
|
142
|
+
}
|
|
143
|
+
}
|
|
144
|
+
|
|
145
|
+
private async relay(body: Record<string, unknown>, signal?: AbortSignal): Promise<Response> {
|
|
146
|
+
const t0 = this.d.now()
|
|
147
|
+
if (body.stream === true) {
|
|
148
|
+
return errorResponse(400, "invalid_request", "/v1/_relay is non-streaming — use /v1/chat/completions for streams")
|
|
149
|
+
}
|
|
150
|
+
const requested = typeof body.model === "string" ? body.model.trim() : ""
|
|
151
|
+
if (!requested) return errorResponse(400, "invalid_request", "body.model is required")
|
|
152
|
+
const cat = this.d.catalog()
|
|
153
|
+
const group = resolveModel(cat, requested)
|
|
154
|
+
|
|
155
|
+
if (!this.d.breakers.get(group).allow().allowed) {
|
|
156
|
+
return this.fail(t0, group, null, 0, false, "upstream_circuit_open",
|
|
157
|
+
errorResponse(429, "upstream_circuit_open",
|
|
158
|
+
`UFR is refusing ${group} right now; the gateway waits instead of hammering it`,
|
|
159
|
+
this.d.breakers.get(group).retryAfterMs()))
|
|
160
|
+
}
|
|
161
|
+
|
|
162
|
+
const slot = this.d.pool.reserve()
|
|
163
|
+
if (slot.verdict === "reject") {
|
|
164
|
+
// allow() above may have consumed a half-open probe that now never gets
|
|
165
|
+
// an outcome — resolve it so the ladder doesn't escalate on a probe timeout.
|
|
166
|
+
this.d.breakers.get(group).onOtherFailure()
|
|
167
|
+
return this.fail(t0, group, null, 0, false, "upstream_pool_cap",
|
|
168
|
+
errorResponse(429, "upstream_pool_cap", `hourly cap of ${this.d.config.limits.poolPerHour} requests reached`, slot.waitMs))
|
|
169
|
+
}
|
|
170
|
+
if (slot.waitMs > 0) {
|
|
171
|
+
try {
|
|
172
|
+
await this.d.sleep(slot.waitMs, signal)
|
|
173
|
+
} catch {
|
|
174
|
+
this.d.pool.release(slot.at)
|
|
175
|
+
this.d.breakers.get(group).onOtherFailure()
|
|
176
|
+
return this.fail(t0, group, null, 0, false, "client_closed",
|
|
177
|
+
errorResponse(499, "client_closed", "client went away while waiting for a slot"))
|
|
178
|
+
}
|
|
179
|
+
}
|
|
180
|
+
|
|
181
|
+
const k = this.d.keys.acquire()
|
|
182
|
+
if (k.kind === "none") {
|
|
183
|
+
this.d.breakers.get(group).onOtherFailure()
|
|
184
|
+
if (k.reason === "no_keys") {
|
|
185
|
+
return this.fail(t0, group, null, 0, true, "no_keys",
|
|
186
|
+
errorResponse(401, "no_keys", "no valid UFR API key configured — reconnect the Uni Freiburg integration in opencode's /connect"))
|
|
187
|
+
}
|
|
188
|
+
// exhausted or all_tried (relay excludes nothing, so all_tried cannot happen — belt and braces)
|
|
189
|
+
return this.fail(t0, group, null, 0, true, "key_pool_exhausted",
|
|
190
|
+
errorResponse(429, "key_pool_exhausted",
|
|
191
|
+
`every UFR key is at its limit of ${this.d.config.limits.keyRpm} requests per ${this.d.config.limits.keyWindowS} s`, k.retryAfterMs))
|
|
192
|
+
}
|
|
193
|
+
if (k.waitMs > 0) {
|
|
194
|
+
try {
|
|
195
|
+
await this.d.sleep(k.waitMs, signal)
|
|
196
|
+
} catch {
|
|
197
|
+
this.d.keys.release(k.alias, k.at)
|
|
198
|
+
this.d.breakers.get(group).onOtherFailure()
|
|
199
|
+
return this.fail(t0, group, null, 0, true, "client_closed",
|
|
200
|
+
errorResponse(499, "client_closed", "client went away while waiting for a key"))
|
|
201
|
+
}
|
|
202
|
+
}
|
|
203
|
+
|
|
204
|
+
let r: UpstreamResult
|
|
205
|
+
try {
|
|
206
|
+
r = await callUpstream({
|
|
207
|
+
transport: this.d.transport,
|
|
208
|
+
baseUrl: this.d.config.upstream.baseUrl,
|
|
209
|
+
key: k.secret,
|
|
210
|
+
body: { ...body, model: group },
|
|
211
|
+
timeoutMs: this.d.config.upstream.requestTimeoutS * 1000,
|
|
212
|
+
signal,
|
|
213
|
+
stream: false,
|
|
214
|
+
})
|
|
215
|
+
} catch (e) {
|
|
216
|
+
this.d.breakers.get(group).onOtherFailure() // the call's outcome never came back
|
|
217
|
+
if (signal?.aborted) {
|
|
218
|
+
return this.fail(t0, group, k.alias, 1, true, "client_closed",
|
|
219
|
+
errorResponse(499, "client_closed", "client went away"))
|
|
220
|
+
}
|
|
221
|
+
return this.fail(t0, group, k.alias, 1, true, "transport_unreachable",
|
|
222
|
+
errorResponse(503, "transport_unreachable", `UFR call failed (${e instanceof Error ? e.message : String(e)})`))
|
|
223
|
+
}
|
|
224
|
+
switch (r.kind) {
|
|
225
|
+
case "ok": {
|
|
226
|
+
this.d.breakers.get(group).onSuccess()
|
|
227
|
+
this.d.onUpstream?.(true, "")
|
|
228
|
+
const text = await r.response.text()
|
|
229
|
+
let json: unknown
|
|
230
|
+
try {
|
|
231
|
+
json = JSON.parse(text)
|
|
232
|
+
} catch {
|
|
233
|
+
// Not JSON — pass it through untouched rather than failing a call UFR answered.
|
|
234
|
+
this.record(t0, group, k.alias, 200, null, 1, null, true)
|
|
235
|
+
return new Response(text, { status: 200, headers: { "content-type": r.response.headers.get("content-type") ?? "application/json" } })
|
|
236
|
+
}
|
|
237
|
+
this.record(t0, group, k.alias, 200, usageOf(json), 1, null, true)
|
|
238
|
+
return Response.json(json)
|
|
239
|
+
}
|
|
240
|
+
case "rate_limited":
|
|
241
|
+
// A wall on the exact model the script asked for — honest breaker signal.
|
|
242
|
+
this.d.keys.onRateLimited(k.alias, group)
|
|
243
|
+
this.d.breakers.get(group).onRateLimited()
|
|
244
|
+
this.record(t0, group, k.alias, 429, null, 1, "upstream_rate_limited", true)
|
|
245
|
+
return new Response(r.body, { status: 429, headers: { "content-type": "application/json" } })
|
|
246
|
+
case "auth_invalid":
|
|
247
|
+
this.d.keys.onInvalid(k.alias)
|
|
248
|
+
this.d.breakers.get(group).onOtherFailure()
|
|
249
|
+
this.record(t0, group, k.alias, r.status, null, 1, "upstream_auth_invalid", true)
|
|
250
|
+
return new Response(r.body, { status: r.status, headers: { "content-type": "application/json" } })
|
|
251
|
+
case "context_overflow":
|
|
252
|
+
// UFR's own 400 naming the limit — the reason scripts use the relay. Not a wall.
|
|
253
|
+
this.d.breakers.get(group).onOtherFailure()
|
|
254
|
+
this.record(t0, group, k.alias, r.status, null, 1, "context_overflow", true)
|
|
255
|
+
return new Response(r.body, { status: r.status, headers: { "content-type": "application/json" } })
|
|
256
|
+
case "unreachable":
|
|
257
|
+
this.d.breakers.get(group).onOtherFailure()
|
|
258
|
+
this.d.onUpstream?.(false, r.message)
|
|
259
|
+
return this.fail(t0, group, k.alias, 1, true, "transport_unreachable", errorResponse(503, "transport_unreachable", r.message))
|
|
260
|
+
case "error":
|
|
261
|
+
this.d.breakers.get(group).onOtherFailure()
|
|
262
|
+
this.record(t0, group, k.alias, r.status, null, 1, `upstream_${r.status}`, true)
|
|
263
|
+
return new Response(r.body, { status: r.status, headers: { "content-type": r.contentType } })
|
|
264
|
+
}
|
|
265
|
+
}
|
|
266
|
+
|
|
118
267
|
/** Upstream calls for one client request: other key, next model, context hub. */
|
|
119
268
|
private async loop(
|
|
120
269
|
body: Record<string, unknown>,
|
|
@@ -164,10 +313,11 @@ export class Router {
|
|
|
164
313
|
if (k.kind === "none") {
|
|
165
314
|
if (k.reason === "no_keys") {
|
|
166
315
|
breakers.get(model).onOtherFailure() // model's probe (if any) got no outcome
|
|
167
|
-
return fail(401, "no_keys", "no valid UFR API key configured —
|
|
316
|
+
return fail(401, "no_keys", "no valid UFR API key configured — reconnect the Uni Freiburg integration in opencode's /connect")
|
|
168
317
|
}
|
|
169
318
|
if (k.reason === "exhausted") {
|
|
170
319
|
breakers.get(model).onOtherFailure() // model's probe (if any) got no outcome
|
|
320
|
+
if (rateLimitedHere) giveUp(model) // the wall counts toward the breaker threshold even here
|
|
171
321
|
return fail(429, "key_pool_exhausted",
|
|
172
322
|
`every UFR key is at its limit of ${config.limits.keyRpm} requests per ${config.limits.keyWindowS} s`, k.retryAfterMs)
|
|
173
323
|
}
|
|
@@ -190,7 +340,12 @@ export class Router {
|
|
|
190
340
|
lastKey = k.alias
|
|
191
341
|
const upstreamBody: Record<string, unknown> = { ...body, model }
|
|
192
342
|
if (stream) {
|
|
193
|
-
|
|
343
|
+
// Only spread a real object — a client sending stream_options as a
|
|
344
|
+
// string (or array) would otherwise be spread into garbage upstream.
|
|
345
|
+
const so = body.stream_options
|
|
346
|
+
const opts =
|
|
347
|
+
typeof so === "object" && so !== null && !Array.isArray(so) ? (so as Record<string, unknown>) : {}
|
|
348
|
+
upstreamBody.stream_options = { ...opts, include_usage: true }
|
|
194
349
|
}
|
|
195
350
|
// Bun's fetch does not propagate a body reader's cancel() into aborting the
|
|
196
351
|
// underlying request — give streaming attempts their own controller so
|
|
@@ -277,30 +432,73 @@ export class Router {
|
|
|
277
432
|
}
|
|
278
433
|
|
|
279
434
|
private async jsonOut(ts: number, body: Record<string, unknown>, group: string, res: LoopOk, max: number, signal?: AbortSignal): Promise<Response> {
|
|
280
|
-
let json: unknown
|
|
435
|
+
let json: unknown
|
|
436
|
+
try {
|
|
437
|
+
json = await res.response.json()
|
|
438
|
+
} catch {
|
|
439
|
+
// A malformed 200 must not escape handleChat as Bun's default 500 with the
|
|
440
|
+
// pool slot never released — turn it into an honest error and give the slot
|
|
441
|
+
// back (poolAdmitted false: the released slot must not seed a phantom
|
|
442
|
+
// admission in the window rebuilt from stats after a restart).
|
|
443
|
+
this.d.pool.release(ts)
|
|
444
|
+
if (signal?.aborted) {
|
|
445
|
+
// A client abort during the body transfer also rejects json() — that is
|
|
446
|
+
// not bad upstream JSON: the call really happened, so record it as the
|
|
447
|
+
// client's doing, in the same shape as the other client-abort paths.
|
|
448
|
+
return this.fail(ts, res.model, res.keyAlias, res.attempts, false, "client_closed",
|
|
449
|
+
errorResponse(499, "client_closed", "client went away while the response body was transferred"))
|
|
450
|
+
}
|
|
451
|
+
return this.fail(ts, res.model, res.keyAlias, res.attempts, false, "upstream_bad_json",
|
|
452
|
+
errorResponse(502, "upstream_bad_json", "UFR answered 200 with a body that is not valid JSON"))
|
|
453
|
+
}
|
|
281
454
|
let usage = usageOf(json)
|
|
282
455
|
let cost = usage ? costUsd(this.priceOf(res.model), usage.prompt, usage.completion) : null
|
|
283
456
|
let attempts = res.attempts
|
|
284
457
|
let model = res.model
|
|
285
458
|
let keyAlias = res.keyAlias
|
|
459
|
+
let retryErrorType: string | null = null
|
|
286
460
|
if (isReasoningStarved(json) && attempts < max) {
|
|
287
|
-
|
|
288
|
-
|
|
289
|
-
|
|
290
|
-
|
|
291
|
-
|
|
292
|
-
|
|
293
|
-
|
|
294
|
-
|
|
295
|
-
|
|
296
|
-
|
|
297
|
-
|
|
298
|
-
|
|
299
|
-
|
|
461
|
+
// The retry is a real upstream call, so it takes a pool slot under the same
|
|
462
|
+
// acquire/wait/abort semantics as the main path — it must not evade the
|
|
463
|
+
// poolPerHour cap. When the pool cannot give one (rejected, or the client
|
|
464
|
+
// went away while waiting), keep the starved answer instead of failing.
|
|
465
|
+
const slot = this.d.pool.reserve()
|
|
466
|
+
let mayRetry = slot.verdict !== "reject"
|
|
467
|
+
if (mayRetry && slot.waitMs > 0) {
|
|
468
|
+
try {
|
|
469
|
+
await this.d.sleep(slot.waitMs, signal)
|
|
470
|
+
} catch {
|
|
471
|
+
this.d.pool.release(slot.at)
|
|
472
|
+
mayRetry = false
|
|
473
|
+
}
|
|
474
|
+
}
|
|
475
|
+
if (mayRetry) {
|
|
476
|
+
const retry = await this.loop({ ...body, reasoning_effort: "none" }, group, res.model, false, max - attempts, signal)
|
|
477
|
+
attempts += retry.attempts
|
|
478
|
+
if (retry.ok) {
|
|
479
|
+
try {
|
|
480
|
+
const again: unknown = await retry.response.json()
|
|
481
|
+
const u2 = usageOf(again)
|
|
482
|
+
// Price each call at the model it was actually answered by — a retry that
|
|
483
|
+
// falls back must not price the first call at the fallback's rate.
|
|
484
|
+
cost = addCost(cost, u2 ? costUsd(this.priceOf(retry.model), u2.prompt, u2.completion) : null)
|
|
485
|
+
usage = addUsage(usage, u2)
|
|
486
|
+
if (!isReasoningStarved(again)) {
|
|
487
|
+
json = again
|
|
488
|
+
model = retry.model
|
|
489
|
+
keyAlias = retry.keyAlias
|
|
490
|
+
}
|
|
491
|
+
} catch {
|
|
492
|
+
retryErrorType = "reasoning_retry_failed"
|
|
493
|
+
}
|
|
494
|
+
} else {
|
|
495
|
+
// The client still gets the starved 200 (intended graceful degradation),
|
|
496
|
+
// but the failed retry must show in stats instead of a clean errorType null.
|
|
497
|
+
retryErrorType = "reasoning_retry_failed"
|
|
300
498
|
}
|
|
301
499
|
}
|
|
302
500
|
}
|
|
303
|
-
this.record(ts, model, keyAlias, 200, usage, attempts,
|
|
501
|
+
this.record(ts, model, keyAlias, 200, usage, attempts, retryErrorType, true, cost)
|
|
304
502
|
return Response.json(json)
|
|
305
503
|
}
|
|
306
504
|
|
package/src/daemon/server.ts
CHANGED
|
@@ -8,8 +8,12 @@ export type ServerDeps = {
|
|
|
8
8
|
router: Router
|
|
9
9
|
models: () => unknown[]
|
|
10
10
|
status: () => unknown
|
|
11
|
+
/** Raw UFR model list + bundled models.json — for scripts that probe every model. */
|
|
12
|
+
catalogData?: () => { ufr: unknown; file: unknown } | null
|
|
11
13
|
onActivity: () => void
|
|
12
14
|
onShutdown: () => void
|
|
15
|
+
/** True once stop() began: new work is fast-failed while in-flight requests drain. */
|
|
16
|
+
draining?: () => boolean
|
|
13
17
|
}
|
|
14
18
|
|
|
15
19
|
export function startServer(d: ServerDeps): { port: number; stop: () => void } {
|
|
@@ -27,6 +31,14 @@ export function startServer(d: ServerDeps): { port: number; stop: () => void } {
|
|
|
27
31
|
}
|
|
28
32
|
d.onActivity()
|
|
29
33
|
const route = `${req.method} ${url.pathname}`
|
|
34
|
+
// Once stop() began, fast-fail new chat/relay work instead of admitting
|
|
35
|
+
// requests the drain grace period would tear down at the deadline.
|
|
36
|
+
// /health, /v1/_status and friends keep answering during the drain.
|
|
37
|
+
if ((route === "POST /v1/chat/completions" || route === "POST /v1/_relay") && d.draining?.()) {
|
|
38
|
+
return errorResponse(503, "draining", "the gateway is shutting down — retry shortly")
|
|
39
|
+
}
|
|
40
|
+
// Dashboard polling of _status must not reset the idle timer — only real work does.
|
|
41
|
+
if (route === "GET /v1/_status") return Response.json(d.status())
|
|
30
42
|
if (route === "GET /v1/models") return Response.json({ object: "list", data: d.models() })
|
|
31
43
|
if (route === "POST /v1/chat/completions") {
|
|
32
44
|
srv.timeout(req, 0) // glm-5.2 can think for minutes before the first byte
|
|
@@ -41,14 +53,34 @@ export function startServer(d: ServerDeps): { port: number; stop: () => void } {
|
|
|
41
53
|
}
|
|
42
54
|
return d.router.handleChat(body as Record<string, unknown>, req.signal)
|
|
43
55
|
}
|
|
56
|
+
if (route === "POST /v1/_relay") {
|
|
57
|
+
// Scripts (context probes, pdf2md): one raw upstream call through the
|
|
58
|
+
// central key rotation and soft rate limiting. Streams live forever.
|
|
59
|
+
srv.timeout(req, 0)
|
|
60
|
+
let body: unknown
|
|
61
|
+
try {
|
|
62
|
+
body = await req.json()
|
|
63
|
+
} catch {
|
|
64
|
+
return errorResponse(400, "invalid_request", "body must be JSON")
|
|
65
|
+
}
|
|
66
|
+
if (typeof body !== "object" || body === null || Array.isArray(body)) {
|
|
67
|
+
return errorResponse(400, "invalid_request", "body must be a JSON object")
|
|
68
|
+
}
|
|
69
|
+
return d.router.handleRelay(body as Record<string, unknown>, req.signal)
|
|
70
|
+
}
|
|
71
|
+
if (route === "GET /v1/_catalog") return Response.json(d.catalogData?.() ?? null)
|
|
44
72
|
if (route === "POST /v1/_client/heartbeat") return Response.json({ ok: true })
|
|
45
|
-
if (route === "GET /v1/_status") return Response.json(d.status())
|
|
46
73
|
if (route === "POST /v1/_shutdown") {
|
|
47
74
|
setTimeout(d.onShutdown, 20)
|
|
48
75
|
return Response.json({ ok: true })
|
|
49
76
|
}
|
|
50
77
|
return errorResponse(404, "not_found", `no route ${route}`)
|
|
51
78
|
},
|
|
79
|
+
// Uncaught throws from the fetch handler (status(), models(), handleChat
|
|
80
|
+
// internals): answer in the gateway's JSON error shape instead of Bun's
|
|
81
|
+
// default HTML 500. Responses already returned — including streams, which
|
|
82
|
+
// surface their own errors — never reach this handler.
|
|
83
|
+
error: (e) => errorResponse(500, "internal_error", `gateway error: ${e instanceof Error ? e.message : String(e)}`),
|
|
52
84
|
})
|
|
53
85
|
return { port: server.port as number, stop: () => server.stop(true) }
|
|
54
86
|
}
|
package/src/daemon/stats.ts
CHANGED
|
@@ -60,7 +60,8 @@ export class Stats {
|
|
|
60
60
|
|
|
61
61
|
/** Admission times for rebuilding the pool window after a restart. */
|
|
62
62
|
poolAdmissionsSince(sinceTs: number): number[] {
|
|
63
|
-
|
|
63
|
+
// ts >= like spendByKeySince/summary — "since" is inclusive everywhere
|
|
64
|
+
const rows = this.db.query("SELECT ts FROM requests WHERE pool_admitted = 1 AND ts >= ? ORDER BY ts").all(sinceTs) as { ts: number }[]
|
|
64
65
|
return rows.map((r) => r.ts)
|
|
65
66
|
}
|
|
66
67
|
|
|
@@ -86,6 +87,17 @@ export class Stats {
|
|
|
86
87
|
return { byModel: q("model"), byKey: q("COALESCE(key_alias, '-')") }
|
|
87
88
|
}
|
|
88
89
|
|
|
90
|
+
/** Request and token totals in a window, for the live rates in /v1/_status. */
|
|
91
|
+
rates(sinceTs: number): { requests: number; promptTokens: number; completionTokens: number } {
|
|
92
|
+
return this.db
|
|
93
|
+
.query(
|
|
94
|
+
`SELECT COUNT(*) AS requests, COALESCE(SUM(prompt_tokens), 0) AS promptTokens,
|
|
95
|
+
COALESCE(SUM(completion_tokens), 0) AS completionTokens
|
|
96
|
+
FROM requests WHERE ts >= ?`,
|
|
97
|
+
)
|
|
98
|
+
.get(sinceTs) as { requests: number; promptTokens: number; completionTokens: number }
|
|
99
|
+
}
|
|
100
|
+
|
|
89
101
|
prune(beforeTs: number): number {
|
|
90
102
|
return this.db.query("DELETE FROM requests WHERE ts < ?").run(beforeTs).changes
|
|
91
103
|
}
|
package/src/daemon/upstream.ts
CHANGED
|
@@ -21,8 +21,17 @@ export async function callUpstream(o: {
|
|
|
21
21
|
signal?: AbortSignal
|
|
22
22
|
stream: boolean
|
|
23
23
|
}): Promise<UpstreamResult> {
|
|
24
|
-
|
|
25
|
-
|
|
24
|
+
// The timeout guards time-to-first-byte: it aborts only until UFR answers
|
|
25
|
+
// with response headers, then is cleared so a long SSE body can stream well
|
|
26
|
+
// past it — a stream longer than requestTimeoutS must not be killed
|
|
27
|
+
// mid-generation (the client's own signal still aborts via the composed
|
|
28
|
+
// signal). Non-streaming calls keep the whole-request guarantee in practice:
|
|
29
|
+
// UFR sends its JSON body immediately after the headers, so a slow generation
|
|
30
|
+
// stalls the headers too; a body stalling after that is the caller's signal's
|
|
31
|
+
// job (it composes into the same signal).
|
|
32
|
+
const ttfb = new AbortController()
|
|
33
|
+
const signal = o.signal ? AbortSignal.any([o.signal, ttfb.signal]) : ttfb.signal
|
|
34
|
+
const timer = setTimeout(() => ttfb.abort(), o.timeoutMs)
|
|
26
35
|
let res: Response
|
|
27
36
|
try {
|
|
28
37
|
res = await o.transport.fetch(`${o.baseUrl}/chat/completions`, {
|
|
@@ -37,8 +46,9 @@ export async function callUpstream(o: {
|
|
|
37
46
|
redirect: "manual",
|
|
38
47
|
})
|
|
39
48
|
} catch (e) {
|
|
49
|
+
clearTimeout(timer)
|
|
40
50
|
if (o.signal?.aborted) throw e
|
|
41
|
-
if (
|
|
51
|
+
if (ttfb.signal.aborted) {
|
|
42
52
|
const message = `UFR did not answer within ${Math.round(o.timeoutMs / 1000)} s`
|
|
43
53
|
return {
|
|
44
54
|
kind: "error",
|
|
@@ -49,6 +59,7 @@ export async function callUpstream(o: {
|
|
|
49
59
|
}
|
|
50
60
|
return { kind: "unreachable", message: `cannot reach UFR (${(e as Error).message}) — are you connected to the uni VPN?` }
|
|
51
61
|
}
|
|
62
|
+
clearTimeout(timer) // headers are in — the body may now stream indefinitely
|
|
52
63
|
const type = res.headers.get("content-type") ?? ""
|
|
53
64
|
if (res.status >= 300 && res.status < 400) {
|
|
54
65
|
// Not proof of the VPN wall (ruling R12): only UFR's HTML page below is.
|
|
@@ -8,6 +8,7 @@
|
|
|
8
8
|
|
|
9
9
|
import tls from "node:tls"
|
|
10
10
|
|
|
11
|
+
import { ipv4Parse } from "./ip"
|
|
11
12
|
import { PppSession, type PppConfig, type PppOptions } from "./ppp"
|
|
12
13
|
|
|
13
14
|
// -- framing -----------------------------------------------------------------
|
|
@@ -232,48 +233,46 @@ export type TunnelConfig = {
|
|
|
232
233
|
}
|
|
233
234
|
|
|
234
235
|
function attr(xml: string, tag: string, attribute: string): string | null {
|
|
235
|
-
|
|
236
|
-
|
|
236
|
+
// The attribute must live INSIDE the requested tag: <tag … attribute="value" …>.
|
|
237
|
+
// (?![\w-]) keeps longer tag names (dtls-config-x) from matching; the leading \s
|
|
238
|
+
// keeps attribute-name suffixes (x-ipv4) from matching. Order, whitespace around
|
|
239
|
+
// "=" and quote style all vary; tag and attribute names are case-insensitive.
|
|
240
|
+
const re = new RegExp(`<${tag}(?![\\w-])[^>]*?\\s${attribute}\\s*=\\s*(["'])([^"'>]*)\\1`, "i")
|
|
241
|
+
return re.exec(xml)?.[2] ?? null
|
|
242
|
+
}
|
|
243
|
+
|
|
244
|
+
/** A positive number, else the fallback ("" → 0 and NaN would silently kill the DPD keepalive). */
|
|
245
|
+
function positiveNumber(v: string | null, fallback: number): number {
|
|
246
|
+
const n = v === null ? Number.NaN : Number(v)
|
|
247
|
+
return Number.isFinite(n) && n > 0 ? n : fallback
|
|
237
248
|
}
|
|
238
249
|
|
|
239
250
|
export function parseTunnelConfig(xml: string): TunnelConfig {
|
|
240
251
|
if (!/<sslvpn-tunnel/i.test(xml)) {
|
|
241
252
|
throw new VpnAuthError("tunnel config: response is not a sslvpn-tunnel document (session dead?)")
|
|
242
253
|
}
|
|
254
|
+
const innerIp = attr(xml, "assigned-addr", "ipv4") ?? ""
|
|
255
|
+
try {
|
|
256
|
+
ipv4Parse(innerIp)
|
|
257
|
+
} catch {
|
|
258
|
+
throw new VpnAuthError(`tunnel config: invalid inner ip "${innerIp}"`)
|
|
259
|
+
}
|
|
243
260
|
const dns = [...xml.matchAll(/<dns[^>]*\bip=["']([^"']*)["']/gi)].map((m) => m[1]!)
|
|
244
261
|
const routes = [...xml.matchAll(/<addr[^>]*\bip=["']([^"']*)["'][^>]*\bmask=["']([^"']*)["']/gi)].map((m) => ({ ip: m[1]!, mask: m[2]! }))
|
|
245
262
|
return {
|
|
246
|
-
innerIp
|
|
263
|
+
innerIp,
|
|
247
264
|
dns,
|
|
248
|
-
dpdS:
|
|
265
|
+
dpdS: positiveNumber(attr(xml, "dtls-config", "heartbeat-interval"), 10),
|
|
249
266
|
idleTimeoutS: Number(attr(xml, "idle-timeout", "val") ?? "0"),
|
|
250
267
|
authTimeoutS: Number(attr(xml, "auth-timeout", "val") ?? "0"),
|
|
251
268
|
routes,
|
|
252
269
|
}
|
|
253
270
|
}
|
|
254
271
|
|
|
255
|
-
export async function fetchTunnelConfig(o: {
|
|
256
|
-
gateway: string
|
|
257
|
-
cookie: string
|
|
258
|
-
fetch?: typeof fetch
|
|
259
|
-
}): Promise<TunnelConfig> {
|
|
260
|
-
const f = o.fetch ?? fetch
|
|
261
|
-
const res = await f(`${o.gateway.replace(/\/$/, "")}/remote/fortisslvpn_xml?dual_stack=1`, {
|
|
262
|
-
headers: { "User-Agent": FORTI_UA, Cookie: o.cookie },
|
|
263
|
-
redirect: "manual",
|
|
264
|
-
})
|
|
265
|
-
const text = await res.text().catch(() => "")
|
|
266
|
-
if (res.status !== 200 || /\/remote\/login/.test(res.headers.get("location") ?? "")) {
|
|
267
|
-
throw new VpnAuthError("tunnel config fetch failed — session invalid (HTTP " + res.status + ")")
|
|
268
|
-
}
|
|
269
|
-
return parseTunnelConfig(text)
|
|
270
|
-
}
|
|
271
|
-
|
|
272
272
|
// -- the tunnel ----------------------------------------------------------------
|
|
273
273
|
|
|
274
274
|
export type TunnelHandle = {
|
|
275
275
|
readonly config: TunnelConfig
|
|
276
|
-
readonly ppp: PppSession
|
|
277
276
|
/** Clean teardown: PPP TERMREQ, close TLS, logout. */
|
|
278
277
|
close(): Promise<void>
|
|
279
278
|
}
|
|
@@ -288,6 +287,8 @@ export async function openTunnel(o: {
|
|
|
288
287
|
onIp: (datagram: Uint8Array) => void
|
|
289
288
|
pppOptions?: PppOptions
|
|
290
289
|
log?: (msg: string) => void
|
|
290
|
+
/** Extra CA certs for tls.connect (tests against a local TLS peer). */
|
|
291
|
+
ca?: string
|
|
291
292
|
}): Promise<TunnelHandle> {
|
|
292
293
|
const log = o.log ?? (() => {})
|
|
293
294
|
const url = new URL(o.gateway)
|
|
@@ -333,7 +334,11 @@ export async function openTunnel(o: {
|
|
|
333
334
|
if (nl === -1) return // wait for more
|
|
334
335
|
const size = parseInt(bodyBuf.subarray(pos, nl).toString(), 16)
|
|
335
336
|
if (Number.isNaN(size)) return fail("broken chunked body in tunnel config response")
|
|
336
|
-
if (size === 0) {
|
|
337
|
+
if (size === 0) {
|
|
338
|
+
if (bodyBuf.length < pos + 5) return // wait for the full "0\r\n\r\n" terminator
|
|
339
|
+
rest = bodyBuf.subarray(pos + 5)
|
|
340
|
+
break
|
|
341
|
+
}
|
|
337
342
|
if (bodyBuf.length < pos + nl + 2 + size + 2) return // wait for more
|
|
338
343
|
out = Buffer.concat([out, bodyBuf.subarray(nl + 2, nl + 2 + size)])
|
|
339
344
|
pos = nl + 2 + size + 2
|
|
@@ -345,9 +350,19 @@ export async function openTunnel(o: {
|
|
|
345
350
|
body = httpBuf.subarray(bodyStart, bodyStart + cl)
|
|
346
351
|
rest = httpBuf.subarray(bodyStart + cl)
|
|
347
352
|
}
|
|
348
|
-
if (status !== 200)
|
|
349
|
-
|
|
350
|
-
|
|
353
|
+
if (status !== 200) {
|
|
354
|
+
if (status >= 300 && status < 400) return fail("tunnel config redirected to login — session invalid")
|
|
355
|
+
return fail(`tunnel config request failed (HTTP ${status}) — session invalid?`)
|
|
356
|
+
}
|
|
357
|
+
if (/\/remote\/login/i.test(head)) return fail("tunnel config redirected to login — session invalid")
|
|
358
|
+
// Bun silently swallows exceptions thrown in socket data handlers — an escaping
|
|
359
|
+
// parse error would stall the tunnel to the establish timeout with zero log
|
|
360
|
+
// output. Every parse failure must go through fail().
|
|
361
|
+
try {
|
|
362
|
+
config = parseTunnelConfig((body as Buffer).toString())
|
|
363
|
+
} catch (e) {
|
|
364
|
+
return fail(`tunnel config invalid: ${(e as Error).message}`)
|
|
365
|
+
}
|
|
351
366
|
log(`tunnel config: inner ip ${config.innerIp}, dns ${config.dns.join(", ") || "none"}, dpd ${config.dpdS}s`)
|
|
352
367
|
// switch into tunnel phase: leftover bytes belong to the tunnel response
|
|
353
368
|
httpBuf = Buffer.from(rest as Buffer)
|
|
@@ -414,6 +429,7 @@ export async function openTunnel(o: {
|
|
|
414
429
|
servername: host,
|
|
415
430
|
ALPNProtocols: ["http/1.1"], // the tunnel speaks HTTP/1.1 then switches to PPP frames
|
|
416
431
|
rejectUnauthorized: true,
|
|
432
|
+
...(o.ca ? { ca: o.ca } : {}),
|
|
417
433
|
})
|
|
418
434
|
socket.on("data", (chunk: Buffer) => {
|
|
419
435
|
if (closed) return
|
|
@@ -466,9 +482,6 @@ export async function openTunnel(o: {
|
|
|
466
482
|
get config(): TunnelConfig {
|
|
467
483
|
return config ?? { innerIp: "", dns: [], dpdS: 10, idleTimeoutS: 0, authTimeoutS: 0, routes: [] }
|
|
468
484
|
},
|
|
469
|
-
get ppp(): PppSession {
|
|
470
|
-
return ppp!
|
|
471
|
-
},
|
|
472
485
|
close,
|
|
473
486
|
}
|
|
474
487
|
}
|