opencode-ufr 0.2.0 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +82 -59
- package/models.json +22 -4
- package/package.json +37 -9
- package/src/cli/catalog.ts +2 -3
- package/src/daemon/catalog-source.ts +8 -34
- package/src/daemon/catalog.ts +6 -2
- package/src/daemon/daemon.ts +52 -7
- package/src/daemon/probe.ts +113 -0
- package/src/daemon/vpn/tcp.ts +46 -18
- package/src/shared/config.ts +0 -5
- package/src/shared/paths.ts +0 -4
package/README.md
CHANGED
|
@@ -1,71 +1,100 @@
|
|
|
1
1
|
# opencode-ufr
|
|
2
2
|
|
|
3
|
-
An [opencode](https://opencode.ai) provider for Uni Freiburg's Open WebUI
|
|
4
|
-
models
|
|
3
|
+
An [opencode](https://opencode.ai) provider for **Uni Freiburg's Open WebUI
|
|
4
|
+
models**. A small local gateway runs on your machine, pools your UFR API
|
|
5
5
|
key(s), and handles UFR's rate limits and outages so opencode doesn't have to.
|
|
6
|
+
Off campus, it connects through the uni's Fortinet VPN **by itself** — with no
|
|
7
|
+
TUN device, no admin rights and no external tools.
|
|
6
8
|
|
|
7
|
-
|
|
8
|
-
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
-
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
login once (`ufr login add <user>`) and the plugin tunnels its UFR calls
|
|
15
|
-
through `fortivpn.uni-freiburg.de` — with no TUN device, no admin rights, no
|
|
16
|
-
openconnect and no changes to your routing. Your own VPNs keep working
|
|
17
|
-
untouched in parallel. See the RZ guide if you prefer a system VPN:
|
|
18
|
-
https://wiki.uni-freiburg.de/rz/doku.php?id=vpn
|
|
9
|
+
```
|
|
10
|
+
opencode ⇄ gateway (127.0.0.1) ⇄ [Fortinet tunnel] ⇄ openwebui.uni-freiburg.de
|
|
11
|
+
│
|
|
12
|
+
├─ key pool (1…n UFR accounts)
|
|
13
|
+
├─ rate-limit pacing, circuit breakers, fallbacks
|
|
14
|
+
└─ VPN when off campus (userspace TCP/IP stack)
|
|
15
|
+
```
|
|
19
16
|
|
|
20
17
|
## Install
|
|
21
18
|
|
|
22
|
-
Not on npm yet — install straight from GitHub:
|
|
23
|
-
|
|
24
19
|
```bash
|
|
25
|
-
bun add -g
|
|
26
|
-
ufr
|
|
20
|
+
bun add -g opencode-ufr # puts `ufr` on PATH
|
|
21
|
+
opencode plugin add opencode-ufr # registers the plugin
|
|
27
22
|
```
|
|
28
23
|
|
|
29
|
-
`ufr
|
|
30
|
-
around keys is filtered, aliases (key1, key2, …) are assigned automatically —
|
|
31
|
-
and optionally your uni login for the built-in VPN. Everything is verified
|
|
32
|
-
against UFR and stored in the OS keyring (never on disk).
|
|
24
|
+
(or add `"plugins": ["opencode-ufr"]` to your `opencode.json` manually.)
|
|
33
25
|
|
|
34
|
-
|
|
35
|
-
`/connect` panel's **Uni Freiburg** entry. It shows the API-key(s) field plus
|
|
36
|
-
**Uni login (optional)** and **Uni password (optional)** — both only needed on
|
|
37
|
-
a machine outside the uni network, where the built-in VPN uses them. Submitted
|
|
38
|
-
credentials land in the OS keyring automatically and the gateway restarts with
|
|
39
|
-
them.
|
|
26
|
+
## Setup
|
|
40
27
|
|
|
41
|
-
|
|
28
|
+
**Inside opencode — the normal way:** open `/connect`, pick **Uni Freiburg**,
|
|
29
|
+
and fill in:
|
|
42
30
|
|
|
43
|
-
|
|
44
|
-
|
|
45
|
-
|
|
31
|
+
| Field | Needed? |
|
|
32
|
+
|---|---|
|
|
33
|
+
| **API keys** | always — paste one key per UFR account, comma-separated (whitespace is filtered) |
|
|
34
|
+
| **Uni login** | only off campus — e.g. `fl240@uni-freiburg.de`, empty on the uni network |
|
|
35
|
+
| **Uni password** | only with a login above |
|
|
36
|
+
|
|
37
|
+
Submitted credentials land in your OS keyring (never on disk), the gateway
|
|
38
|
+
restarts with them, and the `unifreiburg/…` models appear in the model picker.
|
|
39
|
+
|
|
40
|
+
**In a terminal — the fallback:** works before opencode ever starts.
|
|
46
41
|
|
|
47
|
-
|
|
48
|
-
|
|
42
|
+
```bash
|
|
43
|
+
ufr connect --keys "<key1>,<key2>" # keys only
|
|
44
|
+
ufr connect --login fl240@uni-freiburg.de --password … --keys "…" # + VPN login
|
|
45
|
+
```
|
|
49
46
|
|
|
50
|
-
|
|
51
|
-
|
|
47
|
+
`ufr connect` verifies every key against UFR, assigns aliases (`key1`, `key2`,
|
|
48
|
+
…) automatically and accepts keys comma- or line-separated.
|
|
52
49
|
|
|
53
50
|
## Use
|
|
54
51
|
|
|
55
52
|
Models show up in opencode as `unifreiburg/<model>`, e.g.
|
|
56
|
-
`unifreiburg/ufr/coding-complex
|
|
53
|
+
`unifreiburg/ufr/coding-complex` or `unifreiburg/glm-5.3-flash-llmlb`. The
|
|
54
|
+
gateway starts itself the first time opencode needs it (the plugin waits up to
|
|
55
|
+
40 s) and shuts down after 5 minutes idle. You never manage it directly.
|
|
57
56
|
|
|
58
57
|
```bash
|
|
59
|
-
ufr status # gateway, keys, limits, breakers, spend today
|
|
58
|
+
ufr status # gateway, keys, limits, breakers, vpn, spend today
|
|
60
59
|
ufr stats # requests, tokens and cost
|
|
61
|
-
ufr keys
|
|
62
|
-
ufr keys
|
|
60
|
+
ufr keys list # stored aliases
|
|
61
|
+
ufr keys test # check your keys against UFR
|
|
63
62
|
ufr login show # the stored uni login
|
|
64
63
|
```
|
|
65
64
|
|
|
66
|
-
The
|
|
67
|
-
|
|
68
|
-
|
|
65
|
+
## The built-in VPN
|
|
66
|
+
|
|
67
|
+
On the campus network (or the VPN you already run) UFR is reached directly and
|
|
68
|
+
no tunnel opens. Off campus, the gateway:
|
|
69
|
+
|
|
70
|
+
1. logs in to `fortivpn.uni-freiburg.de` with your stored uni login (TLS),
|
|
71
|
+
2. negotiates PPP/IPCP over the Fortinet SSL-VPN channel (this is the same
|
|
72
|
+
protocol `openconnect --protocol=fortinet` speaks, implemented in pure
|
|
73
|
+
TypeScript — see `docs/fortinet-protocol.md`),
|
|
74
|
+
3. runs a **userspace TCP/IP stack inside the gateway process** and routes its
|
|
75
|
+
UFR calls through it via a localhost CONNECT proxy.
|
|
76
|
+
|
|
77
|
+
What that means in practice:
|
|
78
|
+
|
|
79
|
+
- **No TUN device, no routing table changes, no admin rights** — identical
|
|
80
|
+
code on Linux, macOS and Windows.
|
|
81
|
+
- **Your own VPNs keep working untouched in parallel** — the plugin's tunnel
|
|
82
|
+
exists only inside its own process and rides on whatever network the OS
|
|
83
|
+
provides.
|
|
84
|
+
- **Everything is encrypted twice**: the tunnel itself is TLS 1.3, and your
|
|
85
|
+
API calls are HTTPS end-to-end to `openwebui.uni-freiburg.de` inside it.
|
|
86
|
+
- `vpn.mode: "auto"` (default) only tunnels when UFR is unreachable directly;
|
|
87
|
+
`"always"` forces the tunnel (useful behind firewalls that block the campus
|
|
88
|
+
route).
|
|
89
|
+
- If the tunnel dies, it reconnects with backoff — no session is lost, the
|
|
90
|
+
gateway just waits until the path is back.
|
|
91
|
+
|
|
92
|
+
## More than one key
|
|
93
|
+
|
|
94
|
+
UFR allows one API key per account, so one key is one account's worth of
|
|
95
|
+
throughput. The key pool round-robins across all stored keys and keeps each
|
|
96
|
+
under UFR's per-key limits; adding more keys only helps if they belong to
|
|
97
|
+
different UFR accounts.
|
|
69
98
|
|
|
70
99
|
## What the gateway does
|
|
71
100
|
|
|
@@ -74,12 +103,6 @@ don't run or manage it directly.
|
|
|
74
103
|
- Enforces a pool-wide cap of 800 requests/hour across all keys and models —
|
|
75
104
|
UFR walls a model group for hours once it sees sustained traffic above
|
|
76
105
|
roughly 900/hour.
|
|
77
|
-
- **Connects to UFR on its own when you are off campus.** It logs in to the
|
|
78
|
-
uni's Fortinet gateway with your stored uni login, negotiates PPP/IPCP and
|
|
79
|
-
runs a userspace TCP/IP stack inside the gateway process: no TUN device, no
|
|
80
|
-
routes, no admin rights, identical on Linux, macOS and Windows. On campus it
|
|
81
|
-
talks to UFR directly and never opens the tunnel; `vpn.mode: "always"`
|
|
82
|
-
forces the tunnel.
|
|
83
106
|
- Runs a circuit breaker per model: after repeated failures it backs off from
|
|
84
107
|
30 s up to 60 minutes before probing again, instead of hammering a walled
|
|
85
108
|
model.
|
|
@@ -97,15 +120,10 @@ don't run or manage it directly.
|
|
|
97
120
|
unclean shutdown, the next start detects it's stale and takes it over
|
|
98
121
|
rather than refusing to start.
|
|
99
122
|
|
|
100
|
-
## More than one key
|
|
101
|
-
|
|
102
|
-
UFR allows one API key per account, so one key is one account's worth of
|
|
103
|
-
throughput. Adding more keys only helps if they belong to different UFR
|
|
104
|
-
accounts — it does not raise any one account's limit.
|
|
105
|
-
|
|
106
123
|
## Configuration
|
|
107
124
|
|
|
108
|
-
Non-secret settings live in a JSON file; keys always stay in
|
|
125
|
+
Non-secret settings live in a JSON file; keys and the uni login always stay in
|
|
126
|
+
the OS keyring.
|
|
109
127
|
|
|
110
128
|
- **Linux / macOS:** `~/.config/opencode-ufr/config.json` (state in
|
|
111
129
|
`~/.local/state/opencode-ufr`, cache in `~/.cache/opencode-ufr`, data in
|
|
@@ -152,9 +170,14 @@ except `port`, which is written on first start):
|
|
|
152
170
|
|
|
153
171
|
`port` is chosen once (preferred `47300`, next free port if taken) and then
|
|
154
172
|
stays fixed across restarts. `poolPerHour: 0` disables the pool limiter.
|
|
155
|
-
`transport.type: "direct"` never opens the tunnel
|
|
156
|
-
|
|
157
|
-
|
|
173
|
+
`transport.type: "direct"` never opens the tunnel.
|
|
174
|
+
|
|
175
|
+
## Releases
|
|
176
|
+
|
|
177
|
+
Releases are tag-driven: `scripts/release.sh 0.2.1` bumps, commits, tags and
|
|
178
|
+
pushes; the workflow then verifies on Ubuntu, macOS and Windows, stages the
|
|
179
|
+
package on npm (trusted publishing via OIDC — no stored tokens), and creates
|
|
180
|
+
the GitHub release after a maintainer approves the staged version with 2FA.
|
|
158
181
|
|
|
159
182
|
## Model data
|
|
160
183
|
|
package/models.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schema": 1,
|
|
3
|
-
"updated": "2026-
|
|
3
|
+
"updated": "2026-10-02",
|
|
4
4
|
"defaults": {
|
|
5
5
|
"context": 131072,
|
|
6
6
|
"max_output": 16384
|
|
@@ -60,7 +60,11 @@
|
|
|
60
60
|
"context": 256000
|
|
61
61
|
},
|
|
62
62
|
"qwen-3.5-397b-llmlb": {
|
|
63
|
-
"context": 262144
|
|
63
|
+
"context": 262144,
|
|
64
|
+
"price": {
|
|
65
|
+
"input": 0.1,
|
|
66
|
+
"output": 0.1
|
|
67
|
+
}
|
|
64
68
|
},
|
|
65
69
|
"qwen-3.6-27b-llmlb": {
|
|
66
70
|
"context": 262144
|
|
@@ -122,7 +126,7 @@
|
|
|
122
126
|
"vision": true
|
|
123
127
|
},
|
|
124
128
|
"ufr/reasoning-complex": {
|
|
125
|
-
"context":
|
|
129
|
+
"context": 1048576,
|
|
126
130
|
"price": {
|
|
127
131
|
"input": 0.1,
|
|
128
132
|
"output": 0.1
|
|
@@ -135,7 +139,12 @@
|
|
|
135
139
|
"context": 256000
|
|
136
140
|
},
|
|
137
141
|
"ufr/vision-complex": {
|
|
138
|
-
"context":
|
|
142
|
+
"context": 1048576,
|
|
143
|
+
"vision": true,
|
|
144
|
+
"price": {
|
|
145
|
+
"input": 0.1,
|
|
146
|
+
"output": 0.1
|
|
147
|
+
}
|
|
139
148
|
},
|
|
140
149
|
"ufr/vision-fast": {
|
|
141
150
|
"context": 131044
|
|
@@ -201,6 +210,15 @@
|
|
|
201
210
|
"input": 0.3,
|
|
202
211
|
"output": 0.9
|
|
203
212
|
}
|
|
213
|
+
},
|
|
214
|
+
"deepseek-v4.1-flash-llmlb": {
|
|
215
|
+
"context": 1048576,
|
|
216
|
+
"vision": true,
|
|
217
|
+
"price": {
|
|
218
|
+
"input": 0.1,
|
|
219
|
+
"output": 0.1
|
|
220
|
+
},
|
|
221
|
+
"note": "context 1048576 probed live 2026-10-02 (vLLM ContextWindowExceededError named the limit); prices from the UFR portal (Lokale Modelle)"
|
|
204
222
|
}
|
|
205
223
|
}
|
|
206
224
|
}
|
package/package.json
CHANGED
|
@@ -1,16 +1,44 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "opencode-ufr",
|
|
3
|
-
"version": "0.2.
|
|
3
|
+
"version": "0.2.2",
|
|
4
4
|
"description": "Uni Freiburg (UFR) Open WebUI models in opencode — local gateway with key pool, UFR-aware rate limiting, circuit breaker and fallbacks",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|
|
7
7
|
"main": "./src/plugin/index.ts",
|
|
8
|
-
"exports": {
|
|
9
|
-
|
|
10
|
-
|
|
11
|
-
"
|
|
12
|
-
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
"
|
|
8
|
+
"exports": {
|
|
9
|
+
".": "./src/plugin/index.ts"
|
|
10
|
+
},
|
|
11
|
+
"bin": {
|
|
12
|
+
"ufr": "bin/ufr.ts",
|
|
13
|
+
"opencode-ufr": "bin/ufr.ts"
|
|
14
|
+
},
|
|
15
|
+
"files": [
|
|
16
|
+
"bin",
|
|
17
|
+
"src",
|
|
18
|
+
"models.json",
|
|
19
|
+
"README.md",
|
|
20
|
+
"LICENSE"
|
|
21
|
+
],
|
|
22
|
+
"engines": {
|
|
23
|
+
"bun": ">=1.3.0"
|
|
24
|
+
},
|
|
25
|
+
"repository": {
|
|
26
|
+
"type": "git",
|
|
27
|
+
"url": "git+https://github.com/FinleyLaempe/opencode-ufr.git"
|
|
28
|
+
},
|
|
29
|
+
"keywords": [
|
|
30
|
+
"opencode",
|
|
31
|
+
"opencode-plugin",
|
|
32
|
+
"uni-freiburg",
|
|
33
|
+
"open-webui"
|
|
34
|
+
],
|
|
35
|
+
"scripts": {
|
|
36
|
+
"test": "bun test",
|
|
37
|
+
"typecheck": "tsc --noEmit",
|
|
38
|
+
"prepublishOnly": "bun run typecheck && bun test"
|
|
39
|
+
},
|
|
40
|
+
"devDependencies": {
|
|
41
|
+
"@types/bun": "^1.3.0",
|
|
42
|
+
"typescript": "^5.9.0"
|
|
43
|
+
}
|
|
16
44
|
}
|
package/src/cli/catalog.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { UfrModel } from "../daemon/catalog"
|
|
2
|
-
import { BUNDLED_MODELS,
|
|
2
|
+
import { BUNDLED_MODELS, loadBundledModels, loadUfrModels } from "../daemon/catalog-source"
|
|
3
3
|
import { loadConfig } from "../shared/config"
|
|
4
4
|
import type { ModelsFile } from "../shared/models-file"
|
|
5
5
|
import type { CliDeps } from "./index"
|
|
@@ -27,8 +27,7 @@ export async function cmdCatalogDiff(d: CliDeps): Promise<number> {
|
|
|
27
27
|
return 1
|
|
28
28
|
}
|
|
29
29
|
const log = (m: string) => d.io.err(`${m}\n`)
|
|
30
|
-
const mf = await
|
|
31
|
-
bundledPath: BUNDLED_MODELS, fetch: d.fetch, log })
|
|
30
|
+
const mf = await loadBundledModels({ bundledPath: BUNDLED_MODELS })
|
|
32
31
|
const ufr = await loadUfrModels({ baseUrl: cfg.upstream.baseUrl, key, cachePath: d.paths.ufrModelsCache, fetch: d.fetch, log })
|
|
33
32
|
if (ufr.source !== "remote") {
|
|
34
33
|
d.io.err(`cannot read UFR's live model list: ${ufr.error}\n`)
|
|
@@ -1,48 +1,22 @@
|
|
|
1
1
|
import { fileURLToPath } from "node:url"
|
|
2
|
-
import { readJson,
|
|
2
|
+
import { readJson, writeFileAtomic } from "../shared/fs"
|
|
3
3
|
import { type ModelsFile, validateModelsFile } from "../shared/models-file"
|
|
4
4
|
import { VPN_MESSAGE, isVpnPage } from "../shared/vpn"
|
|
5
5
|
import { type UfrModel, parseUfrModels } from "./catalog"
|
|
6
6
|
|
|
7
7
|
export type FetchLike = (url: string, init?: RequestInit) => Promise<Response>
|
|
8
|
-
export type ModelsSource = "
|
|
8
|
+
export type ModelsSource = "bundled"
|
|
9
9
|
|
|
10
10
|
export const BUNDLED_MODELS = fileURLToPath(new URL("../../models.json", import.meta.url))
|
|
11
11
|
|
|
12
|
-
/**
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
12
|
+
/**
|
|
13
|
+
* The fixes file ships with the package (models.json in the release) — no
|
|
14
|
+
* remote pull. Model data updates arrive with plugin updates; UFR's live
|
|
15
|
+
* model list supplies everything else.
|
|
16
|
+
*/
|
|
17
|
+
export async function loadBundledModels(o: {
|
|
17
18
|
bundledPath?: string
|
|
18
|
-
fetch: FetchLike
|
|
19
|
-
log: (m: string) => void
|
|
20
19
|
}): Promise<{ file: ModelsFile; source: ModelsSource }> {
|
|
21
|
-
const cached = await readJson(o.cachePath)
|
|
22
|
-
const etag = cached ? (await readText(o.etagPath))?.trim() : undefined
|
|
23
|
-
try {
|
|
24
|
-
const res = await o.fetch(o.url, {
|
|
25
|
-
headers: etag ? { "If-None-Match": etag } : {},
|
|
26
|
-
signal: AbortSignal.timeout(15_000),
|
|
27
|
-
})
|
|
28
|
-
if (res.status === 304 && cached) return { file: validateModelsFile(cached), source: "remote" }
|
|
29
|
-
if (!res.ok) throw new Error(`HTTP ${res.status}`)
|
|
30
|
-
const text = await res.text()
|
|
31
|
-
const file = validateModelsFile(JSON.parse(text)) // validate before it can replace a good cache
|
|
32
|
-
await writeFileAtomic(o.cachePath, text)
|
|
33
|
-
const tag = res.headers.get("etag")
|
|
34
|
-
if (tag) await writeFileAtomic(o.etagPath, tag)
|
|
35
|
-
return { file, source: "remote" }
|
|
36
|
-
} catch (e) {
|
|
37
|
-
o.log(`models.json: remote copy unavailable or invalid (${(e as Error).message}) — using the ${cached ? "cached" : "bundled"} copy`)
|
|
38
|
-
}
|
|
39
|
-
if (cached) {
|
|
40
|
-
try {
|
|
41
|
-
return { file: validateModelsFile(cached), source: "cache" }
|
|
42
|
-
} catch (e) {
|
|
43
|
-
o.log(`models.json: cached copy invalid (${(e as Error).message})`)
|
|
44
|
-
}
|
|
45
|
-
}
|
|
46
20
|
const bundled: unknown = JSON.parse(await Bun.file(o.bundledPath ?? BUNDLED_MODELS).text())
|
|
47
21
|
return { file: validateModelsFile(bundled), source: "bundled" }
|
|
48
22
|
}
|
package/src/daemon/catalog.ts
CHANGED
|
@@ -46,19 +46,23 @@ function toPrice(p: Price | undefined): ModelPrice | null {
|
|
|
46
46
|
return { input: p.input, output: p.output, cacheRead: p.cache_read ?? p.input, cacheWrite: p.cache_write ?? p.input }
|
|
47
47
|
}
|
|
48
48
|
|
|
49
|
-
export function buildCatalog(ufr: UfrModel[], file: ModelsFile, opts: { allowPaid: boolean }): Catalog {
|
|
49
|
+
export function buildCatalog(ufr: UfrModel[], file: ModelsFile, opts: { allowPaid: boolean; probes?: Record<string, { context: number; at: number }> }): Catalog {
|
|
50
50
|
const warnings: string[] = []
|
|
51
51
|
const exclude = new Set(file.exclude)
|
|
52
|
+
const probes = opts.probes ?? {}
|
|
52
53
|
const models = new Map<string, Model>()
|
|
53
54
|
for (const u of ufr) {
|
|
54
55
|
const e = file.models[u.id]
|
|
56
|
+
// precedence: models.json entry (measured, ships with releases) > auto-probe > defaults
|
|
57
|
+
const probed = probes[u.id]?.context
|
|
58
|
+
const context = e?.context ?? probed ?? file.defaults.context
|
|
55
59
|
models.set(u.id, {
|
|
56
60
|
id: u.id,
|
|
57
61
|
name: u.name,
|
|
58
62
|
tier: u.tier,
|
|
59
63
|
vision: e?.vision ?? u.vision,
|
|
60
64
|
tools: e?.tools ?? true,
|
|
61
|
-
context
|
|
65
|
+
context,
|
|
62
66
|
maxOutput: e?.max_output ?? file.defaults.max_output,
|
|
63
67
|
price: toPrice(e?.price),
|
|
64
68
|
hasEntry: e !== undefined,
|
package/src/daemon/daemon.ts
CHANGED
|
@@ -11,8 +11,11 @@ import { startOfLocalDay } from "../shared/time"
|
|
|
11
11
|
import { VERSION } from "../shared/version"
|
|
12
12
|
import { type BreakerState, BreakerRegistry } from "./breaker"
|
|
13
13
|
import { type Catalog, EMPTY_CATALOG, buildCatalog, listModels } from "./catalog"
|
|
14
|
-
import { type FetchLike, type ModelsSource,
|
|
14
|
+
import { type FetchLike, type ModelsSource, loadBundledModels, loadUfrModels } from "./catalog-source"
|
|
15
|
+
import type { ModelsFile } from "../shared/models-file"
|
|
15
16
|
import { type KeyInfo, KeyPool, type KeySnapshot } from "./keypool"
|
|
17
|
+
import { loadProbes, probeContextLimit, saveProbe } from "./probe"
|
|
18
|
+
import { type UfrModel } from "./catalog"
|
|
16
19
|
import { Router } from "./router"
|
|
17
20
|
import { startServer } from "./server"
|
|
18
21
|
import { Stats } from "./stats"
|
|
@@ -58,6 +61,8 @@ export type DaemonOptions = {
|
|
|
58
61
|
idleCheckMs?: number
|
|
59
62
|
/** Backoff after UFR's model list failed to load (VPN down, UFR unreachable); the last delay repeats. */
|
|
60
63
|
catalogRetryMs?: number[]
|
|
64
|
+
/** Auto-probe the context limit of unknown models (default: on). */
|
|
65
|
+
probes?: boolean
|
|
61
66
|
onStopped?: () => void
|
|
62
67
|
}
|
|
63
68
|
|
|
@@ -243,7 +248,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
243
248
|
const timers: ReturnType<typeof setInterval>[] = []
|
|
244
249
|
let stopping = false
|
|
245
250
|
// UFR's model list failed (VPN not up yet, UFR unreachable): retry with backoff
|
|
246
|
-
// (60 s → 120 s → 300 s → 900 s, then every 900 s
|
|
251
|
+
// (60 s → 120 s → 300 s → 900 s, then every 900 s)
|
|
247
252
|
// so connecting the VPN later needs no daemon restart.
|
|
248
253
|
const retryMs = o.catalogRetryMs ?? [60_000, 120_000, 300_000, 900_000]
|
|
249
254
|
let retryTimer: ReturnType<typeof setTimeout> | null = null
|
|
@@ -259,7 +264,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
259
264
|
}
|
|
260
265
|
const scheduleCatalogRetry = () => {
|
|
261
266
|
if (stopping || retryTimer) return
|
|
262
|
-
const delay = Math.min(retryMs[Math.min(retries, retryMs.length - 1)]!,
|
|
267
|
+
const delay = Math.min(retryMs[Math.min(retries, retryMs.length - 1)]!, 900_000)
|
|
263
268
|
retries++
|
|
264
269
|
retryTimer = setTimeout(() => {
|
|
265
270
|
retryTimer = null
|
|
@@ -273,12 +278,50 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
273
278
|
const reach: StatusJson["upstream"] = { ok: null, message: "", at: 0 }
|
|
274
279
|
let catalog: Catalog = EMPTY_CATALOG
|
|
275
280
|
let catalogInfo: Omit<StatusJson["catalog"], "models" | "warnings"> = { source: "none", ufrSource: "none", loadedAt: 0 }
|
|
281
|
+
let probeStore = loadProbes((k) => db.getKv(k))
|
|
282
|
+
let probing = false
|
|
283
|
+
/** Measure models UFR serves but models.json does not describe (one at a time, off the hot path). */
|
|
284
|
+
const scheduleProbes = async (ufrModels: UfrModel[], file: ModelsFile): Promise<void> => {
|
|
285
|
+
if (probing || keyInfos.length === 0 || o.probes === false) return
|
|
286
|
+
const unknown = ufrModels
|
|
287
|
+
.map((u) => u.id)
|
|
288
|
+
.filter((id) => file.models[id] === undefined && probeStore[id] === undefined)
|
|
289
|
+
if (unknown.length === 0) return
|
|
290
|
+
probing = true
|
|
291
|
+
try {
|
|
292
|
+
for (const id of unknown.slice(0, 2)) {
|
|
293
|
+
if (stopping) break
|
|
294
|
+
const result = await probeContextLimit({
|
|
295
|
+
model: id,
|
|
296
|
+
baseUrl: config.upstream.baseUrl,
|
|
297
|
+
key: keyInfos[0]!.secret,
|
|
298
|
+
transport,
|
|
299
|
+
log,
|
|
300
|
+
})
|
|
301
|
+
if (result) {
|
|
302
|
+
if (stopping) break
|
|
303
|
+
saveProbe(probeStore, id, result, (k, v) => {
|
|
304
|
+
if (stopping) return
|
|
305
|
+
try { db.setKv(k, v) } catch { /* db already closed */ }
|
|
306
|
+
})
|
|
307
|
+
log(`context probe ${id}: ${result.context} tokens (${result.how})`)
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
if (unknown.length > 0) {
|
|
311
|
+
const mf2 = await loadBundledModels({ bundledPath: o.bundledModelsPath })
|
|
312
|
+
catalog = buildCatalog(ufrModels, mf2.file, { allowPaid: config.allowPaid, probes: probeStore })
|
|
313
|
+
catalogInfo = { ...catalogInfo, loadedAt: now() }
|
|
314
|
+
}
|
|
315
|
+
} finally {
|
|
316
|
+
probing = false
|
|
317
|
+
}
|
|
318
|
+
}
|
|
276
319
|
const refreshCatalog = async () => {
|
|
277
|
-
|
|
278
|
-
|
|
320
|
+
// fixes ship with the package; UFR's live list is fetched through the transport (vpn-aware)
|
|
321
|
+
const mf = await loadBundledModels({ bundledPath: o.bundledModelsPath })
|
|
279
322
|
const ufr = await loadUfrModels({ baseUrl: config.upstream.baseUrl, key: keyInfos[0]?.secret ?? null,
|
|
280
323
|
cachePath: p.ufrModelsCache, fetch: (u, i) => transport.fetch(u, i), log })
|
|
281
|
-
catalog = buildCatalog(ufr.models, mf.file, { allowPaid: config.allowPaid })
|
|
324
|
+
catalog = buildCatalog(ufr.models, mf.file, { allowPaid: config.allowPaid, probes: probeStore })
|
|
282
325
|
catalogInfo = { source: mf.source, ufrSource: ufr.source, loadedAt: now() }
|
|
283
326
|
if (ufr.error) Object.assign(reach, { ok: false, message: ufr.error, at: now() })
|
|
284
327
|
else if (reach.ok !== true) Object.assign(reach, { ok: true, message: "", at: now() })
|
|
@@ -286,6 +329,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
286
329
|
// Keys are read once at start: without one a retry can never succeed.
|
|
287
330
|
if (ufr.error && keyInfos.length > 0) scheduleCatalogRetry()
|
|
288
331
|
else clearCatalogRetry()
|
|
332
|
+
void scheduleProbes(ufr.models, mf.file).catch(() => {}) // must not outlive a shutdown
|
|
289
333
|
}
|
|
290
334
|
await refreshCatalog()
|
|
291
335
|
|
|
@@ -375,8 +419,9 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
375
419
|
await writeFileAtomic(p.daemonFile, JSON.stringify({ port, pid: process.pid, version: VERSION, startedAt }) + "\n", 0o600)
|
|
376
420
|
|
|
377
421
|
timers.push(setInterval(() => {
|
|
422
|
+
// pick up new UFR models periodically (the fixes file only changes with a plugin update)
|
|
378
423
|
refreshCatalog().catch((e) => log(`catalog refresh failed: ${(e as Error).message}`))
|
|
379
|
-
},
|
|
424
|
+
}, 6 * 3_600_000))
|
|
380
425
|
timers.push(setInterval(() => db.setKv("breakers", JSON.stringify(breakers.snapshot())), 30_000))
|
|
381
426
|
timers.push(setInterval(() => db.prune(now() - 90 * 86_400_000), 86_400_000))
|
|
382
427
|
if (o.exitOnIdle !== false) {
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Auto-probe for models UFR serves but models.json does not describe yet:
|
|
3
|
+
* one oversized request makes vLLM name its real context limit in the error
|
|
4
|
+
* ("This model's maximum context length is N tokens"). Results are stored in
|
|
5
|
+
* the stats DB and survive restarts — new models measure themselves, no
|
|
6
|
+
* patches needed.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import type { FetchLike } from "./catalog-source"
|
|
10
|
+
|
|
11
|
+
const LIMIT_RES = [
|
|
12
|
+
/context length is (\d+)/i,
|
|
13
|
+
/maximum context length[^\d]{0,40}(\d+)/i,
|
|
14
|
+
/context window[^\d]{0,30}(\d+)/i,
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
export function parseLimitFromBody(body: string): number | null {
|
|
18
|
+
for (const re of LIMIT_RES) {
|
|
19
|
+
const m = re.exec(body)
|
|
20
|
+
if (m) {
|
|
21
|
+
const n = Number(m[1])
|
|
22
|
+
if (Number.isFinite(n) && n >= 1024) return n
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
return null
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** Prompt filler: ~4.5 chars per token for prose-like text. */
|
|
29
|
+
function fillerForTokens(tokens: number): string {
|
|
30
|
+
return "The quick brown fox jumps over the lazy dog. ".repeat(Math.ceil((tokens * 4.5) / 45))
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export type ProbeResult = { context: number; how: "error-named" | "accepted-floor" } | null
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Ladder probe: sizes in tokens. The first error either names the limit
|
|
37
|
+
* (done) or marks the ceiling; the last success sets the floor.
|
|
38
|
+
*/
|
|
39
|
+
export async function probeContextLimit(o: {
|
|
40
|
+
model: string
|
|
41
|
+
baseUrl: string
|
|
42
|
+
key: string
|
|
43
|
+
transport: { name: string; fetch(url: string, init?: RequestInit): Promise<Response> }
|
|
44
|
+
sizes?: number[] // token counts to try, in order
|
|
45
|
+
log: (m: string) => void
|
|
46
|
+
}): Promise<ProbeResult> {
|
|
47
|
+
const sizes = o.sizes ?? [300_000, 600_000, 1_048_000, 1_500_000]
|
|
48
|
+
let floor = 0
|
|
49
|
+
for (const tokens of sizes) {
|
|
50
|
+
const body = JSON.stringify({
|
|
51
|
+
model: o.model,
|
|
52
|
+
max_tokens: 1,
|
|
53
|
+
messages: [{ role: "user", content: fillerForTokens(tokens) + "\n\nReply with exactly: OK" }],
|
|
54
|
+
})
|
|
55
|
+
let res: Response
|
|
56
|
+
try {
|
|
57
|
+
res = await o.transport.fetch(`${o.baseUrl}/chat/completions`, {
|
|
58
|
+
method: "POST",
|
|
59
|
+
headers: { Authorization: `Bearer ${o.key}`, "Content-Type": "application/json" },
|
|
60
|
+
body,
|
|
61
|
+
signal: AbortSignal.timeout(180_000),
|
|
62
|
+
redirect: "manual",
|
|
63
|
+
})
|
|
64
|
+
} catch (e) {
|
|
65
|
+
o.log(`context probe ${o.model}: transport error at ${tokens} tokens (${(e as Error).message})`)
|
|
66
|
+
return floor > 0 ? { context: floor, how: "accepted-floor" } : null
|
|
67
|
+
}
|
|
68
|
+
if (res.ok) {
|
|
69
|
+
const j = (await res.json().catch(() => null)) as { usage?: { prompt_tokens?: number } } | null
|
|
70
|
+
floor = j?.usage?.prompt_tokens ?? tokens
|
|
71
|
+
continue // accepted: try the next rung
|
|
72
|
+
}
|
|
73
|
+
const text = await res.text().catch(() => "")
|
|
74
|
+
const named = parseLimitFromBody(text)
|
|
75
|
+
if (named) {
|
|
76
|
+
o.log(`context probe ${o.model}: the server names the limit: ${named} tokens`)
|
|
77
|
+
return { context: named, how: "error-named" }
|
|
78
|
+
}
|
|
79
|
+
if (res.status === 400 || res.status === 413) {
|
|
80
|
+
o.log(`context probe ${o.model}: rejected at ${tokens} tokens without naming a limit`)
|
|
81
|
+
return floor > 0 ? { context: floor, how: "accepted-floor" } : null
|
|
82
|
+
}
|
|
83
|
+
// rate limit / auth / anything else: not a context answer, retry later
|
|
84
|
+
o.log(`context probe ${o.model}: HTTP ${res.status} — will retry later`)
|
|
85
|
+
return null
|
|
86
|
+
}
|
|
87
|
+
return floor > 0 ? { context: floor, how: "accepted-floor" } : null
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export const PROBE_KV_KEY = "ctxprobes"
|
|
91
|
+
|
|
92
|
+
type ProbeStore = Record<string, { context: number; at: number }>
|
|
93
|
+
|
|
94
|
+
export function loadProbes(getKv: (k: string) => string | null): ProbeStore {
|
|
95
|
+
try {
|
|
96
|
+
const raw = getKv(PROBE_KV_KEY)
|
|
97
|
+
const parsed = raw ? JSON.parse(raw) : null
|
|
98
|
+
return parsed && typeof parsed === "object" ? (parsed as ProbeStore) : {}
|
|
99
|
+
} catch {
|
|
100
|
+
return {}
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function saveProbe(
|
|
105
|
+
store: ProbeStore,
|
|
106
|
+
model: string,
|
|
107
|
+
result: ProbeResult,
|
|
108
|
+
setKv: (k: string, v: string) => void,
|
|
109
|
+
): void {
|
|
110
|
+
if (!result) return
|
|
111
|
+
store[model] = { context: result.context, at: Date.now() }
|
|
112
|
+
setKv(PROBE_KV_KEY, JSON.stringify(store))
|
|
113
|
+
}
|
package/src/daemon/vpn/tcp.ts
CHANGED
|
@@ -49,6 +49,8 @@ const DEFAULT_MSS = 1360 // fits a 1400-byte tunnel MTU with IP+TCP headers
|
|
|
49
49
|
const DEFAULT_RTO_MS = 500
|
|
50
50
|
const DEFAULT_MAX_RETRANSMITS = 6
|
|
51
51
|
const RECV_WINDOW = 65535
|
|
52
|
+
const CWND_START = 4 // segments in flight before the first ACK (slow start)
|
|
53
|
+
const CWND_MAX = 64
|
|
52
54
|
|
|
53
55
|
export function freshLocalPort(): number {
|
|
54
56
|
return 32768 + Math.floor(Math.random() * 20000)
|
|
@@ -65,8 +67,11 @@ export class TcpConn {
|
|
|
65
67
|
private rcvNext = 0 // next byte we expect from the peer (RCV.NXT)
|
|
66
68
|
private peerWindow = 0
|
|
67
69
|
private peerMss = 536
|
|
68
|
-
private outQueue: QueueEntry[] = []
|
|
70
|
+
private outQueue: QueueEntry[] = [] // not yet sent
|
|
71
|
+
private flightQueue: QueueEntry[] = [] // sent, not yet acked (retransmit source)
|
|
69
72
|
private earlyQueue: Uint8Array[] = [] // application data written while syn-sent
|
|
73
|
+
private cwnd = CWND_START // congestion window in segments (slow start)
|
|
74
|
+
private dupAcks = 0
|
|
70
75
|
private finQueued = false
|
|
71
76
|
private finSent = false
|
|
72
77
|
private closeFired = false
|
|
@@ -145,7 +150,7 @@ export class TcpConn {
|
|
|
145
150
|
}
|
|
146
151
|
|
|
147
152
|
get buffered(): number {
|
|
148
|
-
return this.outQueue.reduce((n, e) => n + e.bytes.length, 0)
|
|
153
|
+
return this.outQueue.reduce((n, e) => n + e.bytes.length, 0) + this.flightBytes()
|
|
149
154
|
}
|
|
150
155
|
|
|
151
156
|
// -- inbound -------------------------------------------------------------
|
|
@@ -219,7 +224,6 @@ export class TcpConn {
|
|
|
219
224
|
return
|
|
220
225
|
}
|
|
221
226
|
|
|
222
|
-
if (this.finQueued && !this.outQueue.some((e) => e.len > 0) && this.unacked === 0) this.maybeSendFin()
|
|
223
227
|
this.flush()
|
|
224
228
|
}
|
|
225
229
|
|
|
@@ -239,17 +243,16 @@ export class TcpConn {
|
|
|
239
243
|
const seg: TcpSegment = { header, payload, options }
|
|
240
244
|
const bytes = buildDatagram({ src: this.info.local, dst: this.info.remote, segment: seg })
|
|
241
245
|
this.sendPkt(bytes)
|
|
242
|
-
if (queueLen > 0) this.
|
|
246
|
+
if (queueLen > 0) this.flightQueue.push({ seq: this.seq, len: queueLen, bytes }) // sent: tracks in flight
|
|
243
247
|
}
|
|
244
248
|
|
|
245
249
|
private flush(): void {
|
|
246
|
-
let inFlight = this.
|
|
250
|
+
let inFlight = this.flightBytes()
|
|
247
251
|
let sent = false
|
|
252
|
+
const limit = Math.min(Math.max(this.peerWindow, 1), this.cwnd * this.effMss())
|
|
248
253
|
while (this.outQueue.length > 0) {
|
|
249
254
|
const e = this.outQueue[0]!
|
|
250
|
-
if (e.
|
|
251
|
-
const window = Math.max(this.peerWindow, 1)
|
|
252
|
-
if (inFlight + e.bytes.length > window) break
|
|
255
|
+
if (inFlight + e.bytes.length > limit) break
|
|
253
256
|
const header = {
|
|
254
257
|
srcPort: this.info.localPort,
|
|
255
258
|
dstPort: this.info.remotePort,
|
|
@@ -262,15 +265,23 @@ export class TcpConn {
|
|
|
262
265
|
this.sendPkt(buildDatagram({ src: this.info.local, dst: this.info.remote, segment: { header, payload: e.bytes } }))
|
|
263
266
|
inFlight += e.bytes.length
|
|
264
267
|
sent = true
|
|
265
|
-
this.outQueue.shift()
|
|
268
|
+
this.flightQueue.push(this.outQueue.shift()!)
|
|
266
269
|
}
|
|
267
270
|
if (sent) this.armRto()
|
|
268
271
|
if (this.finQueued && this.buffered === 0) this.maybeSendFin()
|
|
269
272
|
if (this.buffered === 0) this.events.onDrain?.(this)
|
|
270
273
|
}
|
|
271
274
|
|
|
275
|
+
private effMss(): number {
|
|
276
|
+
return Math.min(this.peerMss, this.opts.mss ?? DEFAULT_MSS)
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
private flightBytes(): number {
|
|
280
|
+
return this.flightQueue.reduce((n, e) => n + e.bytes.length, 0)
|
|
281
|
+
}
|
|
282
|
+
|
|
272
283
|
private maybeSendFin(): void {
|
|
273
|
-
if (this.finSent || this.
|
|
284
|
+
if (this.finSent || this.flightQueue.some((e) => e.len > 0)) return // FIN already in flight
|
|
274
285
|
this.finSent = true
|
|
275
286
|
this.sendSegment({ flags: FIN | ACK }, new Uint8Array(0), 1)
|
|
276
287
|
this.seq = (this.seq + 1) >>> 0
|
|
@@ -297,17 +308,29 @@ export class TcpConn {
|
|
|
297
308
|
if (window > 0) this.peerWindow = window
|
|
298
309
|
const unacked = this.unacked
|
|
299
310
|
const newly = (ack - this.acked) >>> 0
|
|
300
|
-
if (newly === 0 || newly > unacked)
|
|
311
|
+
if (newly === 0 || newly > unacked) {
|
|
312
|
+
// duplicate ACK: the peer is missing something after `acked`
|
|
313
|
+
if (unacked > 0 && ++this.dupAcks >= 3) {
|
|
314
|
+
this.dupAcks = 0
|
|
315
|
+
this.cwnd = Math.max(Math.floor(this.cwnd / 2), CWND_START) // multiplicative decrease
|
|
316
|
+
const first = this.flightQueue[0]!
|
|
317
|
+
if (first) this.sendPkt(first.bytes) // fast retransmit, no RTO wait
|
|
318
|
+
}
|
|
319
|
+
return
|
|
320
|
+
}
|
|
321
|
+
this.dupAcks = 0
|
|
322
|
+
// slow start: one segment of window per acked segment, bounded by the peer window
|
|
323
|
+
this.cwnd = Math.min(Math.ceil(this.cwnd + newly / this.effMss()), CWND_MAX)
|
|
301
324
|
this.acked = ack
|
|
302
325
|
this.retransmits = 0
|
|
303
326
|
this.rto = this.opts.rtoMs ?? DEFAULT_RTO_MS
|
|
304
|
-
// drop fully acknowledged queue
|
|
305
|
-
while (this.
|
|
306
|
-
const e = this.
|
|
327
|
+
// drop fully acknowledged entries from the flight queue
|
|
328
|
+
while (this.flightQueue.length > 0) {
|
|
329
|
+
const e = this.flightQueue[0]!
|
|
307
330
|
if ((e.seq + e.len) >>> 0 > ack || (e.len === 0 && e.seq + e.bytes.length > ack)) break
|
|
308
|
-
this.
|
|
331
|
+
this.flightQueue.shift()
|
|
309
332
|
}
|
|
310
|
-
if (this.
|
|
333
|
+
if (this.flightQueue.length === 0 && this.unacked === 0) {
|
|
311
334
|
if (this.timer) { clearTimeout(this.timer); this.timer = null }
|
|
312
335
|
}
|
|
313
336
|
if (this.state === "fin-wait-1" && this.unacked === 0) this.state = "fin-wait-2"
|
|
@@ -335,8 +358,13 @@ export class TcpConn {
|
|
|
335
358
|
return
|
|
336
359
|
}
|
|
337
360
|
this.rto = Math.min(this.rto * 2, 5_000)
|
|
338
|
-
|
|
339
|
-
this.
|
|
361
|
+
this.cwnd = CWND_START // collapse the window: the burst was too much
|
|
362
|
+
this.dupAcks = 0
|
|
363
|
+
// true go-back-N: resend EVERY unacked segment, not just the first —
|
|
364
|
+
// a burst can lose several at once, one-per-RTO would never recover
|
|
365
|
+
for (const e of this.flightQueue) {
|
|
366
|
+
this.sendPkt(e.bytes)
|
|
367
|
+
}
|
|
340
368
|
this.armRto()
|
|
341
369
|
}
|
|
342
370
|
|
package/src/shared/config.ts
CHANGED
|
@@ -19,7 +19,6 @@ export type Config = {
|
|
|
19
19
|
breaker: { tripThreshold: number; ladderS: number[]; probeTimeoutS: number }
|
|
20
20
|
allowPaid: boolean
|
|
21
21
|
dailyBudgetUsd: number
|
|
22
|
-
catalog: { url: string; refreshHours: number }
|
|
23
22
|
idleShutdownMin: number
|
|
24
23
|
}
|
|
25
24
|
|
|
@@ -44,7 +43,6 @@ export const DEFAULTS: Config = {
|
|
|
44
43
|
breaker: { tripThreshold: 3, ladderS: [30, 120, 300, 900, 1800, 3600], probeTimeoutS: 120 },
|
|
45
44
|
allowPaid: false,
|
|
46
45
|
dailyBudgetUsd: 20,
|
|
47
|
-
catalog: { url: "https://raw.githubusercontent.com/FinleyLaempe/opencode-ufr/main/models.json", refreshHours: 6 },
|
|
48
46
|
idleShutdownMin: 5,
|
|
49
47
|
}
|
|
50
48
|
|
|
@@ -72,8 +70,6 @@ const RULES: Record<string, Rule> = {
|
|
|
72
70
|
"breaker.probeTimeoutS": "int>=1",
|
|
73
71
|
allowPaid: "bool",
|
|
74
72
|
dailyBudgetUsd: "num>=0",
|
|
75
|
-
"catalog.url": "str",
|
|
76
|
-
"catalog.refreshHours": "int>=1",
|
|
77
73
|
idleShutdownMin: "int>=1",
|
|
78
74
|
}
|
|
79
75
|
|
|
@@ -88,7 +84,6 @@ const MAX: Record<string, number> = {
|
|
|
88
84
|
"limits.poolMaxWaitS": 86_400,
|
|
89
85
|
"breaker.ladderS": 86_400,
|
|
90
86
|
"breaker.probeTimeoutS": 86_400,
|
|
91
|
-
"catalog.refreshHours": 168,
|
|
92
87
|
idleShutdownMin: 1_440,
|
|
93
88
|
}
|
|
94
89
|
|
package/src/shared/paths.ts
CHANGED
|
@@ -12,8 +12,6 @@ export type Paths = {
|
|
|
12
12
|
lockFile: string
|
|
13
13
|
logFile: string
|
|
14
14
|
statsDb: string
|
|
15
|
-
modelsCache: string
|
|
16
|
-
modelsEtag: string
|
|
17
15
|
ufrModelsCache: string
|
|
18
16
|
}
|
|
19
17
|
|
|
@@ -55,8 +53,6 @@ export function resolvePaths(
|
|
|
55
53
|
lockFile: join(stateDir, "daemon.lock"),
|
|
56
54
|
logFile: join(stateDir, "daemon.log"),
|
|
57
55
|
statsDb: join(dataDir, "stats.db"),
|
|
58
|
-
modelsCache: join(cacheDir, "models.json"),
|
|
59
|
-
modelsEtag: join(cacheDir, "models.etag"),
|
|
60
56
|
ufrModelsCache: join(cacheDir, "ufr-models.json"),
|
|
61
57
|
}
|
|
62
58
|
}
|