opencode-ufr 0.2.1 → 0.2.2
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/models.json +22 -4
- package/package.json +1 -1
- package/src/cli/catalog.ts +2 -3
- package/src/daemon/catalog-source.ts +8 -34
- package/src/daemon/catalog.ts +6 -2
- package/src/daemon/daemon.ts +52 -7
- package/src/daemon/probe.ts +113 -0
- package/src/daemon/vpn/tcp.ts +46 -18
- package/src/shared/config.ts +0 -5
- package/src/shared/paths.ts +0 -4
package/models.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"schema": 1,
|
|
3
|
-
"updated": "2026-
|
|
3
|
+
"updated": "2026-10-02",
|
|
4
4
|
"defaults": {
|
|
5
5
|
"context": 131072,
|
|
6
6
|
"max_output": 16384
|
|
@@ -60,7 +60,11 @@
|
|
|
60
60
|
"context": 256000
|
|
61
61
|
},
|
|
62
62
|
"qwen-3.5-397b-llmlb": {
|
|
63
|
-
"context": 262144
|
|
63
|
+
"context": 262144,
|
|
64
|
+
"price": {
|
|
65
|
+
"input": 0.1,
|
|
66
|
+
"output": 0.1
|
|
67
|
+
}
|
|
64
68
|
},
|
|
65
69
|
"qwen-3.6-27b-llmlb": {
|
|
66
70
|
"context": 262144
|
|
@@ -122,7 +126,7 @@
|
|
|
122
126
|
"vision": true
|
|
123
127
|
},
|
|
124
128
|
"ufr/reasoning-complex": {
|
|
125
|
-
"context":
|
|
129
|
+
"context": 1048576,
|
|
126
130
|
"price": {
|
|
127
131
|
"input": 0.1,
|
|
128
132
|
"output": 0.1
|
|
@@ -135,7 +139,12 @@
|
|
|
135
139
|
"context": 256000
|
|
136
140
|
},
|
|
137
141
|
"ufr/vision-complex": {
|
|
138
|
-
"context":
|
|
142
|
+
"context": 1048576,
|
|
143
|
+
"vision": true,
|
|
144
|
+
"price": {
|
|
145
|
+
"input": 0.1,
|
|
146
|
+
"output": 0.1
|
|
147
|
+
}
|
|
139
148
|
},
|
|
140
149
|
"ufr/vision-fast": {
|
|
141
150
|
"context": 131044
|
|
@@ -201,6 +210,15 @@
|
|
|
201
210
|
"input": 0.3,
|
|
202
211
|
"output": 0.9
|
|
203
212
|
}
|
|
213
|
+
},
|
|
214
|
+
"deepseek-v4.1-flash-llmlb": {
|
|
215
|
+
"context": 1048576,
|
|
216
|
+
"vision": true,
|
|
217
|
+
"price": {
|
|
218
|
+
"input": 0.1,
|
|
219
|
+
"output": 0.1
|
|
220
|
+
},
|
|
221
|
+
"note": "context 1048576 probed live 2026-10-02 (vLLM ContextWindowExceededError named the limit); prices from the UFR portal (Lokale Modelle)"
|
|
204
222
|
}
|
|
205
223
|
}
|
|
206
224
|
}
|
package/package.json
CHANGED
package/src/cli/catalog.ts
CHANGED
|
@@ -1,5 +1,5 @@
|
|
|
1
1
|
import type { UfrModel } from "../daemon/catalog"
|
|
2
|
-
import { BUNDLED_MODELS,
|
|
2
|
+
import { BUNDLED_MODELS, loadBundledModels, loadUfrModels } from "../daemon/catalog-source"
|
|
3
3
|
import { loadConfig } from "../shared/config"
|
|
4
4
|
import type { ModelsFile } from "../shared/models-file"
|
|
5
5
|
import type { CliDeps } from "./index"
|
|
@@ -27,8 +27,7 @@ export async function cmdCatalogDiff(d: CliDeps): Promise<number> {
|
|
|
27
27
|
return 1
|
|
28
28
|
}
|
|
29
29
|
const log = (m: string) => d.io.err(`${m}\n`)
|
|
30
|
-
const mf = await
|
|
31
|
-
bundledPath: BUNDLED_MODELS, fetch: d.fetch, log })
|
|
30
|
+
const mf = await loadBundledModels({ bundledPath: BUNDLED_MODELS })
|
|
32
31
|
const ufr = await loadUfrModels({ baseUrl: cfg.upstream.baseUrl, key, cachePath: d.paths.ufrModelsCache, fetch: d.fetch, log })
|
|
33
32
|
if (ufr.source !== "remote") {
|
|
34
33
|
d.io.err(`cannot read UFR's live model list: ${ufr.error}\n`)
|
|
@@ -1,48 +1,22 @@
|
|
|
1
1
|
import { fileURLToPath } from "node:url"
|
|
2
|
-
import { readJson,
|
|
2
|
+
import { readJson, writeFileAtomic } from "../shared/fs"
|
|
3
3
|
import { type ModelsFile, validateModelsFile } from "../shared/models-file"
|
|
4
4
|
import { VPN_MESSAGE, isVpnPage } from "../shared/vpn"
|
|
5
5
|
import { type UfrModel, parseUfrModels } from "./catalog"
|
|
6
6
|
|
|
7
7
|
export type FetchLike = (url: string, init?: RequestInit) => Promise<Response>
|
|
8
|
-
export type ModelsSource = "
|
|
8
|
+
export type ModelsSource = "bundled"
|
|
9
9
|
|
|
10
10
|
export const BUNDLED_MODELS = fileURLToPath(new URL("../../models.json", import.meta.url))
|
|
11
11
|
|
|
12
|
-
/**
|
|
13
|
-
|
|
14
|
-
|
|
15
|
-
|
|
16
|
-
|
|
12
|
+
/**
|
|
13
|
+
* The fixes file ships with the package (models.json in the release) — no
|
|
14
|
+
* remote pull. Model data updates arrive with plugin updates; UFR's live
|
|
15
|
+
* model list supplies everything else.
|
|
16
|
+
*/
|
|
17
|
+
export async function loadBundledModels(o: {
|
|
17
18
|
bundledPath?: string
|
|
18
|
-
fetch: FetchLike
|
|
19
|
-
log: (m: string) => void
|
|
20
19
|
}): Promise<{ file: ModelsFile; source: ModelsSource }> {
|
|
21
|
-
const cached = await readJson(o.cachePath)
|
|
22
|
-
const etag = cached ? (await readText(o.etagPath))?.trim() : undefined
|
|
23
|
-
try {
|
|
24
|
-
const res = await o.fetch(o.url, {
|
|
25
|
-
headers: etag ? { "If-None-Match": etag } : {},
|
|
26
|
-
signal: AbortSignal.timeout(15_000),
|
|
27
|
-
})
|
|
28
|
-
if (res.status === 304 && cached) return { file: validateModelsFile(cached), source: "remote" }
|
|
29
|
-
if (!res.ok) throw new Error(`HTTP ${res.status}`)
|
|
30
|
-
const text = await res.text()
|
|
31
|
-
const file = validateModelsFile(JSON.parse(text)) // validate before it can replace a good cache
|
|
32
|
-
await writeFileAtomic(o.cachePath, text)
|
|
33
|
-
const tag = res.headers.get("etag")
|
|
34
|
-
if (tag) await writeFileAtomic(o.etagPath, tag)
|
|
35
|
-
return { file, source: "remote" }
|
|
36
|
-
} catch (e) {
|
|
37
|
-
o.log(`models.json: remote copy unavailable or invalid (${(e as Error).message}) — using the ${cached ? "cached" : "bundled"} copy`)
|
|
38
|
-
}
|
|
39
|
-
if (cached) {
|
|
40
|
-
try {
|
|
41
|
-
return { file: validateModelsFile(cached), source: "cache" }
|
|
42
|
-
} catch (e) {
|
|
43
|
-
o.log(`models.json: cached copy invalid (${(e as Error).message})`)
|
|
44
|
-
}
|
|
45
|
-
}
|
|
46
20
|
const bundled: unknown = JSON.parse(await Bun.file(o.bundledPath ?? BUNDLED_MODELS).text())
|
|
47
21
|
return { file: validateModelsFile(bundled), source: "bundled" }
|
|
48
22
|
}
|
package/src/daemon/catalog.ts
CHANGED
|
@@ -46,19 +46,23 @@ function toPrice(p: Price | undefined): ModelPrice | null {
|
|
|
46
46
|
return { input: p.input, output: p.output, cacheRead: p.cache_read ?? p.input, cacheWrite: p.cache_write ?? p.input }
|
|
47
47
|
}
|
|
48
48
|
|
|
49
|
-
export function buildCatalog(ufr: UfrModel[], file: ModelsFile, opts: { allowPaid: boolean }): Catalog {
|
|
49
|
+
export function buildCatalog(ufr: UfrModel[], file: ModelsFile, opts: { allowPaid: boolean; probes?: Record<string, { context: number; at: number }> }): Catalog {
|
|
50
50
|
const warnings: string[] = []
|
|
51
51
|
const exclude = new Set(file.exclude)
|
|
52
|
+
const probes = opts.probes ?? {}
|
|
52
53
|
const models = new Map<string, Model>()
|
|
53
54
|
for (const u of ufr) {
|
|
54
55
|
const e = file.models[u.id]
|
|
56
|
+
// precedence: models.json entry (measured, ships with releases) > auto-probe > defaults
|
|
57
|
+
const probed = probes[u.id]?.context
|
|
58
|
+
const context = e?.context ?? probed ?? file.defaults.context
|
|
55
59
|
models.set(u.id, {
|
|
56
60
|
id: u.id,
|
|
57
61
|
name: u.name,
|
|
58
62
|
tier: u.tier,
|
|
59
63
|
vision: e?.vision ?? u.vision,
|
|
60
64
|
tools: e?.tools ?? true,
|
|
61
|
-
context
|
|
65
|
+
context,
|
|
62
66
|
maxOutput: e?.max_output ?? file.defaults.max_output,
|
|
63
67
|
price: toPrice(e?.price),
|
|
64
68
|
hasEntry: e !== undefined,
|
package/src/daemon/daemon.ts
CHANGED
|
@@ -11,8 +11,11 @@ import { startOfLocalDay } from "../shared/time"
|
|
|
11
11
|
import { VERSION } from "../shared/version"
|
|
12
12
|
import { type BreakerState, BreakerRegistry } from "./breaker"
|
|
13
13
|
import { type Catalog, EMPTY_CATALOG, buildCatalog, listModels } from "./catalog"
|
|
14
|
-
import { type FetchLike, type ModelsSource,
|
|
14
|
+
import { type FetchLike, type ModelsSource, loadBundledModels, loadUfrModels } from "./catalog-source"
|
|
15
|
+
import type { ModelsFile } from "../shared/models-file"
|
|
15
16
|
import { type KeyInfo, KeyPool, type KeySnapshot } from "./keypool"
|
|
17
|
+
import { loadProbes, probeContextLimit, saveProbe } from "./probe"
|
|
18
|
+
import { type UfrModel } from "./catalog"
|
|
16
19
|
import { Router } from "./router"
|
|
17
20
|
import { startServer } from "./server"
|
|
18
21
|
import { Stats } from "./stats"
|
|
@@ -58,6 +61,8 @@ export type DaemonOptions = {
|
|
|
58
61
|
idleCheckMs?: number
|
|
59
62
|
/** Backoff after UFR's model list failed to load (VPN down, UFR unreachable); the last delay repeats. */
|
|
60
63
|
catalogRetryMs?: number[]
|
|
64
|
+
/** Auto-probe the context limit of unknown models (default: on). */
|
|
65
|
+
probes?: boolean
|
|
61
66
|
onStopped?: () => void
|
|
62
67
|
}
|
|
63
68
|
|
|
@@ -243,7 +248,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
243
248
|
const timers: ReturnType<typeof setInterval>[] = []
|
|
244
249
|
let stopping = false
|
|
245
250
|
// UFR's model list failed (VPN not up yet, UFR unreachable): retry with backoff
|
|
246
|
-
// (60 s → 120 s → 300 s → 900 s, then every 900 s
|
|
251
|
+
// (60 s → 120 s → 300 s → 900 s, then every 900 s)
|
|
247
252
|
// so connecting the VPN later needs no daemon restart.
|
|
248
253
|
const retryMs = o.catalogRetryMs ?? [60_000, 120_000, 300_000, 900_000]
|
|
249
254
|
let retryTimer: ReturnType<typeof setTimeout> | null = null
|
|
@@ -259,7 +264,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
259
264
|
}
|
|
260
265
|
const scheduleCatalogRetry = () => {
|
|
261
266
|
if (stopping || retryTimer) return
|
|
262
|
-
const delay = Math.min(retryMs[Math.min(retries, retryMs.length - 1)]!,
|
|
267
|
+
const delay = Math.min(retryMs[Math.min(retries, retryMs.length - 1)]!, 900_000)
|
|
263
268
|
retries++
|
|
264
269
|
retryTimer = setTimeout(() => {
|
|
265
270
|
retryTimer = null
|
|
@@ -273,12 +278,50 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
273
278
|
const reach: StatusJson["upstream"] = { ok: null, message: "", at: 0 }
|
|
274
279
|
let catalog: Catalog = EMPTY_CATALOG
|
|
275
280
|
let catalogInfo: Omit<StatusJson["catalog"], "models" | "warnings"> = { source: "none", ufrSource: "none", loadedAt: 0 }
|
|
281
|
+
let probeStore = loadProbes((k) => db.getKv(k))
|
|
282
|
+
let probing = false
|
|
283
|
+
/** Measure models UFR serves but models.json does not describe (one at a time, off the hot path). */
|
|
284
|
+
const scheduleProbes = async (ufrModels: UfrModel[], file: ModelsFile): Promise<void> => {
|
|
285
|
+
if (probing || keyInfos.length === 0 || o.probes === false) return
|
|
286
|
+
const unknown = ufrModels
|
|
287
|
+
.map((u) => u.id)
|
|
288
|
+
.filter((id) => file.models[id] === undefined && probeStore[id] === undefined)
|
|
289
|
+
if (unknown.length === 0) return
|
|
290
|
+
probing = true
|
|
291
|
+
try {
|
|
292
|
+
for (const id of unknown.slice(0, 2)) {
|
|
293
|
+
if (stopping) break
|
|
294
|
+
const result = await probeContextLimit({
|
|
295
|
+
model: id,
|
|
296
|
+
baseUrl: config.upstream.baseUrl,
|
|
297
|
+
key: keyInfos[0]!.secret,
|
|
298
|
+
transport,
|
|
299
|
+
log,
|
|
300
|
+
})
|
|
301
|
+
if (result) {
|
|
302
|
+
if (stopping) break
|
|
303
|
+
saveProbe(probeStore, id, result, (k, v) => {
|
|
304
|
+
if (stopping) return
|
|
305
|
+
try { db.setKv(k, v) } catch { /* db already closed */ }
|
|
306
|
+
})
|
|
307
|
+
log(`context probe ${id}: ${result.context} tokens (${result.how})`)
|
|
308
|
+
}
|
|
309
|
+
}
|
|
310
|
+
if (unknown.length > 0) {
|
|
311
|
+
const mf2 = await loadBundledModels({ bundledPath: o.bundledModelsPath })
|
|
312
|
+
catalog = buildCatalog(ufrModels, mf2.file, { allowPaid: config.allowPaid, probes: probeStore })
|
|
313
|
+
catalogInfo = { ...catalogInfo, loadedAt: now() }
|
|
314
|
+
}
|
|
315
|
+
} finally {
|
|
316
|
+
probing = false
|
|
317
|
+
}
|
|
318
|
+
}
|
|
276
319
|
const refreshCatalog = async () => {
|
|
277
|
-
|
|
278
|
-
|
|
320
|
+
// fixes ship with the package; UFR's live list is fetched through the transport (vpn-aware)
|
|
321
|
+
const mf = await loadBundledModels({ bundledPath: o.bundledModelsPath })
|
|
279
322
|
const ufr = await loadUfrModels({ baseUrl: config.upstream.baseUrl, key: keyInfos[0]?.secret ?? null,
|
|
280
323
|
cachePath: p.ufrModelsCache, fetch: (u, i) => transport.fetch(u, i), log })
|
|
281
|
-
catalog = buildCatalog(ufr.models, mf.file, { allowPaid: config.allowPaid })
|
|
324
|
+
catalog = buildCatalog(ufr.models, mf.file, { allowPaid: config.allowPaid, probes: probeStore })
|
|
282
325
|
catalogInfo = { source: mf.source, ufrSource: ufr.source, loadedAt: now() }
|
|
283
326
|
if (ufr.error) Object.assign(reach, { ok: false, message: ufr.error, at: now() })
|
|
284
327
|
else if (reach.ok !== true) Object.assign(reach, { ok: true, message: "", at: now() })
|
|
@@ -286,6 +329,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
286
329
|
// Keys are read once at start: without one a retry can never succeed.
|
|
287
330
|
if (ufr.error && keyInfos.length > 0) scheduleCatalogRetry()
|
|
288
331
|
else clearCatalogRetry()
|
|
332
|
+
void scheduleProbes(ufr.models, mf.file).catch(() => {}) // must not outlive a shutdown
|
|
289
333
|
}
|
|
290
334
|
await refreshCatalog()
|
|
291
335
|
|
|
@@ -375,8 +419,9 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
375
419
|
await writeFileAtomic(p.daemonFile, JSON.stringify({ port, pid: process.pid, version: VERSION, startedAt }) + "\n", 0o600)
|
|
376
420
|
|
|
377
421
|
timers.push(setInterval(() => {
|
|
422
|
+
// pick up new UFR models periodically (the fixes file only changes with a plugin update)
|
|
378
423
|
refreshCatalog().catch((e) => log(`catalog refresh failed: ${(e as Error).message}`))
|
|
379
|
-
},
|
|
424
|
+
}, 6 * 3_600_000))
|
|
380
425
|
timers.push(setInterval(() => db.setKv("breakers", JSON.stringify(breakers.snapshot())), 30_000))
|
|
381
426
|
timers.push(setInterval(() => db.prune(now() - 90 * 86_400_000), 86_400_000))
|
|
382
427
|
if (o.exitOnIdle !== false) {
|
|
@@ -0,0 +1,113 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Auto-probe for models UFR serves but models.json does not describe yet:
|
|
3
|
+
* one oversized request makes vLLM name its real context limit in the error
|
|
4
|
+
* ("This model's maximum context length is N tokens"). Results are stored in
|
|
5
|
+
* the stats DB and survive restarts — new models measure themselves, no
|
|
6
|
+
* patches needed.
|
|
7
|
+
*/
|
|
8
|
+
|
|
9
|
+
import type { FetchLike } from "./catalog-source"
|
|
10
|
+
|
|
11
|
+
const LIMIT_RES = [
|
|
12
|
+
/context length is (\d+)/i,
|
|
13
|
+
/maximum context length[^\d]{0,40}(\d+)/i,
|
|
14
|
+
/context window[^\d]{0,30}(\d+)/i,
|
|
15
|
+
]
|
|
16
|
+
|
|
17
|
+
export function parseLimitFromBody(body: string): number | null {
|
|
18
|
+
for (const re of LIMIT_RES) {
|
|
19
|
+
const m = re.exec(body)
|
|
20
|
+
if (m) {
|
|
21
|
+
const n = Number(m[1])
|
|
22
|
+
if (Number.isFinite(n) && n >= 1024) return n
|
|
23
|
+
}
|
|
24
|
+
}
|
|
25
|
+
return null
|
|
26
|
+
}
|
|
27
|
+
|
|
28
|
+
/** Prompt filler: ~4.5 chars per token for prose-like text. */
|
|
29
|
+
function fillerForTokens(tokens: number): string {
|
|
30
|
+
return "The quick brown fox jumps over the lazy dog. ".repeat(Math.ceil((tokens * 4.5) / 45))
|
|
31
|
+
}
|
|
32
|
+
|
|
33
|
+
export type ProbeResult = { context: number; how: "error-named" | "accepted-floor" } | null
|
|
34
|
+
|
|
35
|
+
/**
|
|
36
|
+
* Ladder probe: sizes in tokens. The first error either names the limit
|
|
37
|
+
* (done) or marks the ceiling; the last success sets the floor.
|
|
38
|
+
*/
|
|
39
|
+
export async function probeContextLimit(o: {
|
|
40
|
+
model: string
|
|
41
|
+
baseUrl: string
|
|
42
|
+
key: string
|
|
43
|
+
transport: { name: string; fetch(url: string, init?: RequestInit): Promise<Response> }
|
|
44
|
+
sizes?: number[] // token counts to try, in order
|
|
45
|
+
log: (m: string) => void
|
|
46
|
+
}): Promise<ProbeResult> {
|
|
47
|
+
const sizes = o.sizes ?? [300_000, 600_000, 1_048_000, 1_500_000]
|
|
48
|
+
let floor = 0
|
|
49
|
+
for (const tokens of sizes) {
|
|
50
|
+
const body = JSON.stringify({
|
|
51
|
+
model: o.model,
|
|
52
|
+
max_tokens: 1,
|
|
53
|
+
messages: [{ role: "user", content: fillerForTokens(tokens) + "\n\nReply with exactly: OK" }],
|
|
54
|
+
})
|
|
55
|
+
let res: Response
|
|
56
|
+
try {
|
|
57
|
+
res = await o.transport.fetch(`${o.baseUrl}/chat/completions`, {
|
|
58
|
+
method: "POST",
|
|
59
|
+
headers: { Authorization: `Bearer ${o.key}`, "Content-Type": "application/json" },
|
|
60
|
+
body,
|
|
61
|
+
signal: AbortSignal.timeout(180_000),
|
|
62
|
+
redirect: "manual",
|
|
63
|
+
})
|
|
64
|
+
} catch (e) {
|
|
65
|
+
o.log(`context probe ${o.model}: transport error at ${tokens} tokens (${(e as Error).message})`)
|
|
66
|
+
return floor > 0 ? { context: floor, how: "accepted-floor" } : null
|
|
67
|
+
}
|
|
68
|
+
if (res.ok) {
|
|
69
|
+
const j = (await res.json().catch(() => null)) as { usage?: { prompt_tokens?: number } } | null
|
|
70
|
+
floor = j?.usage?.prompt_tokens ?? tokens
|
|
71
|
+
continue // accepted: try the next rung
|
|
72
|
+
}
|
|
73
|
+
const text = await res.text().catch(() => "")
|
|
74
|
+
const named = parseLimitFromBody(text)
|
|
75
|
+
if (named) {
|
|
76
|
+
o.log(`context probe ${o.model}: the server names the limit: ${named} tokens`)
|
|
77
|
+
return { context: named, how: "error-named" }
|
|
78
|
+
}
|
|
79
|
+
if (res.status === 400 || res.status === 413) {
|
|
80
|
+
o.log(`context probe ${o.model}: rejected at ${tokens} tokens without naming a limit`)
|
|
81
|
+
return floor > 0 ? { context: floor, how: "accepted-floor" } : null
|
|
82
|
+
}
|
|
83
|
+
// rate limit / auth / anything else: not a context answer, retry later
|
|
84
|
+
o.log(`context probe ${o.model}: HTTP ${res.status} — will retry later`)
|
|
85
|
+
return null
|
|
86
|
+
}
|
|
87
|
+
return floor > 0 ? { context: floor, how: "accepted-floor" } : null
|
|
88
|
+
}
|
|
89
|
+
|
|
90
|
+
export const PROBE_KV_KEY = "ctxprobes"
|
|
91
|
+
|
|
92
|
+
type ProbeStore = Record<string, { context: number; at: number }>
|
|
93
|
+
|
|
94
|
+
export function loadProbes(getKv: (k: string) => string | null): ProbeStore {
|
|
95
|
+
try {
|
|
96
|
+
const raw = getKv(PROBE_KV_KEY)
|
|
97
|
+
const parsed = raw ? JSON.parse(raw) : null
|
|
98
|
+
return parsed && typeof parsed === "object" ? (parsed as ProbeStore) : {}
|
|
99
|
+
} catch {
|
|
100
|
+
return {}
|
|
101
|
+
}
|
|
102
|
+
}
|
|
103
|
+
|
|
104
|
+
export function saveProbe(
|
|
105
|
+
store: ProbeStore,
|
|
106
|
+
model: string,
|
|
107
|
+
result: ProbeResult,
|
|
108
|
+
setKv: (k: string, v: string) => void,
|
|
109
|
+
): void {
|
|
110
|
+
if (!result) return
|
|
111
|
+
store[model] = { context: result.context, at: Date.now() }
|
|
112
|
+
setKv(PROBE_KV_KEY, JSON.stringify(store))
|
|
113
|
+
}
|
package/src/daemon/vpn/tcp.ts
CHANGED
|
@@ -49,6 +49,8 @@ const DEFAULT_MSS = 1360 // fits a 1400-byte tunnel MTU with IP+TCP headers
|
|
|
49
49
|
const DEFAULT_RTO_MS = 500
|
|
50
50
|
const DEFAULT_MAX_RETRANSMITS = 6
|
|
51
51
|
const RECV_WINDOW = 65535
|
|
52
|
+
const CWND_START = 4 // segments in flight before the first ACK (slow start)
|
|
53
|
+
const CWND_MAX = 64
|
|
52
54
|
|
|
53
55
|
export function freshLocalPort(): number {
|
|
54
56
|
return 32768 + Math.floor(Math.random() * 20000)
|
|
@@ -65,8 +67,11 @@ export class TcpConn {
|
|
|
65
67
|
private rcvNext = 0 // next byte we expect from the peer (RCV.NXT)
|
|
66
68
|
private peerWindow = 0
|
|
67
69
|
private peerMss = 536
|
|
68
|
-
private outQueue: QueueEntry[] = []
|
|
70
|
+
private outQueue: QueueEntry[] = [] // not yet sent
|
|
71
|
+
private flightQueue: QueueEntry[] = [] // sent, not yet acked (retransmit source)
|
|
69
72
|
private earlyQueue: Uint8Array[] = [] // application data written while syn-sent
|
|
73
|
+
private cwnd = CWND_START // congestion window in segments (slow start)
|
|
74
|
+
private dupAcks = 0
|
|
70
75
|
private finQueued = false
|
|
71
76
|
private finSent = false
|
|
72
77
|
private closeFired = false
|
|
@@ -145,7 +150,7 @@ export class TcpConn {
|
|
|
145
150
|
}
|
|
146
151
|
|
|
147
152
|
get buffered(): number {
|
|
148
|
-
return this.outQueue.reduce((n, e) => n + e.bytes.length, 0)
|
|
153
|
+
return this.outQueue.reduce((n, e) => n + e.bytes.length, 0) + this.flightBytes()
|
|
149
154
|
}
|
|
150
155
|
|
|
151
156
|
// -- inbound -------------------------------------------------------------
|
|
@@ -219,7 +224,6 @@ export class TcpConn {
|
|
|
219
224
|
return
|
|
220
225
|
}
|
|
221
226
|
|
|
222
|
-
if (this.finQueued && !this.outQueue.some((e) => e.len > 0) && this.unacked === 0) this.maybeSendFin()
|
|
223
227
|
this.flush()
|
|
224
228
|
}
|
|
225
229
|
|
|
@@ -239,17 +243,16 @@ export class TcpConn {
|
|
|
239
243
|
const seg: TcpSegment = { header, payload, options }
|
|
240
244
|
const bytes = buildDatagram({ src: this.info.local, dst: this.info.remote, segment: seg })
|
|
241
245
|
this.sendPkt(bytes)
|
|
242
|
-
if (queueLen > 0) this.
|
|
246
|
+
if (queueLen > 0) this.flightQueue.push({ seq: this.seq, len: queueLen, bytes }) // sent: tracks in flight
|
|
243
247
|
}
|
|
244
248
|
|
|
245
249
|
private flush(): void {
|
|
246
|
-
let inFlight = this.
|
|
250
|
+
let inFlight = this.flightBytes()
|
|
247
251
|
let sent = false
|
|
252
|
+
const limit = Math.min(Math.max(this.peerWindow, 1), this.cwnd * this.effMss())
|
|
248
253
|
while (this.outQueue.length > 0) {
|
|
249
254
|
const e = this.outQueue[0]!
|
|
250
|
-
if (e.
|
|
251
|
-
const window = Math.max(this.peerWindow, 1)
|
|
252
|
-
if (inFlight + e.bytes.length > window) break
|
|
255
|
+
if (inFlight + e.bytes.length > limit) break
|
|
253
256
|
const header = {
|
|
254
257
|
srcPort: this.info.localPort,
|
|
255
258
|
dstPort: this.info.remotePort,
|
|
@@ -262,15 +265,23 @@ export class TcpConn {
|
|
|
262
265
|
this.sendPkt(buildDatagram({ src: this.info.local, dst: this.info.remote, segment: { header, payload: e.bytes } }))
|
|
263
266
|
inFlight += e.bytes.length
|
|
264
267
|
sent = true
|
|
265
|
-
this.outQueue.shift()
|
|
268
|
+
this.flightQueue.push(this.outQueue.shift()!)
|
|
266
269
|
}
|
|
267
270
|
if (sent) this.armRto()
|
|
268
271
|
if (this.finQueued && this.buffered === 0) this.maybeSendFin()
|
|
269
272
|
if (this.buffered === 0) this.events.onDrain?.(this)
|
|
270
273
|
}
|
|
271
274
|
|
|
275
|
+
private effMss(): number {
|
|
276
|
+
return Math.min(this.peerMss, this.opts.mss ?? DEFAULT_MSS)
|
|
277
|
+
}
|
|
278
|
+
|
|
279
|
+
private flightBytes(): number {
|
|
280
|
+
return this.flightQueue.reduce((n, e) => n + e.bytes.length, 0)
|
|
281
|
+
}
|
|
282
|
+
|
|
272
283
|
private maybeSendFin(): void {
|
|
273
|
-
if (this.finSent || this.
|
|
284
|
+
if (this.finSent || this.flightQueue.some((e) => e.len > 0)) return // FIN already in flight
|
|
274
285
|
this.finSent = true
|
|
275
286
|
this.sendSegment({ flags: FIN | ACK }, new Uint8Array(0), 1)
|
|
276
287
|
this.seq = (this.seq + 1) >>> 0
|
|
@@ -297,17 +308,29 @@ export class TcpConn {
|
|
|
297
308
|
if (window > 0) this.peerWindow = window
|
|
298
309
|
const unacked = this.unacked
|
|
299
310
|
const newly = (ack - this.acked) >>> 0
|
|
300
|
-
if (newly === 0 || newly > unacked)
|
|
311
|
+
if (newly === 0 || newly > unacked) {
|
|
312
|
+
// duplicate ACK: the peer is missing something after `acked`
|
|
313
|
+
if (unacked > 0 && ++this.dupAcks >= 3) {
|
|
314
|
+
this.dupAcks = 0
|
|
315
|
+
this.cwnd = Math.max(Math.floor(this.cwnd / 2), CWND_START) // multiplicative decrease
|
|
316
|
+
const first = this.flightQueue[0]!
|
|
317
|
+
if (first) this.sendPkt(first.bytes) // fast retransmit, no RTO wait
|
|
318
|
+
}
|
|
319
|
+
return
|
|
320
|
+
}
|
|
321
|
+
this.dupAcks = 0
|
|
322
|
+
// slow start: one segment of window per acked segment, bounded by the peer window
|
|
323
|
+
this.cwnd = Math.min(Math.ceil(this.cwnd + newly / this.effMss()), CWND_MAX)
|
|
301
324
|
this.acked = ack
|
|
302
325
|
this.retransmits = 0
|
|
303
326
|
this.rto = this.opts.rtoMs ?? DEFAULT_RTO_MS
|
|
304
|
-
// drop fully acknowledged queue
|
|
305
|
-
while (this.
|
|
306
|
-
const e = this.
|
|
327
|
+
// drop fully acknowledged entries from the flight queue
|
|
328
|
+
while (this.flightQueue.length > 0) {
|
|
329
|
+
const e = this.flightQueue[0]!
|
|
307
330
|
if ((e.seq + e.len) >>> 0 > ack || (e.len === 0 && e.seq + e.bytes.length > ack)) break
|
|
308
|
-
this.
|
|
331
|
+
this.flightQueue.shift()
|
|
309
332
|
}
|
|
310
|
-
if (this.
|
|
333
|
+
if (this.flightQueue.length === 0 && this.unacked === 0) {
|
|
311
334
|
if (this.timer) { clearTimeout(this.timer); this.timer = null }
|
|
312
335
|
}
|
|
313
336
|
if (this.state === "fin-wait-1" && this.unacked === 0) this.state = "fin-wait-2"
|
|
@@ -335,8 +358,13 @@ export class TcpConn {
|
|
|
335
358
|
return
|
|
336
359
|
}
|
|
337
360
|
this.rto = Math.min(this.rto * 2, 5_000)
|
|
338
|
-
|
|
339
|
-
this.
|
|
361
|
+
this.cwnd = CWND_START // collapse the window: the burst was too much
|
|
362
|
+
this.dupAcks = 0
|
|
363
|
+
// true go-back-N: resend EVERY unacked segment, not just the first —
|
|
364
|
+
// a burst can lose several at once, one-per-RTO would never recover
|
|
365
|
+
for (const e of this.flightQueue) {
|
|
366
|
+
this.sendPkt(e.bytes)
|
|
367
|
+
}
|
|
340
368
|
this.armRto()
|
|
341
369
|
}
|
|
342
370
|
|
package/src/shared/config.ts
CHANGED
|
@@ -19,7 +19,6 @@ export type Config = {
|
|
|
19
19
|
breaker: { tripThreshold: number; ladderS: number[]; probeTimeoutS: number }
|
|
20
20
|
allowPaid: boolean
|
|
21
21
|
dailyBudgetUsd: number
|
|
22
|
-
catalog: { url: string; refreshHours: number }
|
|
23
22
|
idleShutdownMin: number
|
|
24
23
|
}
|
|
25
24
|
|
|
@@ -44,7 +43,6 @@ export const DEFAULTS: Config = {
|
|
|
44
43
|
breaker: { tripThreshold: 3, ladderS: [30, 120, 300, 900, 1800, 3600], probeTimeoutS: 120 },
|
|
45
44
|
allowPaid: false,
|
|
46
45
|
dailyBudgetUsd: 20,
|
|
47
|
-
catalog: { url: "https://raw.githubusercontent.com/FinleyLaempe/opencode-ufr/main/models.json", refreshHours: 6 },
|
|
48
46
|
idleShutdownMin: 5,
|
|
49
47
|
}
|
|
50
48
|
|
|
@@ -72,8 +70,6 @@ const RULES: Record<string, Rule> = {
|
|
|
72
70
|
"breaker.probeTimeoutS": "int>=1",
|
|
73
71
|
allowPaid: "bool",
|
|
74
72
|
dailyBudgetUsd: "num>=0",
|
|
75
|
-
"catalog.url": "str",
|
|
76
|
-
"catalog.refreshHours": "int>=1",
|
|
77
73
|
idleShutdownMin: "int>=1",
|
|
78
74
|
}
|
|
79
75
|
|
|
@@ -88,7 +84,6 @@ const MAX: Record<string, number> = {
|
|
|
88
84
|
"limits.poolMaxWaitS": 86_400,
|
|
89
85
|
"breaker.ladderS": 86_400,
|
|
90
86
|
"breaker.probeTimeoutS": 86_400,
|
|
91
|
-
"catalog.refreshHours": 168,
|
|
92
87
|
idleShutdownMin: 1_440,
|
|
93
88
|
}
|
|
94
89
|
|
package/src/shared/paths.ts
CHANGED
|
@@ -12,8 +12,6 @@ export type Paths = {
|
|
|
12
12
|
lockFile: string
|
|
13
13
|
logFile: string
|
|
14
14
|
statsDb: string
|
|
15
|
-
modelsCache: string
|
|
16
|
-
modelsEtag: string
|
|
17
15
|
ufrModelsCache: string
|
|
18
16
|
}
|
|
19
17
|
|
|
@@ -55,8 +53,6 @@ export function resolvePaths(
|
|
|
55
53
|
lockFile: join(stateDir, "daemon.lock"),
|
|
56
54
|
logFile: join(stateDir, "daemon.log"),
|
|
57
55
|
statsDb: join(dataDir, "stats.db"),
|
|
58
|
-
modelsCache: join(cacheDir, "models.json"),
|
|
59
|
-
modelsEtag: join(cacheDir, "models.etag"),
|
|
60
56
|
ufrModelsCache: join(cacheDir, "ufr-models.json"),
|
|
61
57
|
}
|
|
62
58
|
}
|