opencode-ufr 0.2.1 → 0.2.2

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/models.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schema": 1,
3
- "updated": "2026-09-28",
3
+ "updated": "2026-10-02",
4
4
  "defaults": {
5
5
  "context": 131072,
6
6
  "max_output": 16384
@@ -60,7 +60,11 @@
60
60
  "context": 256000
61
61
  },
62
62
  "qwen-3.5-397b-llmlb": {
63
- "context": 262144
63
+ "context": 262144,
64
+ "price": {
65
+ "input": 0.1,
66
+ "output": 0.1
67
+ }
64
68
  },
65
69
  "qwen-3.6-27b-llmlb": {
66
70
  "context": 262144
@@ -122,7 +126,7 @@
122
126
  "vision": true
123
127
  },
124
128
  "ufr/reasoning-complex": {
125
- "context": 262144,
129
+ "context": 1048576,
126
130
  "price": {
127
131
  "input": 0.1,
128
132
  "output": 0.1
@@ -135,7 +139,12 @@
135
139
  "context": 256000
136
140
  },
137
141
  "ufr/vision-complex": {
138
- "context": 262144
142
+ "context": 1048576,
143
+ "vision": true,
144
+ "price": {
145
+ "input": 0.1,
146
+ "output": 0.1
147
+ }
139
148
  },
140
149
  "ufr/vision-fast": {
141
150
  "context": 131044
@@ -201,6 +210,15 @@
201
210
  "input": 0.3,
202
211
  "output": 0.9
203
212
  }
213
+ },
214
+ "deepseek-v4.1-flash-llmlb": {
215
+ "context": 1048576,
216
+ "vision": true,
217
+ "price": {
218
+ "input": 0.1,
219
+ "output": 0.1
220
+ },
221
+ "note": "context 1048576 probed live 2026-10-02 (vLLM ContextWindowExceededError named the limit); prices from the UFR portal (Lokale Modelle)"
204
222
  }
205
223
  }
206
224
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "opencode-ufr",
3
- "version": "0.2.1",
3
+ "version": "0.2.2",
4
4
  "description": "Uni Freiburg (UFR) Open WebUI models in opencode — local gateway with key pool, UFR-aware rate limiting, circuit breaker and fallbacks",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -1,5 +1,5 @@
1
1
  import type { UfrModel } from "../daemon/catalog"
2
- import { BUNDLED_MODELS, loadModelsFile, loadUfrModels } from "../daemon/catalog-source"
2
+ import { BUNDLED_MODELS, loadBundledModels, loadUfrModels } from "../daemon/catalog-source"
3
3
  import { loadConfig } from "../shared/config"
4
4
  import type { ModelsFile } from "../shared/models-file"
5
5
  import type { CliDeps } from "./index"
@@ -27,8 +27,7 @@ export async function cmdCatalogDiff(d: CliDeps): Promise<number> {
27
27
  return 1
28
28
  }
29
29
  const log = (m: string) => d.io.err(`${m}\n`)
30
- const mf = await loadModelsFile({ url: cfg.catalog.url, cachePath: d.paths.modelsCache, etagPath: d.paths.modelsEtag,
31
- bundledPath: BUNDLED_MODELS, fetch: d.fetch, log })
30
+ const mf = await loadBundledModels({ bundledPath: BUNDLED_MODELS })
32
31
  const ufr = await loadUfrModels({ baseUrl: cfg.upstream.baseUrl, key, cachePath: d.paths.ufrModelsCache, fetch: d.fetch, log })
33
32
  if (ufr.source !== "remote") {
34
33
  d.io.err(`cannot read UFR's live model list: ${ufr.error}\n`)
@@ -1,48 +1,22 @@
1
1
  import { fileURLToPath } from "node:url"
2
- import { readJson, readText, writeFileAtomic } from "../shared/fs"
2
+ import { readJson, writeFileAtomic } from "../shared/fs"
3
3
  import { type ModelsFile, validateModelsFile } from "../shared/models-file"
4
4
  import { VPN_MESSAGE, isVpnPage } from "../shared/vpn"
5
5
  import { type UfrModel, parseUfrModels } from "./catalog"
6
6
 
7
7
  export type FetchLike = (url: string, init?: RequestInit) => Promise<Response>
8
- export type ModelsSource = "remote" | "cache" | "bundled"
8
+ export type ModelsSource = "bundled"
9
9
 
10
10
  export const BUNDLED_MODELS = fileURLToPath(new URL("../../models.json", import.meta.url))
11
11
 
12
- /** models.json: remote (ETag) → last good cache → copy shipped in the package. */
13
- export async function loadModelsFile(o: {
14
- url: string
15
- cachePath: string
16
- etagPath: string
12
+ /**
13
+ * The fixes file ships with the package (models.json in the release) — no
14
+ * remote pull. Model data updates arrive with plugin updates; UFR's live
15
+ * model list supplies everything else.
16
+ */
17
+ export async function loadBundledModels(o: {
17
18
  bundledPath?: string
18
- fetch: FetchLike
19
- log: (m: string) => void
20
19
  }): Promise<{ file: ModelsFile; source: ModelsSource }> {
21
- const cached = await readJson(o.cachePath)
22
- const etag = cached ? (await readText(o.etagPath))?.trim() : undefined
23
- try {
24
- const res = await o.fetch(o.url, {
25
- headers: etag ? { "If-None-Match": etag } : {},
26
- signal: AbortSignal.timeout(15_000),
27
- })
28
- if (res.status === 304 && cached) return { file: validateModelsFile(cached), source: "remote" }
29
- if (!res.ok) throw new Error(`HTTP ${res.status}`)
30
- const text = await res.text()
31
- const file = validateModelsFile(JSON.parse(text)) // validate before it can replace a good cache
32
- await writeFileAtomic(o.cachePath, text)
33
- const tag = res.headers.get("etag")
34
- if (tag) await writeFileAtomic(o.etagPath, tag)
35
- return { file, source: "remote" }
36
- } catch (e) {
37
- o.log(`models.json: remote copy unavailable or invalid (${(e as Error).message}) — using the ${cached ? "cached" : "bundled"} copy`)
38
- }
39
- if (cached) {
40
- try {
41
- return { file: validateModelsFile(cached), source: "cache" }
42
- } catch (e) {
43
- o.log(`models.json: cached copy invalid (${(e as Error).message})`)
44
- }
45
- }
46
20
  const bundled: unknown = JSON.parse(await Bun.file(o.bundledPath ?? BUNDLED_MODELS).text())
47
21
  return { file: validateModelsFile(bundled), source: "bundled" }
48
22
  }
@@ -46,19 +46,23 @@ function toPrice(p: Price | undefined): ModelPrice | null {
46
46
  return { input: p.input, output: p.output, cacheRead: p.cache_read ?? p.input, cacheWrite: p.cache_write ?? p.input }
47
47
  }
48
48
 
49
- export function buildCatalog(ufr: UfrModel[], file: ModelsFile, opts: { allowPaid: boolean }): Catalog {
49
+ export function buildCatalog(ufr: UfrModel[], file: ModelsFile, opts: { allowPaid: boolean; probes?: Record<string, { context: number; at: number }> }): Catalog {
50
50
  const warnings: string[] = []
51
51
  const exclude = new Set(file.exclude)
52
+ const probes = opts.probes ?? {}
52
53
  const models = new Map<string, Model>()
53
54
  for (const u of ufr) {
54
55
  const e = file.models[u.id]
56
+ // precedence: models.json entry (measured, ships with releases) > auto-probe > defaults
57
+ const probed = probes[u.id]?.context
58
+ const context = e?.context ?? probed ?? file.defaults.context
55
59
  models.set(u.id, {
56
60
  id: u.id,
57
61
  name: u.name,
58
62
  tier: u.tier,
59
63
  vision: e?.vision ?? u.vision,
60
64
  tools: e?.tools ?? true,
61
- context: e?.context ?? file.defaults.context,
65
+ context,
62
66
  maxOutput: e?.max_output ?? file.defaults.max_output,
63
67
  price: toPrice(e?.price),
64
68
  hasEntry: e !== undefined,
@@ -11,8 +11,11 @@ import { startOfLocalDay } from "../shared/time"
11
11
  import { VERSION } from "../shared/version"
12
12
  import { type BreakerState, BreakerRegistry } from "./breaker"
13
13
  import { type Catalog, EMPTY_CATALOG, buildCatalog, listModels } from "./catalog"
14
- import { type FetchLike, type ModelsSource, loadModelsFile, loadUfrModels } from "./catalog-source"
14
+ import { type FetchLike, type ModelsSource, loadBundledModels, loadUfrModels } from "./catalog-source"
15
+ import type { ModelsFile } from "../shared/models-file"
15
16
  import { type KeyInfo, KeyPool, type KeySnapshot } from "./keypool"
17
+ import { loadProbes, probeContextLimit, saveProbe } from "./probe"
18
+ import { type UfrModel } from "./catalog"
16
19
  import { Router } from "./router"
17
20
  import { startServer } from "./server"
18
21
  import { Stats } from "./stats"
@@ -58,6 +61,8 @@ export type DaemonOptions = {
58
61
  idleCheckMs?: number
59
62
  /** Backoff after UFR's model list failed to load (VPN down, UFR unreachable); the last delay repeats. */
60
63
  catalogRetryMs?: number[]
64
+ /** Auto-probe the context limit of unknown models (default: on). */
65
+ probes?: boolean
61
66
  onStopped?: () => void
62
67
  }
63
68
 
@@ -243,7 +248,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
243
248
  const timers: ReturnType<typeof setInterval>[] = []
244
249
  let stopping = false
245
250
  // UFR's model list failed (VPN not up yet, UFR unreachable): retry with backoff
246
- // (60 s → 120 s → 300 s → 900 s, then every 900 s; no step longer than refreshHours)
251
+ // (60 s → 120 s → 300 s → 900 s, then every 900 s)
247
252
  // so connecting the VPN later needs no daemon restart.
248
253
  const retryMs = o.catalogRetryMs ?? [60_000, 120_000, 300_000, 900_000]
249
254
  let retryTimer: ReturnType<typeof setTimeout> | null = null
@@ -259,7 +264,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
259
264
  }
260
265
  const scheduleCatalogRetry = () => {
261
266
  if (stopping || retryTimer) return
262
- const delay = Math.min(retryMs[Math.min(retries, retryMs.length - 1)]!, config.catalog.refreshHours * 3_600_000)
267
+ const delay = Math.min(retryMs[Math.min(retries, retryMs.length - 1)]!, 900_000)
263
268
  retries++
264
269
  retryTimer = setTimeout(() => {
265
270
  retryTimer = null
@@ -273,12 +278,50 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
273
278
  const reach: StatusJson["upstream"] = { ok: null, message: "", at: 0 }
274
279
  let catalog: Catalog = EMPTY_CATALOG
275
280
  let catalogInfo: Omit<StatusJson["catalog"], "models" | "warnings"> = { source: "none", ufrSource: "none", loadedAt: 0 }
281
+ let probeStore = loadProbes((k) => db.getKv(k))
282
+ let probing = false
283
+ /** Measure models UFR serves but models.json does not describe (one at a time, off the hot path). */
284
+ const scheduleProbes = async (ufrModels: UfrModel[], file: ModelsFile): Promise<void> => {
285
+ if (probing || keyInfos.length === 0 || o.probes === false) return
286
+ const unknown = ufrModels
287
+ .map((u) => u.id)
288
+ .filter((id) => file.models[id] === undefined && probeStore[id] === undefined)
289
+ if (unknown.length === 0) return
290
+ probing = true
291
+ try {
292
+ for (const id of unknown.slice(0, 2)) {
293
+ if (stopping) break
294
+ const result = await probeContextLimit({
295
+ model: id,
296
+ baseUrl: config.upstream.baseUrl,
297
+ key: keyInfos[0]!.secret,
298
+ transport,
299
+ log,
300
+ })
301
+ if (result) {
302
+ if (stopping) break
303
+ saveProbe(probeStore, id, result, (k, v) => {
304
+ if (stopping) return
305
+ try { db.setKv(k, v) } catch { /* db already closed */ }
306
+ })
307
+ log(`context probe ${id}: ${result.context} tokens (${result.how})`)
308
+ }
309
+ }
310
+ if (unknown.length > 0) {
311
+ const mf2 = await loadBundledModels({ bundledPath: o.bundledModelsPath })
312
+ catalog = buildCatalog(ufrModels, mf2.file, { allowPaid: config.allowPaid, probes: probeStore })
313
+ catalogInfo = { ...catalogInfo, loadedAt: now() }
314
+ }
315
+ } finally {
316
+ probing = false
317
+ }
318
+ }
276
319
  const refreshCatalog = async () => {
277
- const mf = await loadModelsFile({ url: config.catalog.url, cachePath: p.modelsCache, etagPath: p.modelsEtag,
278
- bundledPath: o.bundledModelsPath, fetch: f, log })
320
+ // fixes ship with the package; UFR's live list is fetched through the transport (vpn-aware)
321
+ const mf = await loadBundledModels({ bundledPath: o.bundledModelsPath })
279
322
  const ufr = await loadUfrModels({ baseUrl: config.upstream.baseUrl, key: keyInfos[0]?.secret ?? null,
280
323
  cachePath: p.ufrModelsCache, fetch: (u, i) => transport.fetch(u, i), log })
281
- catalog = buildCatalog(ufr.models, mf.file, { allowPaid: config.allowPaid })
324
+ catalog = buildCatalog(ufr.models, mf.file, { allowPaid: config.allowPaid, probes: probeStore })
282
325
  catalogInfo = { source: mf.source, ufrSource: ufr.source, loadedAt: now() }
283
326
  if (ufr.error) Object.assign(reach, { ok: false, message: ufr.error, at: now() })
284
327
  else if (reach.ok !== true) Object.assign(reach, { ok: true, message: "", at: now() })
@@ -286,6 +329,7 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
286
329
  // Keys are read once at start: without one a retry can never succeed.
287
330
  if (ufr.error && keyInfos.length > 0) scheduleCatalogRetry()
288
331
  else clearCatalogRetry()
332
+ void scheduleProbes(ufr.models, mf.file).catch(() => {}) // must not outlive a shutdown
289
333
  }
290
334
  await refreshCatalog()
291
335
 
@@ -375,8 +419,9 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
375
419
  await writeFileAtomic(p.daemonFile, JSON.stringify({ port, pid: process.pid, version: VERSION, startedAt }) + "\n", 0o600)
376
420
 
377
421
  timers.push(setInterval(() => {
422
+ // pick up new UFR models periodically (the fixes file only changes with a plugin update)
378
423
  refreshCatalog().catch((e) => log(`catalog refresh failed: ${(e as Error).message}`))
379
- }, config.catalog.refreshHours * 3_600_000))
424
+ }, 6 * 3_600_000))
380
425
  timers.push(setInterval(() => db.setKv("breakers", JSON.stringify(breakers.snapshot())), 30_000))
381
426
  timers.push(setInterval(() => db.prune(now() - 90 * 86_400_000), 86_400_000))
382
427
  if (o.exitOnIdle !== false) {
@@ -0,0 +1,113 @@
1
+ /**
2
+ * Auto-probe for models UFR serves but models.json does not describe yet:
3
+ * one oversized request makes vLLM name its real context limit in the error
4
+ * ("This model's maximum context length is N tokens"). Results are stored in
5
+ * the stats DB and survive restarts — new models measure themselves, no
6
+ * patches needed.
7
+ */
8
+
9
+ import type { FetchLike } from "./catalog-source"
10
+
11
+ const LIMIT_RES = [
12
+ /context length is (\d+)/i,
13
+ /maximum context length[^\d]{0,40}(\d+)/i,
14
+ /context window[^\d]{0,30}(\d+)/i,
15
+ ]
16
+
17
+ export function parseLimitFromBody(body: string): number | null {
18
+ for (const re of LIMIT_RES) {
19
+ const m = re.exec(body)
20
+ if (m) {
21
+ const n = Number(m[1])
22
+ if (Number.isFinite(n) && n >= 1024) return n
23
+ }
24
+ }
25
+ return null
26
+ }
27
+
28
+ /** Prompt filler: ~4.5 chars per token for prose-like text. */
29
+ function fillerForTokens(tokens: number): string {
30
+ return "The quick brown fox jumps over the lazy dog. ".repeat(Math.ceil((tokens * 4.5) / 45))
31
+ }
32
+
33
+ export type ProbeResult = { context: number; how: "error-named" | "accepted-floor" } | null
34
+
35
+ /**
36
+ * Ladder probe: sizes in tokens. The first error either names the limit
37
+ * (done) or marks the ceiling; the last success sets the floor.
38
+ */
39
+ export async function probeContextLimit(o: {
40
+ model: string
41
+ baseUrl: string
42
+ key: string
43
+ transport: { name: string; fetch(url: string, init?: RequestInit): Promise<Response> }
44
+ sizes?: number[] // token counts to try, in order
45
+ log: (m: string) => void
46
+ }): Promise<ProbeResult> {
47
+ const sizes = o.sizes ?? [300_000, 600_000, 1_048_000, 1_500_000]
48
+ let floor = 0
49
+ for (const tokens of sizes) {
50
+ const body = JSON.stringify({
51
+ model: o.model,
52
+ max_tokens: 1,
53
+ messages: [{ role: "user", content: fillerForTokens(tokens) + "\n\nReply with exactly: OK" }],
54
+ })
55
+ let res: Response
56
+ try {
57
+ res = await o.transport.fetch(`${o.baseUrl}/chat/completions`, {
58
+ method: "POST",
59
+ headers: { Authorization: `Bearer ${o.key}`, "Content-Type": "application/json" },
60
+ body,
61
+ signal: AbortSignal.timeout(180_000),
62
+ redirect: "manual",
63
+ })
64
+ } catch (e) {
65
+ o.log(`context probe ${o.model}: transport error at ${tokens} tokens (${(e as Error).message})`)
66
+ return floor > 0 ? { context: floor, how: "accepted-floor" } : null
67
+ }
68
+ if (res.ok) {
69
+ const j = (await res.json().catch(() => null)) as { usage?: { prompt_tokens?: number } } | null
70
+ floor = j?.usage?.prompt_tokens ?? tokens
71
+ continue // accepted: try the next rung
72
+ }
73
+ const text = await res.text().catch(() => "")
74
+ const named = parseLimitFromBody(text)
75
+ if (named) {
76
+ o.log(`context probe ${o.model}: the server names the limit: ${named} tokens`)
77
+ return { context: named, how: "error-named" }
78
+ }
79
+ if (res.status === 400 || res.status === 413) {
80
+ o.log(`context probe ${o.model}: rejected at ${tokens} tokens without naming a limit`)
81
+ return floor > 0 ? { context: floor, how: "accepted-floor" } : null
82
+ }
83
+ // rate limit / auth / anything else: not a context answer, retry later
84
+ o.log(`context probe ${o.model}: HTTP ${res.status} — will retry later`)
85
+ return null
86
+ }
87
+ return floor > 0 ? { context: floor, how: "accepted-floor" } : null
88
+ }
89
+
90
+ export const PROBE_KV_KEY = "ctxprobes"
91
+
92
+ type ProbeStore = Record<string, { context: number; at: number }>
93
+
94
+ export function loadProbes(getKv: (k: string) => string | null): ProbeStore {
95
+ try {
96
+ const raw = getKv(PROBE_KV_KEY)
97
+ const parsed = raw ? JSON.parse(raw) : null
98
+ return parsed && typeof parsed === "object" ? (parsed as ProbeStore) : {}
99
+ } catch {
100
+ return {}
101
+ }
102
+ }
103
+
104
+ export function saveProbe(
105
+ store: ProbeStore,
106
+ model: string,
107
+ result: ProbeResult,
108
+ setKv: (k: string, v: string) => void,
109
+ ): void {
110
+ if (!result) return
111
+ store[model] = { context: result.context, at: Date.now() }
112
+ setKv(PROBE_KV_KEY, JSON.stringify(store))
113
+ }
@@ -49,6 +49,8 @@ const DEFAULT_MSS = 1360 // fits a 1400-byte tunnel MTU with IP+TCP headers
49
49
  const DEFAULT_RTO_MS = 500
50
50
  const DEFAULT_MAX_RETRANSMITS = 6
51
51
  const RECV_WINDOW = 65535
52
+ const CWND_START = 4 // segments in flight before the first ACK (slow start)
53
+ const CWND_MAX = 64
52
54
 
53
55
  export function freshLocalPort(): number {
54
56
  return 32768 + Math.floor(Math.random() * 20000)
@@ -65,8 +67,11 @@ export class TcpConn {
65
67
  private rcvNext = 0 // next byte we expect from the peer (RCV.NXT)
66
68
  private peerWindow = 0
67
69
  private peerMss = 536
68
- private outQueue: QueueEntry[] = []
70
+ private outQueue: QueueEntry[] = [] // not yet sent
71
+ private flightQueue: QueueEntry[] = [] // sent, not yet acked (retransmit source)
69
72
  private earlyQueue: Uint8Array[] = [] // application data written while syn-sent
73
+ private cwnd = CWND_START // congestion window in segments (slow start)
74
+ private dupAcks = 0
70
75
  private finQueued = false
71
76
  private finSent = false
72
77
  private closeFired = false
@@ -145,7 +150,7 @@ export class TcpConn {
145
150
  }
146
151
 
147
152
  get buffered(): number {
148
- return this.outQueue.reduce((n, e) => n + e.bytes.length, 0)
153
+ return this.outQueue.reduce((n, e) => n + e.bytes.length, 0) + this.flightBytes()
149
154
  }
150
155
 
151
156
  // -- inbound -------------------------------------------------------------
@@ -219,7 +224,6 @@ export class TcpConn {
219
224
  return
220
225
  }
221
226
 
222
- if (this.finQueued && !this.outQueue.some((e) => e.len > 0) && this.unacked === 0) this.maybeSendFin()
223
227
  this.flush()
224
228
  }
225
229
 
@@ -239,17 +243,16 @@ export class TcpConn {
239
243
  const seg: TcpSegment = { header, payload, options }
240
244
  const bytes = buildDatagram({ src: this.info.local, dst: this.info.remote, segment: seg })
241
245
  this.sendPkt(bytes)
242
- if (queueLen > 0) this.outQueue.push({ seq: this.seq, len: queueLen, bytes })
246
+ if (queueLen > 0) this.flightQueue.push({ seq: this.seq, len: queueLen, bytes }) // sent: tracks in flight
243
247
  }
244
248
 
245
249
  private flush(): void {
246
- let inFlight = this.unacked
250
+ let inFlight = this.flightBytes()
247
251
  let sent = false
252
+ const limit = Math.min(Math.max(this.peerWindow, 1), this.cwnd * this.effMss())
248
253
  while (this.outQueue.length > 0) {
249
254
  const e = this.outQueue[0]!
250
- if (e.len > 0) break // SYN/FIN marker: only retransmits move it
251
- const window = Math.max(this.peerWindow, 1)
252
- if (inFlight + e.bytes.length > window) break
255
+ if (inFlight + e.bytes.length > limit) break
253
256
  const header = {
254
257
  srcPort: this.info.localPort,
255
258
  dstPort: this.info.remotePort,
@@ -262,15 +265,23 @@ export class TcpConn {
262
265
  this.sendPkt(buildDatagram({ src: this.info.local, dst: this.info.remote, segment: { header, payload: e.bytes } }))
263
266
  inFlight += e.bytes.length
264
267
  sent = true
265
- this.outQueue.shift()
268
+ this.flightQueue.push(this.outQueue.shift()!)
266
269
  }
267
270
  if (sent) this.armRto()
268
271
  if (this.finQueued && this.buffered === 0) this.maybeSendFin()
269
272
  if (this.buffered === 0) this.events.onDrain?.(this)
270
273
  }
271
274
 
275
+ private effMss(): number {
276
+ return Math.min(this.peerMss, this.opts.mss ?? DEFAULT_MSS)
277
+ }
278
+
279
+ private flightBytes(): number {
280
+ return this.flightQueue.reduce((n, e) => n + e.bytes.length, 0)
281
+ }
282
+
272
283
  private maybeSendFin(): void {
273
- if (this.finSent || this.outQueue.some((e) => e.len > 0)) return // FIN already in flight
284
+ if (this.finSent || this.flightQueue.some((e) => e.len > 0)) return // FIN already in flight
274
285
  this.finSent = true
275
286
  this.sendSegment({ flags: FIN | ACK }, new Uint8Array(0), 1)
276
287
  this.seq = (this.seq + 1) >>> 0
@@ -297,17 +308,29 @@ export class TcpConn {
297
308
  if (window > 0) this.peerWindow = window
298
309
  const unacked = this.unacked
299
310
  const newly = (ack - this.acked) >>> 0
300
- if (newly === 0 || newly > unacked) return // stale or covering more than we sent
311
+ if (newly === 0 || newly > unacked) {
312
+ // duplicate ACK: the peer is missing something after `acked`
313
+ if (unacked > 0 && ++this.dupAcks >= 3) {
314
+ this.dupAcks = 0
315
+ this.cwnd = Math.max(Math.floor(this.cwnd / 2), CWND_START) // multiplicative decrease
316
+ const first = this.flightQueue[0]!
317
+ if (first) this.sendPkt(first.bytes) // fast retransmit, no RTO wait
318
+ }
319
+ return
320
+ }
321
+ this.dupAcks = 0
322
+ // slow start: one segment of window per acked segment, bounded by the peer window
323
+ this.cwnd = Math.min(Math.ceil(this.cwnd + newly / this.effMss()), CWND_MAX)
301
324
  this.acked = ack
302
325
  this.retransmits = 0
303
326
  this.rto = this.opts.rtoMs ?? DEFAULT_RTO_MS
304
- // drop fully acknowledged queue entries
305
- while (this.outQueue.length > 0) {
306
- const e = this.outQueue[0]!
327
+ // drop fully acknowledged entries from the flight queue
328
+ while (this.flightQueue.length > 0) {
329
+ const e = this.flightQueue[0]!
307
330
  if ((e.seq + e.len) >>> 0 > ack || (e.len === 0 && e.seq + e.bytes.length > ack)) break
308
- this.outQueue.shift()
331
+ this.flightQueue.shift()
309
332
  }
310
- if (this.outQueue.length === 0 && this.unacked === 0) {
333
+ if (this.flightQueue.length === 0 && this.unacked === 0) {
311
334
  if (this.timer) { clearTimeout(this.timer); this.timer = null }
312
335
  }
313
336
  if (this.state === "fin-wait-1" && this.unacked === 0) this.state = "fin-wait-2"
@@ -335,8 +358,13 @@ export class TcpConn {
335
358
  return
336
359
  }
337
360
  this.rto = Math.min(this.rto * 2, 5_000)
338
- const first = this.outQueue[0]!
339
- this.sendPkt(first.bytes) // go-back-N: resend oldest unacked segment
361
+ this.cwnd = CWND_START // collapse the window: the burst was too much
362
+ this.dupAcks = 0
363
+ // true go-back-N: resend EVERY unacked segment, not just the first —
364
+ // a burst can lose several at once, one-per-RTO would never recover
365
+ for (const e of this.flightQueue) {
366
+ this.sendPkt(e.bytes)
367
+ }
340
368
  this.armRto()
341
369
  }
342
370
 
@@ -19,7 +19,6 @@ export type Config = {
19
19
  breaker: { tripThreshold: number; ladderS: number[]; probeTimeoutS: number }
20
20
  allowPaid: boolean
21
21
  dailyBudgetUsd: number
22
- catalog: { url: string; refreshHours: number }
23
22
  idleShutdownMin: number
24
23
  }
25
24
 
@@ -44,7 +43,6 @@ export const DEFAULTS: Config = {
44
43
  breaker: { tripThreshold: 3, ladderS: [30, 120, 300, 900, 1800, 3600], probeTimeoutS: 120 },
45
44
  allowPaid: false,
46
45
  dailyBudgetUsd: 20,
47
- catalog: { url: "https://raw.githubusercontent.com/FinleyLaempe/opencode-ufr/main/models.json", refreshHours: 6 },
48
46
  idleShutdownMin: 5,
49
47
  }
50
48
 
@@ -72,8 +70,6 @@ const RULES: Record<string, Rule> = {
72
70
  "breaker.probeTimeoutS": "int>=1",
73
71
  allowPaid: "bool",
74
72
  dailyBudgetUsd: "num>=0",
75
- "catalog.url": "str",
76
- "catalog.refreshHours": "int>=1",
77
73
  idleShutdownMin: "int>=1",
78
74
  }
79
75
 
@@ -88,7 +84,6 @@ const MAX: Record<string, number> = {
88
84
  "limits.poolMaxWaitS": 86_400,
89
85
  "breaker.ladderS": 86_400,
90
86
  "breaker.probeTimeoutS": 86_400,
91
- "catalog.refreshHours": 168,
92
87
  idleShutdownMin: 1_440,
93
88
  }
94
89
 
@@ -12,8 +12,6 @@ export type Paths = {
12
12
  lockFile: string
13
13
  logFile: string
14
14
  statsDb: string
15
- modelsCache: string
16
- modelsEtag: string
17
15
  ufrModelsCache: string
18
16
  }
19
17
 
@@ -55,8 +53,6 @@ export function resolvePaths(
55
53
  lockFile: join(stateDir, "daemon.lock"),
56
54
  logFile: join(stateDir, "daemon.log"),
57
55
  statsDb: join(dataDir, "stats.db"),
58
- modelsCache: join(cacheDir, "models.json"),
59
- modelsEtag: join(cacheDir, "models.etag"),
60
56
  ufrModelsCache: join(cacheDir, "ufr-models.json"),
61
57
  }
62
58
  }