opencode-ufr 0.2.8 → 0.2.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/README.md CHANGED
@@ -23,7 +23,7 @@ opencode plugin add opencode-ufr # registers the plugin
23
23
 
24
24
  (or add `"plugins": ["opencode-ufr"]` to your `opencode.json` manually.)
25
25
 
26
- ## Updating
26
+ ## Install
27
27
 
28
28
  opencode installs the plugin once into its cache and does not look for a newer
29
29
  version while that copy exists — restarting opencode alone keeps the old one.
@@ -34,7 +34,7 @@ and stop that server:
34
34
  ```bash
35
35
  bun add -g opencode-ufr@latest # update the `ufr` CLI
36
36
  rm -rf ~/.cache/opencode/npm/opencode-ufr@latest # opencode's cached plugin copy
37
- pkill -f "opencode serve --service" # opencode's background server
37
+ /restart in opencode eintippen # opencode's background server
38
38
  ```
39
39
 
40
40
  The next opencode start installs the latest version and replaces the old
@@ -207,10 +207,20 @@ the GitHub release after a maintainer approves the staged version with 2FA.
207
207
 
208
208
  ## Model data
209
209
 
210
- The gateway fetches `models.json` from this repository every 6 hours —
211
- context windows, prices, vision/tool flags, alias spellings and fallback
212
- order that UFR's own API doesn't expose or gets wrong. Fixes reach every
213
- user without a release.
210
+ `models.json` ships with every release — context windows, prices,
211
+ vision/tool flags, alias spellings and fallback order that UFR's own API
212
+ doesn't expose or gets wrong. Fixes reach users with the next plugin update.
213
+
214
+ Context windows are measured, not guessed: the `probe contexts` workflow
215
+ (weekly, on demand, and after every release tag) sends one oversized — and
216
+ therefore free, since UFR rejects it before pricing — request per model and
217
+ commits the exact limit the server names back to `models.json` on main. Run
218
+ it yourself with a stored key:
219
+
220
+ ```bash
221
+ ufr context-probe # report: measured context vs. models.json
222
+ ufr context-probe --write # also patch your local models.json
223
+ ```
214
224
 
215
225
  To contribute a measurement (a price, a context window, a missing alias),
216
226
  open a PR against `models.json` with the source of the measurement in a
package/models.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "schema": 1,
3
- "updated": "2026-10-02",
3
+ "updated": "2026-10-05",
4
4
  "defaults": {
5
5
  "context": 131072,
6
6
  "max_output": 16384
@@ -60,11 +60,12 @@
60
60
  "context": 256000
61
61
  },
62
62
  "qwen-3.5-397b-llmlb": {
63
- "context": 262144,
63
+ "context": 1048576,
64
64
  "price": {
65
65
  "input": 0.1,
66
66
  "output": 0.1
67
- }
67
+ },
68
+ "note": "context 1048576 probed live 2026-10-05 (probe-all-contexts)"
68
69
  },
69
70
  "qwen-3.6-27b-llmlb": {
70
71
  "context": 262144
@@ -139,12 +140,13 @@
139
140
  "context": 256000
140
141
  },
141
142
  "ufr/vision-complex": {
142
- "context": 1048576,
143
+ "context": 262144,
143
144
  "vision": true,
144
145
  "price": {
145
146
  "input": 0.1,
146
147
  "output": 0.1
147
- }
148
+ },
149
+ "note": "context 262144 probed live 2026-10-05 (probe-all-contexts)"
148
150
  },
149
151
  "ufr/vision-fast": {
150
152
  "context": 131044
@@ -219,6 +221,10 @@
219
221
  "output": 0.1
220
222
  },
221
223
  "note": "context 1048576 probed live 2026-10-02 (vLLM ContextWindowExceededError named the limit); prices from the UFR portal (Lokale Modelle)"
224
+ },
225
+ "kolibri-1-llmlb": {
226
+ "context": 262144,
227
+ "note": "context 262144 probed live 2026-10-05 (probe-all-contexts)"
222
228
  }
223
229
  }
224
230
  }
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "opencode-ufr",
3
- "version": "0.2.8",
3
+ "version": "0.2.9",
4
4
  "description": "Uni Freiburg (UFR) Open WebUI models in opencode — local gateway with key pool, UFR-aware rate limiting, circuit breaker and fallbacks",
5
5
  "license": "MIT",
6
6
  "type": "module",
@@ -0,0 +1,96 @@
1
+ import { BUNDLED_MODELS, loadBundledModels } from "../daemon/catalog-source"
2
+ import { applyProbeResults, formatProbeReport, probeAllContexts } from "../daemon/probe-all"
3
+ import { VpnManager } from "../daemon/vpn/manager"
4
+ import { loadConfig } from "../shared/config"
5
+ import { VPN_PASS, VPN_USER } from "../shared/secrets"
6
+ import type { CliDeps } from "./index"
7
+
8
+ /**
9
+ * ufr context-probe — measure every UFR model's context limit with one of the
10
+ * stored keys. Rejected probes are free and don't consume the key's bucket;
11
+ * the report flags values that drifted from models.json. `--write` patches
12
+ * the local models.json copy (the repo copy when run from a checkout).
13
+ *
14
+ * `--vpn` routes the probes through the built-in Fortinet tunnel with the
15
+ * stored uni login (vpn-login/vpn-pass) — for off campus machines without
16
+ * their own VPN connection.
17
+ */
18
+ export async function cmdContextProbe(d: CliDeps, args: string[]): Promise<number> {
19
+ const write = args.includes("--write")
20
+ const viaVpn = args.includes("--vpn")
21
+ const cfg = await loadConfig(d.paths.configFile)
22
+ const alias = cfg.keys[0]
23
+ if (!alias) {
24
+ d.io.err("no stored key — run `ufr connect` first\n")
25
+ return 2
26
+ }
27
+ const key = await d.secrets.get(alias)
28
+ if (!key) {
29
+ d.io.err(`key "${alias}" is configured but missing from the keyring\n`)
30
+ return 2
31
+ }
32
+ let fetch = d.fetch
33
+ let vpn: { stop(): Promise<void> } | null = null
34
+ if (viaVpn) {
35
+ const [user, pass] = await Promise.all([d.secrets.get(VPN_USER), d.secrets.get(VPN_PASS)])
36
+ if (!user || !pass) {
37
+ d.io.err("no stored uni login — run `ufr login` or `ufr connect --login …` first\n")
38
+ return 2
39
+ }
40
+ const manager = new VpnManager({
41
+ gateway: cfg.vpn.gateway,
42
+ upstreamHost: new URL(cfg.upstream.baseUrl).hostname,
43
+ baseUrl: cfg.upstream.baseUrl,
44
+ credentials: { user, pass },
45
+ mode: "always",
46
+ log: (m) => d.io.out(`${m}\n`),
47
+ fetch: d.fetch,
48
+ })
49
+ vpn = manager
50
+ fetch = manager.transport().fetch
51
+ }
52
+ const { file } = await loadBundledModels({})
53
+ d.io.out(`probing ${cfg.keys.length} key(s) via "${alias}"${viaVpn ? " through the built-in VPN" : ""} — rejected probes are free, this takes a few minutes\n`)
54
+ try {
55
+ const rows = await probeAllContexts({
56
+ baseUrl: cfg.upstream.baseUrl,
57
+ key,
58
+ fetch,
59
+ file,
60
+ log: (m) => d.io.out(`${m}\n`),
61
+ })
62
+ d.io.out("\n" + formatProbeReport(rows) + "\n")
63
+ return await report(d, file, rows, write)
64
+ } finally {
65
+ if (vpn) await vpn.stop()
66
+ }
67
+ }
68
+
69
+ async function report(
70
+ d: CliDeps,
71
+ file: Awaited<ReturnType<typeof loadBundledModels>>["file"],
72
+ rows: Awaited<ReturnType<typeof probeAllContexts>>,
73
+ write: boolean,
74
+ ): Promise<number> {
75
+ if (write) {
76
+ const date = new Date().toISOString().slice(0, 10)
77
+ const { file: updated, changed } = applyProbeResults(file, rows, date)
78
+ if (changed.length === 0) {
79
+ d.io.out("\nmodels.json is already current — nothing written\n")
80
+ } else {
81
+ await Bun.write(BUNDLED_MODELS, JSON.stringify(updated, null, 2) + "\n")
82
+ d.io.out(`\n${BUNDLED_MODELS} updated (${changed.length}):\n`)
83
+ for (const c of changed) d.io.out(` ${c.id}: ${c.from ?? "—"} -> ${c.to}\n`)
84
+ d.io.out("\nIf this is a repo checkout: commit models.json (PR welcome per README).\n")
85
+ d.io.out("On an installed copy this patches your local file until the next plugin/CLI update;\n")
86
+ d.io.out("upstream fixes reach everyone through models.json PRs and releases.\n")
87
+ }
88
+ } else {
89
+ const needsUpdate = rows.filter((r) => r.how === "error-named" && r.probed !== null && r.probed !== r.known)
90
+ if (needsUpdate.length > 0) {
91
+ d.io.out("\nrun `ufr context-probe --write` to patch your local models.json, or PR the values upstream\n")
92
+ return 1
93
+ }
94
+ }
95
+ return 0
96
+ }
package/src/cli/index.ts CHANGED
@@ -4,6 +4,7 @@ import { type Paths, resolvePaths } from "../shared/paths"
4
4
  import { KeyringStore, type SecretStore } from "../shared/secrets"
5
5
  import { cmdCatalogDiff } from "./catalog"
6
6
  import { cmdConnect } from "./connect"
7
+ import { cmdContextProbe } from "./context-probe"
7
8
  import { type Io, terminalIo } from "./io"
8
9
  import { cmdDisconnect, cmdKeys, cmdLogin } from "./keys"
9
10
  import { cmdStats } from "./stats"
@@ -32,6 +33,11 @@ const HELP = `ufr — opencode-ufr gateway
32
33
  ufr status gateway, keys, limits, breakers, spend today
33
34
  ufr stats [--days N] requests, tokens and cost (default: today)
34
35
  ufr catalog diff UFR's model list vs. models.json
36
+ ufr context-probe [--write] [--vpn]
37
+ measure every model's context limit live
38
+ (rejected probes are free; --write patches
39
+ the local models.json; --vpn routes the
40
+ probes through the built-in tunnel)
35
41
  ufr stop stop the gateway (opencode starts it again)
36
42
 
37
43
  UFR needs the uni VPN off campus: https://wiki.uni-freiburg.de/rz/doku.php?id=vpn
@@ -78,6 +84,8 @@ export async function main(argv: string[], partial: Partial<CliDeps> = {}): Prom
78
84
  if (rest[0] === "diff") return await cmdCatalogDiff(d)
79
85
  d.io.err("usage: ufr catalog diff\n")
80
86
  return 2
87
+ case "context-probe":
88
+ return await cmdContextProbe(d, rest)
81
89
  case "stop":
82
90
  return await cmdStop(d)
83
91
  case undefined:
@@ -0,0 +1,190 @@
1
+ /**
2
+ * Probe the context limit of every model UFR serves, reusing probe.ts's
3
+ * oversized-request trick: one request above a model's known context makes
4
+ * the server name the exact limit ("maximum context length is N tokens").
5
+ *
6
+ * Cost profile matters here: a REJECTED request is refused before any tokens
7
+ * are priced (measured 2026-10-05: 740 ms, x-process-time 0), so probing a
8
+ * model whose recorded context is still right costs nothing and does not
9
+ * count against the 20/min key bucket. Money is only spent when a probe rung
10
+ * is ACCEPTED (context grew since the last measurement) or a paid model is
11
+ * probed on an accepted rung — which is why paid models are skipped unless
12
+ * asked for.
13
+ */
14
+
15
+ import type { ModelsFile } from "../shared/models-file"
16
+ import type { FetchLike } from "./catalog-source"
17
+ import { parseUfrModels } from "./catalog"
18
+ import { fillerForTokens, parseLimitFromBody } from "./probe"
19
+
20
+ /** No UFR model is known to exceed this; probing higher only risks an accepted (paid) rung. */
21
+ export const PROBE_CEILING = 1_600_000
22
+
23
+ export type ProbeHow =
24
+ | "error-named" // the server named the exact limit — the trustworthy result
25
+ | "accepted-floor" // every rung up to `probed` was accepted — `probed` is a lower bound only
26
+ | "rejected-unnamed" // rejected, but the body named no limit
27
+ | "http-error" // something other than 400/413/429 — not a context answer
28
+ | "skipped-paid" // external model, probing costs money
29
+ | "skipped-hidden" // on models.json's exclude list (not user-facing)
30
+
31
+ export type ProbeRow = {
32
+ id: string
33
+ tier: "free" | "paid"
34
+ known: number | null // value in models.json, null = no entry
35
+ probed: number | null
36
+ how: ProbeHow
37
+ detail?: string
38
+ }
39
+
40
+ export type ProbeAllDeps = {
41
+ baseUrl: string
42
+ key: string
43
+ fetch: FetchLike
44
+ file: ModelsFile // known values + exclude list
45
+ includePaid?: boolean
46
+ /** Pause between models — the per-key bucket is 20/min across ALL models. */
47
+ paceMs?: number
48
+ log: (m: string) => void
49
+ now?: () => number
50
+ sleep?: (ms: number) => Promise<void>
51
+ }
52
+
53
+ const sleep = (ms: number) => new Promise<void>((r) => setTimeout(r, ms))
54
+
55
+ /** Rungs for one model. Known context → just above it (cheapest reject, minimal
56
+ * server tokenization). Unknown → one oversized rung: a rejection is free and
57
+ * names the exact limit, and acceptance only happens for contexts above 1.5M,
58
+ * which no UFR model has — so a ladder would only ever spend money on accepted
59
+ * rungs without adding information. */
60
+ export function sizesFor(known: number | null): number[] {
61
+ if (known !== null) {
62
+ const out = [known + 64]
63
+ for (let s = 2 * known + 64; s <= PROBE_CEILING && out.length < 3; s *= 2) out.push(s)
64
+ return out
65
+ }
66
+ return [1_500_000]
67
+ }
68
+
69
+ async function probeOne(o: ProbeAllDeps, id: string, known: number | null): Promise<ProbeRow> {
70
+ const doSleep = o.sleep ?? sleep
71
+ const floor = { value: 0 }
72
+ let last = { status: 0, body: "" }
73
+ for (const size of sizesFor(known)) {
74
+ for (let tries = 0; ; tries++) {
75
+ let res: Response
76
+ try {
77
+ res = await o.fetch(`${o.baseUrl}/chat/completions`, {
78
+ method: "POST",
79
+ headers: { Authorization: `Bearer ${o.key}`, "Content-Type": "application/json" },
80
+ body: JSON.stringify({
81
+ model: id,
82
+ max_tokens: 1,
83
+ messages: [{ role: "user", content: fillerForTokens(size) + "\n\nReply with exactly: OK" }],
84
+ }),
85
+ signal: AbortSignal.timeout(180_000),
86
+ redirect: "manual",
87
+ })
88
+ } catch (e) {
89
+ return { id, tier: "free", known, probed: floor.value || null,
90
+ how: floor.value ? "accepted-floor" : "http-error", detail: `transport: ${(e as Error).message}` }
91
+ }
92
+ if (res.ok) {
93
+ const j = (await res.json().catch(() => null)) as { usage?: { prompt_tokens?: number } } | null
94
+ floor.value = j?.usage?.prompt_tokens ?? size
95
+ break // accepted — try the next rung
96
+ }
97
+ const text = await res.text().catch(() => "")
98
+ if (res.status === 429 && tries < 2) {
99
+ o.log(` ${id}: 429 at ${size} tokens — backing off 20 s`)
100
+ await doSleep(20_000)
101
+ continue // bucket, not a context answer — retry the same rung
102
+ }
103
+ last = { status: res.status, body: text }
104
+ if (res.status === 400 || res.status === 413) {
105
+ const named = parseLimitFromBody(text)
106
+ if (named) return { id, tier: "free", known, probed: named, how: "error-named" }
107
+ return { id, tier: "free", known, probed: floor.value || null,
108
+ how: floor.value ? "accepted-floor" : "rejected-unnamed", detail: text.slice(0, 200) }
109
+ }
110
+ return { id, tier: "free", known, probed: floor.value || null,
111
+ how: floor.value ? "accepted-floor" : "http-error", detail: `HTTP ${res.status}: ${text.slice(0, 200)}` }
112
+ }
113
+ }
114
+ return { id, tier: "free", known, probed: floor.value || null, how: "accepted-floor",
115
+ detail: last.status ? `last: HTTP ${last.status}` : undefined }
116
+ }
117
+
118
+ export async function probeAllContexts(o: ProbeAllDeps): Promise<ProbeRow[]> {
119
+ const res = await o.fetch(`${o.baseUrl}/models`, {
120
+ headers: { Authorization: `Bearer ${o.key}` },
121
+ signal: AbortSignal.timeout(20_000),
122
+ redirect: "manual",
123
+ })
124
+ if (!res.ok) throw new Error(`UFR /api/models: HTTP ${res.status}`)
125
+ const ufr = parseUfrModels(await res.json())
126
+ const excluded = new Set(o.file.exclude)
127
+ const rows: ProbeRow[] = []
128
+ for (const m of ufr) {
129
+ if (excluded.has(m.id)) {
130
+ rows.push({ id: m.id, tier: m.tier, known: o.file.models[m.id]?.context ?? null, probed: null, how: "skipped-hidden" })
131
+ continue
132
+ }
133
+ if (m.tier === "paid" && !o.includePaid) {
134
+ rows.push({ id: m.id, tier: "paid", known: o.file.models[m.id]?.context ?? null, probed: null, how: "skipped-paid" })
135
+ continue
136
+ }
137
+ rows.push(await probeOne(o, m.id, o.file.models[m.id]?.context ?? null))
138
+ await (o.sleep ?? sleep)(o.paceMs ?? 3_000) // stay under the 20/min key bucket
139
+ }
140
+ return rows
141
+ }
142
+
143
+ /** Human-readable report; `status` flags what a models.json update would change. */
144
+ export function formatProbeReport(rows: ProbeRow[]): string {
145
+ const pad = (s: string, n: number) => (s.length >= n ? s : s + " ".repeat(n - s.length))
146
+ const lines = [
147
+ pad("model", 38) + pad("context", 12) + pad("how", 17) + pad("models.json", 12) + "status",
148
+ "-".repeat(94),
149
+ ]
150
+ let mismatches = 0
151
+ for (const r of rows) {
152
+ const context = r.probed === null ? "—" : r.probed.toLocaleString("en-US")
153
+ const status =
154
+ r.how === "error-named" && r.probed !== null
155
+ ? r.known === null ? "NEW — needs models.json entry" : r.probed === r.known ? "ok" : `MISMATCH (was ${r.known.toLocaleString("en-US")})`
156
+ : r.how
157
+ if (status.startsWith("NEW") || status.startsWith("MISMATCH")) mismatches++
158
+ lines.push(pad(r.id, 38) + pad(context, 12) + pad(r.how, 17) + pad(r.known === null ? "—" : r.known.toLocaleString("en-US"), 12) + status)
159
+ }
160
+ lines.push("-".repeat(94))
161
+ lines.push(`${rows.length} models, ${mismatches} needing a models.json update`)
162
+ return lines.join("\n")
163
+ }
164
+
165
+ /**
166
+ * Applies error-named results to a ModelsFile copy. Only `error-named` is
167
+ * exact; accepted floors are bounds and must not overwrite curated values.
168
+ * Existing notes are kept — a context measurement is appended, a previous
169
+ * context-probe note is replaced.
170
+ */
171
+ export function applyProbeResults(
172
+ file: ModelsFile,
173
+ rows: ProbeRow[],
174
+ date: string,
175
+ ): { file: ModelsFile; changed: { id: string; from: number | null; to: number }[] } {
176
+ const out: ModelsFile = JSON.parse(JSON.stringify(file))
177
+ out.updated = date
178
+ const changed: { id: string; from: number | null; to: number }[] = []
179
+ for (const r of rows) {
180
+ if (r.how !== "error-named" || r.probed === null) continue
181
+ const e = (out.models[r.id] ??= {})
182
+ const from = e.context ?? null
183
+ if (from === r.probed) continue
184
+ e.context = r.probed
185
+ const probeNote = `context ${r.probed} probed live ${date} (probe-all-contexts)`
186
+ e.note = e.note?.includes("probed live") ? probeNote : e.note ? `${e.note} | ${probeNote}` : probeNote
187
+ changed.push({ id: r.id, from, to: r.probed })
188
+ }
189
+ return { file: out, changed }
190
+ }
@@ -26,7 +26,7 @@ export function parseLimitFromBody(body: string): number | null {
26
26
  }
27
27
 
28
28
  /** Prompt filler: ~4.5 chars per token for prose-like text. */
29
- function fillerForTokens(tokens: number): string {
29
+ export function fillerForTokens(tokens: number): string {
30
30
  return "The quick brown fox jumps over the lazy dog. ".repeat(Math.ceil((tokens * 4.5) / 45))
31
31
  }
32
32