opencode-ufr 0.2.8 → 0.2.10
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +65 -62
- package/models.json +11 -5
- package/package.json +14 -6
- package/skills/pdf2md/SKILL.md +32 -0
- package/src/client/gateway.ts +101 -0
- package/src/client/pdf2md.ts +747 -0
- package/src/daemon/catalog-source.ts +9 -0
- package/src/daemon/daemon.ts +70 -18
- package/src/daemon/keypool.ts +12 -7
- package/src/daemon/main.ts +7 -1
- package/src/daemon/probe-all.ts +212 -0
- package/src/daemon/probe.ts +1 -1
- package/src/daemon/router.ts +215 -17
- package/src/daemon/server.ts +33 -1
- package/src/daemon/stats.ts +13 -1
- package/src/daemon/upstream.ts +14 -3
- package/src/daemon/vpn/fortinet.ts +42 -29
- package/src/daemon/vpn/manager.ts +50 -16
- package/src/daemon/vpn/proxy.ts +3 -1
- package/src/plugin/connect.ts +4 -4
- package/src/plugin/ensure-daemon.ts +48 -19
- package/src/plugin/index.ts +121 -46
- package/src/plugin/tui.tsx +77 -0
- package/src/shared/config.ts +53 -4
- package/src/shared/connect.ts +3 -4
- package/src/shared/daemon-client.ts +29 -0
- package/src/shared/errors.ts +1 -1
- package/src/shared/fs.ts +8 -3
- package/src/shared/keys.ts +65 -0
- package/src/shared/models-file.ts +1 -1
- package/src/shared/secrets.ts +2 -6
- package/bin/ufr.ts +0 -4
- package/src/cli/catalog.ts +0 -44
- package/src/cli/connect.ts +0 -146
- package/src/cli/daemon-client.ts +0 -19
- package/src/cli/index.ts +0 -97
- package/src/cli/io.ts +0 -98
- package/src/cli/keys.ts +0 -130
- package/src/cli/stats.ts +0 -37
- package/src/cli/status.ts +0 -48
- package/src/cli/stop.ts +0 -8
- package/src/cli/ufr-check.ts +0 -43
|
@@ -40,6 +40,15 @@ export async function loadUfrModels(o: {
|
|
|
40
40
|
redirect: "manual",
|
|
41
41
|
})
|
|
42
42
|
const type = res.headers.get("content-type") ?? ""
|
|
43
|
+
// The 3xx check must come first: real redirects carry text/html, so the HTML
|
|
44
|
+
// branch below would otherwise swallow them (and could even misfire isVpnPage
|
|
45
|
+
// on a redirect page). Same ordering as callUpstream.
|
|
46
|
+
if (res.status >= 300 && res.status < 400) {
|
|
47
|
+
// Not proof of the VPN wall — cancel the body (like callUpstream does) so
|
|
48
|
+
// the connection is not left hanging on a redirect we will not follow.
|
|
49
|
+
await res.body?.cancel()
|
|
50
|
+
throw new Error(`UFR answered with a redirect (HTTP ${res.status}) instead of JSON`)
|
|
51
|
+
}
|
|
43
52
|
if (type.includes("text/html")) {
|
|
44
53
|
const html = await res.text()
|
|
45
54
|
throw new Error(isVpnPage(html) ? VPN_MESSAGE : `HTML instead of JSON (HTTP ${res.status})`)
|
package/src/daemon/daemon.ts
CHANGED
|
@@ -2,6 +2,7 @@ import { randomBytes } from "node:crypto"
|
|
|
2
2
|
import { mkdir, open, rm, stat } from "node:fs/promises"
|
|
3
3
|
import { dirname } from "node:path"
|
|
4
4
|
import { loadConfig, saveConfig } from "../shared/config"
|
|
5
|
+
import { readDaemonInfo } from "../shared/daemon-client"
|
|
5
6
|
import { readJson, readText, writeFileAtomic } from "../shared/fs"
|
|
6
7
|
import type { Paths } from "../shared/paths"
|
|
7
8
|
import type { SecretStore } from "../shared/secrets"
|
|
@@ -44,6 +45,8 @@ export type StatusJson = {
|
|
|
44
45
|
vpn: { mode: string; detail: string } | null
|
|
45
46
|
spendToday: Record<string, number>
|
|
46
47
|
dailyBudgetUsd: number
|
|
48
|
+
/** Rolling-window throughput — what /v1/_status dashboards display as req/s and tok/s. */
|
|
49
|
+
rates: { windowMs: number; requests: number; reqPerSec: number; tokensInPerSec: number; tokensOutPerSec: number }
|
|
47
50
|
}
|
|
48
51
|
|
|
49
52
|
export type DaemonOptions = {
|
|
@@ -116,8 +119,11 @@ export async function acquireLock(lockFile: string, o: AcquireLockOptions = {}):
|
|
|
116
119
|
heldNonces.add(nonce)
|
|
117
120
|
try {
|
|
118
121
|
const fh = await open(lockFile, "wx")
|
|
119
|
-
|
|
120
|
-
|
|
122
|
+
try {
|
|
123
|
+
await fh.writeFile(`${process.pid} ${nonce}`)
|
|
124
|
+
} finally {
|
|
125
|
+
await fh.close() // a failed writeFile must not leak the fd
|
|
126
|
+
}
|
|
121
127
|
nonceByLock.set(lockFile, nonce)
|
|
122
128
|
return true
|
|
123
129
|
} catch (e) {
|
|
@@ -173,8 +179,8 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
173
179
|
const p = o.paths
|
|
174
180
|
const config = await loadConfig(p.configFile)
|
|
175
181
|
const confirmDaemon = async (pid: number): Promise<boolean> => {
|
|
176
|
-
const saved =
|
|
177
|
-
if (!saved || saved.pid !== pid
|
|
182
|
+
const saved = await readDaemonInfo(p)
|
|
183
|
+
if (!saved || saved.pid !== pid) return false
|
|
178
184
|
try {
|
|
179
185
|
const res = await fetch(`http://127.0.0.1:${saved.port}/health`, { signal: AbortSignal.timeout(2_000) })
|
|
180
186
|
if (!res.ok) return false
|
|
@@ -290,11 +296,18 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
290
296
|
probing = true
|
|
291
297
|
try {
|
|
292
298
|
for (const id of unknown.slice(0, 2)) {
|
|
299
|
+
if (stopping) break
|
|
300
|
+
// Probes share UFR's rate-limit bucket with real traffic: reserve a
|
|
301
|
+
// key slot per probe and keep the admission (the router keeps a
|
|
302
|
+
// completed call's too), so probes cannot oversubscribe the bucket.
|
|
303
|
+
const reserved = keys.acquire()
|
|
304
|
+
if (reserved.kind === "none") break // pool exhausted — the next refresh probes again
|
|
305
|
+
if (reserved.waitMs > 0) await sleep(reserved.waitMs)
|
|
293
306
|
if (stopping) break
|
|
294
307
|
const result = await probeContextLimit({
|
|
295
308
|
model: id,
|
|
296
309
|
baseUrl: config.upstream.baseUrl,
|
|
297
|
-
key:
|
|
310
|
+
key: reserved.secret,
|
|
298
311
|
transport,
|
|
299
312
|
log,
|
|
300
313
|
})
|
|
@@ -308,8 +321,10 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
308
321
|
}
|
|
309
322
|
}
|
|
310
323
|
if (unknown.length > 0) {
|
|
311
|
-
|
|
312
|
-
|
|
324
|
+
// Rebuild from the latest raw inputs: a newer catalog refresh may
|
|
325
|
+
// have replaced the catalog while these probes were in flight — the
|
|
326
|
+
// captured arguments are stale by now.
|
|
327
|
+
catalog = buildCatalog(lastUfrModels, lastModelsFile, { allowPaid: config.allowPaid, probes: probeStore })
|
|
313
328
|
catalogInfo = { ...catalogInfo, loadedAt: now() }
|
|
314
329
|
}
|
|
315
330
|
} finally {
|
|
@@ -323,6 +338,8 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
323
338
|
cachePath: p.ufrModelsCache, fetch: (u, i) => transport.fetch(u, i), log })
|
|
324
339
|
catalog = buildCatalog(ufr.models, mf.file, { allowPaid: config.allowPaid, probes: probeStore })
|
|
325
340
|
catalogInfo = { source: mf.source, ufrSource: ufr.source, loadedAt: now() }
|
|
341
|
+
lastUfrModels = ufr.models
|
|
342
|
+
lastModelsFile = mf.file
|
|
326
343
|
if (ufr.error) Object.assign(reach, { ok: false, message: ufr.error, at: now() })
|
|
327
344
|
else if (reach.ok !== true) Object.assign(reach, { ok: true, message: "", at: now() })
|
|
328
345
|
for (const w of catalog.warnings) log(w)
|
|
@@ -331,6 +348,13 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
331
348
|
else clearCatalogRetry()
|
|
332
349
|
void scheduleProbes(ufr.models, mf.file).catch(() => {}) // must not outlive a shutdown
|
|
333
350
|
}
|
|
351
|
+
// Raw inputs of the catalog, for scripts that probe every model (GET /v1/_catalog).
|
|
352
|
+
// lastModelsFile is assigned by refreshCatalog() right below, before any
|
|
353
|
+
// reader runs: scheduleProbes only fires from refreshCatalog (after the
|
|
354
|
+
// assignment) and catalogData is only served once the server has started,
|
|
355
|
+
// which happens after this awaited call.
|
|
356
|
+
let lastUfrModels: UfrModel[] = []
|
|
357
|
+
let lastModelsFile: ModelsFile
|
|
334
358
|
await refreshCatalog()
|
|
335
359
|
|
|
336
360
|
const router = new Router({
|
|
@@ -349,6 +373,19 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
349
373
|
const startedAt = now()
|
|
350
374
|
let lastActivity = now()
|
|
351
375
|
let port = 0
|
|
376
|
+
/** Throughput over the last 60 s, from completed requests (a running stream's tokens land here when it ends). */
|
|
377
|
+
const ratesSnapshot = (): StatusJson["rates"] => {
|
|
378
|
+
const windowMs = 60_000
|
|
379
|
+
const r = db.rates(now() - windowMs)
|
|
380
|
+
const sec = windowMs / 1000
|
|
381
|
+
return {
|
|
382
|
+
windowMs,
|
|
383
|
+
requests: r.requests,
|
|
384
|
+
reqPerSec: Math.round((r.requests / sec) * 100) / 100,
|
|
385
|
+
tokensInPerSec: Math.round(r.promptTokens / sec),
|
|
386
|
+
tokensOutPerSec: Math.round(r.completionTokens / sec),
|
|
387
|
+
}
|
|
388
|
+
}
|
|
352
389
|
const status = (): StatusJson => ({
|
|
353
390
|
version: VERSION,
|
|
354
391
|
pid: process.pid,
|
|
@@ -364,27 +401,40 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
364
401
|
vpn: vpn ? { mode: vpn.status.mode, detail: vpn.status.detail } : null,
|
|
365
402
|
spendToday: db.spendByKeySince(startOfLocalDay(now())),
|
|
366
403
|
dailyBudgetUsd: config.dailyBudgetUsd,
|
|
404
|
+
rates: ratesSnapshot(),
|
|
367
405
|
})
|
|
368
406
|
|
|
369
407
|
let server: ReturnType<typeof startServer> | null = null
|
|
408
|
+
/** How long in-flight requests get to finish before connections and the db are force-closed. */
|
|
409
|
+
const drainGraceMs = 10_000
|
|
370
410
|
const stop = async () => {
|
|
371
411
|
if (stopping) return
|
|
372
412
|
stopping = true
|
|
373
413
|
for (const t of timers) clearInterval(t)
|
|
374
414
|
clearCatalogRetry()
|
|
375
415
|
try {
|
|
376
|
-
|
|
377
|
-
|
|
378
|
-
//
|
|
416
|
+
// Drain first: in-flight requests still write stats.record() through the
|
|
417
|
+
// db handle and owe their clients a response that server.stop(true)
|
|
418
|
+
// would reset. Idle shutdowns reach this loop with nothing in flight.
|
|
419
|
+
const deadline = now() + drainGraceMs
|
|
420
|
+
while (router.inFlight > 0 && now() < deadline) await sleep(50)
|
|
421
|
+
try {
|
|
422
|
+
db.setKv("breakers", JSON.stringify(breakers.snapshot()))
|
|
423
|
+
} catch {
|
|
424
|
+
// db already closed
|
|
425
|
+
}
|
|
426
|
+
server?.stop()
|
|
427
|
+
db.close()
|
|
428
|
+
await vpn?.stop()
|
|
429
|
+
} finally {
|
|
430
|
+
// Cleanup always runs — even when vpn.stop() or anything above throws —
|
|
431
|
+
// so no daemon-file/lock-file is left behind and onStopped still fires.
|
|
432
|
+
await rm(p.daemonFile, { force: true }).catch(() => {})
|
|
433
|
+
await rm(p.lockFile, { force: true }).catch(() => {})
|
|
434
|
+
releaseLockNonce(p.lockFile)
|
|
435
|
+
log("stopped")
|
|
436
|
+
o.onStopped?.()
|
|
379
437
|
}
|
|
380
|
-
server?.stop()
|
|
381
|
-
db.close()
|
|
382
|
-
await vpn?.stop()
|
|
383
|
-
await rm(p.daemonFile, { force: true })
|
|
384
|
-
await rm(p.lockFile, { force: true })
|
|
385
|
-
releaseLockNonce(p.lockFile)
|
|
386
|
-
log("stopped")
|
|
387
|
-
o.onStopped?.()
|
|
388
438
|
}
|
|
389
439
|
|
|
390
440
|
const fixed = o.port !== undefined || config.port !== null
|
|
@@ -398,6 +448,8 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
|
|
|
398
448
|
router,
|
|
399
449
|
models: () => listModels(catalog),
|
|
400
450
|
status,
|
|
451
|
+
draining: () => stopping,
|
|
452
|
+
catalogData: () => ({ ufr: lastUfrModels, file: lastModelsFile }),
|
|
401
453
|
onActivity: () => {
|
|
402
454
|
lastActivity = now()
|
|
403
455
|
},
|
package/src/daemon/keypool.ts
CHANGED
|
@@ -5,10 +5,15 @@ export type Acquired = { kind: "ok"; alias: string; secret: string; waitMs: numb
|
|
|
5
5
|
export type NotAcquired = { kind: "none"; reason: "no_keys" | "all_tried" | "exhausted"; retryAfterMs: number }
|
|
6
6
|
export type KeySnapshot = { alias: string; used: number; cap: number; blockedForMs: number; invalid: boolean }
|
|
7
7
|
|
|
8
|
+
/** A 401 can be transient (portal hiccup, short-lived token rotation): the key
|
|
9
|
+
* is skipped for this long, then retried — instead of being dead until restart. */
|
|
10
|
+
export const INVALID_KEY_TTL_MS = 10 * 60_000
|
|
11
|
+
|
|
8
12
|
type Slot = KeyInfo & {
|
|
9
13
|
window: SlidingWindow
|
|
10
14
|
blockedUntil: number
|
|
11
|
-
invalid
|
|
15
|
+
/** Until when the key is considered invalid (0 = valid); a transient 401 expires. */
|
|
16
|
+
invalidUntil: number
|
|
12
17
|
recent429: { at: number; model: string }[]
|
|
13
18
|
}
|
|
14
19
|
|
|
@@ -24,14 +29,14 @@ export class KeyPool {
|
|
|
24
29
|
...k,
|
|
25
30
|
window: new SlidingWindow({ cap: o.cap, windowMs: o.windowMs, maxWaitMs: o.maxWaitMs, now: o.now }),
|
|
26
31
|
blockedUntil: 0,
|
|
27
|
-
|
|
32
|
+
invalidUntil: 0,
|
|
28
33
|
recent429: [],
|
|
29
34
|
}))
|
|
30
35
|
}
|
|
31
36
|
|
|
32
|
-
/** Number of keys not marked invalid. */
|
|
37
|
+
/** Number of keys not currently marked invalid. */
|
|
33
38
|
get size(): number {
|
|
34
|
-
return this.slots.filter((s) =>
|
|
39
|
+
return this.slots.filter((s) => s.invalidUntil <= this.o.now()).length
|
|
35
40
|
}
|
|
36
41
|
|
|
37
42
|
private find(alias: string): Slot | undefined {
|
|
@@ -44,7 +49,7 @@ export class KeyPool {
|
|
|
44
49
|
|
|
45
50
|
acquire(exclude: ReadonlySet<string> = new Set()): Acquired | NotAcquired {
|
|
46
51
|
const now = this.o.now()
|
|
47
|
-
const valid = this.slots.filter((s) =>
|
|
52
|
+
const valid = this.slots.filter((s) => s.invalidUntil <= now)
|
|
48
53
|
if (valid.length === 0) return { kind: "none", reason: "no_keys", retryAfterMs: 0 }
|
|
49
54
|
const usable = valid.filter((s) => !exclude.has(s.alias))
|
|
50
55
|
if (usable.length === 0) return { kind: "none", reason: "all_tried", retryAfterMs: 0 }
|
|
@@ -93,7 +98,7 @@ export class KeyPool {
|
|
|
93
98
|
|
|
94
99
|
onInvalid(alias: string): void {
|
|
95
100
|
const s = this.find(alias)
|
|
96
|
-
if (s) s.
|
|
101
|
+
if (s) s.invalidUntil = this.o.now() + INVALID_KEY_TTL_MS
|
|
97
102
|
}
|
|
98
103
|
|
|
99
104
|
snapshot(): KeySnapshot[] {
|
|
@@ -103,7 +108,7 @@ export class KeyPool {
|
|
|
103
108
|
used: s.window.inWindow(),
|
|
104
109
|
cap: this.o.cap,
|
|
105
110
|
blockedForMs: Math.max(0, s.blockedUntil - now),
|
|
106
|
-
invalid: s.
|
|
111
|
+
invalid: s.invalidUntil > now,
|
|
107
112
|
}))
|
|
108
113
|
}
|
|
109
114
|
}
|
package/src/daemon/main.ts
CHANGED
|
@@ -7,7 +7,13 @@ const paths = resolvePaths()
|
|
|
7
7
|
const log = createLogger(paths.logFile)
|
|
8
8
|
try {
|
|
9
9
|
const d = await startDaemon({ paths, secrets: new KeyringStore(), log, onStopped: () => process.exit(0) })
|
|
10
|
-
|
|
10
|
+
let stopping = false
|
|
11
|
+
for (const sig of ["SIGINT", "SIGTERM"] as const)
|
|
12
|
+
process.on(sig, () => {
|
|
13
|
+
if (stopping) process.exit(1) // second signal: stop() is hanging — get out now
|
|
14
|
+
stopping = true
|
|
15
|
+
void d.stop()
|
|
16
|
+
})
|
|
11
17
|
} catch (e) {
|
|
12
18
|
if (e instanceof AlreadyRunningError) {
|
|
13
19
|
log("another daemon is already running — exiting")
|
|
@@ -0,0 +1,212 @@
|
|
|
1
|
+
/**
|
|
2
|
+
* Probe the context limit of every model UFR serves, reusing probe.ts's
|
|
3
|
+
* oversized-request trick: one request above a model's known context makes
|
|
4
|
+
* the server name the exact limit ("maximum context length is N tokens").
|
|
5
|
+
*
|
|
6
|
+
* Cost profile matters here: a REJECTED request is refused before any tokens
|
|
7
|
+
* are priced (measured 2026-10-05: 740 ms, x-process-time 0), so probing a
|
|
8
|
+
* model whose recorded context is still right costs nothing and does not
|
|
9
|
+
* count against the 20/min key bucket. Money is only spent when a probe rung
|
|
10
|
+
* is ACCEPTED (context grew since the last measurement) or a paid model is
|
|
11
|
+
* probed on an accepted rung — which is why paid models are skipped unless
|
|
12
|
+
* asked for.
|
|
13
|
+
*/
|
|
14
|
+
|
|
15
|
+
import type { ModelsFile } from "../shared/models-file"
|
|
16
|
+
import type { UfrModel } from "./catalog"
|
|
17
|
+
import { fillerForTokens, parseLimitFromBody } from "./probe"
|
|
18
|
+
|
|
19
|
+
/** No UFR model is known to exceed this; probing higher only risks an accepted (paid) rung. */
|
|
20
|
+
export const PROBE_CEILING = 1_600_000
|
|
21
|
+
|
|
22
|
+
/**
|
|
23
|
+
* Sub-1% deltas are noise, not context changes: some backends name the
|
|
24
|
+
* input-only limit (context minus reserved output tokens), so the same model
|
|
25
|
+
* can answer 262,144 locally and 261,144 from CI (observed 2026-10-05).
|
|
26
|
+
* Without a tolerance the weekly run would flip-flop such values forever.
|
|
27
|
+
*/
|
|
28
|
+
export const PROBE_TOLERANCE = 0.01
|
|
29
|
+
|
|
30
|
+
export const withinTolerance = (probed: number, known: number): boolean =>
|
|
31
|
+
Math.abs(probed - known) / known < PROBE_TOLERANCE
|
|
32
|
+
|
|
33
|
+
export type ProbeHow =
|
|
34
|
+
| "error-named" // the server named the exact limit — the trustworthy result
|
|
35
|
+
| "accepted-floor" // every rung up to `probed` was accepted — `probed` is a lower bound only
|
|
36
|
+
| "rejected-unnamed" // rejected, but the body named no limit
|
|
37
|
+
| "http-error" // something other than 400/413/429 — not a context answer
|
|
38
|
+
| "skipped-paid" // external model, probing costs money
|
|
39
|
+
| "skipped-hidden" // on models.json's exclude list (not user-facing)
|
|
40
|
+
|
|
41
|
+
export type ProbeRow = {
|
|
42
|
+
id: string
|
|
43
|
+
tier: "free" | "paid"
|
|
44
|
+
known: number | null // value in models.json, null = no entry
|
|
45
|
+
probed: number | null
|
|
46
|
+
how: ProbeHow
|
|
47
|
+
detail?: string
|
|
48
|
+
}
|
|
49
|
+
|
|
50
|
+
/**
|
|
51
|
+
* One model call. Through the gateway's relay (POST /v1/_relay) this gets the
|
|
52
|
+
* central key rotation and soft rate limiting for free — the call waits for a
|
|
53
|
+
* key slot instead of being told to back off, and UFR's status and body come
|
|
54
|
+
* back verbatim (a context probe needs the 400 body that names the limit).
|
|
55
|
+
*/
|
|
56
|
+
export type ProbeCallResult = { status: number; body: string; retryAfterMs?: number }
|
|
57
|
+
export type ProbeCall = (body: Record<string, unknown>) => Promise<ProbeCallResult>
|
|
58
|
+
|
|
59
|
+
export type ProbeAllDeps = {
|
|
60
|
+
call: ProbeCall
|
|
61
|
+
/** UFR's model list — the caller's job to fetch (gateway catalog or direct). */
|
|
62
|
+
ufr: UfrModel[]
|
|
63
|
+
file: ModelsFile // known values + exclude list
|
|
64
|
+
includePaid?: boolean
|
|
65
|
+
/** Pause between models for direct callers that pace themselves. The gateway
|
|
66
|
+
* relay paces on its own — pass 0 there and the key pool decides. */
|
|
67
|
+
paceMs?: number
|
|
68
|
+
log: (m: string) => void
|
|
69
|
+
now?: () => number
|
|
70
|
+
sleep?: (ms: number) => Promise<void>
|
|
71
|
+
}
|
|
72
|
+
|
|
73
|
+
const sleep = (ms: number) => new Promise<void>((r) => setTimeout(r, ms))
|
|
74
|
+
|
|
75
|
+
/** Rungs for one model. Known context → just above it (cheapest reject, minimal
|
|
76
|
+
* server tokenization). Unknown → one oversized rung: a rejection is free and
|
|
77
|
+
* names the exact limit, and acceptance only happens for contexts above 1.5M,
|
|
78
|
+
* which no UFR model has — so a ladder would only ever spend money on accepted
|
|
79
|
+
* rungs without adding information. */
|
|
80
|
+
export function sizesFor(known: number | null): number[] {
|
|
81
|
+
if (known !== null) {
|
|
82
|
+
const out = [known + 64]
|
|
83
|
+
for (let s = 2 * known + 64; s <= PROBE_CEILING && out.length < 3; s *= 2) out.push(s)
|
|
84
|
+
return out
|
|
85
|
+
}
|
|
86
|
+
return [1_500_000]
|
|
87
|
+
}
|
|
88
|
+
|
|
89
|
+
async function probeOne(o: ProbeAllDeps, id: string, known: number | null): Promise<ProbeRow> {
|
|
90
|
+
const doSleep = o.sleep ?? sleep
|
|
91
|
+
const floor = { value: 0 }
|
|
92
|
+
let last = { status: 0, body: "" }
|
|
93
|
+
for (const size of sizesFor(known)) {
|
|
94
|
+
for (let tries = 0; ; tries++) {
|
|
95
|
+
let res: ProbeCallResult
|
|
96
|
+
try {
|
|
97
|
+
res = await o.call({
|
|
98
|
+
model: id,
|
|
99
|
+
max_tokens: 1,
|
|
100
|
+
messages: [{ role: "user", content: fillerForTokens(size) + "\n\nReply with exactly: OK" }],
|
|
101
|
+
})
|
|
102
|
+
} catch (e) {
|
|
103
|
+
return { id, tier: "free", known, probed: floor.value || null,
|
|
104
|
+
how: floor.value ? "accepted-floor" : "http-error", detail: `transport: ${(e as Error).message}` }
|
|
105
|
+
}
|
|
106
|
+
if (res.status >= 200 && res.status < 300) {
|
|
107
|
+
const j = parseJson(res.body) as { usage?: { prompt_tokens?: number } } | null
|
|
108
|
+
floor.value = j?.usage?.prompt_tokens ?? size
|
|
109
|
+
break // accepted — try the next rung
|
|
110
|
+
}
|
|
111
|
+
if (res.status === 429 && tries < 2) {
|
|
112
|
+
// 429 through the relay means walled, pool-capped or budget: all rare.
|
|
113
|
+
// Respect the gateway's retry-after when it names one (capped so a
|
|
114
|
+
// 30-minute ladder doesn't stall a whole probe run).
|
|
115
|
+
const wait = Math.min(res.retryAfterMs ?? 20_000, 120_000)
|
|
116
|
+
o.log(` ${id}: 429 at ${size} tokens — backing off ${Math.round(wait / 1000)} s`)
|
|
117
|
+
await doSleep(wait)
|
|
118
|
+
continue // not a context answer — retry the same rung
|
|
119
|
+
}
|
|
120
|
+
last = { status: res.status, body: res.body }
|
|
121
|
+
if (res.status === 400 || res.status === 413) {
|
|
122
|
+
const named = parseLimitFromBody(res.body)
|
|
123
|
+
if (named) return { id, tier: "free", known, probed: named, how: "error-named" }
|
|
124
|
+
return { id, tier: "free", known, probed: floor.value || null,
|
|
125
|
+
how: floor.value ? "accepted-floor" : "rejected-unnamed", detail: res.body.slice(0, 200) }
|
|
126
|
+
}
|
|
127
|
+
return { id, tier: "free", known, probed: floor.value || null,
|
|
128
|
+
how: floor.value ? "accepted-floor" : "http-error", detail: `HTTP ${res.status}: ${res.body.slice(0, 200)}` }
|
|
129
|
+
}
|
|
130
|
+
}
|
|
131
|
+
return { id, tier: "free", known, probed: floor.value || null, how: "accepted-floor",
|
|
132
|
+
detail: last.status ? `last: HTTP ${last.status}` : undefined }
|
|
133
|
+
}
|
|
134
|
+
|
|
135
|
+
function parseJson(text: string): unknown {
|
|
136
|
+
try {
|
|
137
|
+
return JSON.parse(text)
|
|
138
|
+
} catch {
|
|
139
|
+
return null
|
|
140
|
+
}
|
|
141
|
+
}
|
|
142
|
+
|
|
143
|
+
export async function probeAllContexts(o: ProbeAllDeps): Promise<ProbeRow[]> {
|
|
144
|
+
const excluded = new Set(o.file.exclude)
|
|
145
|
+
const rows: ProbeRow[] = []
|
|
146
|
+
for (const m of o.ufr) {
|
|
147
|
+
if (excluded.has(m.id)) {
|
|
148
|
+
rows.push({ id: m.id, tier: m.tier, known: o.file.models[m.id]?.context ?? null, probed: null, how: "skipped-hidden" })
|
|
149
|
+
continue
|
|
150
|
+
}
|
|
151
|
+
if (m.tier === "paid" && !o.includePaid) {
|
|
152
|
+
rows.push({ id: m.id, tier: "paid", known: o.file.models[m.id]?.context ?? null, probed: null, how: "skipped-paid" })
|
|
153
|
+
continue
|
|
154
|
+
}
|
|
155
|
+
rows.push(await probeOne(o, m.id, o.file.models[m.id]?.context ?? null))
|
|
156
|
+
if ((o.paceMs ?? 0) > 0) await (o.sleep ?? sleep)(o.paceMs!) // direct mode only: the relay paces itself
|
|
157
|
+
}
|
|
158
|
+
return rows
|
|
159
|
+
}
|
|
160
|
+
|
|
161
|
+
/** Human-readable report; `status` flags what a models.json update would change. */
|
|
162
|
+
export function formatProbeReport(rows: ProbeRow[]): string {
|
|
163
|
+
const pad = (s: string, n: number) => (s.length >= n ? s : s + " ".repeat(n - s.length))
|
|
164
|
+
const lines = [
|
|
165
|
+
pad("model", 38) + pad("context", 12) + pad("how", 17) + pad("models.json", 12) + "status",
|
|
166
|
+
"-".repeat(94),
|
|
167
|
+
]
|
|
168
|
+
let mismatches = 0
|
|
169
|
+
for (const r of rows) {
|
|
170
|
+
const context = r.probed === null ? "—" : r.probed.toLocaleString("en-US")
|
|
171
|
+
const status =
|
|
172
|
+
r.how === "error-named" && r.probed !== null
|
|
173
|
+
? r.known === null
|
|
174
|
+
? "NEW — needs models.json entry"
|
|
175
|
+
: withinTolerance(r.probed, r.known)
|
|
176
|
+
? r.probed === r.known ? "ok" : `ok (±${Math.abs(r.probed - r.known).toLocaleString("en-US")})`
|
|
177
|
+
: `MISMATCH (was ${r.known.toLocaleString("en-US")})`
|
|
178
|
+
: r.how
|
|
179
|
+
if (status.startsWith("NEW") || status.startsWith("MISMATCH")) mismatches++
|
|
180
|
+
lines.push(pad(r.id, 38) + pad(context, 12) + pad(r.how, 17) + pad(r.known === null ? "—" : r.known.toLocaleString("en-US"), 12) + status)
|
|
181
|
+
}
|
|
182
|
+
lines.push("-".repeat(94))
|
|
183
|
+
lines.push(`${rows.length} models, ${mismatches} needing a models.json update`)
|
|
184
|
+
return lines.join("\n")
|
|
185
|
+
}
|
|
186
|
+
|
|
187
|
+
/**
|
|
188
|
+
* Applies error-named results to a ModelsFile copy. Only `error-named` is
|
|
189
|
+
* exact; accepted floors are bounds and must not overwrite curated values.
|
|
190
|
+
* Existing notes are kept — a context measurement is appended, a previous
|
|
191
|
+
* context-probe note is replaced.
|
|
192
|
+
*/
|
|
193
|
+
export function applyProbeResults(
|
|
194
|
+
file: ModelsFile,
|
|
195
|
+
rows: ProbeRow[],
|
|
196
|
+
date: string,
|
|
197
|
+
): { file: ModelsFile; changed: { id: string; from: number | null; to: number }[] } {
|
|
198
|
+
const out: ModelsFile = JSON.parse(JSON.stringify(file))
|
|
199
|
+
out.updated = date
|
|
200
|
+
const changed: { id: string; from: number | null; to: number }[] = []
|
|
201
|
+
for (const r of rows) {
|
|
202
|
+
if (r.how !== "error-named" || r.probed === null) continue
|
|
203
|
+
const e = (out.models[r.id] ??= {})
|
|
204
|
+
const from = e.context ?? null
|
|
205
|
+
if (from === r.probed || (from !== null && withinTolerance(r.probed, from))) continue
|
|
206
|
+
e.context = r.probed
|
|
207
|
+
const probeNote = `context ${r.probed} probed live ${date} (probe-all-contexts)`
|
|
208
|
+
e.note = e.note?.includes("probed live") ? probeNote : e.note ? `${e.note} | ${probeNote}` : probeNote
|
|
209
|
+
changed.push({ id: r.id, from, to: r.probed })
|
|
210
|
+
}
|
|
211
|
+
return { file: out, changed }
|
|
212
|
+
}
|
package/src/daemon/probe.ts
CHANGED
|
@@ -26,7 +26,7 @@ export function parseLimitFromBody(body: string): number | null {
|
|
|
26
26
|
}
|
|
27
27
|
|
|
28
28
|
/** Prompt filler: ~4.5 chars per token for prose-like text. */
|
|
29
|
-
function fillerForTokens(tokens: number): string {
|
|
29
|
+
export function fillerForTokens(tokens: number): string {
|
|
30
30
|
return "The quick brown fox jumps over the lazy dog. ".repeat(Math.ceil((tokens * 4.5) / 45))
|
|
31
31
|
}
|
|
32
32
|
|