opencode-ufr 0.2.9 → 0.2.11

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
Files changed (42) hide show
  1. package/README.md +55 -62
  2. package/dist/tui.js +155 -0
  3. package/package.json +20 -7
  4. package/skills/pdf2md/SKILL.md +32 -0
  5. package/src/client/gateway.ts +101 -0
  6. package/src/client/pdf2md.ts +747 -0
  7. package/src/daemon/catalog-source.ts +9 -0
  8. package/src/daemon/daemon.ts +70 -18
  9. package/src/daemon/keypool.ts +12 -7
  10. package/src/daemon/main.ts +7 -1
  11. package/src/daemon/probe-all.ts +60 -38
  12. package/src/daemon/router.ts +215 -17
  13. package/src/daemon/server.ts +33 -1
  14. package/src/daemon/stats.ts +13 -1
  15. package/src/daemon/upstream.ts +14 -3
  16. package/src/daemon/vpn/fortinet.ts +42 -29
  17. package/src/daemon/vpn/manager.ts +50 -16
  18. package/src/daemon/vpn/proxy.ts +3 -1
  19. package/src/plugin/connect.ts +4 -4
  20. package/src/plugin/ensure-daemon.ts +48 -19
  21. package/src/plugin/index.ts +121 -46
  22. package/src/plugin/tui.tsx +77 -0
  23. package/src/shared/config.ts +53 -4
  24. package/src/shared/connect.ts +3 -4
  25. package/src/shared/daemon-client.ts +29 -0
  26. package/src/shared/errors.ts +1 -1
  27. package/src/shared/fs.ts +8 -3
  28. package/src/shared/keys.ts +65 -0
  29. package/src/shared/models-file.ts +1 -1
  30. package/src/shared/secrets.ts +2 -6
  31. package/bin/ufr.ts +0 -4
  32. package/src/cli/catalog.ts +0 -44
  33. package/src/cli/connect.ts +0 -146
  34. package/src/cli/context-probe.ts +0 -96
  35. package/src/cli/daemon-client.ts +0 -19
  36. package/src/cli/index.ts +0 -105
  37. package/src/cli/io.ts +0 -98
  38. package/src/cli/keys.ts +0 -130
  39. package/src/cli/stats.ts +0 -37
  40. package/src/cli/status.ts +0 -48
  41. package/src/cli/stop.ts +0 -8
  42. package/src/cli/ufr-check.ts +0 -43
@@ -40,6 +40,15 @@ export async function loadUfrModels(o: {
40
40
  redirect: "manual",
41
41
  })
42
42
  const type = res.headers.get("content-type") ?? ""
43
+ // The 3xx check must come first: real redirects carry text/html, so the HTML
44
+ // branch below would otherwise swallow them (and could even misfire isVpnPage
45
+ // on a redirect page). Same ordering as callUpstream.
46
+ if (res.status >= 300 && res.status < 400) {
47
+ // Not proof of the VPN wall — cancel the body (like callUpstream does) so
48
+ // the connection is not left hanging on a redirect we will not follow.
49
+ await res.body?.cancel()
50
+ throw new Error(`UFR answered with a redirect (HTTP ${res.status}) instead of JSON`)
51
+ }
43
52
  if (type.includes("text/html")) {
44
53
  const html = await res.text()
45
54
  throw new Error(isVpnPage(html) ? VPN_MESSAGE : `HTML instead of JSON (HTTP ${res.status})`)
@@ -2,6 +2,7 @@ import { randomBytes } from "node:crypto"
2
2
  import { mkdir, open, rm, stat } from "node:fs/promises"
3
3
  import { dirname } from "node:path"
4
4
  import { loadConfig, saveConfig } from "../shared/config"
5
+ import { readDaemonInfo } from "../shared/daemon-client"
5
6
  import { readJson, readText, writeFileAtomic } from "../shared/fs"
6
7
  import type { Paths } from "../shared/paths"
7
8
  import type { SecretStore } from "../shared/secrets"
@@ -44,6 +45,8 @@ export type StatusJson = {
44
45
  vpn: { mode: string; detail: string } | null
45
46
  spendToday: Record<string, number>
46
47
  dailyBudgetUsd: number
48
+ /** Rolling-window throughput — what /v1/_status dashboards display as req/s and tok/s. */
49
+ rates: { windowMs: number; requests: number; reqPerSec: number; tokensInPerSec: number; tokensOutPerSec: number }
47
50
  }
48
51
 
49
52
  export type DaemonOptions = {
@@ -116,8 +119,11 @@ export async function acquireLock(lockFile: string, o: AcquireLockOptions = {}):
116
119
  heldNonces.add(nonce)
117
120
  try {
118
121
  const fh = await open(lockFile, "wx")
119
- await fh.writeFile(`${process.pid} ${nonce}`)
120
- await fh.close()
122
+ try {
123
+ await fh.writeFile(`${process.pid} ${nonce}`)
124
+ } finally {
125
+ await fh.close() // a failed writeFile must not leak the fd
126
+ }
121
127
  nonceByLock.set(lockFile, nonce)
122
128
  return true
123
129
  } catch (e) {
@@ -173,8 +179,8 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
173
179
  const p = o.paths
174
180
  const config = await loadConfig(p.configFile)
175
181
  const confirmDaemon = async (pid: number): Promise<boolean> => {
176
- const saved = (await readJson(p.daemonFile)) as { pid?: number; port?: number } | null
177
- if (!saved || saved.pid !== pid || typeof saved.port !== "number") return false
182
+ const saved = await readDaemonInfo(p)
183
+ if (!saved || saved.pid !== pid) return false
178
184
  try {
179
185
  const res = await fetch(`http://127.0.0.1:${saved.port}/health`, { signal: AbortSignal.timeout(2_000) })
180
186
  if (!res.ok) return false
@@ -290,11 +296,18 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
290
296
  probing = true
291
297
  try {
292
298
  for (const id of unknown.slice(0, 2)) {
299
+ if (stopping) break
300
+ // Probes share UFR's rate-limit bucket with real traffic: reserve a
301
+ // key slot per probe and keep the admission (the router keeps a
302
+ // completed call's too), so probes cannot oversubscribe the bucket.
303
+ const reserved = keys.acquire()
304
+ if (reserved.kind === "none") break // pool exhausted — the next refresh probes again
305
+ if (reserved.waitMs > 0) await sleep(reserved.waitMs)
293
306
  if (stopping) break
294
307
  const result = await probeContextLimit({
295
308
  model: id,
296
309
  baseUrl: config.upstream.baseUrl,
297
- key: keyInfos[0]!.secret,
310
+ key: reserved.secret,
298
311
  transport,
299
312
  log,
300
313
  })
@@ -308,8 +321,10 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
308
321
  }
309
322
  }
310
323
  if (unknown.length > 0) {
311
- const mf2 = await loadBundledModels({ bundledPath: o.bundledModelsPath })
312
- catalog = buildCatalog(ufrModels, mf2.file, { allowPaid: config.allowPaid, probes: probeStore })
324
+ // Rebuild from the latest raw inputs: a newer catalog refresh may
325
+ // have replaced the catalog while these probes were in flight — the
326
+ // captured arguments are stale by now.
327
+ catalog = buildCatalog(lastUfrModels, lastModelsFile, { allowPaid: config.allowPaid, probes: probeStore })
313
328
  catalogInfo = { ...catalogInfo, loadedAt: now() }
314
329
  }
315
330
  } finally {
@@ -323,6 +338,8 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
323
338
  cachePath: p.ufrModelsCache, fetch: (u, i) => transport.fetch(u, i), log })
324
339
  catalog = buildCatalog(ufr.models, mf.file, { allowPaid: config.allowPaid, probes: probeStore })
325
340
  catalogInfo = { source: mf.source, ufrSource: ufr.source, loadedAt: now() }
341
+ lastUfrModels = ufr.models
342
+ lastModelsFile = mf.file
326
343
  if (ufr.error) Object.assign(reach, { ok: false, message: ufr.error, at: now() })
327
344
  else if (reach.ok !== true) Object.assign(reach, { ok: true, message: "", at: now() })
328
345
  for (const w of catalog.warnings) log(w)
@@ -331,6 +348,13 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
331
348
  else clearCatalogRetry()
332
349
  void scheduleProbes(ufr.models, mf.file).catch(() => {}) // must not outlive a shutdown
333
350
  }
351
+ // Raw inputs of the catalog, for scripts that probe every model (GET /v1/_catalog).
352
+ // lastModelsFile is assigned by refreshCatalog() right below, before any
353
+ // reader runs: scheduleProbes only fires from refreshCatalog (after the
354
+ // assignment) and catalogData is only served once the server has started,
355
+ // which happens after this awaited call.
356
+ let lastUfrModels: UfrModel[] = []
357
+ let lastModelsFile: ModelsFile
334
358
  await refreshCatalog()
335
359
 
336
360
  const router = new Router({
@@ -349,6 +373,19 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
349
373
  const startedAt = now()
350
374
  let lastActivity = now()
351
375
  let port = 0
376
+ /** Throughput over the last 60 s, from completed requests (a running stream's tokens land here when it ends). */
377
+ const ratesSnapshot = (): StatusJson["rates"] => {
378
+ const windowMs = 60_000
379
+ const r = db.rates(now() - windowMs)
380
+ const sec = windowMs / 1000
381
+ return {
382
+ windowMs,
383
+ requests: r.requests,
384
+ reqPerSec: Math.round((r.requests / sec) * 100) / 100,
385
+ tokensInPerSec: Math.round(r.promptTokens / sec),
386
+ tokensOutPerSec: Math.round(r.completionTokens / sec),
387
+ }
388
+ }
352
389
  const status = (): StatusJson => ({
353
390
  version: VERSION,
354
391
  pid: process.pid,
@@ -364,27 +401,40 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
364
401
  vpn: vpn ? { mode: vpn.status.mode, detail: vpn.status.detail } : null,
365
402
  spendToday: db.spendByKeySince(startOfLocalDay(now())),
366
403
  dailyBudgetUsd: config.dailyBudgetUsd,
404
+ rates: ratesSnapshot(),
367
405
  })
368
406
 
369
407
  let server: ReturnType<typeof startServer> | null = null
408
+ /** How long in-flight requests get to finish before connections and the db are force-closed. */
409
+ const drainGraceMs = 10_000
370
410
  const stop = async () => {
371
411
  if (stopping) return
372
412
  stopping = true
373
413
  for (const t of timers) clearInterval(t)
374
414
  clearCatalogRetry()
375
415
  try {
376
- db.setKv("breakers", JSON.stringify(breakers.snapshot()))
377
- } catch {
378
- // db already closed
416
+ // Drain first: in-flight requests still write stats.record() through the
417
+ // db handle and owe their clients a response that server.stop(true)
418
+ // would reset. Idle shutdowns reach this loop with nothing in flight.
419
+ const deadline = now() + drainGraceMs
420
+ while (router.inFlight > 0 && now() < deadline) await sleep(50)
421
+ try {
422
+ db.setKv("breakers", JSON.stringify(breakers.snapshot()))
423
+ } catch {
424
+ // db already closed
425
+ }
426
+ server?.stop()
427
+ db.close()
428
+ await vpn?.stop()
429
+ } finally {
430
+ // Cleanup always runs — even when vpn.stop() or anything above throws —
431
+ // so no daemon-file/lock-file is left behind and onStopped still fires.
432
+ await rm(p.daemonFile, { force: true }).catch(() => {})
433
+ await rm(p.lockFile, { force: true }).catch(() => {})
434
+ releaseLockNonce(p.lockFile)
435
+ log("stopped")
436
+ o.onStopped?.()
379
437
  }
380
- server?.stop()
381
- db.close()
382
- await vpn?.stop()
383
- await rm(p.daemonFile, { force: true })
384
- await rm(p.lockFile, { force: true })
385
- releaseLockNonce(p.lockFile)
386
- log("stopped")
387
- o.onStopped?.()
388
438
  }
389
439
 
390
440
  const fixed = o.port !== undefined || config.port !== null
@@ -398,6 +448,8 @@ export async function startDaemon(o: DaemonOptions): Promise<RunningDaemon> {
398
448
  router,
399
449
  models: () => listModels(catalog),
400
450
  status,
451
+ draining: () => stopping,
452
+ catalogData: () => ({ ufr: lastUfrModels, file: lastModelsFile }),
401
453
  onActivity: () => {
402
454
  lastActivity = now()
403
455
  },
@@ -5,10 +5,15 @@ export type Acquired = { kind: "ok"; alias: string; secret: string; waitMs: numb
5
5
  export type NotAcquired = { kind: "none"; reason: "no_keys" | "all_tried" | "exhausted"; retryAfterMs: number }
6
6
  export type KeySnapshot = { alias: string; used: number; cap: number; blockedForMs: number; invalid: boolean }
7
7
 
8
+ /** A 401 can be transient (portal hiccup, short-lived token rotation): the key
9
+ * is skipped for this long, then retried — instead of being dead until restart. */
10
+ export const INVALID_KEY_TTL_MS = 10 * 60_000
11
+
8
12
  type Slot = KeyInfo & {
9
13
  window: SlidingWindow
10
14
  blockedUntil: number
11
- invalid: boolean
15
+ /** Until when the key is considered invalid (0 = valid); a transient 401 expires. */
16
+ invalidUntil: number
12
17
  recent429: { at: number; model: string }[]
13
18
  }
14
19
 
@@ -24,14 +29,14 @@ export class KeyPool {
24
29
  ...k,
25
30
  window: new SlidingWindow({ cap: o.cap, windowMs: o.windowMs, maxWaitMs: o.maxWaitMs, now: o.now }),
26
31
  blockedUntil: 0,
27
- invalid: false,
32
+ invalidUntil: 0,
28
33
  recent429: [],
29
34
  }))
30
35
  }
31
36
 
32
- /** Number of keys not marked invalid. */
37
+ /** Number of keys not currently marked invalid. */
33
38
  get size(): number {
34
- return this.slots.filter((s) => !s.invalid).length
39
+ return this.slots.filter((s) => s.invalidUntil <= this.o.now()).length
35
40
  }
36
41
 
37
42
  private find(alias: string): Slot | undefined {
@@ -44,7 +49,7 @@ export class KeyPool {
44
49
 
45
50
  acquire(exclude: ReadonlySet<string> = new Set()): Acquired | NotAcquired {
46
51
  const now = this.o.now()
47
- const valid = this.slots.filter((s) => !s.invalid)
52
+ const valid = this.slots.filter((s) => s.invalidUntil <= now)
48
53
  if (valid.length === 0) return { kind: "none", reason: "no_keys", retryAfterMs: 0 }
49
54
  const usable = valid.filter((s) => !exclude.has(s.alias))
50
55
  if (usable.length === 0) return { kind: "none", reason: "all_tried", retryAfterMs: 0 }
@@ -93,7 +98,7 @@ export class KeyPool {
93
98
 
94
99
  onInvalid(alias: string): void {
95
100
  const s = this.find(alias)
96
- if (s) s.invalid = true
101
+ if (s) s.invalidUntil = this.o.now() + INVALID_KEY_TTL_MS
97
102
  }
98
103
 
99
104
  snapshot(): KeySnapshot[] {
@@ -103,7 +108,7 @@ export class KeyPool {
103
108
  used: s.window.inWindow(),
104
109
  cap: this.o.cap,
105
110
  blockedForMs: Math.max(0, s.blockedUntil - now),
106
- invalid: s.invalid,
111
+ invalid: s.invalidUntil > now,
107
112
  }))
108
113
  }
109
114
  }
@@ -7,7 +7,13 @@ const paths = resolvePaths()
7
7
  const log = createLogger(paths.logFile)
8
8
  try {
9
9
  const d = await startDaemon({ paths, secrets: new KeyringStore(), log, onStopped: () => process.exit(0) })
10
- for (const sig of ["SIGINT", "SIGTERM"] as const) process.on(sig, () => void d.stop())
10
+ let stopping = false
11
+ for (const sig of ["SIGINT", "SIGTERM"] as const)
12
+ process.on(sig, () => {
13
+ if (stopping) process.exit(1) // second signal: stop() is hanging — get out now
14
+ stopping = true
15
+ void d.stop()
16
+ })
11
17
  } catch (e) {
12
18
  if (e instanceof AlreadyRunningError) {
13
19
  log("another daemon is already running — exiting")
@@ -13,13 +13,23 @@
13
13
  */
14
14
 
15
15
  import type { ModelsFile } from "../shared/models-file"
16
- import type { FetchLike } from "./catalog-source"
17
- import { parseUfrModels } from "./catalog"
16
+ import type { UfrModel } from "./catalog"
18
17
  import { fillerForTokens, parseLimitFromBody } from "./probe"
19
18
 
20
19
  /** No UFR model is known to exceed this; probing higher only risks an accepted (paid) rung. */
21
20
  export const PROBE_CEILING = 1_600_000
22
21
 
22
+ /**
23
+ * Sub-1% deltas are noise, not context changes: some backends name the
24
+ * input-only limit (context minus reserved output tokens), so the same model
25
+ * can answer 262,144 locally and 261,144 from CI (observed 2026-10-05).
26
+ * Without a tolerance the weekly run would flip-flop such values forever.
27
+ */
28
+ export const PROBE_TOLERANCE = 0.01
29
+
30
+ export const withinTolerance = (probed: number, known: number): boolean =>
31
+ Math.abs(probed - known) / known < PROBE_TOLERANCE
32
+
23
33
  export type ProbeHow =
24
34
  | "error-named" // the server named the exact limit — the trustworthy result
25
35
  | "accepted-floor" // every rung up to `probed` was accepted — `probed` is a lower bound only
@@ -37,13 +47,23 @@ export type ProbeRow = {
37
47
  detail?: string
38
48
  }
39
49
 
50
+ /**
51
+ * One model call. Through the gateway's relay (POST /v1/_relay) this gets the
52
+ * central key rotation and soft rate limiting for free — the call waits for a
53
+ * key slot instead of being told to back off, and UFR's status and body come
54
+ * back verbatim (a context probe needs the 400 body that names the limit).
55
+ */
56
+ export type ProbeCallResult = { status: number; body: string; retryAfterMs?: number }
57
+ export type ProbeCall = (body: Record<string, unknown>) => Promise<ProbeCallResult>
58
+
40
59
  export type ProbeAllDeps = {
41
- baseUrl: string
42
- key: string
43
- fetch: FetchLike
60
+ call: ProbeCall
61
+ /** UFR's model list — the caller's job to fetch (gateway catalog or direct). */
62
+ ufr: UfrModel[]
44
63
  file: ModelsFile // known values + exclude list
45
64
  includePaid?: boolean
46
- /** Pause between models — the per-key bucket is 20/min across ALL models. */
65
+ /** Pause between models for direct callers that pace themselves. The gateway
66
+ * relay paces on its own — pass 0 there and the key pool decides. */
47
67
  paceMs?: number
48
68
  log: (m: string) => void
49
69
  now?: () => number
@@ -72,60 +92,58 @@ async function probeOne(o: ProbeAllDeps, id: string, known: number | null): Prom
72
92
  let last = { status: 0, body: "" }
73
93
  for (const size of sizesFor(known)) {
74
94
  for (let tries = 0; ; tries++) {
75
- let res: Response
95
+ let res: ProbeCallResult
76
96
  try {
77
- res = await o.fetch(`${o.baseUrl}/chat/completions`, {
78
- method: "POST",
79
- headers: { Authorization: `Bearer ${o.key}`, "Content-Type": "application/json" },
80
- body: JSON.stringify({
81
- model: id,
82
- max_tokens: 1,
83
- messages: [{ role: "user", content: fillerForTokens(size) + "\n\nReply with exactly: OK" }],
84
- }),
85
- signal: AbortSignal.timeout(180_000),
86
- redirect: "manual",
97
+ res = await o.call({
98
+ model: id,
99
+ max_tokens: 1,
100
+ messages: [{ role: "user", content: fillerForTokens(size) + "\n\nReply with exactly: OK" }],
87
101
  })
88
102
  } catch (e) {
89
103
  return { id, tier: "free", known, probed: floor.value || null,
90
104
  how: floor.value ? "accepted-floor" : "http-error", detail: `transport: ${(e as Error).message}` }
91
105
  }
92
- if (res.ok) {
93
- const j = (await res.json().catch(() => null)) as { usage?: { prompt_tokens?: number } } | null
106
+ if (res.status >= 200 && res.status < 300) {
107
+ const j = parseJson(res.body) as { usage?: { prompt_tokens?: number } } | null
94
108
  floor.value = j?.usage?.prompt_tokens ?? size
95
109
  break // accepted — try the next rung
96
110
  }
97
- const text = await res.text().catch(() => "")
98
111
  if (res.status === 429 && tries < 2) {
99
- o.log(` ${id}: 429 at ${size} tokens — backing off 20 s`)
100
- await doSleep(20_000)
101
- continue // bucket, not a context answer — retry the same rung
112
+ // 429 through the relay means walled, pool-capped or budget: all rare.
113
+ // Respect the gateway's retry-after when it names one (capped so a
114
+ // 30-minute ladder doesn't stall a whole probe run).
115
+ const wait = Math.min(res.retryAfterMs ?? 20_000, 120_000)
116
+ o.log(` ${id}: 429 at ${size} tokens — backing off ${Math.round(wait / 1000)} s`)
117
+ await doSleep(wait)
118
+ continue // not a context answer — retry the same rung
102
119
  }
103
- last = { status: res.status, body: text }
120
+ last = { status: res.status, body: res.body }
104
121
  if (res.status === 400 || res.status === 413) {
105
- const named = parseLimitFromBody(text)
122
+ const named = parseLimitFromBody(res.body)
106
123
  if (named) return { id, tier: "free", known, probed: named, how: "error-named" }
107
124
  return { id, tier: "free", known, probed: floor.value || null,
108
- how: floor.value ? "accepted-floor" : "rejected-unnamed", detail: text.slice(0, 200) }
125
+ how: floor.value ? "accepted-floor" : "rejected-unnamed", detail: res.body.slice(0, 200) }
109
126
  }
110
127
  return { id, tier: "free", known, probed: floor.value || null,
111
- how: floor.value ? "accepted-floor" : "http-error", detail: `HTTP ${res.status}: ${text.slice(0, 200)}` }
128
+ how: floor.value ? "accepted-floor" : "http-error", detail: `HTTP ${res.status}: ${res.body.slice(0, 200)}` }
112
129
  }
113
130
  }
114
131
  return { id, tier: "free", known, probed: floor.value || null, how: "accepted-floor",
115
132
  detail: last.status ? `last: HTTP ${last.status}` : undefined }
116
133
  }
117
134
 
135
+ function parseJson(text: string): unknown {
136
+ try {
137
+ return JSON.parse(text)
138
+ } catch {
139
+ return null
140
+ }
141
+ }
142
+
118
143
  export async function probeAllContexts(o: ProbeAllDeps): Promise<ProbeRow[]> {
119
- const res = await o.fetch(`${o.baseUrl}/models`, {
120
- headers: { Authorization: `Bearer ${o.key}` },
121
- signal: AbortSignal.timeout(20_000),
122
- redirect: "manual",
123
- })
124
- if (!res.ok) throw new Error(`UFR /api/models: HTTP ${res.status}`)
125
- const ufr = parseUfrModels(await res.json())
126
144
  const excluded = new Set(o.file.exclude)
127
145
  const rows: ProbeRow[] = []
128
- for (const m of ufr) {
146
+ for (const m of o.ufr) {
129
147
  if (excluded.has(m.id)) {
130
148
  rows.push({ id: m.id, tier: m.tier, known: o.file.models[m.id]?.context ?? null, probed: null, how: "skipped-hidden" })
131
149
  continue
@@ -135,7 +153,7 @@ export async function probeAllContexts(o: ProbeAllDeps): Promise<ProbeRow[]> {
135
153
  continue
136
154
  }
137
155
  rows.push(await probeOne(o, m.id, o.file.models[m.id]?.context ?? null))
138
- await (o.sleep ?? sleep)(o.paceMs ?? 3_000) // stay under the 20/min key bucket
156
+ if ((o.paceMs ?? 0) > 0) await (o.sleep ?? sleep)(o.paceMs!) // direct mode only: the relay paces itself
139
157
  }
140
158
  return rows
141
159
  }
@@ -152,7 +170,11 @@ export function formatProbeReport(rows: ProbeRow[]): string {
152
170
  const context = r.probed === null ? "—" : r.probed.toLocaleString("en-US")
153
171
  const status =
154
172
  r.how === "error-named" && r.probed !== null
155
- ? r.known === null ? "NEW — needs models.json entry" : r.probed === r.known ? "ok" : `MISMATCH (was ${r.known.toLocaleString("en-US")})`
173
+ ? r.known === null
174
+ ? "NEW — needs models.json entry"
175
+ : withinTolerance(r.probed, r.known)
176
+ ? r.probed === r.known ? "ok" : `ok (±${Math.abs(r.probed - r.known).toLocaleString("en-US")})`
177
+ : `MISMATCH (was ${r.known.toLocaleString("en-US")})`
156
178
  : r.how
157
179
  if (status.startsWith("NEW") || status.startsWith("MISMATCH")) mismatches++
158
180
  lines.push(pad(r.id, 38) + pad(context, 12) + pad(r.how, 17) + pad(r.known === null ? "—" : r.known.toLocaleString("en-US"), 12) + status)
@@ -180,7 +202,7 @@ export function applyProbeResults(
180
202
  if (r.how !== "error-named" || r.probed === null) continue
181
203
  const e = (out.models[r.id] ??= {})
182
204
  const from = e.context ?? null
183
- if (from === r.probed) continue
205
+ if (from === r.probed || (from !== null && withinTolerance(r.probed, from))) continue
184
206
  e.context = r.probed
185
207
  const probeNote = `context ${r.probed} probed live ${date} (probe-all-contexts)`
186
208
  e.note = e.note?.includes("probed live") ? probeNote : e.note ? `${e.note} | ${probeNote}` : probeNote