openzoo 0.48.18 → 0.48.22

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/namespace.js CHANGED
@@ -36,19 +36,45 @@ import { loadOrCreateWallet } from './wallet.js';
36
36
  * BREAKING: corpora bound before this lived in the shared tenant and are not
37
37
  * reachable from a signed request. Re-bind them.
38
38
  */
39
- let cached = null;
39
+ /**
40
+ * ONE NAMESPACE ACROSS EVERY STACC APP (2026-08-18).
41
+ *
42
+ * This used to be sha256("openzoo-ns:" + pubkey) — stable, per-user, and
43
+ * DIFFERENT from what every other stacc app sent. openzoo brain sent
44
+ * HMAC(OPENZOO_TENANT_SECRET, pubkey); the open-webui wallet sent a literal
45
+ * app name. Since the gateway keys a tenant on sha256(chain:signer:namespace),
46
+ * one wallet therefore landed in THREE tenants: three separate leCore
47
+ * memories and three separate credit balances, for the same person. Bind a
48
+ * corpus in the CLI and it was invisible from the browser.
49
+ *
50
+ * The constant is safe because the reason for the per-user hash is gone. It
51
+ * existed to stop someone addressing your tenant by guessing your namespace —
52
+ * but the gateway now folds the VERIFIED SIGNER into the tenant hash (see
53
+ * nsauth.ts / tenantFor), so the namespace string is no longer the
54
+ * access-control boundary. Signing is what proves ownership; this string only
55
+ * chooses WHICH of your namespaces you mean.
56
+ *
57
+ * A constant is in fact MORE private than what it replaces. A per-user hash is
58
+ * stable and unique, i.e. a tracking identifier that correlates one user's
59
+ * requests across time. A value every caller sends identically discloses
60
+ * nothing at all.
61
+ *
62
+ * BREAKING: corpora bound under the old per-wallet namespace live in a
63
+ * different tenant and will not be found. The gateway's tenantsToTry fallback
64
+ * reaches pre-existing context ids by id, but anything relying on the old
65
+ * namespace should be re-bound.
66
+ */
67
+ export const STACC_NAMESPACE = 'stacc';
40
68
 
41
69
  export function namespaceHeaderValue() {
42
- if (cached) return cached;
43
70
  try {
44
- const w = loadOrCreateWallet();
45
- cached = crypto.createHash('sha256')
46
- .update(`openzoo-ns:${w.keypair.publicKey.toBase58()}`)
47
- .digest('hex');
71
+ // Still require a wallet: the namespace is meaningless without a signer to
72
+ // prove it, and sending one unsigned drops you into the SHARED tenant.
73
+ loadOrCreateWallet();
74
+ return STACC_NAMESPACE;
48
75
  } catch {
49
- cached = ''; // no wallet (read-only use): fall back to the shared tenant
76
+ return ''; // no wallet (read-only use): fall back to the shared tenant
50
77
  }
51
- return cached;
52
78
  }
53
79
 
54
80
  /**
package/lib/podagent.mjs CHANGED
@@ -249,13 +249,40 @@ async function httpErrorNote(status) {
249
249
  // same rail, second call pays fine). Surfacing that as a chat message makes
250
250
  // the user do the retry by hand — so do it here instead.
251
251
  const PAYMENT_RETRIES = 3;
252
- async function postChat(body, contextId) {
252
+ /**
253
+ * ADAPTIVE top_k. We have learned this one the expensive way already.
254
+ *
255
+ * On leCore the miss was never the ranker — BM25 ranked correctly. It was
256
+ * top_k=16 against a 7,000-chunk corpus: we only ever ASKED for sixteen. Same
257
+ * shape here, and worse: grokui never set the header at all, so every call fell
258
+ * to the gateway default of EIGHT, while a whole project's bots write into one
259
+ * shared context. Six bots working for an hour and the model sees eight chunks
260
+ * of it.
261
+ *
262
+ * So scale with the corpus instead of picking a number. sqrt keeps it sane at
263
+ * both ends — 100 chunks -> 20, 1k -> 63, 7k -> 167, and it saturates at the
264
+ * gateway's 256 ceiling rather than growing without bound. Floor of 16 so a
265
+ * brand-new thread is never worse off than the old default.
266
+ *
267
+ * Cost is real and proportional (measured on leCore: top_k 16 = $0.0070,
268
+ * top_k 128 = $0.0489 on the same question) — which is the point. The extra
269
+ * spend IS the extra corpus actually being read.
270
+ */
271
+ export function adaptiveTopK(boundItems) {
272
+ const n = Math.max(0, Number(boundItems) || 0);
273
+ return Math.max(16, Math.min(256, Math.ceil(Math.sqrt(n) * 2)));
274
+ }
275
+
276
+ async function postChat(body, contextId, topK) {
253
277
  let r;
254
278
  for (let attempt = 0; attempt <= PAYMENT_RETRIES; attempt++) {
255
279
  r = await fetch(`${PROXY}/chat/completions`, {
256
280
  method: 'POST',
257
281
  headers: {
258
282
  'content-type': 'application/json', authorization: 'Bearer sk-openzoo',
283
+ // Only sent when we actually know the corpus size; without it the
284
+ // gateway keeps its own default rather than getting a made-up number.
285
+ ...(topK ? { 'x-hrr-top-k': String(topK) } : {}),
259
286
  // real leCore memory for this thread, bound via POST /v1/hrr/bind — NOT
260
287
  // a fabricated mechanism. Retrieval runs automatically once this header
261
288
  // is set; nothing more for the model to invent or explain.
@@ -284,7 +311,7 @@ function withModelId(messages, model) {
284
311
  : m));
285
312
  }
286
313
 
287
- export async function brain(messages, contextId, modelOverride) {
314
+ export async function brain(messages, contextId, modelOverride, topK) {
288
315
  // explicit plugins, not relying on the gateway's "inject when caller said
289
316
  // nothing" default — an explicit array is always respected as-is, so every
290
317
  // bot on every model actually has web search. max_tokens 900 was cutting
@@ -299,24 +326,53 @@ export async function brain(messages, contextId, modelOverride) {
299
326
  messages = vision ? messages : stripImages(messages);
300
327
  const r = await postChat(
301
328
  { model, max_tokens: 4096, messages: withModelId(messages, model), plugins: [{ id: 'web' }] },
302
- contextId,
329
+ contextId, topK,
303
330
  );
304
331
  const j = await r.json().catch(() => ({}));
305
332
  const content = j?.choices?.[0]?.message?.content;
333
+ // Same truncation catch as the streaming path (see brainStream): a reply that
334
+ // stops because the budget ran out is not a finished reply, and this path is
335
+ // what non-streaming callers — including every SPAWNed subagent — go through.
336
+ if (content && j?.choices?.[0]?.finish_reason === 'length') {
337
+ const rest = await brainContinue(messages, content, contextId, modelOverride, 0);
338
+ return content + rest;
339
+ }
306
340
  return content || (r.ok ? '' : await httpErrorNote(r.status));
307
341
  }
308
342
 
343
+ /** Resume a reply that hit the output cap, non-streaming. Bounded by
344
+ * CONTINUE_ROUNDS for the same runaway reason brainStream is. */
345
+ async function brainContinue(messages, sofar, contextId, modelOverride, round) {
346
+ if (round >= CONTINUE_ROUNDS) return '';
347
+ const vision = hasImages(messages);
348
+ const model = vision ? VISION_MODEL : (modelOverride || MODEL);
349
+ const next = [...messages,
350
+ { role: 'assistant', content: sofar },
351
+ { role: 'user', content: CONTINUE_NUDGE }];
352
+ const r = await postChat(
353
+ { model, max_tokens: Math.min(4096 * (2 ** (round + 1)), MAX_CONTINUE_TOKENS),
354
+ messages: withModelId(vision ? next : stripImages(next), model), plugins: [{ id: 'web' }] },
355
+ contextId,
356
+ );
357
+ const j = await r.json().catch(() => ({}));
358
+ const more = j?.choices?.[0]?.message?.content || '';
359
+ if (more && j?.choices?.[0]?.finish_reason === 'length') {
360
+ return more + await brainContinue(messages, sofar + more, contextId, modelOverride, round + 1);
361
+ }
362
+ return more;
363
+ }
364
+
309
365
  /** Same call, but streamed — invokes onDelta(text) as tokens arrive (for a
310
366
  * live-typing UI) and resolves with the full accumulated text at the end, so
311
367
  * callers that need to parse a directive out of the complete reply still can. */
312
- export async function brainStream(messages, onDelta, contextId, modelOverride, maxTokens) {
368
+ export async function brainStream(messages, onDelta, contextId, modelOverride, maxTokens, round = 0, topK = 0) {
313
369
  const vision = hasImages(messages);
314
370
  const model = vision ? VISION_MODEL : (modelOverride || MODEL);
315
371
  messages = vision ? messages : stripImages(messages);
316
372
  const budget = maxTokens || MAX_TOKENS;
317
373
  const r = await postChat(
318
374
  { model, max_tokens: budget, messages: withModelId(messages, model), plugins: [{ id: 'web' }], stream: true },
319
- contextId,
375
+ contextId, topK,
320
376
  );
321
377
  if (!r.ok || !r.body) {
322
378
  // fall back to the non-streaming path rather than fail outright
@@ -327,7 +383,7 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
327
383
  }
328
384
  const reader = r.body.getReader();
329
385
  const decoder = new TextDecoder();
330
- let buf = '', full = '', reasonedChars = 0;
386
+ let buf = '', full = '', reasonedChars = 0, finish = '';
331
387
  for (;;) {
332
388
  const { value, done } = await reader.read();
333
389
  if (done) break;
@@ -340,7 +396,12 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
340
396
  const payload = s.slice(5).trim();
341
397
  if (payload === '[DONE]') continue;
342
398
  try {
343
- const d = JSON.parse(payload)?.choices?.[0]?.delta;
399
+ const c = JSON.parse(payload)?.choices?.[0];
400
+ const d = c?.delta;
401
+ // The LAST chunk carries why generation stopped. "length" means the
402
+ // budget ran out mid-answer — the only way to tell a finished reply
403
+ // from a guillotined one.
404
+ if (c?.finish_reason) finish = c.finish_reason;
344
405
  if (d?.content) { full += d.content; onDelta(d.content); }
345
406
  // Reasoning models emit their chain of thought on a SEPARATE field and
346
407
  // only then start producing content. Count it — not to show it, but to
@@ -358,9 +419,267 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
358
419
  if (!full && reasonedChars > 0 && !maxTokens) {
359
420
  return brainStream(messages, onDelta, contextId, modelOverride, budget * 4);
360
421
  }
422
+
423
+ // CUT OFF MID-ANSWER. finish_reason "length" means the model had more to say
424
+ // and the budget ended the sentence for it — seen live as a reply that stops
425
+ // inside `for (`. Nothing above catches this, because `full` is non-empty:
426
+ // by every other measure the turn succeeded.
427
+ //
428
+ // CONTINUE rather than retry. Re-running the turn with a bigger budget makes
429
+ // the user pay twice for the half we already have (and on a reasoning model,
430
+ // pay for the whole chain of thought again). Handing the model back its own
431
+ // partial and asking for the rest costs only the rest.
432
+ //
433
+ // Bounded, because a model that ignores the nudge would otherwise continue
434
+ // forever on the user's wallet.
435
+ if (full && finish === 'length' && round < CONTINUE_ROUNDS) {
436
+ const more = await brainStream(
437
+ [...messages,
438
+ { role: 'assistant', content: full },
439
+ { role: 'user', content: CONTINUE_NUDGE }],
440
+ onDelta, contextId, modelOverride,
441
+ Math.min(budget * 2, MAX_CONTINUE_TOKENS), round + 1,
442
+ );
443
+ return full + (more || '');
444
+ }
361
445
  return full;
362
446
  }
363
447
 
448
+ // How many times a single answer may be resumed after hitting the cap. Three
449
+ // doublings off 4096 is ~57k tokens of answer, which is past any real reply and
450
+ // well short of a runaway.
451
+ const CONTINUE_ROUNDS = Number(process.env.OZ_CONTINUE_ROUNDS || 3);
452
+ const MAX_CONTINUE_TOKENS = Number(process.env.OZ_MAX_CONTINUE_TOKENS || 32768);
453
+ // Deliberately blunt about the seam: the partial usually ends mid-token, and a
454
+ // model that "helpfully" restarts the sentence produces a visible stutter in
455
+ // the middle of the user's code.
456
+ const CONTINUE_NUDGE = 'You were cut off — your previous message hit the output limit mid-way. '
457
+ + 'Continue from EXACTLY where it stopped. Do not repeat any of it, do not summarise it, '
458
+ + 'do not add a preamble or an apology, and do not re-open a code fence that is already open. '
459
+ + 'Resume mid-word if that is where it ended.';
460
+
461
+ // ---------------------------------------------------------------------------
462
+ // MODEL TIERS · cheap / medium / expensive
463
+ // ---------------------------------------------------------------------------
464
+ // Ranking the live catalog by price alone picks garbage at both ends: the most
465
+ // expensive served model is o1-pro at $1800/Mtok (a bad coding model that would
466
+ // drain the box wallet in a handful of turns), and the cheapest is a roleplay
467
+ // finetune. Price is a proxy for capability only within a band, never across
468
+ // the whole catalog. So each tier is a CURATED, ordered preference list, and
469
+ // the catalog is used to check what is actually served today — the zoo's model
470
+ // list changes under us, and a tier that resolves to a 404 is worse than no
471
+ // tier at all.
472
+ //
473
+ // Each list is ordered best-first (that is what a non-racing "auto" picks) but
474
+ // deliberately WIDE, because a race samples from it at random: a pool of three
475
+ // would race the same three models every time, which is neither a real hedge
476
+ // against a single provider having a bad minute nor a real sample of the tier.
477
+ // Prices in the comments are completion USD per Mtok as served, measured.
478
+ const TIERS = {
479
+ // ≲ $3/Mtok. Fast, good enough for glue work, cheap enough to race widely.
480
+ cheap: [
481
+ 'deepseek/deepseek-v4-flash', // 0.45
482
+ 'meta-llama/llama-4-scout', // 0.90
483
+ 'z-ai/glm-4.7-flash', // 1.20
484
+ 'bytedance-seed/seed-2.0-mini', // 1.20
485
+ 'meta-llama/llama-4-maverick', // 2.40
486
+ 'z-ai/glm-4.5-air', // 2.55
487
+ 'minimax/minimax-m2.5', // 2.70
488
+ 'z-ai/glm-4.6v', // 2.70
489
+ 'minimax/minimax-m2', // 3.06
490
+ 'inclusionai/ling-3.0-flash', // 0.19
491
+ ],
492
+ // ~$4.5–11/Mtok. The default band; deepseek-v4-pro is the app default.
493
+ medium: [
494
+ 'deepseek/deepseek-v4-pro-0813', // 5.94
495
+ 'z-ai/glm-4.7', // 5.25
496
+ 'google/gemini-3.7-flash', // 5.63
497
+ 'x-ai/grok-4.3', // 7.50
498
+ 'moonshotai/kimi-k2.7-code', // 10.50
499
+ 'z-ai/glm-5', // 5.76
500
+ 'moonshotai/kimi-k2.6', // 7.08
501
+ 'mistralai/mistral-large-2512', // 4.50
502
+ 'bytedance-seed/seed-2.0-code', // 9.00
503
+ 'qwen/qwen3.8-27b', // 9.60
504
+ ],
505
+ // ≥ $18/Mtok. Frontier. NOTE the ceiling: o1-pro ($1800) and the *-pro tiers
506
+ // ($240–540) are deliberately NOT here. A race of four across that band can
507
+ // cost dollars per turn on a box funded with a few cents.
508
+ expensive: [
509
+ 'anthropic/claude-opus-5', // 75
510
+ 'openai/gpt-5.5', // 90
511
+ 'anthropic/claude-sonnet-5', // 30
512
+ 'x-ai/grok-4.6', // 18
513
+ 'moonshotai/kimi-k3', // 45
514
+ 'anthropic/claude-opus-4.8', // 75
515
+ 'openai/gpt-5.4', // 45
516
+ 'qwen/qwen3.8-max', // 18
517
+ 'x-ai/grok-4.5', // 18
518
+ ],
519
+ };
520
+ export const TIER_NAMES = Object.keys(TIERS);
521
+
522
+ let catalogCache = { at: 0, ids: null };
523
+ async function servedIds() {
524
+ // 5 minutes: long enough that a race does not re-fetch per model, short
525
+ // enough that a model coming back after an outage is picked up the same
526
+ // session.
527
+ if (catalogCache.ids && Date.now() - catalogCache.at < 300_000) return catalogCache.ids;
528
+ try {
529
+ const r = await fetch(`${PROXY}/models`);
530
+ const j = await r.json();
531
+ const ids = new Set((j?.data || []).map((m) => m.id).filter(Boolean));
532
+ if (ids.size) catalogCache = { at: Date.now(), ids };
533
+ } catch { /* proxy down — fall through to whatever we had, or null */ }
534
+ return catalogCache.ids;
535
+ }
536
+
537
+ /**
538
+ * The models a tier resolves to right now, only ones actually served.
539
+ *
540
+ * `random` is what a race uses: pick n from the whole tier at random rather
541
+ * than always the top n. Two reasons it must be random and not top-n — a fixed
542
+ * trio is not a hedge (they can share an upstream having a bad minute, which is
543
+ * precisely the failure racing is meant to survive), and it silently reduces a
544
+ * ten-model tier to three models the user never chose.
545
+ *
546
+ * Falls back to the curated list unchecked if the catalog is unreachable — a
547
+ * stale-but-plausible id beats refusing to answer.
548
+ */
549
+ export async function tierModels(tier, n = 1, random = false) {
550
+ const want = TIERS[tier] || TIERS.medium;
551
+ const ids = await servedIds();
552
+ const live = ids ? want.filter((m) => ids.has(m)) : want;
553
+ const pool = live.length ? live : want;
554
+ const take = Math.max(1, Math.min(n, pool.length));
555
+ if (!random) return pool.slice(0, take);
556
+ // Fisher-Yates on a copy: sampling without replacement, because racing a
557
+ // model against itself buys nothing and still bills twice.
558
+ const a = pool.slice();
559
+ for (let i = a.length - 1; i > 0; i--) {
560
+ const j = Math.floor(Math.random() * (i + 1));
561
+ [a[i], a[j]] = [a[j], a[i]];
562
+ }
563
+ return a.slice(0, take);
564
+ }
565
+
566
+ /**
567
+ * Launch N models at once, judge the FIRST K that come back.
568
+ *
569
+ * "/race 2 3" — start three, and the moment two of them have returned a real
570
+ * answer, judge those two and ship the winner. The third is abandoned mid-flight.
571
+ *
572
+ * This is the useful shape, and it is neither of the obvious two:
573
+ * - first-past-the-post (K=1) optimises latency only, and on a hard question
574
+ * it rewards whichever model thought LEAST.
575
+ * - wait-for-all-then-judge (K=N) buys quality with the slowest entrant's
576
+ * latency, and one wedged provider stalls the whole turn.
577
+ * Taking the first K bounds the wait at the Kth-fastest while still giving the
578
+ * judge something to compare. The straggler is exactly the entrant you were
579
+ * least likely to want anyway.
580
+ *
581
+ * Reliability comes free with it: empty completions and provider 5xx are
582
+ * per-model and uncorrelated, which is why "the model returned nothing 4 times"
583
+ * was never fixable by a fourth try at the same model. An empty reply does NOT
584
+ * count toward K — otherwise the fastest model to FAIL would decide the race,
585
+ * the exact bug this exists to fix.
586
+ *
587
+ * Streaming is deliberately not forwarded while the race runs: nobody knows who
588
+ * is winning until they finish, and interleaving deltas from three models would
589
+ * render as noise. The winner's text is emitted whole.
590
+ *
591
+ * Every entrant is paid for, including the abandoned one — this trades money
592
+ * for latency and quality, which is why it is opt-in and capped.
593
+ */
594
+ export async function brainRace(messages, onDelta, contextId, models, need = 1, maxTokens) {
595
+ const list = (models || []).filter(Boolean).slice(0, RACE_MAX);
596
+ if (list.length < 2) return brainStream(messages, onDelta, contextId, list[0], maxTokens);
597
+ const want = Math.max(1, Math.min(Number(need) || 1, list.length));
598
+
599
+ const done = [];
600
+ let finished = 0;
601
+ let release;
602
+ const enough = new Promise((r) => { release = r; });
603
+
604
+ const attempts = list.map((m) => brainStream(messages, () => {}, contextId, m, maxTokens)
605
+ .then((text) => { if (text && text.trim()) done.push({ model: m, text }); })
606
+ .catch(() => { /* one entrant dying is not the race dying */ })
607
+ .finally(() => {
608
+ finished += 1;
609
+ // Either we have what we asked for, or everyone is done and no more is
610
+ // coming — without the second condition a race where two of three fail
611
+ // would hang forever waiting for a K that can never arrive.
612
+ if (done.length >= want || finished === list.length) release();
613
+ }));
614
+ // Losers keep running; swallow their rejections so one cannot take the
615
+ // process down after the winner has already been returned.
616
+ for (const p of attempts) p.catch(() => {});
617
+
618
+ await enough;
619
+ // Completion order, so this really is the first K back — not the first K
620
+ // launched.
621
+ const cands = done.slice(0, want);
622
+ if (!cands.length) return '';
623
+ // Nothing to compare — do not spend a judging call to rubber-stamp one answer.
624
+ if (cands.length === 1) { onDelta(cands[0].text); return cands[0].text; }
625
+
626
+ const winner = await judge(messages, cands);
627
+ onDelta(winner.text);
628
+ return winner.text;
629
+ }
630
+
631
+ /**
632
+ * Pick the best of several finished answers with a small model.
633
+ *
634
+ * BLIND, as A/B/C/D. A judge told "this one is Claude and this one is a 4B
635
+ * llama" is being handed the answer and will take it, which would turn the
636
+ * whole thing into an expensive way to re-pick the tier's first entry.
637
+ *
638
+ * Cheap on purpose: reading finished replies and comparing them against a
639
+ * question is a far easier task than answering it, and paying frontier prices
640
+ * to referee frontier models would roughly double the cost of the expensive
641
+ * tier for no measured gain.
642
+ */
643
+ async function judge(messages, cands) {
644
+ const letters = cands.map((_, i) => String.fromCharCode(65 + i));
645
+ // The question, not the transcript: the judge needs to know what was ASKED,
646
+ // and a full history would cost more to judge than the turn cost to answer.
647
+ const asked = [...messages].reverse().find((m) => m.role === 'user')?.content;
648
+ const question = typeof asked === 'string' ? asked : '(see candidates)';
649
+ const prompt = 'You are judging answers to one question. Pick the single best one.\n\n'
650
+ + 'QUESTION:\n' + String(question).slice(0, 4000) + '\n\n'
651
+ + cands.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + c.text.slice(0, 6000)).join('\n\n')
652
+ + '\n\nJudge on: correctness first, then completeness, then whether it actually did what was asked '
653
+ + '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
654
+ + 'Ignore length and confidence of tone.\n'
655
+ + 'Reply with ONE letter and nothing else: ' + letters.join(' or ') + '.';
656
+ try {
657
+ const verdict = await brainStream([{ role: 'user', content: prompt }], () => {}, undefined, JUDGE_MODEL, 8);
658
+ // First in-range letter anywhere in the reply. A judge that ignores "one
659
+ // letter and nothing else" and writes "The best is B." still counts, which
660
+ // is most of them.
661
+ const hit = String(verdict).toUpperCase().split('').find((ch) => {
662
+ const n = ch.charCodeAt(0) - 65;
663
+ return n >= 0 && n < cands.length;
664
+ });
665
+ if (hit) return cands[hit.charCodeAt(0) - 65];
666
+ } catch { /* fall through */ }
667
+ // A dead or delisted judge must not lose the answers. Falling back to the
668
+ // first finisher degrades this to "fastest wins" — worse than judged, far
669
+ // better than empty.
670
+ return cands[0];
671
+ }
672
+
673
+ const RACE_MAX = Number(process.env.OZ_RACE_MAX || 4);
674
+
675
+ // brainBest is GONE — brainRace(models, need) subsumes it. "wait for all N
676
+ // then judge" is exactly need === N, and keeping a second judged-race entry
677
+ // point meant two call sites that could disagree about what a race is.
678
+
679
+ // Cheapest thing that can reliably output one letter. Overridable because the
680
+ // catalog moves; if it is delisted the try/catch above falls back cleanly.
681
+ const JUDGE_MODEL = process.env.OZ_JUDGE_MODEL || 'deepseek/deepseek-v4-flash';
682
+
364
683
  const SYSTEM = `You are the brain of a Grok-Bot-style coding/ops agent. The polished chat UI
365
684
  the user sees is Grok Bot (Anysphere's app); its "sandbox" has been pointed at THIS box, and
366
685
  your reasoning is served by openzoo (pay-per-call access to ~435 models over x402 — no API key,
package/lib/x402.js CHANGED
@@ -216,7 +216,22 @@ export async function tokenBalance(connection, owner, mintStr) {
216
216
  export function receiptLine(accept, settle) {
217
217
  const x = accept.extra || {};
218
218
  const usd = x.billedUsd != null ? `$${Number(x.billedUsd).toFixed(6)}` : `${accept.maxAmountRequired} raw units`;
219
- const saves = x.savesVsDirect != null ? ` (${Number(x.savesVsDirect).toFixed(1)}× cheaper than direct)` : (x.markup != null ? ` (markup ${x.markup}×, short body)` : '');
219
+ // "1.0× cheaper than direct" is a sentence that means nothing, and a user
220
+ // read it as a bug — rightly. Since the gateway repriced to an OpenRouter
221
+ // CEILING, an uncompressed call bills exactly the direct rate, so the ratio
222
+ // is 1.0 by design rather than by accident. Say what actually happened:
223
+ // below 1.05× there is no saving to report, so report the price instead.
224
+ //
225
+ // The saving comes from leCore forwarding fewer tokens. A short body never
226
+ // reaches the spill threshold, so there is nothing to compress and nothing
227
+ // to save — which is worth saying out loud, because the fix on the caller's
228
+ // side is to BIND a corpus, not to change models.
229
+ const ratio = x.savesVsDirect != null ? Number(x.savesVsDirect) : null;
230
+ const saves = ratio != null
231
+ ? (ratio >= 1.05
232
+ ? ` (${ratio.toFixed(1)}× cheaper than direct)`
233
+ : ' (at direct price — nothing to compress; bind a corpus to save)')
234
+ : (x.markup != null ? ` (markup ${x.markup}×, short body)` : '');
220
235
  const tx = settle?.transaction || settle?.txHash || settle?.signature;
221
236
  const rail = railOf(accept);
222
237
  return `paid ${usd}${saves}${rail ? ` · rail ${rail}` : ''}${tx ? ` · tx ${tx}` : ''}`;
package/package.json CHANGED
@@ -1,6 +1,6 @@
1
1
  {
2
2
  "name": "openzoo",
3
- "version": "0.48.18",
3
+ "version": "0.48.22",
4
4
  "description": "Local x402-paying proxy + MCP server for openzoo.fun — point any OpenAI-compatible harness (Cursor, Claude Code, aider, SDKs) at localhost and it pays per call from a local burner wallet. Solana and Base rails live; Robinhood experimental.",
5
5
  "license": "MIT",
6
6
  "type": "module",