openzoo 0.48.18 → 0.48.22
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/lib/ctxalias.js +127 -0
- package/lib/grokui.mjs +1687 -59
- package/lib/namespace.js +34 -8
- package/lib/podagent.mjs +326 -7
- package/lib/x402.js +16 -1
- package/package.json +1 -1
package/lib/namespace.js
CHANGED
|
@@ -36,19 +36,45 @@ import { loadOrCreateWallet } from './wallet.js';
|
|
|
36
36
|
* BREAKING: corpora bound before this lived in the shared tenant and are not
|
|
37
37
|
* reachable from a signed request. Re-bind them.
|
|
38
38
|
*/
|
|
39
|
-
|
|
39
|
+
/**
|
|
40
|
+
* ONE NAMESPACE ACROSS EVERY STACC APP (2026-08-18).
|
|
41
|
+
*
|
|
42
|
+
* This used to be sha256("openzoo-ns:" + pubkey) — stable, per-user, and
|
|
43
|
+
* DIFFERENT from what every other stacc app sent. openzoo brain sent
|
|
44
|
+
* HMAC(OPENZOO_TENANT_SECRET, pubkey); the open-webui wallet sent a literal
|
|
45
|
+
* app name. Since the gateway keys a tenant on sha256(chain:signer:namespace),
|
|
46
|
+
* one wallet therefore landed in THREE tenants: three separate leCore
|
|
47
|
+
* memories and three separate credit balances, for the same person. Bind a
|
|
48
|
+
* corpus in the CLI and it was invisible from the browser.
|
|
49
|
+
*
|
|
50
|
+
* The constant is safe because the reason for the per-user hash is gone. It
|
|
51
|
+
* existed to stop someone addressing your tenant by guessing your namespace —
|
|
52
|
+
* but the gateway now folds the VERIFIED SIGNER into the tenant hash (see
|
|
53
|
+
* nsauth.ts / tenantFor), so the namespace string is no longer the
|
|
54
|
+
* access-control boundary. Signing is what proves ownership; this string only
|
|
55
|
+
* chooses WHICH of your namespaces you mean.
|
|
56
|
+
*
|
|
57
|
+
* A constant is in fact MORE private than what it replaces. A per-user hash is
|
|
58
|
+
* stable and unique, i.e. a tracking identifier that correlates one user's
|
|
59
|
+
* requests across time. A value every caller sends identically discloses
|
|
60
|
+
* nothing at all.
|
|
61
|
+
*
|
|
62
|
+
* BREAKING: corpora bound under the old per-wallet namespace live in a
|
|
63
|
+
* different tenant and will not be found. The gateway's tenantsToTry fallback
|
|
64
|
+
* reaches pre-existing context ids by id, but anything relying on the old
|
|
65
|
+
* namespace should be re-bound.
|
|
66
|
+
*/
|
|
67
|
+
export const STACC_NAMESPACE = 'stacc';
|
|
40
68
|
|
|
41
69
|
export function namespaceHeaderValue() {
|
|
42
|
-
if (cached) return cached;
|
|
43
70
|
try {
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
71
|
+
// Still require a wallet: the namespace is meaningless without a signer to
|
|
72
|
+
// prove it, and sending one unsigned drops you into the SHARED tenant.
|
|
73
|
+
loadOrCreateWallet();
|
|
74
|
+
return STACC_NAMESPACE;
|
|
48
75
|
} catch {
|
|
49
|
-
|
|
76
|
+
return ''; // no wallet (read-only use): fall back to the shared tenant
|
|
50
77
|
}
|
|
51
|
-
return cached;
|
|
52
78
|
}
|
|
53
79
|
|
|
54
80
|
/**
|
package/lib/podagent.mjs
CHANGED
|
@@ -249,13 +249,40 @@ async function httpErrorNote(status) {
|
|
|
249
249
|
// same rail, second call pays fine). Surfacing that as a chat message makes
|
|
250
250
|
// the user do the retry by hand — so do it here instead.
|
|
251
251
|
const PAYMENT_RETRIES = 3;
|
|
252
|
-
|
|
252
|
+
/**
|
|
253
|
+
* ADAPTIVE top_k. We have learned this one the expensive way already.
|
|
254
|
+
*
|
|
255
|
+
* On leCore the miss was never the ranker — BM25 ranked correctly. It was
|
|
256
|
+
* top_k=16 against a 7,000-chunk corpus: we only ever ASKED for sixteen. Same
|
|
257
|
+
* shape here, and worse: grokui never set the header at all, so every call fell
|
|
258
|
+
* to the gateway default of EIGHT, while a whole project's bots write into one
|
|
259
|
+
* shared context. Six bots working for an hour and the model sees eight chunks
|
|
260
|
+
* of it.
|
|
261
|
+
*
|
|
262
|
+
* So scale with the corpus instead of picking a number. sqrt keeps it sane at
|
|
263
|
+
* both ends — 100 chunks -> 20, 1k -> 63, 7k -> 167, and it saturates at the
|
|
264
|
+
* gateway's 256 ceiling rather than growing without bound. Floor of 16 so a
|
|
265
|
+
* brand-new thread is never worse off than the old default.
|
|
266
|
+
*
|
|
267
|
+
* Cost is real and proportional (measured on leCore: top_k 16 = $0.0070,
|
|
268
|
+
* top_k 128 = $0.0489 on the same question) — which is the point. The extra
|
|
269
|
+
* spend IS the extra corpus actually being read.
|
|
270
|
+
*/
|
|
271
|
+
export function adaptiveTopK(boundItems) {
|
|
272
|
+
const n = Math.max(0, Number(boundItems) || 0);
|
|
273
|
+
return Math.max(16, Math.min(256, Math.ceil(Math.sqrt(n) * 2)));
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
async function postChat(body, contextId, topK) {
|
|
253
277
|
let r;
|
|
254
278
|
for (let attempt = 0; attempt <= PAYMENT_RETRIES; attempt++) {
|
|
255
279
|
r = await fetch(`${PROXY}/chat/completions`, {
|
|
256
280
|
method: 'POST',
|
|
257
281
|
headers: {
|
|
258
282
|
'content-type': 'application/json', authorization: 'Bearer sk-openzoo',
|
|
283
|
+
// Only sent when we actually know the corpus size; without it the
|
|
284
|
+
// gateway keeps its own default rather than getting a made-up number.
|
|
285
|
+
...(topK ? { 'x-hrr-top-k': String(topK) } : {}),
|
|
259
286
|
// real leCore memory for this thread, bound via POST /v1/hrr/bind — NOT
|
|
260
287
|
// a fabricated mechanism. Retrieval runs automatically once this header
|
|
261
288
|
// is set; nothing more for the model to invent or explain.
|
|
@@ -284,7 +311,7 @@ function withModelId(messages, model) {
|
|
|
284
311
|
: m));
|
|
285
312
|
}
|
|
286
313
|
|
|
287
|
-
export async function brain(messages, contextId, modelOverride) {
|
|
314
|
+
export async function brain(messages, contextId, modelOverride, topK) {
|
|
288
315
|
// explicit plugins, not relying on the gateway's "inject when caller said
|
|
289
316
|
// nothing" default — an explicit array is always respected as-is, so every
|
|
290
317
|
// bot on every model actually has web search. max_tokens 900 was cutting
|
|
@@ -299,24 +326,53 @@ export async function brain(messages, contextId, modelOverride) {
|
|
|
299
326
|
messages = vision ? messages : stripImages(messages);
|
|
300
327
|
const r = await postChat(
|
|
301
328
|
{ model, max_tokens: 4096, messages: withModelId(messages, model), plugins: [{ id: 'web' }] },
|
|
302
|
-
contextId,
|
|
329
|
+
contextId, topK,
|
|
303
330
|
);
|
|
304
331
|
const j = await r.json().catch(() => ({}));
|
|
305
332
|
const content = j?.choices?.[0]?.message?.content;
|
|
333
|
+
// Same truncation catch as the streaming path (see brainStream): a reply that
|
|
334
|
+
// stops because the budget ran out is not a finished reply, and this path is
|
|
335
|
+
// what non-streaming callers — including every SPAWNed subagent — go through.
|
|
336
|
+
if (content && j?.choices?.[0]?.finish_reason === 'length') {
|
|
337
|
+
const rest = await brainContinue(messages, content, contextId, modelOverride, 0);
|
|
338
|
+
return content + rest;
|
|
339
|
+
}
|
|
306
340
|
return content || (r.ok ? '' : await httpErrorNote(r.status));
|
|
307
341
|
}
|
|
308
342
|
|
|
343
|
+
/** Resume a reply that hit the output cap, non-streaming. Bounded by
|
|
344
|
+
* CONTINUE_ROUNDS for the same runaway reason brainStream is. */
|
|
345
|
+
async function brainContinue(messages, sofar, contextId, modelOverride, round) {
|
|
346
|
+
if (round >= CONTINUE_ROUNDS) return '';
|
|
347
|
+
const vision = hasImages(messages);
|
|
348
|
+
const model = vision ? VISION_MODEL : (modelOverride || MODEL);
|
|
349
|
+
const next = [...messages,
|
|
350
|
+
{ role: 'assistant', content: sofar },
|
|
351
|
+
{ role: 'user', content: CONTINUE_NUDGE }];
|
|
352
|
+
const r = await postChat(
|
|
353
|
+
{ model, max_tokens: Math.min(4096 * (2 ** (round + 1)), MAX_CONTINUE_TOKENS),
|
|
354
|
+
messages: withModelId(vision ? next : stripImages(next), model), plugins: [{ id: 'web' }] },
|
|
355
|
+
contextId,
|
|
356
|
+
);
|
|
357
|
+
const j = await r.json().catch(() => ({}));
|
|
358
|
+
const more = j?.choices?.[0]?.message?.content || '';
|
|
359
|
+
if (more && j?.choices?.[0]?.finish_reason === 'length') {
|
|
360
|
+
return more + await brainContinue(messages, sofar + more, contextId, modelOverride, round + 1);
|
|
361
|
+
}
|
|
362
|
+
return more;
|
|
363
|
+
}
|
|
364
|
+
|
|
309
365
|
/** Same call, but streamed — invokes onDelta(text) as tokens arrive (for a
|
|
310
366
|
* live-typing UI) and resolves with the full accumulated text at the end, so
|
|
311
367
|
* callers that need to parse a directive out of the complete reply still can. */
|
|
312
|
-
export async function brainStream(messages, onDelta, contextId, modelOverride, maxTokens) {
|
|
368
|
+
export async function brainStream(messages, onDelta, contextId, modelOverride, maxTokens, round = 0, topK = 0) {
|
|
313
369
|
const vision = hasImages(messages);
|
|
314
370
|
const model = vision ? VISION_MODEL : (modelOverride || MODEL);
|
|
315
371
|
messages = vision ? messages : stripImages(messages);
|
|
316
372
|
const budget = maxTokens || MAX_TOKENS;
|
|
317
373
|
const r = await postChat(
|
|
318
374
|
{ model, max_tokens: budget, messages: withModelId(messages, model), plugins: [{ id: 'web' }], stream: true },
|
|
319
|
-
contextId,
|
|
375
|
+
contextId, topK,
|
|
320
376
|
);
|
|
321
377
|
if (!r.ok || !r.body) {
|
|
322
378
|
// fall back to the non-streaming path rather than fail outright
|
|
@@ -327,7 +383,7 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
327
383
|
}
|
|
328
384
|
const reader = r.body.getReader();
|
|
329
385
|
const decoder = new TextDecoder();
|
|
330
|
-
let buf = '', full = '', reasonedChars = 0;
|
|
386
|
+
let buf = '', full = '', reasonedChars = 0, finish = '';
|
|
331
387
|
for (;;) {
|
|
332
388
|
const { value, done } = await reader.read();
|
|
333
389
|
if (done) break;
|
|
@@ -340,7 +396,12 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
340
396
|
const payload = s.slice(5).trim();
|
|
341
397
|
if (payload === '[DONE]') continue;
|
|
342
398
|
try {
|
|
343
|
-
const
|
|
399
|
+
const c = JSON.parse(payload)?.choices?.[0];
|
|
400
|
+
const d = c?.delta;
|
|
401
|
+
// The LAST chunk carries why generation stopped. "length" means the
|
|
402
|
+
// budget ran out mid-answer — the only way to tell a finished reply
|
|
403
|
+
// from a guillotined one.
|
|
404
|
+
if (c?.finish_reason) finish = c.finish_reason;
|
|
344
405
|
if (d?.content) { full += d.content; onDelta(d.content); }
|
|
345
406
|
// Reasoning models emit their chain of thought on a SEPARATE field and
|
|
346
407
|
// only then start producing content. Count it — not to show it, but to
|
|
@@ -358,9 +419,267 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
358
419
|
if (!full && reasonedChars > 0 && !maxTokens) {
|
|
359
420
|
return brainStream(messages, onDelta, contextId, modelOverride, budget * 4);
|
|
360
421
|
}
|
|
422
|
+
|
|
423
|
+
// CUT OFF MID-ANSWER. finish_reason "length" means the model had more to say
|
|
424
|
+
// and the budget ended the sentence for it — seen live as a reply that stops
|
|
425
|
+
// inside `for (`. Nothing above catches this, because `full` is non-empty:
|
|
426
|
+
// by every other measure the turn succeeded.
|
|
427
|
+
//
|
|
428
|
+
// CONTINUE rather than retry. Re-running the turn with a bigger budget makes
|
|
429
|
+
// the user pay twice for the half we already have (and on a reasoning model,
|
|
430
|
+
// pay for the whole chain of thought again). Handing the model back its own
|
|
431
|
+
// partial and asking for the rest costs only the rest.
|
|
432
|
+
//
|
|
433
|
+
// Bounded, because a model that ignores the nudge would otherwise continue
|
|
434
|
+
// forever on the user's wallet.
|
|
435
|
+
if (full && finish === 'length' && round < CONTINUE_ROUNDS) {
|
|
436
|
+
const more = await brainStream(
|
|
437
|
+
[...messages,
|
|
438
|
+
{ role: 'assistant', content: full },
|
|
439
|
+
{ role: 'user', content: CONTINUE_NUDGE }],
|
|
440
|
+
onDelta, contextId, modelOverride,
|
|
441
|
+
Math.min(budget * 2, MAX_CONTINUE_TOKENS), round + 1,
|
|
442
|
+
);
|
|
443
|
+
return full + (more || '');
|
|
444
|
+
}
|
|
361
445
|
return full;
|
|
362
446
|
}
|
|
363
447
|
|
|
448
|
+
// How many times a single answer may be resumed after hitting the cap. Three
|
|
449
|
+
// doublings off 4096 is ~57k tokens of answer, which is past any real reply and
|
|
450
|
+
// well short of a runaway.
|
|
451
|
+
const CONTINUE_ROUNDS = Number(process.env.OZ_CONTINUE_ROUNDS || 3);
|
|
452
|
+
const MAX_CONTINUE_TOKENS = Number(process.env.OZ_MAX_CONTINUE_TOKENS || 32768);
|
|
453
|
+
// Deliberately blunt about the seam: the partial usually ends mid-token, and a
|
|
454
|
+
// model that "helpfully" restarts the sentence produces a visible stutter in
|
|
455
|
+
// the middle of the user's code.
|
|
456
|
+
const CONTINUE_NUDGE = 'You were cut off — your previous message hit the output limit mid-way. '
|
|
457
|
+
+ 'Continue from EXACTLY where it stopped. Do not repeat any of it, do not summarise it, '
|
|
458
|
+
+ 'do not add a preamble or an apology, and do not re-open a code fence that is already open. '
|
|
459
|
+
+ 'Resume mid-word if that is where it ended.';
|
|
460
|
+
|
|
461
|
+
// ---------------------------------------------------------------------------
|
|
462
|
+
// MODEL TIERS · cheap / medium / expensive
|
|
463
|
+
// ---------------------------------------------------------------------------
|
|
464
|
+
// Ranking the live catalog by price alone picks garbage at both ends: the most
|
|
465
|
+
// expensive served model is o1-pro at $1800/Mtok (a bad coding model that would
|
|
466
|
+
// drain the box wallet in a handful of turns), and the cheapest is a roleplay
|
|
467
|
+
// finetune. Price is a proxy for capability only within a band, never across
|
|
468
|
+
// the whole catalog. So each tier is a CURATED, ordered preference list, and
|
|
469
|
+
// the catalog is used to check what is actually served today — the zoo's model
|
|
470
|
+
// list changes under us, and a tier that resolves to a 404 is worse than no
|
|
471
|
+
// tier at all.
|
|
472
|
+
//
|
|
473
|
+
// Each list is ordered best-first (that is what a non-racing "auto" picks) but
|
|
474
|
+
// deliberately WIDE, because a race samples from it at random: a pool of three
|
|
475
|
+
// would race the same three models every time, which is neither a real hedge
|
|
476
|
+
// against a single provider having a bad minute nor a real sample of the tier.
|
|
477
|
+
// Prices in the comments are completion USD per Mtok as served, measured.
|
|
478
|
+
const TIERS = {
|
|
479
|
+
// ≲ $3/Mtok. Fast, good enough for glue work, cheap enough to race widely.
|
|
480
|
+
cheap: [
|
|
481
|
+
'deepseek/deepseek-v4-flash', // 0.45
|
|
482
|
+
'meta-llama/llama-4-scout', // 0.90
|
|
483
|
+
'z-ai/glm-4.7-flash', // 1.20
|
|
484
|
+
'bytedance-seed/seed-2.0-mini', // 1.20
|
|
485
|
+
'meta-llama/llama-4-maverick', // 2.40
|
|
486
|
+
'z-ai/glm-4.5-air', // 2.55
|
|
487
|
+
'minimax/minimax-m2.5', // 2.70
|
|
488
|
+
'z-ai/glm-4.6v', // 2.70
|
|
489
|
+
'minimax/minimax-m2', // 3.06
|
|
490
|
+
'inclusionai/ling-3.0-flash', // 0.19
|
|
491
|
+
],
|
|
492
|
+
// ~$4.5–11/Mtok. The default band; deepseek-v4-pro is the app default.
|
|
493
|
+
medium: [
|
|
494
|
+
'deepseek/deepseek-v4-pro-0813', // 5.94
|
|
495
|
+
'z-ai/glm-4.7', // 5.25
|
|
496
|
+
'google/gemini-3.7-flash', // 5.63
|
|
497
|
+
'x-ai/grok-4.3', // 7.50
|
|
498
|
+
'moonshotai/kimi-k2.7-code', // 10.50
|
|
499
|
+
'z-ai/glm-5', // 5.76
|
|
500
|
+
'moonshotai/kimi-k2.6', // 7.08
|
|
501
|
+
'mistralai/mistral-large-2512', // 4.50
|
|
502
|
+
'bytedance-seed/seed-2.0-code', // 9.00
|
|
503
|
+
'qwen/qwen3.8-27b', // 9.60
|
|
504
|
+
],
|
|
505
|
+
// ≥ $18/Mtok. Frontier. NOTE the ceiling: o1-pro ($1800) and the *-pro tiers
|
|
506
|
+
// ($240–540) are deliberately NOT here. A race of four across that band can
|
|
507
|
+
// cost dollars per turn on a box funded with a few cents.
|
|
508
|
+
expensive: [
|
|
509
|
+
'anthropic/claude-opus-5', // 75
|
|
510
|
+
'openai/gpt-5.5', // 90
|
|
511
|
+
'anthropic/claude-sonnet-5', // 30
|
|
512
|
+
'x-ai/grok-4.6', // 18
|
|
513
|
+
'moonshotai/kimi-k3', // 45
|
|
514
|
+
'anthropic/claude-opus-4.8', // 75
|
|
515
|
+
'openai/gpt-5.4', // 45
|
|
516
|
+
'qwen/qwen3.8-max', // 18
|
|
517
|
+
'x-ai/grok-4.5', // 18
|
|
518
|
+
],
|
|
519
|
+
};
|
|
520
|
+
export const TIER_NAMES = Object.keys(TIERS);
|
|
521
|
+
|
|
522
|
+
let catalogCache = { at: 0, ids: null };
|
|
523
|
+
async function servedIds() {
|
|
524
|
+
// 5 minutes: long enough that a race does not re-fetch per model, short
|
|
525
|
+
// enough that a model coming back after an outage is picked up the same
|
|
526
|
+
// session.
|
|
527
|
+
if (catalogCache.ids && Date.now() - catalogCache.at < 300_000) return catalogCache.ids;
|
|
528
|
+
try {
|
|
529
|
+
const r = await fetch(`${PROXY}/models`);
|
|
530
|
+
const j = await r.json();
|
|
531
|
+
const ids = new Set((j?.data || []).map((m) => m.id).filter(Boolean));
|
|
532
|
+
if (ids.size) catalogCache = { at: Date.now(), ids };
|
|
533
|
+
} catch { /* proxy down — fall through to whatever we had, or null */ }
|
|
534
|
+
return catalogCache.ids;
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
/**
|
|
538
|
+
* The models a tier resolves to right now, only ones actually served.
|
|
539
|
+
*
|
|
540
|
+
* `random` is what a race uses: pick n from the whole tier at random rather
|
|
541
|
+
* than always the top n. Two reasons it must be random and not top-n — a fixed
|
|
542
|
+
* trio is not a hedge (they can share an upstream having a bad minute, which is
|
|
543
|
+
* precisely the failure racing is meant to survive), and it silently reduces a
|
|
544
|
+
* ten-model tier to three models the user never chose.
|
|
545
|
+
*
|
|
546
|
+
* Falls back to the curated list unchecked if the catalog is unreachable — a
|
|
547
|
+
* stale-but-plausible id beats refusing to answer.
|
|
548
|
+
*/
|
|
549
|
+
export async function tierModels(tier, n = 1, random = false) {
|
|
550
|
+
const want = TIERS[tier] || TIERS.medium;
|
|
551
|
+
const ids = await servedIds();
|
|
552
|
+
const live = ids ? want.filter((m) => ids.has(m)) : want;
|
|
553
|
+
const pool = live.length ? live : want;
|
|
554
|
+
const take = Math.max(1, Math.min(n, pool.length));
|
|
555
|
+
if (!random) return pool.slice(0, take);
|
|
556
|
+
// Fisher-Yates on a copy: sampling without replacement, because racing a
|
|
557
|
+
// model against itself buys nothing and still bills twice.
|
|
558
|
+
const a = pool.slice();
|
|
559
|
+
for (let i = a.length - 1; i > 0; i--) {
|
|
560
|
+
const j = Math.floor(Math.random() * (i + 1));
|
|
561
|
+
[a[i], a[j]] = [a[j], a[i]];
|
|
562
|
+
}
|
|
563
|
+
return a.slice(0, take);
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
/**
|
|
567
|
+
* Launch N models at once, judge the FIRST K that come back.
|
|
568
|
+
*
|
|
569
|
+
* "/race 2 3" — start three, and the moment two of them have returned a real
|
|
570
|
+
* answer, judge those two and ship the winner. The third is abandoned mid-flight.
|
|
571
|
+
*
|
|
572
|
+
* This is the useful shape, and it is neither of the obvious two:
|
|
573
|
+
* - first-past-the-post (K=1) optimises latency only, and on a hard question
|
|
574
|
+
* it rewards whichever model thought LEAST.
|
|
575
|
+
* - wait-for-all-then-judge (K=N) buys quality with the slowest entrant's
|
|
576
|
+
* latency, and one wedged provider stalls the whole turn.
|
|
577
|
+
* Taking the first K bounds the wait at the Kth-fastest while still giving the
|
|
578
|
+
* judge something to compare. The straggler is exactly the entrant you were
|
|
579
|
+
* least likely to want anyway.
|
|
580
|
+
*
|
|
581
|
+
* Reliability comes free with it: empty completions and provider 5xx are
|
|
582
|
+
* per-model and uncorrelated, which is why "the model returned nothing 4 times"
|
|
583
|
+
* was never fixable by a fourth try at the same model. An empty reply does NOT
|
|
584
|
+
* count toward K — otherwise the fastest model to FAIL would decide the race,
|
|
585
|
+
* the exact bug this exists to fix.
|
|
586
|
+
*
|
|
587
|
+
* Streaming is deliberately not forwarded while the race runs: nobody knows who
|
|
588
|
+
* is winning until they finish, and interleaving deltas from three models would
|
|
589
|
+
* render as noise. The winner's text is emitted whole.
|
|
590
|
+
*
|
|
591
|
+
* Every entrant is paid for, including the abandoned one — this trades money
|
|
592
|
+
* for latency and quality, which is why it is opt-in and capped.
|
|
593
|
+
*/
|
|
594
|
+
export async function brainRace(messages, onDelta, contextId, models, need = 1, maxTokens) {
|
|
595
|
+
const list = (models || []).filter(Boolean).slice(0, RACE_MAX);
|
|
596
|
+
if (list.length < 2) return brainStream(messages, onDelta, contextId, list[0], maxTokens);
|
|
597
|
+
const want = Math.max(1, Math.min(Number(need) || 1, list.length));
|
|
598
|
+
|
|
599
|
+
const done = [];
|
|
600
|
+
let finished = 0;
|
|
601
|
+
let release;
|
|
602
|
+
const enough = new Promise((r) => { release = r; });
|
|
603
|
+
|
|
604
|
+
const attempts = list.map((m) => brainStream(messages, () => {}, contextId, m, maxTokens)
|
|
605
|
+
.then((text) => { if (text && text.trim()) done.push({ model: m, text }); })
|
|
606
|
+
.catch(() => { /* one entrant dying is not the race dying */ })
|
|
607
|
+
.finally(() => {
|
|
608
|
+
finished += 1;
|
|
609
|
+
// Either we have what we asked for, or everyone is done and no more is
|
|
610
|
+
// coming — without the second condition a race where two of three fail
|
|
611
|
+
// would hang forever waiting for a K that can never arrive.
|
|
612
|
+
if (done.length >= want || finished === list.length) release();
|
|
613
|
+
}));
|
|
614
|
+
// Losers keep running; swallow their rejections so one cannot take the
|
|
615
|
+
// process down after the winner has already been returned.
|
|
616
|
+
for (const p of attempts) p.catch(() => {});
|
|
617
|
+
|
|
618
|
+
await enough;
|
|
619
|
+
// Completion order, so this really is the first K back — not the first K
|
|
620
|
+
// launched.
|
|
621
|
+
const cands = done.slice(0, want);
|
|
622
|
+
if (!cands.length) return '';
|
|
623
|
+
// Nothing to compare — do not spend a judging call to rubber-stamp one answer.
|
|
624
|
+
if (cands.length === 1) { onDelta(cands[0].text); return cands[0].text; }
|
|
625
|
+
|
|
626
|
+
const winner = await judge(messages, cands);
|
|
627
|
+
onDelta(winner.text);
|
|
628
|
+
return winner.text;
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
/**
|
|
632
|
+
* Pick the best of several finished answers with a small model.
|
|
633
|
+
*
|
|
634
|
+
* BLIND, as A/B/C/D. A judge told "this one is Claude and this one is a 4B
|
|
635
|
+
* llama" is being handed the answer and will take it, which would turn the
|
|
636
|
+
* whole thing into an expensive way to re-pick the tier's first entry.
|
|
637
|
+
*
|
|
638
|
+
* Cheap on purpose: reading finished replies and comparing them against a
|
|
639
|
+
* question is a far easier task than answering it, and paying frontier prices
|
|
640
|
+
* to referee frontier models would roughly double the cost of the expensive
|
|
641
|
+
* tier for no measured gain.
|
|
642
|
+
*/
|
|
643
|
+
async function judge(messages, cands) {
|
|
644
|
+
const letters = cands.map((_, i) => String.fromCharCode(65 + i));
|
|
645
|
+
// The question, not the transcript: the judge needs to know what was ASKED,
|
|
646
|
+
// and a full history would cost more to judge than the turn cost to answer.
|
|
647
|
+
const asked = [...messages].reverse().find((m) => m.role === 'user')?.content;
|
|
648
|
+
const question = typeof asked === 'string' ? asked : '(see candidates)';
|
|
649
|
+
const prompt = 'You are judging answers to one question. Pick the single best one.\n\n'
|
|
650
|
+
+ 'QUESTION:\n' + String(question).slice(0, 4000) + '\n\n'
|
|
651
|
+
+ cands.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + c.text.slice(0, 6000)).join('\n\n')
|
|
652
|
+
+ '\n\nJudge on: correctness first, then completeness, then whether it actually did what was asked '
|
|
653
|
+
+ '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
|
|
654
|
+
+ 'Ignore length and confidence of tone.\n'
|
|
655
|
+
+ 'Reply with ONE letter and nothing else: ' + letters.join(' or ') + '.';
|
|
656
|
+
try {
|
|
657
|
+
const verdict = await brainStream([{ role: 'user', content: prompt }], () => {}, undefined, JUDGE_MODEL, 8);
|
|
658
|
+
// First in-range letter anywhere in the reply. A judge that ignores "one
|
|
659
|
+
// letter and nothing else" and writes "The best is B." still counts, which
|
|
660
|
+
// is most of them.
|
|
661
|
+
const hit = String(verdict).toUpperCase().split('').find((ch) => {
|
|
662
|
+
const n = ch.charCodeAt(0) - 65;
|
|
663
|
+
return n >= 0 && n < cands.length;
|
|
664
|
+
});
|
|
665
|
+
if (hit) return cands[hit.charCodeAt(0) - 65];
|
|
666
|
+
} catch { /* fall through */ }
|
|
667
|
+
// A dead or delisted judge must not lose the answers. Falling back to the
|
|
668
|
+
// first finisher degrades this to "fastest wins" — worse than judged, far
|
|
669
|
+
// better than empty.
|
|
670
|
+
return cands[0];
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
const RACE_MAX = Number(process.env.OZ_RACE_MAX || 4);
|
|
674
|
+
|
|
675
|
+
// brainBest is GONE — brainRace(models, need) subsumes it. "wait for all N
|
|
676
|
+
// then judge" is exactly need === N, and keeping a second judged-race entry
|
|
677
|
+
// point meant two call sites that could disagree about what a race is.
|
|
678
|
+
|
|
679
|
+
// Cheapest thing that can reliably output one letter. Overridable because the
|
|
680
|
+
// catalog moves; if it is delisted the try/catch above falls back cleanly.
|
|
681
|
+
const JUDGE_MODEL = process.env.OZ_JUDGE_MODEL || 'deepseek/deepseek-v4-flash';
|
|
682
|
+
|
|
364
683
|
const SYSTEM = `You are the brain of a Grok-Bot-style coding/ops agent. The polished chat UI
|
|
365
684
|
the user sees is Grok Bot (Anysphere's app); its "sandbox" has been pointed at THIS box, and
|
|
366
685
|
your reasoning is served by openzoo (pay-per-call access to ~435 models over x402 — no API key,
|
package/lib/x402.js
CHANGED
|
@@ -216,7 +216,22 @@ export async function tokenBalance(connection, owner, mintStr) {
|
|
|
216
216
|
export function receiptLine(accept, settle) {
|
|
217
217
|
const x = accept.extra || {};
|
|
218
218
|
const usd = x.billedUsd != null ? `$${Number(x.billedUsd).toFixed(6)}` : `${accept.maxAmountRequired} raw units`;
|
|
219
|
-
|
|
219
|
+
// "1.0× cheaper than direct" is a sentence that means nothing, and a user
|
|
220
|
+
// read it as a bug — rightly. Since the gateway repriced to an OpenRouter
|
|
221
|
+
// CEILING, an uncompressed call bills exactly the direct rate, so the ratio
|
|
222
|
+
// is 1.0 by design rather than by accident. Say what actually happened:
|
|
223
|
+
// below 1.05× there is no saving to report, so report the price instead.
|
|
224
|
+
//
|
|
225
|
+
// The saving comes from leCore forwarding fewer tokens. A short body never
|
|
226
|
+
// reaches the spill threshold, so there is nothing to compress and nothing
|
|
227
|
+
// to save — which is worth saying out loud, because the fix on the caller's
|
|
228
|
+
// side is to BIND a corpus, not to change models.
|
|
229
|
+
const ratio = x.savesVsDirect != null ? Number(x.savesVsDirect) : null;
|
|
230
|
+
const saves = ratio != null
|
|
231
|
+
? (ratio >= 1.05
|
|
232
|
+
? ` (${ratio.toFixed(1)}× cheaper than direct)`
|
|
233
|
+
: ' (at direct price — nothing to compress; bind a corpus to save)')
|
|
234
|
+
: (x.markup != null ? ` (markup ${x.markup}×, short body)` : '');
|
|
220
235
|
const tx = settle?.transaction || settle?.txHash || settle?.signature;
|
|
221
236
|
const rail = railOf(accept);
|
|
222
237
|
return `paid ${usd}${saves}${rail ? ` · rail ${rail}` : ''}${tx ? ` · tx ${tx}` : ''}`;
|
package/package.json
CHANGED
|
@@ -1,6 +1,6 @@
|
|
|
1
1
|
{
|
|
2
2
|
"name": "openzoo",
|
|
3
|
-
"version": "0.48.
|
|
3
|
+
"version": "0.48.22",
|
|
4
4
|
"description": "Local x402-paying proxy + MCP server for openzoo.fun — point any OpenAI-compatible harness (Cursor, Claude Code, aider, SDKs) at localhost and it pays per call from a local burner wallet. Solana and Base rails live; Robinhood experimental.",
|
|
5
5
|
"license": "MIT",
|
|
6
6
|
"type": "module",
|