openzoo 0.48.18 → 0.48.23
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/bin/openzoo.js +54 -0
- package/lib/ctxalias.js +127 -0
- package/lib/grokui.mjs +1687 -59
- package/lib/launch.js +38 -5
- package/lib/namespace.js +34 -8
- package/lib/podagent.mjs +326 -7
- package/lib/proxy.js +106 -5
- package/lib/proxy.js.bak +980 -0
- package/lib/responses.js +425 -0
- package/lib/tunnel.js +8 -1
- package/lib/x402.js +16 -1
- package/package.json +1 -1
package/lib/launch.js
CHANGED
|
@@ -49,21 +49,54 @@ function resolveClaudeCli() {
|
|
|
49
49
|
* Both get ANTHROPIC_BASE_URL so inference pays x402.
|
|
50
50
|
*/
|
|
51
51
|
export async function launchClaude(argv) {
|
|
52
|
-
const
|
|
52
|
+
// let, not const: startProxy can heal onto a different port and every URL
|
|
53
|
+
// below must follow the port we actually bound.
|
|
54
|
+
let base = `http://localhost:${config.port}/v1`;
|
|
53
55
|
// AUTO-START THE PROXY. One command should just work — if nothing is listening,
|
|
54
56
|
// boot the proxy in THIS process (it stays alive because claude runs in the
|
|
55
57
|
// foreground below), rather than making the user run `npx openzoo` first.
|
|
56
58
|
let up = false;
|
|
57
59
|
try { up = (await fetch(`${base}/models`, { signal: AbortSignal.timeout(3000) })).ok; } catch { up = false; }
|
|
58
60
|
if (!up) {
|
|
59
|
-
|
|
60
|
-
|
|
61
|
-
|
|
61
|
+
// NEVER GO SILENT DURING STARTUP. silent:true routes the proxy's own lines
|
|
62
|
+
// to ~/.openzoo/proxy.log so payment receipts cannot corrupt Claude Code's
|
|
63
|
+
// stdio — correct DURING the session, wrong BEFORE it, because it made a
|
|
64
|
+
// slow step and a dead step look identical. Reported from the wild as
|
|
65
|
+
// "always gets stuck at 'starting the proxy in the background...'": there
|
|
66
|
+
// was nothing on screen for up to 46s and then, at worst, one terse line.
|
|
67
|
+
// A ticking elapsed counter is the difference between "hung" and "working".
|
|
68
|
+
process.stderr.write('openzoo: starting the proxy...');
|
|
69
|
+
const t0 = Date.now();
|
|
70
|
+
const tick = setInterval(() => {
|
|
71
|
+
process.stderr.write(`\ropenzoo: starting the proxy... ${((Date.now() - t0) / 1000).toFixed(0)}s`);
|
|
72
|
+
}, 1000);
|
|
73
|
+
tick.unref?.();
|
|
74
|
+
const done = (msg) => { clearInterval(tick); process.stderr.write(`\r\x1b[2Kopenzoo: ${msg}\n`); };
|
|
75
|
+
try {
|
|
76
|
+
const { startProxy } = await import('./proxy.js');
|
|
77
|
+
await startProxy({ silent: true, autoTunnel: true });
|
|
78
|
+
} catch (err) {
|
|
79
|
+
// An exception here used to surface as an eternal spinner. Say what broke.
|
|
80
|
+
done(`proxy failed to start: ${err?.message || err}`);
|
|
81
|
+
console.error(' full log: ~/.openzoo/proxy.log');
|
|
82
|
+
console.error(' try: OPENZOO_NO_TUNNEL=1 npx openzoo claude (skips the cloudflared download)');
|
|
83
|
+
process.exit(1);
|
|
84
|
+
}
|
|
85
|
+
// The proxy may have healed onto a different port (8402 busy). config.port
|
|
86
|
+
// is the one it ACTUALLY bound, so re-derive every URL from it — the old
|
|
87
|
+
// code kept polling the port it wished for and timed out on a live proxy.
|
|
88
|
+
base = `http://localhost:${config.port}/v1`;
|
|
62
89
|
for (let i = 0; i < 20 && !up; i++) {
|
|
63
90
|
await new Promise((r) => setTimeout(r, 300));
|
|
64
91
|
try { up = (await fetch(`${base}/models`, { signal: AbortSignal.timeout(2000) })).ok; } catch { /* keep waiting */ }
|
|
65
92
|
}
|
|
66
|
-
if (!up) {
|
|
93
|
+
if (!up) {
|
|
94
|
+
done(`proxy did not answer on ${base}`);
|
|
95
|
+
console.error(' full log: ~/.openzoo/proxy.log');
|
|
96
|
+
console.error(' try: OPENZOO_NO_TUNNEL=1 npx openzoo claude (skips the cloudflared download)');
|
|
97
|
+
process.exit(1);
|
|
98
|
+
}
|
|
99
|
+
done(`proxy up on :${config.port} (${((Date.now() - t0) / 1000).toFixed(1)}s)`);
|
|
67
100
|
}
|
|
68
101
|
// TERMINAL (Claude Code CLI) IS THE DEFAULT — it is the guaranteed-x402 path
|
|
69
102
|
// and honours ANTHROPIC_BASE_URL. --desktop explicitly opens the desktop app
|
package/lib/namespace.js
CHANGED
|
@@ -36,19 +36,45 @@ import { loadOrCreateWallet } from './wallet.js';
|
|
|
36
36
|
* BREAKING: corpora bound before this lived in the shared tenant and are not
|
|
37
37
|
* reachable from a signed request. Re-bind them.
|
|
38
38
|
*/
|
|
39
|
-
|
|
39
|
+
/**
|
|
40
|
+
* ONE NAMESPACE ACROSS EVERY STACC APP (2026-08-18).
|
|
41
|
+
*
|
|
42
|
+
* This used to be sha256("openzoo-ns:" + pubkey) — stable, per-user, and
|
|
43
|
+
* DIFFERENT from what every other stacc app sent. openzoo brain sent
|
|
44
|
+
* HMAC(OPENZOO_TENANT_SECRET, pubkey); the open-webui wallet sent a literal
|
|
45
|
+
* app name. Since the gateway keys a tenant on sha256(chain:signer:namespace),
|
|
46
|
+
* one wallet therefore landed in THREE tenants: three separate leCore
|
|
47
|
+
* memories and three separate credit balances, for the same person. Bind a
|
|
48
|
+
* corpus in the CLI and it was invisible from the browser.
|
|
49
|
+
*
|
|
50
|
+
* The constant is safe because the reason for the per-user hash is gone. It
|
|
51
|
+
* existed to stop someone addressing your tenant by guessing your namespace —
|
|
52
|
+
* but the gateway now folds the VERIFIED SIGNER into the tenant hash (see
|
|
53
|
+
* nsauth.ts / tenantFor), so the namespace string is no longer the
|
|
54
|
+
* access-control boundary. Signing is what proves ownership; this string only
|
|
55
|
+
* chooses WHICH of your namespaces you mean.
|
|
56
|
+
*
|
|
57
|
+
* A constant is in fact MORE private than what it replaces. A per-user hash is
|
|
58
|
+
* stable and unique, i.e. a tracking identifier that correlates one user's
|
|
59
|
+
* requests across time. A value every caller sends identically discloses
|
|
60
|
+
* nothing at all.
|
|
61
|
+
*
|
|
62
|
+
* BREAKING: corpora bound under the old per-wallet namespace live in a
|
|
63
|
+
* different tenant and will not be found. The gateway's tenantsToTry fallback
|
|
64
|
+
* reaches pre-existing context ids by id, but anything relying on the old
|
|
65
|
+
* namespace should be re-bound.
|
|
66
|
+
*/
|
|
67
|
+
export const STACC_NAMESPACE = 'stacc';
|
|
40
68
|
|
|
41
69
|
export function namespaceHeaderValue() {
|
|
42
|
-
if (cached) return cached;
|
|
43
70
|
try {
|
|
44
|
-
|
|
45
|
-
|
|
46
|
-
|
|
47
|
-
|
|
71
|
+
// Still require a wallet: the namespace is meaningless without a signer to
|
|
72
|
+
// prove it, and sending one unsigned drops you into the SHARED tenant.
|
|
73
|
+
loadOrCreateWallet();
|
|
74
|
+
return STACC_NAMESPACE;
|
|
48
75
|
} catch {
|
|
49
|
-
|
|
76
|
+
return ''; // no wallet (read-only use): fall back to the shared tenant
|
|
50
77
|
}
|
|
51
|
-
return cached;
|
|
52
78
|
}
|
|
53
79
|
|
|
54
80
|
/**
|
package/lib/podagent.mjs
CHANGED
|
@@ -249,13 +249,40 @@ async function httpErrorNote(status) {
|
|
|
249
249
|
// same rail, second call pays fine). Surfacing that as a chat message makes
|
|
250
250
|
// the user do the retry by hand — so do it here instead.
|
|
251
251
|
const PAYMENT_RETRIES = 3;
|
|
252
|
-
|
|
252
|
+
/**
|
|
253
|
+
* ADAPTIVE top_k. We have learned this one the expensive way already.
|
|
254
|
+
*
|
|
255
|
+
* On leCore the miss was never the ranker — BM25 ranked correctly. It was
|
|
256
|
+
* top_k=16 against a 7,000-chunk corpus: we only ever ASKED for sixteen. Same
|
|
257
|
+
* shape here, and worse: grokui never set the header at all, so every call fell
|
|
258
|
+
* to the gateway default of EIGHT, while a whole project's bots write into one
|
|
259
|
+
* shared context. Six bots working for an hour and the model sees eight chunks
|
|
260
|
+
* of it.
|
|
261
|
+
*
|
|
262
|
+
* So scale with the corpus instead of picking a number. sqrt keeps it sane at
|
|
263
|
+
* both ends — 100 chunks -> 20, 1k -> 63, 7k -> 167, and it saturates at the
|
|
264
|
+
* gateway's 256 ceiling rather than growing without bound. Floor of 16 so a
|
|
265
|
+
* brand-new thread is never worse off than the old default.
|
|
266
|
+
*
|
|
267
|
+
* Cost is real and proportional (measured on leCore: top_k 16 = $0.0070,
|
|
268
|
+
* top_k 128 = $0.0489 on the same question) — which is the point. The extra
|
|
269
|
+
* spend IS the extra corpus actually being read.
|
|
270
|
+
*/
|
|
271
|
+
export function adaptiveTopK(boundItems) {
|
|
272
|
+
const n = Math.max(0, Number(boundItems) || 0);
|
|
273
|
+
return Math.max(16, Math.min(256, Math.ceil(Math.sqrt(n) * 2)));
|
|
274
|
+
}
|
|
275
|
+
|
|
276
|
+
async function postChat(body, contextId, topK) {
|
|
253
277
|
let r;
|
|
254
278
|
for (let attempt = 0; attempt <= PAYMENT_RETRIES; attempt++) {
|
|
255
279
|
r = await fetch(`${PROXY}/chat/completions`, {
|
|
256
280
|
method: 'POST',
|
|
257
281
|
headers: {
|
|
258
282
|
'content-type': 'application/json', authorization: 'Bearer sk-openzoo',
|
|
283
|
+
// Only sent when we actually know the corpus size; without it the
|
|
284
|
+
// gateway keeps its own default rather than getting a made-up number.
|
|
285
|
+
...(topK ? { 'x-hrr-top-k': String(topK) } : {}),
|
|
259
286
|
// real leCore memory for this thread, bound via POST /v1/hrr/bind — NOT
|
|
260
287
|
// a fabricated mechanism. Retrieval runs automatically once this header
|
|
261
288
|
// is set; nothing more for the model to invent or explain.
|
|
@@ -284,7 +311,7 @@ function withModelId(messages, model) {
|
|
|
284
311
|
: m));
|
|
285
312
|
}
|
|
286
313
|
|
|
287
|
-
export async function brain(messages, contextId, modelOverride) {
|
|
314
|
+
export async function brain(messages, contextId, modelOverride, topK) {
|
|
288
315
|
// explicit plugins, not relying on the gateway's "inject when caller said
|
|
289
316
|
// nothing" default — an explicit array is always respected as-is, so every
|
|
290
317
|
// bot on every model actually has web search. max_tokens 900 was cutting
|
|
@@ -299,24 +326,53 @@ export async function brain(messages, contextId, modelOverride) {
|
|
|
299
326
|
messages = vision ? messages : stripImages(messages);
|
|
300
327
|
const r = await postChat(
|
|
301
328
|
{ model, max_tokens: 4096, messages: withModelId(messages, model), plugins: [{ id: 'web' }] },
|
|
302
|
-
contextId,
|
|
329
|
+
contextId, topK,
|
|
303
330
|
);
|
|
304
331
|
const j = await r.json().catch(() => ({}));
|
|
305
332
|
const content = j?.choices?.[0]?.message?.content;
|
|
333
|
+
// Same truncation catch as the streaming path (see brainStream): a reply that
|
|
334
|
+
// stops because the budget ran out is not a finished reply, and this path is
|
|
335
|
+
// what non-streaming callers — including every SPAWNed subagent — go through.
|
|
336
|
+
if (content && j?.choices?.[0]?.finish_reason === 'length') {
|
|
337
|
+
const rest = await brainContinue(messages, content, contextId, modelOverride, 0);
|
|
338
|
+
return content + rest;
|
|
339
|
+
}
|
|
306
340
|
return content || (r.ok ? '' : await httpErrorNote(r.status));
|
|
307
341
|
}
|
|
308
342
|
|
|
343
|
+
/** Resume a reply that hit the output cap, non-streaming. Bounded by
|
|
344
|
+
* CONTINUE_ROUNDS for the same runaway reason brainStream is. */
|
|
345
|
+
async function brainContinue(messages, sofar, contextId, modelOverride, round) {
|
|
346
|
+
if (round >= CONTINUE_ROUNDS) return '';
|
|
347
|
+
const vision = hasImages(messages);
|
|
348
|
+
const model = vision ? VISION_MODEL : (modelOverride || MODEL);
|
|
349
|
+
const next = [...messages,
|
|
350
|
+
{ role: 'assistant', content: sofar },
|
|
351
|
+
{ role: 'user', content: CONTINUE_NUDGE }];
|
|
352
|
+
const r = await postChat(
|
|
353
|
+
{ model, max_tokens: Math.min(4096 * (2 ** (round + 1)), MAX_CONTINUE_TOKENS),
|
|
354
|
+
messages: withModelId(vision ? next : stripImages(next), model), plugins: [{ id: 'web' }] },
|
|
355
|
+
contextId,
|
|
356
|
+
);
|
|
357
|
+
const j = await r.json().catch(() => ({}));
|
|
358
|
+
const more = j?.choices?.[0]?.message?.content || '';
|
|
359
|
+
if (more && j?.choices?.[0]?.finish_reason === 'length') {
|
|
360
|
+
return more + await brainContinue(messages, sofar + more, contextId, modelOverride, round + 1);
|
|
361
|
+
}
|
|
362
|
+
return more;
|
|
363
|
+
}
|
|
364
|
+
|
|
309
365
|
/** Same call, but streamed — invokes onDelta(text) as tokens arrive (for a
|
|
310
366
|
* live-typing UI) and resolves with the full accumulated text at the end, so
|
|
311
367
|
* callers that need to parse a directive out of the complete reply still can. */
|
|
312
|
-
export async function brainStream(messages, onDelta, contextId, modelOverride, maxTokens) {
|
|
368
|
+
export async function brainStream(messages, onDelta, contextId, modelOverride, maxTokens, round = 0, topK = 0) {
|
|
313
369
|
const vision = hasImages(messages);
|
|
314
370
|
const model = vision ? VISION_MODEL : (modelOverride || MODEL);
|
|
315
371
|
messages = vision ? messages : stripImages(messages);
|
|
316
372
|
const budget = maxTokens || MAX_TOKENS;
|
|
317
373
|
const r = await postChat(
|
|
318
374
|
{ model, max_tokens: budget, messages: withModelId(messages, model), plugins: [{ id: 'web' }], stream: true },
|
|
319
|
-
contextId,
|
|
375
|
+
contextId, topK,
|
|
320
376
|
);
|
|
321
377
|
if (!r.ok || !r.body) {
|
|
322
378
|
// fall back to the non-streaming path rather than fail outright
|
|
@@ -327,7 +383,7 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
327
383
|
}
|
|
328
384
|
const reader = r.body.getReader();
|
|
329
385
|
const decoder = new TextDecoder();
|
|
330
|
-
let buf = '', full = '', reasonedChars = 0;
|
|
386
|
+
let buf = '', full = '', reasonedChars = 0, finish = '';
|
|
331
387
|
for (;;) {
|
|
332
388
|
const { value, done } = await reader.read();
|
|
333
389
|
if (done) break;
|
|
@@ -340,7 +396,12 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
340
396
|
const payload = s.slice(5).trim();
|
|
341
397
|
if (payload === '[DONE]') continue;
|
|
342
398
|
try {
|
|
343
|
-
const
|
|
399
|
+
const c = JSON.parse(payload)?.choices?.[0];
|
|
400
|
+
const d = c?.delta;
|
|
401
|
+
// The LAST chunk carries why generation stopped. "length" means the
|
|
402
|
+
// budget ran out mid-answer — the only way to tell a finished reply
|
|
403
|
+
// from a guillotined one.
|
|
404
|
+
if (c?.finish_reason) finish = c.finish_reason;
|
|
344
405
|
if (d?.content) { full += d.content; onDelta(d.content); }
|
|
345
406
|
// Reasoning models emit their chain of thought on a SEPARATE field and
|
|
346
407
|
// only then start producing content. Count it — not to show it, but to
|
|
@@ -358,9 +419,267 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
358
419
|
if (!full && reasonedChars > 0 && !maxTokens) {
|
|
359
420
|
return brainStream(messages, onDelta, contextId, modelOverride, budget * 4);
|
|
360
421
|
}
|
|
422
|
+
|
|
423
|
+
// CUT OFF MID-ANSWER. finish_reason "length" means the model had more to say
|
|
424
|
+
// and the budget ended the sentence for it — seen live as a reply that stops
|
|
425
|
+
// inside `for (`. Nothing above catches this, because `full` is non-empty:
|
|
426
|
+
// by every other measure the turn succeeded.
|
|
427
|
+
//
|
|
428
|
+
// CONTINUE rather than retry. Re-running the turn with a bigger budget makes
|
|
429
|
+
// the user pay twice for the half we already have (and on a reasoning model,
|
|
430
|
+
// pay for the whole chain of thought again). Handing the model back its own
|
|
431
|
+
// partial and asking for the rest costs only the rest.
|
|
432
|
+
//
|
|
433
|
+
// Bounded, because a model that ignores the nudge would otherwise continue
|
|
434
|
+
// forever on the user's wallet.
|
|
435
|
+
if (full && finish === 'length' && round < CONTINUE_ROUNDS) {
|
|
436
|
+
const more = await brainStream(
|
|
437
|
+
[...messages,
|
|
438
|
+
{ role: 'assistant', content: full },
|
|
439
|
+
{ role: 'user', content: CONTINUE_NUDGE }],
|
|
440
|
+
onDelta, contextId, modelOverride,
|
|
441
|
+
Math.min(budget * 2, MAX_CONTINUE_TOKENS), round + 1,
|
|
442
|
+
);
|
|
443
|
+
return full + (more || '');
|
|
444
|
+
}
|
|
361
445
|
return full;
|
|
362
446
|
}
|
|
363
447
|
|
|
448
|
+
// How many times a single answer may be resumed after hitting the cap. Three
|
|
449
|
+
// doublings off 4096 is ~57k tokens of answer, which is past any real reply and
|
|
450
|
+
// well short of a runaway.
|
|
451
|
+
const CONTINUE_ROUNDS = Number(process.env.OZ_CONTINUE_ROUNDS || 3);
|
|
452
|
+
const MAX_CONTINUE_TOKENS = Number(process.env.OZ_MAX_CONTINUE_TOKENS || 32768);
|
|
453
|
+
// Deliberately blunt about the seam: the partial usually ends mid-token, and a
|
|
454
|
+
// model that "helpfully" restarts the sentence produces a visible stutter in
|
|
455
|
+
// the middle of the user's code.
|
|
456
|
+
const CONTINUE_NUDGE = 'You were cut off — your previous message hit the output limit mid-way. '
|
|
457
|
+
+ 'Continue from EXACTLY where it stopped. Do not repeat any of it, do not summarise it, '
|
|
458
|
+
+ 'do not add a preamble or an apology, and do not re-open a code fence that is already open. '
|
|
459
|
+
+ 'Resume mid-word if that is where it ended.';
|
|
460
|
+
|
|
461
|
+
// ---------------------------------------------------------------------------
|
|
462
|
+
// MODEL TIERS · cheap / medium / expensive
|
|
463
|
+
// ---------------------------------------------------------------------------
|
|
464
|
+
// Ranking the live catalog by price alone picks garbage at both ends: the most
|
|
465
|
+
// expensive served model is o1-pro at $1800/Mtok (a bad coding model that would
|
|
466
|
+
// drain the box wallet in a handful of turns), and the cheapest is a roleplay
|
|
467
|
+
// finetune. Price is a proxy for capability only within a band, never across
|
|
468
|
+
// the whole catalog. So each tier is a CURATED, ordered preference list, and
|
|
469
|
+
// the catalog is used to check what is actually served today — the zoo's model
|
|
470
|
+
// list changes under us, and a tier that resolves to a 404 is worse than no
|
|
471
|
+
// tier at all.
|
|
472
|
+
//
|
|
473
|
+
// Each list is ordered best-first (that is what a non-racing "auto" picks) but
|
|
474
|
+
// deliberately WIDE, because a race samples from it at random: a pool of three
|
|
475
|
+
// would race the same three models every time, which is neither a real hedge
|
|
476
|
+
// against a single provider having a bad minute nor a real sample of the tier.
|
|
477
|
+
// Prices in the comments are completion USD per Mtok as served, measured.
|
|
478
|
+
const TIERS = {
|
|
479
|
+
// ≲ $3/Mtok. Fast, good enough for glue work, cheap enough to race widely.
|
|
480
|
+
cheap: [
|
|
481
|
+
'deepseek/deepseek-v4-flash', // 0.45
|
|
482
|
+
'meta-llama/llama-4-scout', // 0.90
|
|
483
|
+
'z-ai/glm-4.7-flash', // 1.20
|
|
484
|
+
'bytedance-seed/seed-2.0-mini', // 1.20
|
|
485
|
+
'meta-llama/llama-4-maverick', // 2.40
|
|
486
|
+
'z-ai/glm-4.5-air', // 2.55
|
|
487
|
+
'minimax/minimax-m2.5', // 2.70
|
|
488
|
+
'z-ai/glm-4.6v', // 2.70
|
|
489
|
+
'minimax/minimax-m2', // 3.06
|
|
490
|
+
'inclusionai/ling-3.0-flash', // 0.19
|
|
491
|
+
],
|
|
492
|
+
// ~$4.5–11/Mtok. The default band; deepseek-v4-pro is the app default.
|
|
493
|
+
medium: [
|
|
494
|
+
'deepseek/deepseek-v4-pro-0813', // 5.94
|
|
495
|
+
'z-ai/glm-4.7', // 5.25
|
|
496
|
+
'google/gemini-3.7-flash', // 5.63
|
|
497
|
+
'x-ai/grok-4.3', // 7.50
|
|
498
|
+
'moonshotai/kimi-k2.7-code', // 10.50
|
|
499
|
+
'z-ai/glm-5', // 5.76
|
|
500
|
+
'moonshotai/kimi-k2.6', // 7.08
|
|
501
|
+
'mistralai/mistral-large-2512', // 4.50
|
|
502
|
+
'bytedance-seed/seed-2.0-code', // 9.00
|
|
503
|
+
'qwen/qwen3.8-27b', // 9.60
|
|
504
|
+
],
|
|
505
|
+
// ≥ $18/Mtok. Frontier. NOTE the ceiling: o1-pro ($1800) and the *-pro tiers
|
|
506
|
+
// ($240–540) are deliberately NOT here. A race of four across that band can
|
|
507
|
+
// cost dollars per turn on a box funded with a few cents.
|
|
508
|
+
expensive: [
|
|
509
|
+
'anthropic/claude-opus-5', // 75
|
|
510
|
+
'openai/gpt-5.5', // 90
|
|
511
|
+
'anthropic/claude-sonnet-5', // 30
|
|
512
|
+
'x-ai/grok-4.6', // 18
|
|
513
|
+
'moonshotai/kimi-k3', // 45
|
|
514
|
+
'anthropic/claude-opus-4.8', // 75
|
|
515
|
+
'openai/gpt-5.4', // 45
|
|
516
|
+
'qwen/qwen3.8-max', // 18
|
|
517
|
+
'x-ai/grok-4.5', // 18
|
|
518
|
+
],
|
|
519
|
+
};
|
|
520
|
+
export const TIER_NAMES = Object.keys(TIERS);
|
|
521
|
+
|
|
522
|
+
let catalogCache = { at: 0, ids: null };
|
|
523
|
+
async function servedIds() {
|
|
524
|
+
// 5 minutes: long enough that a race does not re-fetch per model, short
|
|
525
|
+
// enough that a model coming back after an outage is picked up the same
|
|
526
|
+
// session.
|
|
527
|
+
if (catalogCache.ids && Date.now() - catalogCache.at < 300_000) return catalogCache.ids;
|
|
528
|
+
try {
|
|
529
|
+
const r = await fetch(`${PROXY}/models`);
|
|
530
|
+
const j = await r.json();
|
|
531
|
+
const ids = new Set((j?.data || []).map((m) => m.id).filter(Boolean));
|
|
532
|
+
if (ids.size) catalogCache = { at: Date.now(), ids };
|
|
533
|
+
} catch { /* proxy down — fall through to whatever we had, or null */ }
|
|
534
|
+
return catalogCache.ids;
|
|
535
|
+
}
|
|
536
|
+
|
|
537
|
+
/**
|
|
538
|
+
* The models a tier resolves to right now, only ones actually served.
|
|
539
|
+
*
|
|
540
|
+
* `random` is what a race uses: pick n from the whole tier at random rather
|
|
541
|
+
* than always the top n. Two reasons it must be random and not top-n — a fixed
|
|
542
|
+
* trio is not a hedge (they can share an upstream having a bad minute, which is
|
|
543
|
+
* precisely the failure racing is meant to survive), and it silently reduces a
|
|
544
|
+
* ten-model tier to three models the user never chose.
|
|
545
|
+
*
|
|
546
|
+
* Falls back to the curated list unchecked if the catalog is unreachable — a
|
|
547
|
+
* stale-but-plausible id beats refusing to answer.
|
|
548
|
+
*/
|
|
549
|
+
export async function tierModels(tier, n = 1, random = false) {
|
|
550
|
+
const want = TIERS[tier] || TIERS.medium;
|
|
551
|
+
const ids = await servedIds();
|
|
552
|
+
const live = ids ? want.filter((m) => ids.has(m)) : want;
|
|
553
|
+
const pool = live.length ? live : want;
|
|
554
|
+
const take = Math.max(1, Math.min(n, pool.length));
|
|
555
|
+
if (!random) return pool.slice(0, take);
|
|
556
|
+
// Fisher-Yates on a copy: sampling without replacement, because racing a
|
|
557
|
+
// model against itself buys nothing and still bills twice.
|
|
558
|
+
const a = pool.slice();
|
|
559
|
+
for (let i = a.length - 1; i > 0; i--) {
|
|
560
|
+
const j = Math.floor(Math.random() * (i + 1));
|
|
561
|
+
[a[i], a[j]] = [a[j], a[i]];
|
|
562
|
+
}
|
|
563
|
+
return a.slice(0, take);
|
|
564
|
+
}
|
|
565
|
+
|
|
566
|
+
/**
|
|
567
|
+
* Launch N models at once, judge the FIRST K that come back.
|
|
568
|
+
*
|
|
569
|
+
* "/race 2 3" — start three, and the moment two of them have returned a real
|
|
570
|
+
* answer, judge those two and ship the winner. The third is abandoned mid-flight.
|
|
571
|
+
*
|
|
572
|
+
* This is the useful shape, and it is neither of the obvious two:
|
|
573
|
+
* - first-past-the-post (K=1) optimises latency only, and on a hard question
|
|
574
|
+
* it rewards whichever model thought LEAST.
|
|
575
|
+
* - wait-for-all-then-judge (K=N) buys quality with the slowest entrant's
|
|
576
|
+
* latency, and one wedged provider stalls the whole turn.
|
|
577
|
+
* Taking the first K bounds the wait at the Kth-fastest while still giving the
|
|
578
|
+
* judge something to compare. The straggler is exactly the entrant you were
|
|
579
|
+
* least likely to want anyway.
|
|
580
|
+
*
|
|
581
|
+
* Reliability comes free with it: empty completions and provider 5xx are
|
|
582
|
+
* per-model and uncorrelated, which is why "the model returned nothing 4 times"
|
|
583
|
+
* was never fixable by a fourth try at the same model. An empty reply does NOT
|
|
584
|
+
* count toward K — otherwise the fastest model to FAIL would decide the race,
|
|
585
|
+
* the exact bug this exists to fix.
|
|
586
|
+
*
|
|
587
|
+
* Streaming is deliberately not forwarded while the race runs: nobody knows who
|
|
588
|
+
* is winning until they finish, and interleaving deltas from three models would
|
|
589
|
+
* render as noise. The winner's text is emitted whole.
|
|
590
|
+
*
|
|
591
|
+
* Every entrant is paid for, including the abandoned one — this trades money
|
|
592
|
+
* for latency and quality, which is why it is opt-in and capped.
|
|
593
|
+
*/
|
|
594
|
+
export async function brainRace(messages, onDelta, contextId, models, need = 1, maxTokens) {
|
|
595
|
+
const list = (models || []).filter(Boolean).slice(0, RACE_MAX);
|
|
596
|
+
if (list.length < 2) return brainStream(messages, onDelta, contextId, list[0], maxTokens);
|
|
597
|
+
const want = Math.max(1, Math.min(Number(need) || 1, list.length));
|
|
598
|
+
|
|
599
|
+
const done = [];
|
|
600
|
+
let finished = 0;
|
|
601
|
+
let release;
|
|
602
|
+
const enough = new Promise((r) => { release = r; });
|
|
603
|
+
|
|
604
|
+
const attempts = list.map((m) => brainStream(messages, () => {}, contextId, m, maxTokens)
|
|
605
|
+
.then((text) => { if (text && text.trim()) done.push({ model: m, text }); })
|
|
606
|
+
.catch(() => { /* one entrant dying is not the race dying */ })
|
|
607
|
+
.finally(() => {
|
|
608
|
+
finished += 1;
|
|
609
|
+
// Either we have what we asked for, or everyone is done and no more is
|
|
610
|
+
// coming — without the second condition a race where two of three fail
|
|
611
|
+
// would hang forever waiting for a K that can never arrive.
|
|
612
|
+
if (done.length >= want || finished === list.length) release();
|
|
613
|
+
}));
|
|
614
|
+
// Losers keep running; swallow their rejections so one cannot take the
|
|
615
|
+
// process down after the winner has already been returned.
|
|
616
|
+
for (const p of attempts) p.catch(() => {});
|
|
617
|
+
|
|
618
|
+
await enough;
|
|
619
|
+
// Completion order, so this really is the first K back — not the first K
|
|
620
|
+
// launched.
|
|
621
|
+
const cands = done.slice(0, want);
|
|
622
|
+
if (!cands.length) return '';
|
|
623
|
+
// Nothing to compare — do not spend a judging call to rubber-stamp one answer.
|
|
624
|
+
if (cands.length === 1) { onDelta(cands[0].text); return cands[0].text; }
|
|
625
|
+
|
|
626
|
+
const winner = await judge(messages, cands);
|
|
627
|
+
onDelta(winner.text);
|
|
628
|
+
return winner.text;
|
|
629
|
+
}
|
|
630
|
+
|
|
631
|
+
/**
|
|
632
|
+
* Pick the best of several finished answers with a small model.
|
|
633
|
+
*
|
|
634
|
+
* BLIND, as A/B/C/D. A judge told "this one is Claude and this one is a 4B
|
|
635
|
+
* llama" is being handed the answer and will take it, which would turn the
|
|
636
|
+
* whole thing into an expensive way to re-pick the tier's first entry.
|
|
637
|
+
*
|
|
638
|
+
* Cheap on purpose: reading finished replies and comparing them against a
|
|
639
|
+
* question is a far easier task than answering it, and paying frontier prices
|
|
640
|
+
* to referee frontier models would roughly double the cost of the expensive
|
|
641
|
+
* tier for no measured gain.
|
|
642
|
+
*/
|
|
643
|
+
async function judge(messages, cands) {
|
|
644
|
+
const letters = cands.map((_, i) => String.fromCharCode(65 + i));
|
|
645
|
+
// The question, not the transcript: the judge needs to know what was ASKED,
|
|
646
|
+
// and a full history would cost more to judge than the turn cost to answer.
|
|
647
|
+
const asked = [...messages].reverse().find((m) => m.role === 'user')?.content;
|
|
648
|
+
const question = typeof asked === 'string' ? asked : '(see candidates)';
|
|
649
|
+
const prompt = 'You are judging answers to one question. Pick the single best one.\n\n'
|
|
650
|
+
+ 'QUESTION:\n' + String(question).slice(0, 4000) + '\n\n'
|
|
651
|
+
+ cands.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + c.text.slice(0, 6000)).join('\n\n')
|
|
652
|
+
+ '\n\nJudge on: correctness first, then completeness, then whether it actually did what was asked '
|
|
653
|
+
+ '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
|
|
654
|
+
+ 'Ignore length and confidence of tone.\n'
|
|
655
|
+
+ 'Reply with ONE letter and nothing else: ' + letters.join(' or ') + '.';
|
|
656
|
+
try {
|
|
657
|
+
const verdict = await brainStream([{ role: 'user', content: prompt }], () => {}, undefined, JUDGE_MODEL, 8);
|
|
658
|
+
// First in-range letter anywhere in the reply. A judge that ignores "one
|
|
659
|
+
// letter and nothing else" and writes "The best is B." still counts, which
|
|
660
|
+
// is most of them.
|
|
661
|
+
const hit = String(verdict).toUpperCase().split('').find((ch) => {
|
|
662
|
+
const n = ch.charCodeAt(0) - 65;
|
|
663
|
+
return n >= 0 && n < cands.length;
|
|
664
|
+
});
|
|
665
|
+
if (hit) return cands[hit.charCodeAt(0) - 65];
|
|
666
|
+
} catch { /* fall through */ }
|
|
667
|
+
// A dead or delisted judge must not lose the answers. Falling back to the
|
|
668
|
+
// first finisher degrades this to "fastest wins" — worse than judged, far
|
|
669
|
+
// better than empty.
|
|
670
|
+
return cands[0];
|
|
671
|
+
}
|
|
672
|
+
|
|
673
|
+
const RACE_MAX = Number(process.env.OZ_RACE_MAX || 4);
|
|
674
|
+
|
|
675
|
+
// brainBest is GONE — brainRace(models, need) subsumes it. "wait for all N
|
|
676
|
+
// then judge" is exactly need === N, and keeping a second judged-race entry
|
|
677
|
+
// point meant two call sites that could disagree about what a race is.
|
|
678
|
+
|
|
679
|
+
// Cheapest thing that can reliably output one letter. Overridable because the
|
|
680
|
+
// catalog moves; if it is delisted the try/catch above falls back cleanly.
|
|
681
|
+
const JUDGE_MODEL = process.env.OZ_JUDGE_MODEL || 'deepseek/deepseek-v4-flash';
|
|
682
|
+
|
|
364
683
|
const SYSTEM = `You are the brain of a Grok-Bot-style coding/ops agent. The polished chat UI
|
|
365
684
|
the user sees is Grok Bot (Anysphere's app); its "sandbox" has been pointed at THIS box, and
|
|
366
685
|
your reasoning is served by openzoo (pay-per-call access to ~435 models over x402 — no API key,
|
package/lib/proxy.js
CHANGED
|
@@ -16,6 +16,7 @@ import { forgetContext } from './contexts.js';
|
|
|
16
16
|
import { injectBrief } from './brief.js';
|
|
17
17
|
import { withNamespace } from './namespace.js';
|
|
18
18
|
import { anthropicToOpenAI, openAIToAnthropic, writeAnthropicSse } from './anthropic.js';
|
|
19
|
+
import { responsesToChat, chatToResponses, writeResponsesSse } from './responses.js';
|
|
19
20
|
|
|
20
21
|
const HOP_BY_HOP = new Set([
|
|
21
22
|
'host', 'connection', 'keep-alive', 'transfer-encoding', 'upgrade',
|
|
@@ -41,7 +42,7 @@ const HOP_BY_HOP = new Set([
|
|
|
41
42
|
function normalizePath(url) {
|
|
42
43
|
const [path, query] = (url || '/').split(/(?=\?)/);
|
|
43
44
|
let p = path.replace(/^(?:\/v1)+(?=\/v1\/)/, ''); // /v1/v1/x -> /v1/x
|
|
44
|
-
if (!/^\/v1(\/|$)/.test(p) && /^\/(hrr|chat|models|completions|embeddings|usage)/.test(p)) p = `/v1${p}`;
|
|
45
|
+
if (!/^\/v1(\/|$)/.test(p) && /^\/(hrr|chat|models|completions|embeddings|usage|responses)/.test(p)) p = `/v1${p}`;
|
|
45
46
|
return p === path ? url : `${p}${query || ''}`;
|
|
46
47
|
}
|
|
47
48
|
|
|
@@ -547,7 +548,51 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
|
|
|
547
548
|
// answer back on the way out. See lib/anthropic.js.
|
|
548
549
|
let anthropicMode = false;
|
|
549
550
|
let anthropicModel = null;
|
|
551
|
+
let responsesMode = false;
|
|
552
|
+
let responsesModel = null;
|
|
553
|
+
let responsesCustom = null; // names of freeform tools needing custom_tool_call on the way back
|
|
550
554
|
const rawPath = (req.url || '').split('?')[0];
|
|
555
|
+
|
|
556
|
+
// RESPONSES API. Some harnesses speak only this wire format — OpenAI's
|
|
557
|
+
// Codex Security CLI pins `wire_api: "responses"` in its provider table, so
|
|
558
|
+
// a bare 404 here made it fall back to wss://api.openai.com and bypass the
|
|
559
|
+
// proxy entirely while reporting the failure as an auth error. Translate
|
|
560
|
+
// in, translate out; everything between stays on the chat path.
|
|
561
|
+
if (req.method === 'POST' && (rawPath === '/v1/responses' || rawPath === '/responses')) {
|
|
562
|
+
try {
|
|
563
|
+
const inbound = JSON.parse(bodyBuf.toString('utf8'));
|
|
564
|
+
// TEMP CAPTURE: dump the first few Responses requests so the agent's
|
|
565
|
+
// actual wire usage (store / previous_response_id / tool shapes) can be
|
|
566
|
+
// read rather than inferred. Guarded by an env var so it is off unless
|
|
567
|
+
// asked for.
|
|
568
|
+
if (process.env.OZ_CAPTURE_RESPONSES) {
|
|
569
|
+
try {
|
|
570
|
+
const fs = await import('node:fs');
|
|
571
|
+
const dir = process.env.OZ_CAPTURE_RESPONSES;
|
|
572
|
+
fs.mkdirSync(dir, { recursive: true });
|
|
573
|
+
const n = fs.readdirSync(dir).length;
|
|
574
|
+
// Capture the LATER turns too. Turn 1 is already understood; the
|
|
575
|
+
// unknown is what codex sends back after running a tool, so bias
|
|
576
|
+
// the capture toward requests that carry a tool result.
|
|
577
|
+
const hasResult = Array.isArray(inbound.input)
|
|
578
|
+
&& inbound.input.some((i) => i && String(i.type || '').endsWith('_call_output'));
|
|
579
|
+
const tag = hasResult ? 'result' : 'plain';
|
|
580
|
+
if (n < 24) fs.writeFileSync(`${dir}/${tag}-${n}.json`, JSON.stringify(inbound, null, 2));
|
|
581
|
+
} catch { /* capture must never break a paid call */ }
|
|
582
|
+
}
|
|
583
|
+
responsesModel = inbound.model;
|
|
584
|
+
const meta = {};
|
|
585
|
+
bodyBuf = Buffer.from(JSON.stringify(responsesToChat(inbound, meta)));
|
|
586
|
+
responsesCustom = meta.custom;
|
|
587
|
+
responsesMode = true;
|
|
588
|
+
req.url = '/v1/chat/completions';
|
|
589
|
+
url = `${config.apiBase}${req.url}`;
|
|
590
|
+
} catch {
|
|
591
|
+
jsonErr(res, 400, 'invalid responses body');
|
|
592
|
+
return;
|
|
593
|
+
}
|
|
594
|
+
}
|
|
595
|
+
|
|
551
596
|
if (req.method === 'POST' && (rawPath === '/v1/messages' || rawPath === '/messages')) {
|
|
552
597
|
try {
|
|
553
598
|
const inbound = JSON.parse(bodyBuf.toString('utf8'));
|
|
@@ -604,6 +649,18 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
|
|
|
604
649
|
const hit = replayGet(rKey);
|
|
605
650
|
if (hit) {
|
|
606
651
|
log('identical request within 30s — served the cached completion, NOT re-paid');
|
|
652
|
+
// The cache stores the CHAT shape. A Responses caller must get its own
|
|
653
|
+
// wire format back, or the replay path silently answers in a format the
|
|
654
|
+
// client cannot parse — a bug that only appears on the SECOND identical
|
|
655
|
+
// request, which is exactly when nobody is watching.
|
|
656
|
+
if (responsesMode) {
|
|
657
|
+
if (wantsStream) { writeResponsesSse(res, hit.data, responsesModel, null, responsesCustom); return; }
|
|
658
|
+
const rh = { 'content-type': 'application/json' };
|
|
659
|
+
if (hit.settle) rh['x-payment-response'] = hit.settle;
|
|
660
|
+
res.writeHead(200, rh);
|
|
661
|
+
res.end(JSON.stringify(chatToResponses(hit.data, responsesModel, responsesCustom)));
|
|
662
|
+
return;
|
|
663
|
+
}
|
|
607
664
|
if (wantsStream) { serveAsSse(res, hit.data, null); return; }
|
|
608
665
|
const h = { 'content-type': 'application/json' };
|
|
609
666
|
if (hit.settle) h['x-payment-response'] = hit.settle;
|
|
@@ -752,6 +809,20 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
|
|
|
752
809
|
res.end(JSON.stringify(msg));
|
|
753
810
|
return;
|
|
754
811
|
}
|
|
812
|
+
if (responsesMode) {
|
|
813
|
+
// A Responses client that asked to stream is WAITING for
|
|
814
|
+
// `response.completed`; handing it a JSON body closes the socket
|
|
815
|
+
// mid-stream and it reports "stream disconnected before
|
|
816
|
+
// completion". Honour the streaming contract when it asked for it.
|
|
817
|
+
if (wantsStream) { writeResponsesSse(res, data, responsesModel, response, responsesCustom); return; }
|
|
818
|
+
const out = chatToResponses(data, responsesModel, responsesCustom);
|
|
819
|
+
const h = { 'content-type': 'application/json' };
|
|
820
|
+
const settleHdr = response.headers.get('x-payment-response');
|
|
821
|
+
if (settleHdr) h['x-payment-response'] = settleHdr;
|
|
822
|
+
res.writeHead(200, h);
|
|
823
|
+
res.end(JSON.stringify(out));
|
|
824
|
+
return;
|
|
825
|
+
}
|
|
755
826
|
if (wantsStream) { serveAsSse(res, data, response); return; }
|
|
756
827
|
const h = { 'content-type': 'application/json' };
|
|
757
828
|
const settleHdr = response.headers.get('x-payment-response');
|
|
@@ -790,10 +861,40 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
|
|
|
790
861
|
// AND a tunnel token, so the RunPod-fronted port stays gated exactly like the
|
|
791
862
|
// public tunnel path.
|
|
792
863
|
const bindHost = process.env.OPENZOO_BIND || '127.0.0.1';
|
|
793
|
-
|
|
794
|
-
|
|
795
|
-
|
|
796
|
-
|
|
864
|
+
// SELF-HEAL A TAKEN PORT. A killed-but-not-reaped run, a second terminal, or
|
|
865
|
+
// anything else already on 8402 made listen() reject and took the whole start
|
|
866
|
+
// down — and because `openzoo claude` starts us with silent:true, the user saw
|
|
867
|
+
// only "starting the proxy in the background..." and no reason. Walk up to the
|
|
868
|
+
// next free port instead of dying; the caller reads config.port back out, so
|
|
869
|
+
// every URL printed afterwards is the one we actually bound.
|
|
870
|
+
//
|
|
871
|
+
// EXCEPTION: if the thing already on the port is a HEALTHY openzoo proxy,
|
|
872
|
+
// reuse it rather than starting a rival that splits spend across two wallets.
|
|
873
|
+
const wanted = config.port;
|
|
874
|
+
for (let attempt = 0; ; attempt++) {
|
|
875
|
+
try {
|
|
876
|
+
await new Promise((resolve, reject) => {
|
|
877
|
+
const onErr = (e) => { server.removeListener('error', onErr); reject(e); };
|
|
878
|
+
server.on('error', onErr);
|
|
879
|
+
server.listen(config.port, bindHost, () => { server.removeListener('error', onErr); resolve(); });
|
|
880
|
+
});
|
|
881
|
+
break;
|
|
882
|
+
} catch (e) {
|
|
883
|
+
if (e?.code !== 'EADDRINUSE' || attempt >= 12) throw e;
|
|
884
|
+
if (attempt === 0) {
|
|
885
|
+
try {
|
|
886
|
+
const probe = await fetch(`http://127.0.0.1:${config.port}/v1/models`, { signal: AbortSignal.timeout(2500) });
|
|
887
|
+
if (probe.ok) {
|
|
888
|
+
say(`openzoo: a healthy proxy is already on :${config.port} — reusing it`);
|
|
889
|
+
return { server: null, client, reused: true, port: config.port, spent: () => 0, publicUrl: null, tunnelToken: null, tunnelError: null };
|
|
890
|
+
}
|
|
891
|
+
} catch { /* not ours, or wedged — take the next port */ }
|
|
892
|
+
}
|
|
893
|
+
config.port += 1;
|
|
894
|
+
say(`openzoo: :${config.port - 1} busy — trying :${config.port}`);
|
|
895
|
+
}
|
|
896
|
+
}
|
|
897
|
+
if (config.port !== wanted) say(`openzoo: listening on :${config.port} (:${wanted} was busy)`);
|
|
797
898
|
|
|
798
899
|
// AUTO-PREPAY. Paying on-chain per call is where the latency lives: the
|
|
799
900
|
// gateway answers its 402 challenge in ~0.12s while a full settled call
|