openzoo 0.49.7 → 0.49.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/podagent.mjs CHANGED
@@ -28,6 +28,14 @@ import {
28
28
  isRaceCountable, raceLastShip, shouldRetryRaceArrival, raceFailKind,
29
29
  summarizeRaceFailures,
30
30
  } from './livestatus.js';
31
+ import { messageReasoning, wrapThink, stripThinkTags, reasoningPresent, splitThink } from './think.js';
32
+ import { guardFindCwd } from './runguard.js';
33
+
34
+ function emitReply(onDelta, raw) {
35
+ const parts = splitThink(raw);
36
+ if (parts.thinking) onDelta(parts.thinking, { think: true });
37
+ if (parts.visible) onDelta(parts.visible);
38
+ }
31
39
  import {
32
40
  probeGatewayRace, capRaceByCredit, inferRaceTier, RACE_NO_CREDIT,
33
41
  recutRaceByHud, sessionDollarX,
@@ -42,7 +50,7 @@ export const PROXY = process.env.OZ_PROXY || 'http://127.0.0.1:8402/v1';
42
50
  function completionsProxy() {
43
51
  return process.env.OZ_PROXY || PROXY;
44
52
  }
45
- export const MODEL = process.env.OZ_BRAIN_MODEL || 'deepseek/deepseek-v4-pro-0813';
53
+ export const MODEL = process.env.OZ_BRAIN_MODEL || 'openzoo/auto';
46
54
  const MAX_STEPS = Number(process.env.OZ_MAX_STEPS || 10);
47
55
 
48
56
  // Matches the Grok Bot chat surface itself (dark canvas, right-aligned grey
@@ -188,7 +196,7 @@ function execFrame(command, cwd = '/tmp') {
188
196
  approvalId: randomUUID(),
189
197
  // agent.v1.ExecServerMessage as protobuf-es JSON (camelCase). The daemon
190
198
  // assigns `id`; we only supply the shell variant.
191
- serverMessage: { shellArgs: { command, workingDirectory: cwd, timeout: 120 } },
199
+ serverMessage: { shellArgs: { command: guardFindCwd(command, cwd), workingDirectory: cwd, timeout: 120 } },
192
200
  };
193
201
  }
194
202
 
@@ -247,7 +255,7 @@ async function httpErrorNote(status) {
247
255
  // funded === false is the genuinely-empty case; funded === true after
248
256
  // the retries above means the rail/quote failed, not the balance
249
257
  if (w.funded === false) {
250
- return `(payment failed — HTTP 402, the wallet is empty. ${w.funding}. EVM (Base/Robinhood): ${w.evm}.)`;
258
+ return `(payment required — HTTP 402, the wallet is empty. ${w.funding}. EVM (Base/Robinhood): ${w.evm}.)`;
251
259
  }
252
260
  return `(payment failed — HTTP 402 after ${PAYMENT_RETRIES} retries, though the wallet holds ${w.balances || 'a balance'}. Send it again; if it keeps failing the quoted asset may not be convertible right now. Fund with: ${w.funding})`;
253
261
  }
@@ -276,7 +284,13 @@ function sanitizeProxiedError(msg) {
276
284
  // this attempt, but the NEXT attempt usually settles (measured: same wallet,
277
285
  // same rail, second call pays fine). Surfacing that as a chat message makes
278
286
  // the user do the retry by hand — so do it here instead.
287
+ // Empty/underfunded is NOT that handshake: retrying just re-walks wrap.
279
288
  const PAYMENT_RETRIES = 3;
289
+
290
+ export function isUnderfunded402Body(body) {
291
+ const msg = String(body?.error?.message || body?.error || body?.message || '');
292
+ return /\b(?:underfunded|wallet is empty|empty wallet)\b/i.test(msg);
293
+ }
280
294
  /**
281
295
  * ADAPTIVE top_k. We have learned this one the expensive way already.
282
296
  *
@@ -329,6 +343,10 @@ async function postChat(body, contextId, topK, onStatus, signal) {
329
343
  ...(signal ? { signal } : {}),
330
344
  });
331
345
  if (r.status !== 402 || attempt === PAYMENT_RETRIES) return r;
346
+ // Handshake 402 (funded, settle flake) → retry. Empty-wallet 402 → stop.
347
+ // Opening Pay / parking happens on the empty body, not on every 402.
348
+ const peek = await r.clone().json().catch(() => null);
349
+ if (isUnderfunded402Body(peek)) return r;
332
350
  // A 402 retry used to be silent — grokui sat on mute "…" for the whole
333
351
  // settle. Tell the watcher this attempt is paying, not wedged.
334
352
  onStatus?.(formatPayStatus(attempt));
@@ -370,15 +388,17 @@ export async function brain(messages, contextId, modelOverride, topK, signal) {
370
388
  contextId, topK, undefined, signal,
371
389
  );
372
390
  const j = await r.json().catch(() => ({}));
373
- const content = j?.choices?.[0]?.message?.content;
391
+ const msg = j?.choices?.[0]?.message;
392
+ const content = msg?.content;
393
+ const thinking = messageReasoning(msg);
374
394
  // Same truncation catch as the streaming path (see brainStream): a reply that
375
395
  // stops because the budget ran out is not a finished reply, and this path is
376
396
  // what non-streaming callers — including every SPAWNed subagent — go through.
377
397
  if (content && j?.choices?.[0]?.finish_reason === 'length') {
378
398
  const rest = await brainContinue(messages, content, contextId, modelOverride, 0);
379
- return content + rest;
399
+ return wrapThink(thinking, content + rest);
380
400
  }
381
- return content || (r.ok ? '' : await httpErrorNote(r.status));
401
+ return wrapThink(thinking, content || (r.ok ? '' : await httpErrorNote(r.status)));
382
402
  }
383
403
 
384
404
  /** Resume a reply that hit the output cap, non-streaming. Bounded by
@@ -420,20 +440,31 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
420
440
  if (!r.ok || !r.body) {
421
441
  // fall back to the non-streaming path rather than fail outright
422
442
  const j = await r.json().catch(() => ({}));
423
- const content = j?.choices?.[0]?.message?.content;
443
+ const msg = j?.choices?.[0]?.message;
444
+ const content = msg?.content;
445
+ const thinking = messageReasoning(msg);
424
446
  const proxied = sanitizeProxiedError(j?.error?.message);
425
447
  const text = content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status)));
448
+ if (thinking) onDelta(thinking, { think: true });
426
449
  if (text) onDelta(text);
427
- return text;
450
+ return wrapThink(thinking, text);
428
451
  }
429
452
  const reader = r.body.getReader();
430
453
  const decoder = new TextDecoder();
431
- let buf = '', full = '', reasonedChars = 0, finish = '';
454
+ let buf = '', full = '', reasonedChars = 0, reasonedText = '', finish = '';
432
455
  let stopWait = startModelWait(onStatus);
433
456
  const noteThinking = () => {
434
457
  stopWait();
435
458
  onStatus?.('thinking…');
436
459
  };
460
+ const takeReasoning = (delta, message) => {
461
+ const chunk = messageReasoning(message, delta);
462
+ if (!chunk) return;
463
+ reasonedChars += chunk.length;
464
+ reasonedText += chunk;
465
+ onDelta(chunk, { think: true });
466
+ if (!full) noteThinking();
467
+ };
437
468
  try {
438
469
  for (;;) {
439
470
  let chunk;
@@ -449,11 +480,11 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
449
480
  if (full) {
450
481
  const note = '\n\n(stream stalled — showing what arrived before the timeout)';
451
482
  onDelta(note);
452
- return full + note;
483
+ return wrapThink(reasonedText, full + note);
453
484
  }
454
485
  onStatus?.('waiting on model…');
455
486
  const fallback = await brain(messages, contextId, modelOverride, topK, signal);
456
- if (fallback) onDelta(fallback);
487
+ if (fallback) emitReply(onDelta, fallback);
457
488
  return fallback || '(stream timed out — no tokens arrived)';
458
489
  }
459
490
  const { value, done } = chunk;
@@ -478,13 +509,18 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
478
509
  full += d.content;
479
510
  onDelta(d.content);
480
511
  }
481
- // Reasoning models emit their chain of thought on a SEPARATE field and
482
- // only then start producing content. Count it — not to show it, but to
483
- // tell "the model said nothing" apart from "the model spent its whole
484
- // budget thinking and got cut off". Surface "thinking…" so the wait
485
- // is not mute dots.
486
- else if (d?.reasoning || d?.reasoning_content) {
487
- reasonedChars += (d.reasoning || d.reasoning_content).length;
512
+ // Reasoning models emit their chain of thought on a SEPARATE field
513
+ // (reasoning_content / reasoning / thinking / thought, or a provider
514
+ // object with a plaintext summary). Forward the plaintext as
515
+ // onDelta(text, { think: true }) so the canvas can fold it — never
516
+ // as visible content. Encrypted blobs stay counted for the empty-
517
+ // after-think retry, but are not forwarded.
518
+ const before = reasonedChars;
519
+ takeReasoning(d, c?.message);
520
+ if (reasonedChars === before && reasoningPresent(d, c?.message)) {
521
+ // Encrypted-only: nothing to fold, but the model DID think —
522
+ // count it so an empty completion retries instead of "(no response)".
523
+ reasonedChars += 1;
488
524
  if (!full) noteThinking();
489
525
  }
490
526
  } catch { /* keep-alive line or partial JSON — ignore */ }
@@ -523,9 +559,9 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
523
559
  onDelta, contextId, modelOverride,
524
560
  Math.min(budget * 2, MAX_CONTINUE_TOKENS), round + 1, topK, onStatus,
525
561
  );
526
- return full + (more || '');
562
+ return wrapThink(reasonedText, full + (more || ''));
527
563
  }
528
- return full;
564
+ return wrapThink(reasonedText, full);
529
565
  }
530
566
 
531
567
  // How many times a single answer may be resumed after hitting the cap. Three
@@ -618,6 +654,7 @@ export const TIER_ALIASES = {
618
654
  };
619
655
  export function normalizeTier(s) {
620
656
  const raw = String(s || '').trim().toLowerCase();
657
+ if (raw === 'auto') return 'auto';
621
658
  if (TIER_NAMES.includes(raw)) return raw;
622
659
  if (TIER_ALIASES[raw]) return TIER_ALIASES[raw];
623
660
  const compact = raw.replace(/[\s_]/g, '');
@@ -776,15 +813,16 @@ async function brainGatewayRace(messages, onDelta, contextId, models, need, maxT
776
813
  const r = await postChat(body, contextId, 0, onStatus, hooks.signal);
777
814
  if (!r.ok || !r.body) {
778
815
  const j = await r.json().catch(() => ({}));
779
- const content = j?.choices?.[0]?.message?.content;
816
+ const msg = j?.choices?.[0]?.message;
817
+ const content = msg?.content;
780
818
  const proxied = sanitizeProxiedError(j?.error?.message);
781
- const text = content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status)));
819
+ const text = wrapThink(messageReasoning(msg), content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status))));
782
820
  lastFail = { model: 'gateway', text: text || '', error: r.ok ? undefined : `HTTP ${r.status}` };
783
821
  if (isRaceCountable(lastFail)) {
784
822
  arrivals.push(lastFail);
785
823
  done.push(lastFail);
786
824
  feed.onBack(lastFail.model);
787
- if (text) onDelta(text);
825
+ if (text) emitReply(onDelta, text);
788
826
  break;
789
827
  }
790
828
  if (!shouldRetryRaceArrival(lastFail) || attempt === 1) break;
@@ -836,6 +874,7 @@ async function readGatewayRaceStream(r, feed, signal) {
836
874
  const decoder = new TextDecoder();
837
875
  let buf = '';
838
876
  const texts = new Map();
877
+ const thinks = new Map();
839
878
  const finished = new Map();
840
879
  const live = { id: null };
841
880
  const stopWait = startModelWait(() => {});
@@ -847,11 +886,21 @@ async function readGatewayRaceStream(r, feed, signal) {
847
886
  texts.set(key, (texts.get(key) || '') + chunk);
848
887
  feed.onToken(key, chunk);
849
888
  };
889
+ const pushThink = (id, chunk) => {
890
+ if (chunk == null || chunk === '') return;
891
+ const key = id || live.id || 'gateway';
892
+ live.id = key;
893
+ thinks.set(key, (thinks.get(key) || '') + chunk);
894
+ };
850
895
  const finishOne = (id, extra = {}) => {
851
896
  const key = id || live.id || 'gateway';
852
897
  if (finished.has(key)) return;
853
898
  const text = extra.text != null ? String(extra.text) : (texts.get(key) || '');
854
- const row = { model: extra.model || key, text, error: extra.error };
899
+ const row = {
900
+ model: extra.model || key,
901
+ text: wrapThink(thinks.get(key) || '', text),
902
+ error: extra.error,
903
+ };
855
904
  finished.set(key, row);
856
905
  };
857
906
 
@@ -899,7 +948,11 @@ async function readGatewayRaceStream(r, feed, signal) {
899
948
  const d = c?.delta;
900
949
  const id = raceRacerId(obj, live.id || 'gateway');
901
950
  if (d?.content) pushText(id, d.content);
902
- else if (d?.reasoning || d?.reasoning_content) { /* thinking — not content */ }
951
+ // Reasoning is not a racer preview. Fold it onto the arrival so
952
+ // the winner's canvas row can show a thinking chip — never dump
953
+ // CoT into the spectator grid.
954
+ const think = messageReasoning(c?.message, d);
955
+ if (think) pushThink(id, think);
903
956
  if (c?.finish_reason) finishOne(id, { model: obj.model });
904
957
  if (ev?.ev === 'back' || ev?.ev === 'done') finishOne(ev.id, { text: ev.text, model: ev.id });
905
958
  if (ev?.ev === 'fail' || ev?.error) finishOne(ev.id, { text: ev.text || '', error: ev.error || 'empty body' });
@@ -1066,7 +1119,10 @@ export async function brainRace(messages, onDelta, contextId, models, need = 1,
1066
1119
  for (let attempt = 0; attempt < 2; attempt++) {
1067
1120
  if (raceAbort.signal.aborted && attempt > 0) break;
1068
1121
  try {
1069
- const text = await stream(messages, (chunk) => feed.onToken(m, chunk), contextId, m, maxTokens, 0, 0, undefined, raceAbort.signal);
1122
+ const text = await stream(messages, (chunk, meta) => {
1123
+ if (meta?.think) return;
1124
+ feed.onToken(m, chunk);
1125
+ }, contextId, m, maxTokens, 0, 0, undefined, raceAbort.signal);
1070
1126
  last = { model: m, text: text == null ? '' : String(text) };
1071
1127
  if (isRaceCountable(last)) {
1072
1128
  arrivals.push(last);
@@ -1136,7 +1192,7 @@ function raceQuestion(messages) {
1136
1192
  async function classifyRaceAnswer(messages, cand) {
1137
1193
  const prompt = 'Score this answer to one question from 0 to 10.\n\n'
1138
1194
  + 'QUESTION:\n' + String(raceQuestion(messages)).slice(0, 4000) + '\n\n'
1139
- + 'ANSWER:\n' + String(cand?.text || '').slice(0, 6000) + '\n\n'
1195
+ + 'ANSWER:\n' + stripThinkTags(String(cand?.text || '')).slice(0, 6000) + '\n\n'
1140
1196
  + 'Judge on: correctness first, then completeness, then whether it actually did what was asked '
1141
1197
  + '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
1142
1198
  + 'Ignore length and confidence of tone.\n'
@@ -1155,7 +1211,7 @@ async function pairwiseTied(messages, tied) {
1155
1211
  const letters = tied.map((_, i) => String.fromCharCode(65 + i));
1156
1212
  const prompt = 'You are judging answers to one question. Pick the single best one.\n\n'
1157
1213
  + 'QUESTION:\n' + String(raceQuestion(messages)).slice(0, 4000) + '\n\n'
1158
- + tied.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + String(c.text || '').slice(0, 6000)).join('\n\n')
1214
+ + tied.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + stripThinkTags(String(c.text || '')).slice(0, 6000)).join('\n\n')
1159
1215
  + '\n\nJudge on: correctness first, then completeness, then whether it actually did what was asked '
1160
1216
  + '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
1161
1217
  + 'Ignore length and confidence of tone.\n'
package/lib/proxy.js CHANGED
@@ -20,7 +20,11 @@ import {
20
20
  corpusRecall,
21
21
  decideChatSpill, isOneShotCorpusAsk,
22
22
  } from './spill.js';
23
- import { rewritablePath, augmentModelList, ALIAS_IDS, rewriteChatModel, zooModelIds, CLASSIFY_MAX_TOKENS } from './models.js';
23
+ import { rewritablePath, modelsListForRequest, isHarnessAliasId, rewriteChatModel, zooModelIds, CLASSIFY_MAX_TOKENS, raiseReasoningMaxTokens, isAutoModel } from './models.js';
24
+ import {
25
+ route as routeTask, routeChatBody, fallbackChain, isRetryableStatus,
26
+ outcomeFromResponse, recordRouteOutcome, autoModelListEntry,
27
+ } from './modelroute.js';
24
28
  import { forgetContext } from './contexts.js';
25
29
  import { injectBrief } from './brief.js';
26
30
  import { withNamespace } from './namespace.js';
@@ -30,8 +34,10 @@ import { loadSessionSpend, saveSessionSpend } from './session.js';
30
34
  import { creditBalance, quotedPrices } from './info.js';
31
35
  import { subscriptionPublicView } from './subscription.js';
32
36
  import { priceHoldings } from './livestatus.js';
33
- import { receiptUsedCogs, receiptDirectUsd } from './racesettle.js';
37
+ import { receiptUsedCogs, receiptDirectUsd, pairActualBilled } from './racesettle.js';
34
38
  import { rewriteWrapClientError } from './wrap.js';
39
+ import { fetchHeaders } from './fetch.js';
40
+ import { relay } from './relay.js';
35
41
 
36
42
  const HOP_BY_HOP = new Set([
37
43
  'host', 'connection', 'keep-alive', 'transfer-encoding', 'upgrade',
@@ -81,46 +87,6 @@ async function readBody(req) {
81
87
  return Buffer.concat(chunks);
82
88
  }
83
89
 
84
- /** Pipe an upstream fetch Response to the client, unbuffered (SSE-safe).
85
- *
86
- * `onReceipt` is called with the gateway's x402 block when it arrives. On a
87
- * STREAMED call there is no JSON body to carry that block, so the gateway
88
- * emits it as an SSE COMMENT (`: x402 {...}`) after the last frame — comments
89
- * are discarded by every compliant client, so nothing downstream sees it, but
90
- * without reading it here every spend and savings figure on the status line
91
- * would silently read zero the moment real streaming was switched on.
92
- *
93
- * Sniffing NEVER delays a byte: each chunk is written to the client first and
94
- * only then scanned. */
95
- function relay(res, upstream, onReceipt) {
96
- const headers = {};
97
- upstream.headers.forEach((v, k) => {
98
- if (!['transfer-encoding', 'connection', 'content-encoding', 'content-length'].includes(k)) headers[k] = v;
99
- });
100
- res.writeHead(upstream.status, headers);
101
- if (!upstream.body) { res.end(); return Promise.resolve(); }
102
- const sse = (upstream.headers.get('content-type') || '').includes('text/event-stream');
103
- return new Promise((resolve) => {
104
- const body = Readable.fromWeb(upstream.body);
105
- body.on('error', () => res.destroy());
106
- res.on('close', () => body.destroy());
107
- body.on('end', resolve);
108
- if (!sse || typeof onReceipt !== 'function') { body.pipe(res); return; }
109
- let pending = '';
110
- body.on('data', (c) => {
111
- res.write(c);
112
- pending += c.toString('utf8');
113
- const lines = pending.split('\n');
114
- pending = lines.pop() ?? ''; // a comment can straddle two chunks
115
- for (const line of lines) {
116
- if (!line.startsWith(': x402 ')) continue;
117
- try { onReceipt(JSON.parse(line.slice(7))); } catch { /* not our frame */ }
118
- }
119
- });
120
- body.on('end', () => res.end());
121
- });
122
- }
123
-
124
90
  function jsonErr(res, status, message, extraFields = {}) {
125
91
  res.writeHead(status, { 'content-type': 'application/json' });
126
92
  res.end(JSON.stringify({ error: { message }, ...extraFields }));
@@ -813,6 +779,9 @@ async function maybeCacheCorpus(req, bodyBuf, log, stats, extra = {}) {
813
779
  * default, so localhost behaviour is unchanged.
814
780
  */
815
781
  export async function startProxy({ silent = false, requireToken = null, sessionMaxUsd = null, autoTunnel = false } = {}) {
782
+ // grokui's packed sidecar spawn sets OPENZOO_SILENT=1 so 400s land in
783
+ // ~/.openzoo/proxy.log instead of Electron's inherited /dev/null.
784
+ if (process.env.OPENZOO_SILENT === '1') silent = true;
816
785
  const client = new PayClient();
817
786
  const log = silent ? () => {} : (...a) => console.log(...a);
818
787
  // ALWAYS-ON. `silent: true` is used by the editor path to keep startup tidy,
@@ -1196,6 +1165,35 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1196
1165
  return;
1197
1166
  }
1198
1167
 
1168
+ // Local decision � no upstream, no payment. Same keys as Python route().
1169
+ {
1170
+ const routePath = (req.url || '').split('?')[0];
1171
+ if (req.method === 'POST' && (routePath === '/route' || routePath === '/v1/route')) {
1172
+ let payload;
1173
+ try { payload = JSON.parse(bodyBuf.toString('utf8') || '{}'); } catch {
1174
+ jsonErr(res, 400, 'invalid /route body');
1175
+ return;
1176
+ }
1177
+ try {
1178
+ const r = routeTask(payload.text ?? '', {
1179
+ allow_free: payload.allow_free ?? false,
1180
+ bindable: payload.bindable ?? true,
1181
+ context: payload.context,
1182
+ input_tokens: payload.input_tokens,
1183
+ has_image: payload.has_image,
1184
+ needs_tools: payload.needs_tools,
1185
+ needs_json: payload.needs_json,
1186
+ ...(payload.constraints && typeof payload.constraints === 'object' ? payload.constraints : {}),
1187
+ });
1188
+ res.writeHead(200, { 'content-type': 'application/json' });
1189
+ res.end(JSON.stringify(r));
1190
+ } catch (err) {
1191
+ jsonErr(res, 500, `openzoo /route: ${err.message}`);
1192
+ }
1193
+ return;
1194
+ }
1195
+ }
1196
+
1199
1197
  // ANTHROPIC MESSAGES SHAPE. A harness pointed here via ANTHROPIC_BASE_URL
1200
1198
  // (Claude Code, the Anthropic SDKs) speaks POST /v1/messages, not chat
1201
1199
  // completions — this is how such a harness routes its inference through
@@ -1298,6 +1296,7 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1298
1296
  // carries a model field, not just chat/completions, so /completions,
1299
1297
  // /responses and future shapes all work. Never silent.
1300
1298
  let wantsStream = false;
1299
+ let autoRoute = null;
1301
1300
  if ((req.url || '').includes('/chat/completions') && req.method === 'POST') {
1302
1301
  servedRequests += 1;
1303
1302
  say(`\n<- request #${servedRequests} from ${(req.headers['user-agent'] || 'unknown').slice(0, 40)}`);
@@ -1331,6 +1330,26 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1331
1330
  try { ids = await zooModelIds(); } catch { /* catalog miss: still skip the floor on a tiny classify */ }
1332
1331
  const policy = rewriteChatModel(parsed, ids, { bodyLen: bodyBuf.length });
1333
1332
  parsed = policy.parsed;
1333
+ if (!policy.tiny && (policy.auto || isAutoModel(parsed?.model))) {
1334
+ autoRoute = routeChatBody(parsed, {
1335
+ allow_free: false,
1336
+ bindable: true,
1337
+ // Live quoteable ids only � Auto must never emit :batch / $0 /
1338
+ // missing-price rows that 500 `bad openrouter price` on Fly.
1339
+ ...(ids.length ? { allow_ids: ids } : {}),
1340
+ });
1341
+ if (!autoRoute.model) {
1342
+ jsonErr(res, 422, autoRoute.reason || 'openzoo/auto: no feasible model', { route: autoRoute });
1343
+ return;
1344
+ }
1345
+ parsed = { ...parsed, model: autoRoute.model };
1346
+ const bump = raiseReasoningMaxTokens(parsed);
1347
+ parsed = bump.parsed;
1348
+ say(`openzoo/auto -> ${autoRoute.model} p=${autoRoute.p_success} cleared=${autoRoute.cleared_bar} ${autoRoute.task_class}${autoRoute.bind_first ? ' bind_first' : ''}`);
1349
+ if (!autoRoute.cleared_bar) {
1350
+ say(`openzoo/auto cleared_bar=false � strongest fallback, not a normal pick`);
1351
+ }
1352
+ }
1334
1353
  if (policy.tiny) {
1335
1354
  // `openzoo claude` starts us silent � `log` is a no-op then. say()
1336
1355
  // is the proxy.log channel and never the Claude Code TTY.
@@ -1380,7 +1399,7 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1380
1399
  log(`reasoning model ${parsed.model}: max_tokens ${policy.raisedFrom} -> ${policy.raisedTo} (thinking shares the budget; OPENZOO_REASONING_MAX_TOKENS_X=1 disables)`);
1381
1400
  }
1382
1401
  }
1383
- if (policy.tiny || policy.raised || (policy.to && policy.to !== policy.from)) {
1402
+ if (policy.tiny || policy.raised || autoRoute || (policy.to && policy.to !== policy.from)) {
1384
1403
  bodyBuf = Buffer.from(JSON.stringify(parsed));
1385
1404
  }
1386
1405
  // Tell the agent what it is actually connected to — in band, where it
@@ -1501,16 +1520,32 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1501
1520
  // rewrite never gets its chance.
1502
1521
  const path = (req.url || '').split('?')[0];
1503
1522
  if (req.method === 'GET' && path === '/v1/models') {
1523
+ // Catalog is chrome, not a paid call. Paying the list would wrap-walk
1524
+ // an empty burner (~4.5s/row) and stall first paint / harness probe.
1504
1525
  try {
1505
- const { response } = await client.fetch(url, init);
1506
- const payload = await response.json();
1507
- res.writeHead(response.status, { 'content-type': 'application/json' });
1508
- res.end(JSON.stringify(response.ok ? augmentModelList(payload) : payload));
1509
- return;
1510
- } catch { /* fall through to the plain relay below */ }
1526
+ const response = await fetchHeaders(url, init);
1527
+ if (response.ok) {
1528
+ const payload = await response.json();
1529
+ // Quoteable catalog only. Raw OpenRouter dump includes :batch,
1530
+ // $0 / missing prices, and we used to mint openzoo-* twins of each
1531
+ // � Claude Code /model then showed 63 clones and Auto 500'd.
1532
+ res.writeHead(200, { 'content-type': 'application/json' });
1533
+ res.end(JSON.stringify(modelsListForRequest(payload, req.headers)));
1534
+ return;
1535
+ }
1536
+ await response.text().catch(() => {});
1537
+ } catch { /* gateway 402/down � serve aliases so chrome still paints */ }
1538
+ res.writeHead(200, { 'content-type': 'application/json' });
1539
+ res.end(JSON.stringify(modelsListForRequest({ object: 'list', data: [] }, req.headers)));
1540
+ return;
1511
1541
  }
1512
1542
  const probe = req.method === 'GET' && /^\/v1\/models\/(.+)$/.exec(path);
1513
- if (probe && ALIAS_IDS.includes(decodeURIComponent(probe[1]))) {
1543
+ if (probe && isAutoModel(decodeURIComponent(probe[1]))) {
1544
+ res.writeHead(200, { 'content-type': 'application/json' });
1545
+ res.end(JSON.stringify(autoModelListEntry()));
1546
+ return;
1547
+ }
1548
+ if (probe && isHarnessAliasId(decodeURIComponent(probe[1]))) {
1514
1549
  res.writeHead(200, { 'content-type': 'application/json' });
1515
1550
  res.end(JSON.stringify({ id: decodeURIComponent(probe[1]), object: 'model', owned_by: 'openzoo-alias' }));
1516
1551
  return;
@@ -1580,6 +1615,32 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1580
1615
  } else {
1581
1616
  result = await client.fetch(url, init);
1582
1617
  }
1618
+ // Walk the auto shortlist on 429/5xx � cheapest-first among models that
1619
+ // cleared the bar. Finite: one pass over fallbackChain(), never a loop.
1620
+ if (autoRoute?.cleared_bar && isRetryableStatus(result.response.status)) {
1621
+ const chain = fallbackChain(autoRoute);
1622
+ for (const next of chain) {
1623
+ const failed = result.response.status;
1624
+ try { await result.response.arrayBuffer(); } catch { /* drain */ }
1625
+ say(`openzoo/auto fallback HTTP ${failed} -> ${next}`);
1626
+ const rewriteModel = (buf) => {
1627
+ try {
1628
+ const b = JSON.parse(buf.toString('utf8'));
1629
+ b.model = next;
1630
+ return Buffer.from(JSON.stringify(b));
1631
+ } catch { return buf; }
1632
+ };
1633
+ bodyBuf = rewriteModel(bodyBuf);
1634
+ init.body = bodyBuf;
1635
+ if (cached) {
1636
+ cached = { ...cached, body: rewriteModel(cached.body) };
1637
+ result = await send(cached.body, cached.contextId, cached.topK, corpusCharsForSend(boundChars, cached.contextId, cached.corpus?.length));
1638
+ } else {
1639
+ result = await client.fetch(url, { ...init, body: bodyBuf });
1640
+ }
1641
+ if (!isRetryableStatus(result.response.status)) break;
1642
+ }
1643
+ }
1583
1644
  const { response, paid, receipt } = result;
1584
1645
  if (paid && receipt) {
1585
1646
  if (receipt.ok && typeof receipt.billedUsd === 'number') {
@@ -1624,6 +1685,13 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1624
1685
  // that contract ourselves. An upstream that someday truly streams (SSE
1625
1686
  // content-type) passes straight through the relay below, untouched.
1626
1687
  const upCt = response.headers.get('content-type') || '';
1688
+ const noteAutoOutcome = (status, data, streamed = false) => {
1689
+ if (!autoRoute) return;
1690
+ let ok = outcomeFromResponse(status, data);
1691
+ if (ok == null) return;
1692
+ if (streamed && status >= 200 && status < 300) ok = true;
1693
+ recordRouteOutcome(autoRoute, ok);
1694
+ };
1627
1695
  if (isChat && response.ok && upCt.includes('application/json')) {
1628
1696
  let data = null;
1629
1697
  try { data = await response.clone().json(); } catch { /* not JSON after all */ }
@@ -1631,18 +1699,26 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1631
1699
  // completion already — no extra call, and unlike the account-level
1632
1700
  // /api/v1/credits total it is attributable to THIS proxy even though the
1633
1701
  // same OpenRouter key also pays for ttfx and everything else.
1634
- if (typeof data?.usage?.cost === 'number' && data.usage.cost >= 0) {
1635
- sessionActual += data.usage.cost;
1636
- actualCalls += 1;
1702
+ {
1637
1703
  // PAIR THE NUMERATOR WITH THE DENOMINATOR. sessionSpent is summed on
1638
1704
  // three paths and sessionActual on two, so markupX divided ALL billed
1639
- // by the SUBSET that reported a real cost — a 402-receipt call added
1705
+ // by the SUBSET that reported a real cost � a 402-receipt call added
1640
1706
  // to billed and nothing to real, and the ratio read 12.55x on a stack
1641
1707
  // running at ~1.0x. Track the billed side of exactly the calls whose
1642
1708
  // cost we actually learned.
1643
1709
  // Both figures ride the SAME response object, so read them together
1644
1710
  // rather than carrying one across sites and hoping the order holds.
1645
- billedWithActual += Number(data?.x402?.billedUsd) || 0;
1711
+ //
1712
+ // x402.billedUsd is often the QUOTE reserve (max_tokens � catalog),
1713
+ // not the settled charge. MEASURED: $0.9858 reserved vs $0.007962
1714
+ // usage.cost -> markupX lied at 124x on a ~1x call. Pair usage.cost
1715
+ // with post-completion billed, never the reserve.
1716
+ const pair = pairActualBilled(data?.x402, data?.usage);
1717
+ if (pair) {
1718
+ sessionActual += pair.upstreamUsd;
1719
+ actualCalls += 1;
1720
+ billedWithActual += pair.billedUsd;
1721
+ }
1646
1722
  }
1647
1723
  // PREPAID CALLS STILL COST MONEY. The block above only meters calls
1648
1724
  // where THIS proxy answered a 402 and paid. When prepaid credit covers
@@ -1674,6 +1750,7 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1674
1750
  say(`credit -> $${x.billedUsd.toFixed(6)} · session $${sessionSpent.toFixed(6)}`);
1675
1751
  }
1676
1752
  if (data?.object === 'chat.completion') {
1753
+ noteAutoOutcome(response.status, data);
1677
1754
  if (rKey) replayPut(rKey, data, response.headers.get('x-payment-response'));
1678
1755
  // Anthropic-shaped caller gets an Anthropic-shaped answer, streamed
1679
1756
  // or not, so Claude Code and the SDKs parse it natively.
@@ -1714,12 +1791,20 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1714
1791
  // same figures the JSON path reads out of `data.x402`, same counters, so
1715
1792
  // the status line does not care which transport served the answer.
1716
1793
  const meterStreamed = (x) => {
1794
+ // Same pairing rule as the JSON path: actualUsd / usage.cost with the
1795
+ // settled billed twin, even on a wallet-paid stream (do not skip just
1796
+ // because `paid` already recorded the quote-time receipt).
1797
+ const pair = pairActualBilled(x, x?.usage);
1798
+ if (pair) {
1799
+ sessionActual += pair.upstreamUsd;
1800
+ actualCalls += 1;
1801
+ billedWithActual += pair.billedUsd;
1802
+ }
1717
1803
  if (paid || typeof x?.billedUsd !== 'number') return;
1718
1804
  sessionSpent += x.billedUsd;
1719
1805
  sessionCogs += receiptUsedCogs(x);
1720
1806
  noteQuote(x);
1721
1807
  sessionDirect += receiptDirectUsd(x);
1722
- if (typeof x.actualUsd === 'number' && x.actualUsd >= 0) { sessionActual += x.actualUsd; actualCalls += 1; billedWithActual += x.billedUsd || 0; }
1723
1808
  if (didSpill) {
1724
1809
  log(spillPricedLine(x, { streamed: true }));
1725
1810
  spill.spillSpend += x.billedUsd;
@@ -1764,6 +1849,9 @@ export async function startProxy({ silent = false, requireToken = null, sessionM
1764
1849
  return;
1765
1850
  }
1766
1851
 
1852
+ if (autoRoute) {
1853
+ noteAutoOutcome(response.status, null, (upCt.includes('text/event-stream') && response.ok));
1854
+ }
1767
1855
  await relay(res, response, meterStreamed);
1768
1856
  } catch (err) {
1769
1857
  if (err instanceof QuoteTooHighError) {