openzoo 0.49.8 → 0.49.9

This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
package/lib/models.js CHANGED
@@ -1,5 +1,11 @@
1
1
  import { config } from './config.js';
2
2
  import { fetchHeaders } from './fetch.js';
3
+ import {
4
+ AUTO_MODEL_ID, autoHasPricedModels, autoModelListEntry, isAutoModel,
5
+ isPricedTokenPair, isUnservableRouteId,
6
+ } from './modelroute.js';
7
+
8
+ export { AUTO_MODEL_ID, isAutoModel, isPricedTokenPair, isUnservableRouteId };
3
9
 
4
10
  /** Same threshold as BIND_MIN_CHARS in hrr.js — kept local so this
5
11
  * module stays importable without the wallet/rpc stack. */
@@ -50,18 +56,43 @@ export function editorSlot(id) {
50
56
  }
51
57
 
52
58
  const CATALOG_TTL_MS = 5 * 60 * 1000;
53
- let cache = { at: 0, ids: null };
59
+ let cache = { at: 0, ids: null, base: null };
60
+
61
+ export function resetZooModelIdsCache() {
62
+ cache = { at: 0, ids: null, base: null };
63
+ }
54
64
 
55
65
  export async function zooModelIds() {
56
- if (cache.ids && Date.now() - cache.at < CATALOG_TTL_MS) return cache.ids;
66
+ if (cache.ids && cache.base === config.apiBase && Date.now() - cache.at < CATALOG_TTL_MS) return cache.ids;
57
67
  const r = await fetchHeaders(`${config.apiBase}/v1/models`);
58
68
  if (!r.ok) throw new Error(`model catalog fetch failed: HTTP ${r.status}`);
59
69
  const d = await r.json();
60
- const ids = (d.data || []).map((m) => m.id).filter(Boolean);
61
- if (ids.length) cache = { at: Date.now(), ids };
70
+ const ids = quoteableRows(d.data).map((m) => m.id).filter((id) => id && !isAutoModel(id));
71
+ if (ids.length) cache = { at: Date.now(), ids, base: config.apiBase };
62
72
  return ids;
63
73
  }
64
74
 
75
+ /** OpenRouter / gateway token price pair. Image/video rows have no prompt. */
76
+ export function tokenPricePair(pricing) {
77
+ if (!pricing || typeof pricing !== 'object') return [NaN, NaN];
78
+ return [Number(pricing.prompt ?? pricing.input), Number(pricing.completion ?? pricing.output)];
79
+ }
80
+
81
+ /**
82
+ * A row the gateway can actually quote for chat. Drops :batch, ~latest
83
+ * pointers, openzoo-* twins, $0 / missing / non-token OpenRouter prices.
84
+ */
85
+ export function isQuoteableModel(m) {
86
+ const id = m?.id;
87
+ if (isAutoModel(id)) return true;
88
+ if (isUnservableRouteId(id)) return false;
89
+ return isPricedTokenPair(...tokenPricePair(m?.pricing));
90
+ }
91
+
92
+ export function quoteableRows(data) {
93
+ return (Array.isArray(data) ? data : []).filter(isQuoteableModel);
94
+ }
95
+
65
96
  /** Vendor fingerprints in harness model ids → zoo catalog prefixes. Order
66
97
  * matters only for overlapping hints; first match wins. */
67
98
  const FAMILIES = [
@@ -119,6 +150,14 @@ const tokensOf = (id) => id.toLowerCase().split(/[^a-z0-9.]+/).filter((t) => t &
119
150
  * OPENZOO_DEFAULT_MODEL is an explicit user override, not a fallback tier.
120
151
  */
121
152
  export function resolveModel(requested, ids) {
153
+ // Virtual router id — never family-match, never steal via OPENZOO_DEFAULT_MODEL.
154
+ if (isAutoModel(requested)) return null;
155
+ // Bare Anthropic / Claude Code ids are never live on Fly/OpenRouter
156
+ // (`claude-opus-5` → 500 unknown model). Rewrite even on a catalog miss
157
+ // or if a gateway row lists the bare name — the request must not leave
158
+ // the sidecar as that id.
159
+ const native = anthropicNativeAlias(requested);
160
+ if (native) return native;
122
161
  if (!requested || !ids?.length || ids.includes(requested)) return null;
123
162
  // `openzoo-` prefixed names exist so an editor cannot mistake them for its
124
163
  // OWN models: Cursor claims any name in its catalog (claude-opus-5, grok-4.6)
@@ -187,6 +226,46 @@ export function resolveModel(requested, ids) {
187
226
  return best;
188
227
  }
189
228
 
229
+ /**
230
+ * Bare Anthropic / Claude Code ids. MEASURED 2026-08-20 against Fly
231
+ * x402-tokens: POST /v1/chat/completions model=claude-opus-5 → 500
232
+ * `unknown model claude-opus-5`. The vendor-prefixed twin
233
+ * (`anthropic/claude-opus-5`) 402s and is priced. Claude Code's Auto
234
+ * permission classifier calls these ids on ANTHROPIC_BASE_URL; if the
235
+ * sidecar forwards the bare name, Bash hangs with
236
+ * "claude-opus-5 is temporarily unavailable, so auto mode cannot
237
+ * determine the safety of Bash". Always rewrite. Never send the bare
238
+ * id to OpenRouter / Fly. Do not push x402-tokens — alias here.
239
+ */
240
+ export const ANTHROPIC_NATIVE_ALIASES = {
241
+ 'claude-opus-5': 'anthropic/claude-opus-5',
242
+ 'claude-opus-5-fast': 'anthropic/claude-opus-5-fast',
243
+ 'claude-3-5-opus': 'anthropic/claude-opus-5',
244
+ 'claude-sonnet-5': 'anthropic/claude-sonnet-5',
245
+ 'claude-sonnet-5-fast': 'anthropic/claude-sonnet-5',
246
+ 'claude-opus-4-8': 'anthropic/claude-opus-4.8',
247
+ 'claude-opus-4.8': 'anthropic/claude-opus-4.8',
248
+ 'claude-fable-5': 'anthropic/claude-fable-5',
249
+ 'claude-haiku-4.5': 'anthropic/claude-haiku-4.5',
250
+ };
251
+
252
+ /** Claude Code sometimes suffixes a window marker (`claude-opus-5[1m]`). */
253
+ function stripAnthropicWindowSuffix(id) {
254
+ return String(id || '').trim().replace(/\[[\d]+m\]$/i, '');
255
+ }
256
+
257
+ /**
258
+ * Priced zoo twin for a bare Anthropic / Claude Code id, or null.
259
+ * Vendor-prefixed ids and openzoo-* twins are left to the rest of resolveModel.
260
+ */
261
+ export function anthropicNativeAlias(requested) {
262
+ if (!requested || typeof requested !== 'string') return null;
263
+ const stripped = stripAnthropicWindowSuffix(requested);
264
+ if (!stripped || stripped.includes('/')) return null;
265
+ if (/^openzoo[-/]/i.test(stripped)) return null;
266
+ return ANTHROPIC_NATIVE_ALIASES[stripped] || ANTHROPIC_NATIVE_ALIASES[stripped.toLowerCase()] || null;
267
+ }
268
+
190
269
  /**
191
270
  * Ids harnesses ship as DEFAULTS (Cursor, Continue, Aider, Codex CLI, Cline,
192
271
  * OpenClaw, LangChain templates…). Merged into GET /v1/models so a harness
@@ -197,58 +276,51 @@ export const ALIAS_IDS = [
197
276
  'gpt-4o', 'gpt-4o-mini', 'gpt-4.1', 'gpt-4.1-mini', 'gpt-4-turbo', 'gpt-3.5-turbo',
198
277
  'gpt-5', 'gpt-5-mini', 'chatgpt-4o-latest', 'o1', 'o3', 'o3-mini', 'o4-mini',
199
278
  'claude-3-5-sonnet-latest', 'claude-sonnet-4-0', 'claude-opus-4-1',
279
+ 'claude-opus-5', 'claude-opus-5-fast', 'claude-3-5-opus', 'claude-sonnet-5',
280
+ 'claude-opus-4-8', 'claude-fable-5',
200
281
  'gemini-2.5-pro', 'gemini-2.5-flash', 'grok-4', 'grok-3',
201
282
  'deepseek-chat', 'deepseek-reasoner', 'qwen-max', 'llama-3.3-70b',
202
283
  ];
203
284
 
285
+ export function isHarnessAliasId(id) {
286
+ const stripped = stripAnthropicWindowSuffix(id);
287
+ return ALIAS_IDS.includes(stripped) || Boolean(anthropicNativeAlias(stripped));
288
+ }
289
+
204
290
  /**
205
291
  * Merge alias rows into a /v1/models payload without duplicating real ids.
206
292
  * Each alias inherits context_length and pricing from the model it RESOLVES
207
293
  * to — a harness sizing its corpus off "gpt-4o" gets the real ceiling of the
208
294
  * model that will actually serve it, not a blank.
295
+ *
296
+ * Does not mint openzoo-* twins. Those duplicated every real id (and every
297
+ * :batch id) in Claude Code's /model picker.
209
298
  */
210
- export function augmentModelList(payload) {
211
- const data = Array.isArray(payload?.data) ? payload.data : [];
299
+ export function augmentModelList(payload, { aliases: withAliases = true } = {}) {
300
+ const data = quoteableRows(payload?.data);
212
301
  const have = new Set(data.map((m) => m.id));
213
302
  const ids = data.map((m) => m.id);
214
- // openzoo-* twins of the popular models. An editor that validates a custom
215
- // model against THIS list (Cursor's "Add model" box reports "No models
216
- // available" for anything missing here) can only offer what we publish — and
217
- // the openzoo- prefix is what stops it claiming the name as one of its own
218
- // built-ins and routing to its backend instead of to us.
219
- const branded = [];
220
- for (const src0 of data) {
221
- const id = src0.id;
222
- // Brand only REAL upstream models. Anything we synthesised (a twin or a
223
- // harness alias) must be skipped, or augmenting an already-augmented
224
- // payload mints openzoo-openzoo-* and the catalog grows every pass.
225
- if (!id || id.startsWith('openzoo-') || String(src0.owned_by || '').startsWith('openzoo')) continue;
226
- const short = id.includes('/') ? id.split('/')[1] : id;
227
- const name = `openzoo-${short}`;
228
- if (have.has(name)) continue;
229
- const src = data.find((m) => m.id === id);
230
- branded.push({
231
- id: name,
232
- object: 'model',
233
- owned_by: 'openzoo',
234
- served_by: id,
235
- ...(src?.context_length ? { context_length: src.context_length, context_window: src.context_window ?? src.context_length } : {}),
236
- ...(src?.pricing ? { pricing: src.pricing } : {}),
237
- });
238
- have.add(name);
303
+ const aliases = withAliases
304
+ ? ALIAS_IDS.filter((id) => !have.has(id)).map((id) => {
305
+ const target = data.find((m) => m.id === resolveModel(id, ids));
306
+ return {
307
+ id,
308
+ object: 'model',
309
+ owned_by: 'openzoo-alias',
310
+ ...(target?.context_length ? { context_length: target.context_length, context_window: target.context_window ?? target.context_length } : {}),
311
+ ...(target?.pricing ? { pricing: target.pricing } : {}),
312
+ ...(target ? { served_by: target.id } : {}),
313
+ };
314
+ })
315
+ : [];
316
+ const virtual = [];
317
+ // Auto is listed only when its own shortlist is priced models — never an
318
+ // unquoted OpenRouter id that 500s `bad openrouter price`.
319
+ if (!have.has(AUTO_MODEL_ID) && autoHasPricedModels(undefined, ids.length ? ids : null)) {
320
+ virtual.push(autoModelListEntry());
321
+ have.add(AUTO_MODEL_ID);
239
322
  }
240
- const aliases = ALIAS_IDS.filter((id) => !have.has(id)).map((id) => {
241
- const target = data.find((m) => m.id === resolveModel(id, ids));
242
- return {
243
- id,
244
- object: 'model',
245
- owned_by: 'openzoo-alias',
246
- ...(target?.context_length ? { context_length: target.context_length, context_window: target.context_window ?? target.context_length } : {}),
247
- ...(target?.pricing ? { pricing: target.pricing } : {}),
248
- ...(target ? { served_by: target.id } : {}),
249
- };
250
- });
251
- return { ...payload, object: payload?.object || 'list', data: [...data, ...branded, ...aliases] };
323
+ return { ...payload, object: payload?.object || 'list', data: [...virtual, ...data, ...aliases] };
252
324
  }
253
325
 
254
326
  /**
@@ -277,14 +349,12 @@ function decorateModelEntry(m) {
277
349
  }
278
350
 
279
351
  /**
280
- * OpenAI-compatible /v1/models body Claude Code and every other harness can
281
- * read. The full zoo catalog is kept — no claude-* filter, no single opus-5
282
- * collapse. Extra Anthropic fields (`type`, `display_name`) sit alongside
283
- * OpenAI ones so a Messages-API client can label rows without a second
284
- * endpoint. Existing OpenAI clients ignore the extras.
352
+ * OpenAI-compatible /v1/models body. Quoteable chat models only — no :batch,
353
+ * no unpriced / image-video rows, no openzoo-* twins. Extra Anthropic fields
354
+ * (`type`, `display_name`) sit alongside OpenAI ones.
285
355
  */
286
- export function publishModelList(payload) {
287
- const merged = augmentModelList(payload);
356
+ export function publishModelList(payload, opts) {
357
+ const merged = augmentModelList(payload, opts);
288
358
  const data = (merged.data || []).map(decorateModelEntry);
289
359
  return {
290
360
  ...merged,
@@ -296,21 +366,37 @@ export function publishModelList(payload) {
296
366
  };
297
367
  }
298
368
 
299
- /**
300
- * Anthropic GET /v1/models shape (id + display_name + type).
301
- * Claude Code gateway discovery reads `data[].id` and optional `display_name`.
302
- * Ids are the zoo's real catalog ids — never rewritten to fake `claude-*`
303
- * aliases for grok / deepseek / gemini / etc.
304
- */
305
- export function anthropicModelList(payload) {
306
- const published = publishModelList(payload);
307
- const data = (published.data || []).map((m) => ({
369
+ function anthropicRows(rows) {
370
+ return rows.map((m) => ({
308
371
  type: 'model',
309
372
  id: m.id,
310
373
  display_name: m.display_name || displayNameFor(m.id),
311
374
  ...(m.created_at ? { created_at: m.created_at } : {}),
312
375
  ...(m.served_by ? { served_by: m.served_by } : {}),
313
376
  }));
377
+ }
378
+
379
+ /** openzoo/auto first so Claude Code's picker default is the router, not opus. */
380
+ function withAutoFirst(rows) {
381
+ const list = Array.isArray(rows) ? rows.slice() : [];
382
+ const i = list.findIndex((m) => isAutoModel(m?.id));
383
+ if (i > 0) {
384
+ const [auto] = list.splice(i, 1);
385
+ list.unshift(auto);
386
+ }
387
+ return list;
388
+ }
389
+
390
+ /**
391
+ * Anthropic GET /v1/models shape (id + display_name + type).
392
+ * Claude Code gateway discovery reads `data[].id` and optional `display_name`.
393
+ * Full quoteable catalog — every priced published id, including OpenRouter
394
+ * grok/gemini/gpt rows. Not a 4-id Anthropic cap, not openzoo-* clones.
395
+ * Picker ≠ classifier: rewriteChatModel still pins 16-token classify off opus-5.
396
+ */
397
+ export function anthropicModelList(payload) {
398
+ const published = publishModelList(payload, { aliases: false });
399
+ const data = anthropicRows(withAutoFirst(published.data || []));
314
400
  return {
315
401
  data,
316
402
  has_more: false,
@@ -329,12 +415,14 @@ export function wantsAnthropicModelList(headers = {}) {
329
415
  }
330
416
 
331
417
  /**
332
- * Body for GET /v1/models. Always the full zoo catalog (never opus-5-only).
333
- * Anthropic-shaped clients get id+display_name; OpenAI clients keep object:list.
418
+ * Body for GET /v1/models. Quoteable catalog only (never opus-5-only, never
419
+ * unpriced / :batch / openzoo-* clones). Anthropic-shaped clients get the
420
+ * same quoteable ids in Anthropic shape (type / id / display_name) so Claude
421
+ * Code can select grok, gemini, gpt, etc. OpenAI clients keep object:list
422
+ * of every quoteable id + harness aliases.
334
423
  */
335
424
  export function modelsListForRequest(payload, headers) {
336
- const published = publishModelList(payload);
337
- return wantsAnthropicModelList(headers) ? anthropicModelList(published) : published;
425
+ return wantsAnthropicModelList(headers) ? anthropicModelList(payload) : publishModelList(payload);
338
426
  }
339
427
 
340
428
  /**
@@ -471,6 +559,9 @@ export function raiseReasoningMaxTokens(parsed, env = process.env) {
471
559
  export function rewriteChatModel(parsed, ids, { bodyLen } = {}) {
472
560
  const from = parsed?.model;
473
561
  const len = bodyLen ?? (parsed == null ? 0 : Buffer.byteLength(JSON.stringify(parsed)));
562
+ if (isAutoModel(from) && !isTinyClassify(parsed, len)) {
563
+ return { parsed, tiny: false, auto: true, from, to: AUTO_MODEL_ID, raised: false };
564
+ }
474
565
  if (isTinyClassify(parsed, len)) {
475
566
  const picked = pickClassifierModel(ids);
476
567
  // pickClassifierModel returns null on an empty catalog or a zoo that
@@ -489,7 +580,7 @@ export function rewriteChatModel(parsed, ids, { bodyLen } = {}) {
489
580
  if (typeof from !== 'string') {
490
581
  return { parsed, tiny: false, from, to: from, raised: false };
491
582
  }
492
- const resolved = resolveModel(from, ids);
583
+ const resolved = resolveModel(from, ids) || anthropicNativeAlias(from);
493
584
  const next = resolved ? { ...parsed, model: resolved } : parsed;
494
585
  const bump = raiseReasoningMaxTokens(next);
495
586
  return {
@@ -0,0 +1,3 @@
1
+ {
2
+ "type": "module"
3
+ }
package/lib/podagent.mjs CHANGED
@@ -28,6 +28,14 @@ import {
28
28
  isRaceCountable, raceLastShip, shouldRetryRaceArrival, raceFailKind,
29
29
  summarizeRaceFailures,
30
30
  } from './livestatus.js';
31
+ import { messageReasoning, wrapThink, stripThinkTags, reasoningPresent, splitThink } from './think.js';
32
+ import { guardFindCwd } from './runguard.js';
33
+
34
+ function emitReply(onDelta, raw) {
35
+ const parts = splitThink(raw);
36
+ if (parts.thinking) onDelta(parts.thinking, { think: true });
37
+ if (parts.visible) onDelta(parts.visible);
38
+ }
31
39
  import {
32
40
  probeGatewayRace, capRaceByCredit, inferRaceTier, RACE_NO_CREDIT,
33
41
  recutRaceByHud, sessionDollarX,
@@ -42,7 +50,7 @@ export const PROXY = process.env.OZ_PROXY || 'http://127.0.0.1:8402/v1';
42
50
  function completionsProxy() {
43
51
  return process.env.OZ_PROXY || PROXY;
44
52
  }
45
- export const MODEL = process.env.OZ_BRAIN_MODEL || 'deepseek/deepseek-v4-pro-0813';
53
+ export const MODEL = process.env.OZ_BRAIN_MODEL || 'openzoo/auto';
46
54
  const MAX_STEPS = Number(process.env.OZ_MAX_STEPS || 10);
47
55
 
48
56
  // Matches the Grok Bot chat surface itself (dark canvas, right-aligned grey
@@ -188,7 +196,7 @@ function execFrame(command, cwd = '/tmp') {
188
196
  approvalId: randomUUID(),
189
197
  // agent.v1.ExecServerMessage as protobuf-es JSON (camelCase). The daemon
190
198
  // assigns `id`; we only supply the shell variant.
191
- serverMessage: { shellArgs: { command, workingDirectory: cwd, timeout: 120 } },
199
+ serverMessage: { shellArgs: { command: guardFindCwd(command, cwd), workingDirectory: cwd, timeout: 120 } },
192
200
  };
193
201
  }
194
202
 
@@ -380,15 +388,17 @@ export async function brain(messages, contextId, modelOverride, topK, signal) {
380
388
  contextId, topK, undefined, signal,
381
389
  );
382
390
  const j = await r.json().catch(() => ({}));
383
- const content = j?.choices?.[0]?.message?.content;
391
+ const msg = j?.choices?.[0]?.message;
392
+ const content = msg?.content;
393
+ const thinking = messageReasoning(msg);
384
394
  // Same truncation catch as the streaming path (see brainStream): a reply that
385
395
  // stops because the budget ran out is not a finished reply, and this path is
386
396
  // what non-streaming callers — including every SPAWNed subagent — go through.
387
397
  if (content && j?.choices?.[0]?.finish_reason === 'length') {
388
398
  const rest = await brainContinue(messages, content, contextId, modelOverride, 0);
389
- return content + rest;
399
+ return wrapThink(thinking, content + rest);
390
400
  }
391
- return content || (r.ok ? '' : await httpErrorNote(r.status));
401
+ return wrapThink(thinking, content || (r.ok ? '' : await httpErrorNote(r.status)));
392
402
  }
393
403
 
394
404
  /** Resume a reply that hit the output cap, non-streaming. Bounded by
@@ -430,20 +440,31 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
430
440
  if (!r.ok || !r.body) {
431
441
  // fall back to the non-streaming path rather than fail outright
432
442
  const j = await r.json().catch(() => ({}));
433
- const content = j?.choices?.[0]?.message?.content;
443
+ const msg = j?.choices?.[0]?.message;
444
+ const content = msg?.content;
445
+ const thinking = messageReasoning(msg);
434
446
  const proxied = sanitizeProxiedError(j?.error?.message);
435
447
  const text = content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status)));
448
+ if (thinking) onDelta(thinking, { think: true });
436
449
  if (text) onDelta(text);
437
- return text;
450
+ return wrapThink(thinking, text);
438
451
  }
439
452
  const reader = r.body.getReader();
440
453
  const decoder = new TextDecoder();
441
- let buf = '', full = '', reasonedChars = 0, finish = '';
454
+ let buf = '', full = '', reasonedChars = 0, reasonedText = '', finish = '';
442
455
  let stopWait = startModelWait(onStatus);
443
456
  const noteThinking = () => {
444
457
  stopWait();
445
458
  onStatus?.('thinking…');
446
459
  };
460
+ const takeReasoning = (delta, message) => {
461
+ const chunk = messageReasoning(message, delta);
462
+ if (!chunk) return;
463
+ reasonedChars += chunk.length;
464
+ reasonedText += chunk;
465
+ onDelta(chunk, { think: true });
466
+ if (!full) noteThinking();
467
+ };
447
468
  try {
448
469
  for (;;) {
449
470
  let chunk;
@@ -459,11 +480,11 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
459
480
  if (full) {
460
481
  const note = '\n\n(stream stalled — showing what arrived before the timeout)';
461
482
  onDelta(note);
462
- return full + note;
483
+ return wrapThink(reasonedText, full + note);
463
484
  }
464
485
  onStatus?.('waiting on model…');
465
486
  const fallback = await brain(messages, contextId, modelOverride, topK, signal);
466
- if (fallback) onDelta(fallback);
487
+ if (fallback) emitReply(onDelta, fallback);
467
488
  return fallback || '(stream timed out — no tokens arrived)';
468
489
  }
469
490
  const { value, done } = chunk;
@@ -488,13 +509,18 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
488
509
  full += d.content;
489
510
  onDelta(d.content);
490
511
  }
491
- // Reasoning models emit their chain of thought on a SEPARATE field and
492
- // only then start producing content. Count it — not to show it, but to
493
- // tell "the model said nothing" apart from "the model spent its whole
494
- // budget thinking and got cut off". Surface "thinking…" so the wait
495
- // is not mute dots.
496
- else if (d?.reasoning || d?.reasoning_content) {
497
- reasonedChars += (d.reasoning || d.reasoning_content).length;
512
+ // Reasoning models emit their chain of thought on a SEPARATE field
513
+ // (reasoning_content / reasoning / thinking / thought, or a provider
514
+ // object with a plaintext summary). Forward the plaintext as
515
+ // onDelta(text, { think: true }) so the canvas can fold it — never
516
+ // as visible content. Encrypted blobs stay counted for the empty-
517
+ // after-think retry, but are not forwarded.
518
+ const before = reasonedChars;
519
+ takeReasoning(d, c?.message);
520
+ if (reasonedChars === before && reasoningPresent(d, c?.message)) {
521
+ // Encrypted-only: nothing to fold, but the model DID think —
522
+ // count it so an empty completion retries instead of "(no response)".
523
+ reasonedChars += 1;
498
524
  if (!full) noteThinking();
499
525
  }
500
526
  } catch { /* keep-alive line or partial JSON — ignore */ }
@@ -533,9 +559,9 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
533
559
  onDelta, contextId, modelOverride,
534
560
  Math.min(budget * 2, MAX_CONTINUE_TOKENS), round + 1, topK, onStatus,
535
561
  );
536
- return full + (more || '');
562
+ return wrapThink(reasonedText, full + (more || ''));
537
563
  }
538
- return full;
564
+ return wrapThink(reasonedText, full);
539
565
  }
540
566
 
541
567
  // How many times a single answer may be resumed after hitting the cap. Three
@@ -628,6 +654,7 @@ export const TIER_ALIASES = {
628
654
  };
629
655
  export function normalizeTier(s) {
630
656
  const raw = String(s || '').trim().toLowerCase();
657
+ if (raw === 'auto') return 'auto';
631
658
  if (TIER_NAMES.includes(raw)) return raw;
632
659
  if (TIER_ALIASES[raw]) return TIER_ALIASES[raw];
633
660
  const compact = raw.replace(/[\s_]/g, '');
@@ -786,15 +813,16 @@ async function brainGatewayRace(messages, onDelta, contextId, models, need, maxT
786
813
  const r = await postChat(body, contextId, 0, onStatus, hooks.signal);
787
814
  if (!r.ok || !r.body) {
788
815
  const j = await r.json().catch(() => ({}));
789
- const content = j?.choices?.[0]?.message?.content;
816
+ const msg = j?.choices?.[0]?.message;
817
+ const content = msg?.content;
790
818
  const proxied = sanitizeProxiedError(j?.error?.message);
791
- const text = content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status)));
819
+ const text = wrapThink(messageReasoning(msg), content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status))));
792
820
  lastFail = { model: 'gateway', text: text || '', error: r.ok ? undefined : `HTTP ${r.status}` };
793
821
  if (isRaceCountable(lastFail)) {
794
822
  arrivals.push(lastFail);
795
823
  done.push(lastFail);
796
824
  feed.onBack(lastFail.model);
797
- if (text) onDelta(text);
825
+ if (text) emitReply(onDelta, text);
798
826
  break;
799
827
  }
800
828
  if (!shouldRetryRaceArrival(lastFail) || attempt === 1) break;
@@ -846,6 +874,7 @@ async function readGatewayRaceStream(r, feed, signal) {
846
874
  const decoder = new TextDecoder();
847
875
  let buf = '';
848
876
  const texts = new Map();
877
+ const thinks = new Map();
849
878
  const finished = new Map();
850
879
  const live = { id: null };
851
880
  const stopWait = startModelWait(() => {});
@@ -857,11 +886,21 @@ async function readGatewayRaceStream(r, feed, signal) {
857
886
  texts.set(key, (texts.get(key) || '') + chunk);
858
887
  feed.onToken(key, chunk);
859
888
  };
889
+ const pushThink = (id, chunk) => {
890
+ if (chunk == null || chunk === '') return;
891
+ const key = id || live.id || 'gateway';
892
+ live.id = key;
893
+ thinks.set(key, (thinks.get(key) || '') + chunk);
894
+ };
860
895
  const finishOne = (id, extra = {}) => {
861
896
  const key = id || live.id || 'gateway';
862
897
  if (finished.has(key)) return;
863
898
  const text = extra.text != null ? String(extra.text) : (texts.get(key) || '');
864
- const row = { model: extra.model || key, text, error: extra.error };
899
+ const row = {
900
+ model: extra.model || key,
901
+ text: wrapThink(thinks.get(key) || '', text),
902
+ error: extra.error,
903
+ };
865
904
  finished.set(key, row);
866
905
  };
867
906
 
@@ -909,7 +948,11 @@ async function readGatewayRaceStream(r, feed, signal) {
909
948
  const d = c?.delta;
910
949
  const id = raceRacerId(obj, live.id || 'gateway');
911
950
  if (d?.content) pushText(id, d.content);
912
- else if (d?.reasoning || d?.reasoning_content) { /* thinking — not content */ }
951
+ // Reasoning is not a racer preview. Fold it onto the arrival so
952
+ // the winner's canvas row can show a thinking chip — never dump
953
+ // CoT into the spectator grid.
954
+ const think = messageReasoning(c?.message, d);
955
+ if (think) pushThink(id, think);
913
956
  if (c?.finish_reason) finishOne(id, { model: obj.model });
914
957
  if (ev?.ev === 'back' || ev?.ev === 'done') finishOne(ev.id, { text: ev.text, model: ev.id });
915
958
  if (ev?.ev === 'fail' || ev?.error) finishOne(ev.id, { text: ev.text || '', error: ev.error || 'empty body' });
@@ -1076,7 +1119,10 @@ export async function brainRace(messages, onDelta, contextId, models, need = 1,
1076
1119
  for (let attempt = 0; attempt < 2; attempt++) {
1077
1120
  if (raceAbort.signal.aborted && attempt > 0) break;
1078
1121
  try {
1079
- const text = await stream(messages, (chunk) => feed.onToken(m, chunk), contextId, m, maxTokens, 0, 0, undefined, raceAbort.signal);
1122
+ const text = await stream(messages, (chunk, meta) => {
1123
+ if (meta?.think) return;
1124
+ feed.onToken(m, chunk);
1125
+ }, contextId, m, maxTokens, 0, 0, undefined, raceAbort.signal);
1080
1126
  last = { model: m, text: text == null ? '' : String(text) };
1081
1127
  if (isRaceCountable(last)) {
1082
1128
  arrivals.push(last);
@@ -1146,7 +1192,7 @@ function raceQuestion(messages) {
1146
1192
  async function classifyRaceAnswer(messages, cand) {
1147
1193
  const prompt = 'Score this answer to one question from 0 to 10.\n\n'
1148
1194
  + 'QUESTION:\n' + String(raceQuestion(messages)).slice(0, 4000) + '\n\n'
1149
- + 'ANSWER:\n' + String(cand?.text || '').slice(0, 6000) + '\n\n'
1195
+ + 'ANSWER:\n' + stripThinkTags(String(cand?.text || '')).slice(0, 6000) + '\n\n'
1150
1196
  + 'Judge on: correctness first, then completeness, then whether it actually did what was asked '
1151
1197
  + '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
1152
1198
  + 'Ignore length and confidence of tone.\n'
@@ -1165,7 +1211,7 @@ async function pairwiseTied(messages, tied) {
1165
1211
  const letters = tied.map((_, i) => String.fromCharCode(65 + i));
1166
1212
  const prompt = 'You are judging answers to one question. Pick the single best one.\n\n'
1167
1213
  + 'QUESTION:\n' + String(raceQuestion(messages)).slice(0, 4000) + '\n\n'
1168
- + tied.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + String(c.text || '').slice(0, 6000)).join('\n\n')
1214
+ + tied.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + stripThinkTags(String(c.text || '')).slice(0, 6000)).join('\n\n')
1169
1215
  + '\n\nJudge on: correctness first, then completeness, then whether it actually did what was asked '
1170
1216
  + '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
1171
1217
  + 'Ignore length and confidence of tone.\n'