openzoo 0.49.8 → 0.49.9
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +26 -7
- package/bin/openzoo.js +3 -2
- package/lib/claudecode.js +847 -0
- package/lib/grokui.mjs +1323 -225
- package/lib/launch.js +220 -91
- package/lib/livestatus.js +4 -2
- package/lib/modelroute/README.md +1 -0
- package/lib/modelroute/catalog.json +1 -0
- package/lib/modelroute/outcomes.json +1566 -0
- package/lib/modelroute/router.json +1 -0
- package/lib/modelroute.js +737 -0
- package/lib/models.js +155 -64
- package/lib/package.json +3 -0
- package/lib/podagent.mjs +73 -27
- package/lib/proxy.js +106 -46
- package/lib/relay.js +275 -0
- package/lib/runguard.js +31 -0
- package/lib/spill.js +9 -1
- package/lib/think.js +126 -0
- package/package.json +5 -4
- package/vendor/modelroute/CURRENT_STATE.md +132 -0
- package/vendor/modelroute/HANDOFF.md +159 -0
- package/vendor/modelroute/catalog.json +1 -0
- package/vendor/modelroute/holographic_modelroute.py +809 -0
- package/vendor/modelroute/outcomes.json +1566 -0
- package/vendor/modelroute/router.json +1 -0
package/lib/models.js
CHANGED
|
@@ -1,5 +1,11 @@
|
|
|
1
1
|
import { config } from './config.js';
|
|
2
2
|
import { fetchHeaders } from './fetch.js';
|
|
3
|
+
import {
|
|
4
|
+
AUTO_MODEL_ID, autoHasPricedModels, autoModelListEntry, isAutoModel,
|
|
5
|
+
isPricedTokenPair, isUnservableRouteId,
|
|
6
|
+
} from './modelroute.js';
|
|
7
|
+
|
|
8
|
+
export { AUTO_MODEL_ID, isAutoModel, isPricedTokenPair, isUnservableRouteId };
|
|
3
9
|
|
|
4
10
|
/** Same threshold as BIND_MIN_CHARS in hrr.js — kept local so this
|
|
5
11
|
* module stays importable without the wallet/rpc stack. */
|
|
@@ -50,18 +56,43 @@ export function editorSlot(id) {
|
|
|
50
56
|
}
|
|
51
57
|
|
|
52
58
|
const CATALOG_TTL_MS = 5 * 60 * 1000;
|
|
53
|
-
let cache = { at: 0, ids: null };
|
|
59
|
+
let cache = { at: 0, ids: null, base: null };
|
|
60
|
+
|
|
61
|
+
export function resetZooModelIdsCache() {
|
|
62
|
+
cache = { at: 0, ids: null, base: null };
|
|
63
|
+
}
|
|
54
64
|
|
|
55
65
|
export async function zooModelIds() {
|
|
56
|
-
if (cache.ids && Date.now() - cache.at < CATALOG_TTL_MS) return cache.ids;
|
|
66
|
+
if (cache.ids && cache.base === config.apiBase && Date.now() - cache.at < CATALOG_TTL_MS) return cache.ids;
|
|
57
67
|
const r = await fetchHeaders(`${config.apiBase}/v1/models`);
|
|
58
68
|
if (!r.ok) throw new Error(`model catalog fetch failed: HTTP ${r.status}`);
|
|
59
69
|
const d = await r.json();
|
|
60
|
-
const ids = (d.data
|
|
61
|
-
if (ids.length) cache = { at: Date.now(), ids };
|
|
70
|
+
const ids = quoteableRows(d.data).map((m) => m.id).filter((id) => id && !isAutoModel(id));
|
|
71
|
+
if (ids.length) cache = { at: Date.now(), ids, base: config.apiBase };
|
|
62
72
|
return ids;
|
|
63
73
|
}
|
|
64
74
|
|
|
75
|
+
/** OpenRouter / gateway token price pair. Image/video rows have no prompt. */
|
|
76
|
+
export function tokenPricePair(pricing) {
|
|
77
|
+
if (!pricing || typeof pricing !== 'object') return [NaN, NaN];
|
|
78
|
+
return [Number(pricing.prompt ?? pricing.input), Number(pricing.completion ?? pricing.output)];
|
|
79
|
+
}
|
|
80
|
+
|
|
81
|
+
/**
|
|
82
|
+
* A row the gateway can actually quote for chat. Drops :batch, ~latest
|
|
83
|
+
* pointers, openzoo-* twins, $0 / missing / non-token OpenRouter prices.
|
|
84
|
+
*/
|
|
85
|
+
export function isQuoteableModel(m) {
|
|
86
|
+
const id = m?.id;
|
|
87
|
+
if (isAutoModel(id)) return true;
|
|
88
|
+
if (isUnservableRouteId(id)) return false;
|
|
89
|
+
return isPricedTokenPair(...tokenPricePair(m?.pricing));
|
|
90
|
+
}
|
|
91
|
+
|
|
92
|
+
export function quoteableRows(data) {
|
|
93
|
+
return (Array.isArray(data) ? data : []).filter(isQuoteableModel);
|
|
94
|
+
}
|
|
95
|
+
|
|
65
96
|
/** Vendor fingerprints in harness model ids → zoo catalog prefixes. Order
|
|
66
97
|
* matters only for overlapping hints; first match wins. */
|
|
67
98
|
const FAMILIES = [
|
|
@@ -119,6 +150,14 @@ const tokensOf = (id) => id.toLowerCase().split(/[^a-z0-9.]+/).filter((t) => t &
|
|
|
119
150
|
* OPENZOO_DEFAULT_MODEL is an explicit user override, not a fallback tier.
|
|
120
151
|
*/
|
|
121
152
|
export function resolveModel(requested, ids) {
|
|
153
|
+
// Virtual router id — never family-match, never steal via OPENZOO_DEFAULT_MODEL.
|
|
154
|
+
if (isAutoModel(requested)) return null;
|
|
155
|
+
// Bare Anthropic / Claude Code ids are never live on Fly/OpenRouter
|
|
156
|
+
// (`claude-opus-5` → 500 unknown model). Rewrite even on a catalog miss
|
|
157
|
+
// or if a gateway row lists the bare name — the request must not leave
|
|
158
|
+
// the sidecar as that id.
|
|
159
|
+
const native = anthropicNativeAlias(requested);
|
|
160
|
+
if (native) return native;
|
|
122
161
|
if (!requested || !ids?.length || ids.includes(requested)) return null;
|
|
123
162
|
// `openzoo-` prefixed names exist so an editor cannot mistake them for its
|
|
124
163
|
// OWN models: Cursor claims any name in its catalog (claude-opus-5, grok-4.6)
|
|
@@ -187,6 +226,46 @@ export function resolveModel(requested, ids) {
|
|
|
187
226
|
return best;
|
|
188
227
|
}
|
|
189
228
|
|
|
229
|
+
/**
|
|
230
|
+
* Bare Anthropic / Claude Code ids. MEASURED 2026-08-20 against Fly
|
|
231
|
+
* x402-tokens: POST /v1/chat/completions model=claude-opus-5 → 500
|
|
232
|
+
* `unknown model claude-opus-5`. The vendor-prefixed twin
|
|
233
|
+
* (`anthropic/claude-opus-5`) 402s and is priced. Claude Code's Auto
|
|
234
|
+
* permission classifier calls these ids on ANTHROPIC_BASE_URL; if the
|
|
235
|
+
* sidecar forwards the bare name, Bash hangs with
|
|
236
|
+
* "claude-opus-5 is temporarily unavailable, so auto mode cannot
|
|
237
|
+
* determine the safety of Bash". Always rewrite. Never send the bare
|
|
238
|
+
* id to OpenRouter / Fly. Do not push x402-tokens — alias here.
|
|
239
|
+
*/
|
|
240
|
+
export const ANTHROPIC_NATIVE_ALIASES = {
|
|
241
|
+
'claude-opus-5': 'anthropic/claude-opus-5',
|
|
242
|
+
'claude-opus-5-fast': 'anthropic/claude-opus-5-fast',
|
|
243
|
+
'claude-3-5-opus': 'anthropic/claude-opus-5',
|
|
244
|
+
'claude-sonnet-5': 'anthropic/claude-sonnet-5',
|
|
245
|
+
'claude-sonnet-5-fast': 'anthropic/claude-sonnet-5',
|
|
246
|
+
'claude-opus-4-8': 'anthropic/claude-opus-4.8',
|
|
247
|
+
'claude-opus-4.8': 'anthropic/claude-opus-4.8',
|
|
248
|
+
'claude-fable-5': 'anthropic/claude-fable-5',
|
|
249
|
+
'claude-haiku-4.5': 'anthropic/claude-haiku-4.5',
|
|
250
|
+
};
|
|
251
|
+
|
|
252
|
+
/** Claude Code sometimes suffixes a window marker (`claude-opus-5[1m]`). */
|
|
253
|
+
function stripAnthropicWindowSuffix(id) {
|
|
254
|
+
return String(id || '').trim().replace(/\[[\d]+m\]$/i, '');
|
|
255
|
+
}
|
|
256
|
+
|
|
257
|
+
/**
|
|
258
|
+
* Priced zoo twin for a bare Anthropic / Claude Code id, or null.
|
|
259
|
+
* Vendor-prefixed ids and openzoo-* twins are left to the rest of resolveModel.
|
|
260
|
+
*/
|
|
261
|
+
export function anthropicNativeAlias(requested) {
|
|
262
|
+
if (!requested || typeof requested !== 'string') return null;
|
|
263
|
+
const stripped = stripAnthropicWindowSuffix(requested);
|
|
264
|
+
if (!stripped || stripped.includes('/')) return null;
|
|
265
|
+
if (/^openzoo[-/]/i.test(stripped)) return null;
|
|
266
|
+
return ANTHROPIC_NATIVE_ALIASES[stripped] || ANTHROPIC_NATIVE_ALIASES[stripped.toLowerCase()] || null;
|
|
267
|
+
}
|
|
268
|
+
|
|
190
269
|
/**
|
|
191
270
|
* Ids harnesses ship as DEFAULTS (Cursor, Continue, Aider, Codex CLI, Cline,
|
|
192
271
|
* OpenClaw, LangChain templates…). Merged into GET /v1/models so a harness
|
|
@@ -197,58 +276,51 @@ export const ALIAS_IDS = [
|
|
|
197
276
|
'gpt-4o', 'gpt-4o-mini', 'gpt-4.1', 'gpt-4.1-mini', 'gpt-4-turbo', 'gpt-3.5-turbo',
|
|
198
277
|
'gpt-5', 'gpt-5-mini', 'chatgpt-4o-latest', 'o1', 'o3', 'o3-mini', 'o4-mini',
|
|
199
278
|
'claude-3-5-sonnet-latest', 'claude-sonnet-4-0', 'claude-opus-4-1',
|
|
279
|
+
'claude-opus-5', 'claude-opus-5-fast', 'claude-3-5-opus', 'claude-sonnet-5',
|
|
280
|
+
'claude-opus-4-8', 'claude-fable-5',
|
|
200
281
|
'gemini-2.5-pro', 'gemini-2.5-flash', 'grok-4', 'grok-3',
|
|
201
282
|
'deepseek-chat', 'deepseek-reasoner', 'qwen-max', 'llama-3.3-70b',
|
|
202
283
|
];
|
|
203
284
|
|
|
285
|
+
export function isHarnessAliasId(id) {
|
|
286
|
+
const stripped = stripAnthropicWindowSuffix(id);
|
|
287
|
+
return ALIAS_IDS.includes(stripped) || Boolean(anthropicNativeAlias(stripped));
|
|
288
|
+
}
|
|
289
|
+
|
|
204
290
|
/**
|
|
205
291
|
* Merge alias rows into a /v1/models payload without duplicating real ids.
|
|
206
292
|
* Each alias inherits context_length and pricing from the model it RESOLVES
|
|
207
293
|
* to — a harness sizing its corpus off "gpt-4o" gets the real ceiling of the
|
|
208
294
|
* model that will actually serve it, not a blank.
|
|
295
|
+
*
|
|
296
|
+
* Does not mint openzoo-* twins. Those duplicated every real id (and every
|
|
297
|
+
* :batch id) in Claude Code's /model picker.
|
|
209
298
|
*/
|
|
210
|
-
export function augmentModelList(payload) {
|
|
211
|
-
const data =
|
|
299
|
+
export function augmentModelList(payload, { aliases: withAliases = true } = {}) {
|
|
300
|
+
const data = quoteableRows(payload?.data);
|
|
212
301
|
const have = new Set(data.map((m) => m.id));
|
|
213
302
|
const ids = data.map((m) => m.id);
|
|
214
|
-
|
|
215
|
-
|
|
216
|
-
|
|
217
|
-
|
|
218
|
-
|
|
219
|
-
|
|
220
|
-
|
|
221
|
-
|
|
222
|
-
|
|
223
|
-
|
|
224
|
-
|
|
225
|
-
|
|
226
|
-
|
|
227
|
-
|
|
228
|
-
|
|
229
|
-
|
|
230
|
-
|
|
231
|
-
|
|
232
|
-
|
|
233
|
-
owned_by: 'openzoo',
|
|
234
|
-
served_by: id,
|
|
235
|
-
...(src?.context_length ? { context_length: src.context_length, context_window: src.context_window ?? src.context_length } : {}),
|
|
236
|
-
...(src?.pricing ? { pricing: src.pricing } : {}),
|
|
237
|
-
});
|
|
238
|
-
have.add(name);
|
|
303
|
+
const aliases = withAliases
|
|
304
|
+
? ALIAS_IDS.filter((id) => !have.has(id)).map((id) => {
|
|
305
|
+
const target = data.find((m) => m.id === resolveModel(id, ids));
|
|
306
|
+
return {
|
|
307
|
+
id,
|
|
308
|
+
object: 'model',
|
|
309
|
+
owned_by: 'openzoo-alias',
|
|
310
|
+
...(target?.context_length ? { context_length: target.context_length, context_window: target.context_window ?? target.context_length } : {}),
|
|
311
|
+
...(target?.pricing ? { pricing: target.pricing } : {}),
|
|
312
|
+
...(target ? { served_by: target.id } : {}),
|
|
313
|
+
};
|
|
314
|
+
})
|
|
315
|
+
: [];
|
|
316
|
+
const virtual = [];
|
|
317
|
+
// Auto is listed only when its own shortlist is priced models — never an
|
|
318
|
+
// unquoted OpenRouter id that 500s `bad openrouter price`.
|
|
319
|
+
if (!have.has(AUTO_MODEL_ID) && autoHasPricedModels(undefined, ids.length ? ids : null)) {
|
|
320
|
+
virtual.push(autoModelListEntry());
|
|
321
|
+
have.add(AUTO_MODEL_ID);
|
|
239
322
|
}
|
|
240
|
-
|
|
241
|
-
const target = data.find((m) => m.id === resolveModel(id, ids));
|
|
242
|
-
return {
|
|
243
|
-
id,
|
|
244
|
-
object: 'model',
|
|
245
|
-
owned_by: 'openzoo-alias',
|
|
246
|
-
...(target?.context_length ? { context_length: target.context_length, context_window: target.context_window ?? target.context_length } : {}),
|
|
247
|
-
...(target?.pricing ? { pricing: target.pricing } : {}),
|
|
248
|
-
...(target ? { served_by: target.id } : {}),
|
|
249
|
-
};
|
|
250
|
-
});
|
|
251
|
-
return { ...payload, object: payload?.object || 'list', data: [...data, ...branded, ...aliases] };
|
|
323
|
+
return { ...payload, object: payload?.object || 'list', data: [...virtual, ...data, ...aliases] };
|
|
252
324
|
}
|
|
253
325
|
|
|
254
326
|
/**
|
|
@@ -277,14 +349,12 @@ function decorateModelEntry(m) {
|
|
|
277
349
|
}
|
|
278
350
|
|
|
279
351
|
/**
|
|
280
|
-
* OpenAI-compatible /v1/models body
|
|
281
|
-
*
|
|
282
|
-
*
|
|
283
|
-
* OpenAI ones so a Messages-API client can label rows without a second
|
|
284
|
-
* endpoint. Existing OpenAI clients ignore the extras.
|
|
352
|
+
* OpenAI-compatible /v1/models body. Quoteable chat models only — no :batch,
|
|
353
|
+
* no unpriced / image-video rows, no openzoo-* twins. Extra Anthropic fields
|
|
354
|
+
* (`type`, `display_name`) sit alongside OpenAI ones.
|
|
285
355
|
*/
|
|
286
|
-
export function publishModelList(payload) {
|
|
287
|
-
const merged = augmentModelList(payload);
|
|
356
|
+
export function publishModelList(payload, opts) {
|
|
357
|
+
const merged = augmentModelList(payload, opts);
|
|
288
358
|
const data = (merged.data || []).map(decorateModelEntry);
|
|
289
359
|
return {
|
|
290
360
|
...merged,
|
|
@@ -296,21 +366,37 @@ export function publishModelList(payload) {
|
|
|
296
366
|
};
|
|
297
367
|
}
|
|
298
368
|
|
|
299
|
-
|
|
300
|
-
|
|
301
|
-
* Claude Code gateway discovery reads `data[].id` and optional `display_name`.
|
|
302
|
-
* Ids are the zoo's real catalog ids — never rewritten to fake `claude-*`
|
|
303
|
-
* aliases for grok / deepseek / gemini / etc.
|
|
304
|
-
*/
|
|
305
|
-
export function anthropicModelList(payload) {
|
|
306
|
-
const published = publishModelList(payload);
|
|
307
|
-
const data = (published.data || []).map((m) => ({
|
|
369
|
+
function anthropicRows(rows) {
|
|
370
|
+
return rows.map((m) => ({
|
|
308
371
|
type: 'model',
|
|
309
372
|
id: m.id,
|
|
310
373
|
display_name: m.display_name || displayNameFor(m.id),
|
|
311
374
|
...(m.created_at ? { created_at: m.created_at } : {}),
|
|
312
375
|
...(m.served_by ? { served_by: m.served_by } : {}),
|
|
313
376
|
}));
|
|
377
|
+
}
|
|
378
|
+
|
|
379
|
+
/** openzoo/auto first so Claude Code's picker default is the router, not opus. */
|
|
380
|
+
function withAutoFirst(rows) {
|
|
381
|
+
const list = Array.isArray(rows) ? rows.slice() : [];
|
|
382
|
+
const i = list.findIndex((m) => isAutoModel(m?.id));
|
|
383
|
+
if (i > 0) {
|
|
384
|
+
const [auto] = list.splice(i, 1);
|
|
385
|
+
list.unshift(auto);
|
|
386
|
+
}
|
|
387
|
+
return list;
|
|
388
|
+
}
|
|
389
|
+
|
|
390
|
+
/**
|
|
391
|
+
* Anthropic GET /v1/models shape (id + display_name + type).
|
|
392
|
+
* Claude Code gateway discovery reads `data[].id` and optional `display_name`.
|
|
393
|
+
* Full quoteable catalog — every priced published id, including OpenRouter
|
|
394
|
+
* grok/gemini/gpt rows. Not a 4-id Anthropic cap, not openzoo-* clones.
|
|
395
|
+
* Picker ≠ classifier: rewriteChatModel still pins 16-token classify off opus-5.
|
|
396
|
+
*/
|
|
397
|
+
export function anthropicModelList(payload) {
|
|
398
|
+
const published = publishModelList(payload, { aliases: false });
|
|
399
|
+
const data = anthropicRows(withAutoFirst(published.data || []));
|
|
314
400
|
return {
|
|
315
401
|
data,
|
|
316
402
|
has_more: false,
|
|
@@ -329,12 +415,14 @@ export function wantsAnthropicModelList(headers = {}) {
|
|
|
329
415
|
}
|
|
330
416
|
|
|
331
417
|
/**
|
|
332
|
-
* Body for GET /v1/models.
|
|
333
|
-
* Anthropic-shaped clients get
|
|
418
|
+
* Body for GET /v1/models. Quoteable catalog only (never opus-5-only, never
|
|
419
|
+
* unpriced / :batch / openzoo-* clones). Anthropic-shaped clients get the
|
|
420
|
+
* same quoteable ids in Anthropic shape (type / id / display_name) so Claude
|
|
421
|
+
* Code can select grok, gemini, gpt, etc. OpenAI clients keep object:list
|
|
422
|
+
* of every quoteable id + harness aliases.
|
|
334
423
|
*/
|
|
335
424
|
export function modelsListForRequest(payload, headers) {
|
|
336
|
-
|
|
337
|
-
return wantsAnthropicModelList(headers) ? anthropicModelList(published) : published;
|
|
425
|
+
return wantsAnthropicModelList(headers) ? anthropicModelList(payload) : publishModelList(payload);
|
|
338
426
|
}
|
|
339
427
|
|
|
340
428
|
/**
|
|
@@ -471,6 +559,9 @@ export function raiseReasoningMaxTokens(parsed, env = process.env) {
|
|
|
471
559
|
export function rewriteChatModel(parsed, ids, { bodyLen } = {}) {
|
|
472
560
|
const from = parsed?.model;
|
|
473
561
|
const len = bodyLen ?? (parsed == null ? 0 : Buffer.byteLength(JSON.stringify(parsed)));
|
|
562
|
+
if (isAutoModel(from) && !isTinyClassify(parsed, len)) {
|
|
563
|
+
return { parsed, tiny: false, auto: true, from, to: AUTO_MODEL_ID, raised: false };
|
|
564
|
+
}
|
|
474
565
|
if (isTinyClassify(parsed, len)) {
|
|
475
566
|
const picked = pickClassifierModel(ids);
|
|
476
567
|
// pickClassifierModel returns null on an empty catalog or a zoo that
|
|
@@ -489,7 +580,7 @@ export function rewriteChatModel(parsed, ids, { bodyLen } = {}) {
|
|
|
489
580
|
if (typeof from !== 'string') {
|
|
490
581
|
return { parsed, tiny: false, from, to: from, raised: false };
|
|
491
582
|
}
|
|
492
|
-
const resolved = resolveModel(from, ids);
|
|
583
|
+
const resolved = resolveModel(from, ids) || anthropicNativeAlias(from);
|
|
493
584
|
const next = resolved ? { ...parsed, model: resolved } : parsed;
|
|
494
585
|
const bump = raiseReasoningMaxTokens(next);
|
|
495
586
|
return {
|
package/lib/package.json
ADDED
package/lib/podagent.mjs
CHANGED
|
@@ -28,6 +28,14 @@ import {
|
|
|
28
28
|
isRaceCountable, raceLastShip, shouldRetryRaceArrival, raceFailKind,
|
|
29
29
|
summarizeRaceFailures,
|
|
30
30
|
} from './livestatus.js';
|
|
31
|
+
import { messageReasoning, wrapThink, stripThinkTags, reasoningPresent, splitThink } from './think.js';
|
|
32
|
+
import { guardFindCwd } from './runguard.js';
|
|
33
|
+
|
|
34
|
+
function emitReply(onDelta, raw) {
|
|
35
|
+
const parts = splitThink(raw);
|
|
36
|
+
if (parts.thinking) onDelta(parts.thinking, { think: true });
|
|
37
|
+
if (parts.visible) onDelta(parts.visible);
|
|
38
|
+
}
|
|
31
39
|
import {
|
|
32
40
|
probeGatewayRace, capRaceByCredit, inferRaceTier, RACE_NO_CREDIT,
|
|
33
41
|
recutRaceByHud, sessionDollarX,
|
|
@@ -42,7 +50,7 @@ export const PROXY = process.env.OZ_PROXY || 'http://127.0.0.1:8402/v1';
|
|
|
42
50
|
function completionsProxy() {
|
|
43
51
|
return process.env.OZ_PROXY || PROXY;
|
|
44
52
|
}
|
|
45
|
-
export const MODEL = process.env.OZ_BRAIN_MODEL || '
|
|
53
|
+
export const MODEL = process.env.OZ_BRAIN_MODEL || 'openzoo/auto';
|
|
46
54
|
const MAX_STEPS = Number(process.env.OZ_MAX_STEPS || 10);
|
|
47
55
|
|
|
48
56
|
// Matches the Grok Bot chat surface itself (dark canvas, right-aligned grey
|
|
@@ -188,7 +196,7 @@ function execFrame(command, cwd = '/tmp') {
|
|
|
188
196
|
approvalId: randomUUID(),
|
|
189
197
|
// agent.v1.ExecServerMessage as protobuf-es JSON (camelCase). The daemon
|
|
190
198
|
// assigns `id`; we only supply the shell variant.
|
|
191
|
-
serverMessage: { shellArgs: { command, workingDirectory: cwd, timeout: 120 } },
|
|
199
|
+
serverMessage: { shellArgs: { command: guardFindCwd(command, cwd), workingDirectory: cwd, timeout: 120 } },
|
|
192
200
|
};
|
|
193
201
|
}
|
|
194
202
|
|
|
@@ -380,15 +388,17 @@ export async function brain(messages, contextId, modelOverride, topK, signal) {
|
|
|
380
388
|
contextId, topK, undefined, signal,
|
|
381
389
|
);
|
|
382
390
|
const j = await r.json().catch(() => ({}));
|
|
383
|
-
const
|
|
391
|
+
const msg = j?.choices?.[0]?.message;
|
|
392
|
+
const content = msg?.content;
|
|
393
|
+
const thinking = messageReasoning(msg);
|
|
384
394
|
// Same truncation catch as the streaming path (see brainStream): a reply that
|
|
385
395
|
// stops because the budget ran out is not a finished reply, and this path is
|
|
386
396
|
// what non-streaming callers — including every SPAWNed subagent — go through.
|
|
387
397
|
if (content && j?.choices?.[0]?.finish_reason === 'length') {
|
|
388
398
|
const rest = await brainContinue(messages, content, contextId, modelOverride, 0);
|
|
389
|
-
return content + rest;
|
|
399
|
+
return wrapThink(thinking, content + rest);
|
|
390
400
|
}
|
|
391
|
-
return content || (r.ok ? '' : await httpErrorNote(r.status));
|
|
401
|
+
return wrapThink(thinking, content || (r.ok ? '' : await httpErrorNote(r.status)));
|
|
392
402
|
}
|
|
393
403
|
|
|
394
404
|
/** Resume a reply that hit the output cap, non-streaming. Bounded by
|
|
@@ -430,20 +440,31 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
430
440
|
if (!r.ok || !r.body) {
|
|
431
441
|
// fall back to the non-streaming path rather than fail outright
|
|
432
442
|
const j = await r.json().catch(() => ({}));
|
|
433
|
-
const
|
|
443
|
+
const msg = j?.choices?.[0]?.message;
|
|
444
|
+
const content = msg?.content;
|
|
445
|
+
const thinking = messageReasoning(msg);
|
|
434
446
|
const proxied = sanitizeProxiedError(j?.error?.message);
|
|
435
447
|
const text = content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status)));
|
|
448
|
+
if (thinking) onDelta(thinking, { think: true });
|
|
436
449
|
if (text) onDelta(text);
|
|
437
|
-
return text;
|
|
450
|
+
return wrapThink(thinking, text);
|
|
438
451
|
}
|
|
439
452
|
const reader = r.body.getReader();
|
|
440
453
|
const decoder = new TextDecoder();
|
|
441
|
-
let buf = '', full = '', reasonedChars = 0, finish = '';
|
|
454
|
+
let buf = '', full = '', reasonedChars = 0, reasonedText = '', finish = '';
|
|
442
455
|
let stopWait = startModelWait(onStatus);
|
|
443
456
|
const noteThinking = () => {
|
|
444
457
|
stopWait();
|
|
445
458
|
onStatus?.('thinking…');
|
|
446
459
|
};
|
|
460
|
+
const takeReasoning = (delta, message) => {
|
|
461
|
+
const chunk = messageReasoning(message, delta);
|
|
462
|
+
if (!chunk) return;
|
|
463
|
+
reasonedChars += chunk.length;
|
|
464
|
+
reasonedText += chunk;
|
|
465
|
+
onDelta(chunk, { think: true });
|
|
466
|
+
if (!full) noteThinking();
|
|
467
|
+
};
|
|
447
468
|
try {
|
|
448
469
|
for (;;) {
|
|
449
470
|
let chunk;
|
|
@@ -459,11 +480,11 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
459
480
|
if (full) {
|
|
460
481
|
const note = '\n\n(stream stalled — showing what arrived before the timeout)';
|
|
461
482
|
onDelta(note);
|
|
462
|
-
return full + note;
|
|
483
|
+
return wrapThink(reasonedText, full + note);
|
|
463
484
|
}
|
|
464
485
|
onStatus?.('waiting on model…');
|
|
465
486
|
const fallback = await brain(messages, contextId, modelOverride, topK, signal);
|
|
466
|
-
if (fallback) onDelta
|
|
487
|
+
if (fallback) emitReply(onDelta, fallback);
|
|
467
488
|
return fallback || '(stream timed out — no tokens arrived)';
|
|
468
489
|
}
|
|
469
490
|
const { value, done } = chunk;
|
|
@@ -488,13 +509,18 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
488
509
|
full += d.content;
|
|
489
510
|
onDelta(d.content);
|
|
490
511
|
}
|
|
491
|
-
// Reasoning models emit their chain of thought on a SEPARATE field
|
|
492
|
-
//
|
|
493
|
-
//
|
|
494
|
-
//
|
|
495
|
-
//
|
|
496
|
-
|
|
497
|
-
|
|
512
|
+
// Reasoning models emit their chain of thought on a SEPARATE field
|
|
513
|
+
// (reasoning_content / reasoning / thinking / thought, or a provider
|
|
514
|
+
// object with a plaintext summary). Forward the plaintext as
|
|
515
|
+
// onDelta(text, { think: true }) so the canvas can fold it — never
|
|
516
|
+
// as visible content. Encrypted blobs stay counted for the empty-
|
|
517
|
+
// after-think retry, but are not forwarded.
|
|
518
|
+
const before = reasonedChars;
|
|
519
|
+
takeReasoning(d, c?.message);
|
|
520
|
+
if (reasonedChars === before && reasoningPresent(d, c?.message)) {
|
|
521
|
+
// Encrypted-only: nothing to fold, but the model DID think —
|
|
522
|
+
// count it so an empty completion retries instead of "(no response)".
|
|
523
|
+
reasonedChars += 1;
|
|
498
524
|
if (!full) noteThinking();
|
|
499
525
|
}
|
|
500
526
|
} catch { /* keep-alive line or partial JSON — ignore */ }
|
|
@@ -533,9 +559,9 @@ export async function brainStream(messages, onDelta, contextId, modelOverride, m
|
|
|
533
559
|
onDelta, contextId, modelOverride,
|
|
534
560
|
Math.min(budget * 2, MAX_CONTINUE_TOKENS), round + 1, topK, onStatus,
|
|
535
561
|
);
|
|
536
|
-
return full + (more || '');
|
|
562
|
+
return wrapThink(reasonedText, full + (more || ''));
|
|
537
563
|
}
|
|
538
|
-
return full;
|
|
564
|
+
return wrapThink(reasonedText, full);
|
|
539
565
|
}
|
|
540
566
|
|
|
541
567
|
// How many times a single answer may be resumed after hitting the cap. Three
|
|
@@ -628,6 +654,7 @@ export const TIER_ALIASES = {
|
|
|
628
654
|
};
|
|
629
655
|
export function normalizeTier(s) {
|
|
630
656
|
const raw = String(s || '').trim().toLowerCase();
|
|
657
|
+
if (raw === 'auto') return 'auto';
|
|
631
658
|
if (TIER_NAMES.includes(raw)) return raw;
|
|
632
659
|
if (TIER_ALIASES[raw]) return TIER_ALIASES[raw];
|
|
633
660
|
const compact = raw.replace(/[\s_]/g, '');
|
|
@@ -786,15 +813,16 @@ async function brainGatewayRace(messages, onDelta, contextId, models, need, maxT
|
|
|
786
813
|
const r = await postChat(body, contextId, 0, onStatus, hooks.signal);
|
|
787
814
|
if (!r.ok || !r.body) {
|
|
788
815
|
const j = await r.json().catch(() => ({}));
|
|
789
|
-
const
|
|
816
|
+
const msg = j?.choices?.[0]?.message;
|
|
817
|
+
const content = msg?.content;
|
|
790
818
|
const proxied = sanitizeProxiedError(j?.error?.message);
|
|
791
|
-
const text = content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status)));
|
|
819
|
+
const text = wrapThink(messageReasoning(msg), content || (r.ok ? '' : (proxied ? `(request failed — HTTP ${r.status}: ${proxied})` : await httpErrorNote(r.status))));
|
|
792
820
|
lastFail = { model: 'gateway', text: text || '', error: r.ok ? undefined : `HTTP ${r.status}` };
|
|
793
821
|
if (isRaceCountable(lastFail)) {
|
|
794
822
|
arrivals.push(lastFail);
|
|
795
823
|
done.push(lastFail);
|
|
796
824
|
feed.onBack(lastFail.model);
|
|
797
|
-
if (text) onDelta
|
|
825
|
+
if (text) emitReply(onDelta, text);
|
|
798
826
|
break;
|
|
799
827
|
}
|
|
800
828
|
if (!shouldRetryRaceArrival(lastFail) || attempt === 1) break;
|
|
@@ -846,6 +874,7 @@ async function readGatewayRaceStream(r, feed, signal) {
|
|
|
846
874
|
const decoder = new TextDecoder();
|
|
847
875
|
let buf = '';
|
|
848
876
|
const texts = new Map();
|
|
877
|
+
const thinks = new Map();
|
|
849
878
|
const finished = new Map();
|
|
850
879
|
const live = { id: null };
|
|
851
880
|
const stopWait = startModelWait(() => {});
|
|
@@ -857,11 +886,21 @@ async function readGatewayRaceStream(r, feed, signal) {
|
|
|
857
886
|
texts.set(key, (texts.get(key) || '') + chunk);
|
|
858
887
|
feed.onToken(key, chunk);
|
|
859
888
|
};
|
|
889
|
+
const pushThink = (id, chunk) => {
|
|
890
|
+
if (chunk == null || chunk === '') return;
|
|
891
|
+
const key = id || live.id || 'gateway';
|
|
892
|
+
live.id = key;
|
|
893
|
+
thinks.set(key, (thinks.get(key) || '') + chunk);
|
|
894
|
+
};
|
|
860
895
|
const finishOne = (id, extra = {}) => {
|
|
861
896
|
const key = id || live.id || 'gateway';
|
|
862
897
|
if (finished.has(key)) return;
|
|
863
898
|
const text = extra.text != null ? String(extra.text) : (texts.get(key) || '');
|
|
864
|
-
const row = {
|
|
899
|
+
const row = {
|
|
900
|
+
model: extra.model || key,
|
|
901
|
+
text: wrapThink(thinks.get(key) || '', text),
|
|
902
|
+
error: extra.error,
|
|
903
|
+
};
|
|
865
904
|
finished.set(key, row);
|
|
866
905
|
};
|
|
867
906
|
|
|
@@ -909,7 +948,11 @@ async function readGatewayRaceStream(r, feed, signal) {
|
|
|
909
948
|
const d = c?.delta;
|
|
910
949
|
const id = raceRacerId(obj, live.id || 'gateway');
|
|
911
950
|
if (d?.content) pushText(id, d.content);
|
|
912
|
-
|
|
951
|
+
// Reasoning is not a racer preview. Fold it onto the arrival so
|
|
952
|
+
// the winner's canvas row can show a thinking chip — never dump
|
|
953
|
+
// CoT into the spectator grid.
|
|
954
|
+
const think = messageReasoning(c?.message, d);
|
|
955
|
+
if (think) pushThink(id, think);
|
|
913
956
|
if (c?.finish_reason) finishOne(id, { model: obj.model });
|
|
914
957
|
if (ev?.ev === 'back' || ev?.ev === 'done') finishOne(ev.id, { text: ev.text, model: ev.id });
|
|
915
958
|
if (ev?.ev === 'fail' || ev?.error) finishOne(ev.id, { text: ev.text || '', error: ev.error || 'empty body' });
|
|
@@ -1076,7 +1119,10 @@ export async function brainRace(messages, onDelta, contextId, models, need = 1,
|
|
|
1076
1119
|
for (let attempt = 0; attempt < 2; attempt++) {
|
|
1077
1120
|
if (raceAbort.signal.aborted && attempt > 0) break;
|
|
1078
1121
|
try {
|
|
1079
|
-
const text = await stream(messages, (chunk
|
|
1122
|
+
const text = await stream(messages, (chunk, meta) => {
|
|
1123
|
+
if (meta?.think) return;
|
|
1124
|
+
feed.onToken(m, chunk);
|
|
1125
|
+
}, contextId, m, maxTokens, 0, 0, undefined, raceAbort.signal);
|
|
1080
1126
|
last = { model: m, text: text == null ? '' : String(text) };
|
|
1081
1127
|
if (isRaceCountable(last)) {
|
|
1082
1128
|
arrivals.push(last);
|
|
@@ -1146,7 +1192,7 @@ function raceQuestion(messages) {
|
|
|
1146
1192
|
async function classifyRaceAnswer(messages, cand) {
|
|
1147
1193
|
const prompt = 'Score this answer to one question from 0 to 10.\n\n'
|
|
1148
1194
|
+ 'QUESTION:\n' + String(raceQuestion(messages)).slice(0, 4000) + '\n\n'
|
|
1149
|
-
+ 'ANSWER:\n' + String(cand?.text || '').slice(0, 6000) + '\n\n'
|
|
1195
|
+
+ 'ANSWER:\n' + stripThinkTags(String(cand?.text || '')).slice(0, 6000) + '\n\n'
|
|
1150
1196
|
+ 'Judge on: correctness first, then completeness, then whether it actually did what was asked '
|
|
1151
1197
|
+ '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
|
|
1152
1198
|
+ 'Ignore length and confidence of tone.\n'
|
|
@@ -1165,7 +1211,7 @@ async function pairwiseTied(messages, tied) {
|
|
|
1165
1211
|
const letters = tied.map((_, i) => String.fromCharCode(65 + i));
|
|
1166
1212
|
const prompt = 'You are judging answers to one question. Pick the single best one.\n\n'
|
|
1167
1213
|
+ 'QUESTION:\n' + String(raceQuestion(messages)).slice(0, 4000) + '\n\n'
|
|
1168
|
-
+ tied.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + String(c.text || '').slice(0, 6000)).join('\n\n')
|
|
1214
|
+
+ tied.map((c, i) => 'ANSWER ' + letters[i] + ':\n' + stripThinkTags(String(c.text || '')).slice(0, 6000)).join('\n\n')
|
|
1169
1215
|
+ '\n\nJudge on: correctness first, then completeness, then whether it actually did what was asked '
|
|
1170
1216
|
+ '(a directive like RUN: or DONE: on one line is the correct format here, not a flaw). '
|
|
1171
1217
|
+ 'Ignore length and confidence of tone.\n'
|