billion-context 0.1.87 → 0.1.88-pr.571.415
This diff represents the content of publicly available package versions that have been released to one of the supported registries. The information contained in this diff is provided for informational purposes only and reflects changes between package versions as they appear in their respective public registries.
- package/README.md +67 -3
- package/README.zh-CN.md +5 -1
- package/dist/agent/omp.js +4 -0
- package/dist/agent/omp.js.map +1 -1
- package/dist/agent/pi.js +4 -0
- package/dist/agent/pi.js.map +1 -1
- package/dist/index.js +1359 -546
- package/dist/index.js.map +1 -1
- package/dist/mcp.js +1 -1
- package/dist/mcp.js.map +1 -1
- package/package.json +1 -1
package/dist/index.js
CHANGED
|
@@ -9365,7 +9365,7 @@ var require_pool = __commonJS({
|
|
|
9365
9365
|
function defaultFactory(origin, opts) {
|
|
9366
9366
|
return new Client(origin, opts);
|
|
9367
9367
|
}
|
|
9368
|
-
var
|
|
9368
|
+
var Pool2 = class extends PoolBase {
|
|
9369
9369
|
constructor(origin, {
|
|
9370
9370
|
connections,
|
|
9371
9371
|
factory = defaultFactory,
|
|
@@ -9438,7 +9438,7 @@ var require_pool = __commonJS({
|
|
|
9438
9438
|
}
|
|
9439
9439
|
}
|
|
9440
9440
|
};
|
|
9441
|
-
module.exports =
|
|
9441
|
+
module.exports = Pool2;
|
|
9442
9442
|
}
|
|
9443
9443
|
});
|
|
9444
9444
|
|
|
@@ -9458,7 +9458,7 @@ var require_balanced_pool = __commonJS({
|
|
|
9458
9458
|
kRemoveClient,
|
|
9459
9459
|
kGetDispatcher
|
|
9460
9460
|
} = require_pool_base();
|
|
9461
|
-
var
|
|
9461
|
+
var Pool2 = require_pool();
|
|
9462
9462
|
var { kUrl } = require_symbols();
|
|
9463
9463
|
var util = require_util();
|
|
9464
9464
|
var kFactory = /* @__PURE__ */ Symbol("factory");
|
|
@@ -9479,7 +9479,7 @@ var require_balanced_pool = __commonJS({
|
|
|
9479
9479
|
return a;
|
|
9480
9480
|
}
|
|
9481
9481
|
function defaultFactory(origin, opts) {
|
|
9482
|
-
return new
|
|
9482
|
+
return new Pool2(origin, opts);
|
|
9483
9483
|
}
|
|
9484
9484
|
var BalancedPool = class extends PoolBase {
|
|
9485
9485
|
constructor(upstreams = [], { factory = defaultFactory, ...opts } = {}) {
|
|
@@ -9714,7 +9714,7 @@ var require_agent = __commonJS({
|
|
|
9714
9714
|
var { InvalidArgumentError, MaxOriginsReachedError } = require_errors();
|
|
9715
9715
|
var { kClients, kRunning, kClose, kDestroy, kDispatch, kUrl } = require_symbols();
|
|
9716
9716
|
var DispatcherBase = require_dispatcher_base();
|
|
9717
|
-
var
|
|
9717
|
+
var Pool2 = require_pool();
|
|
9718
9718
|
var Client = require_client();
|
|
9719
9719
|
var util = require_util();
|
|
9720
9720
|
var kOnConnect = /* @__PURE__ */ Symbol("onConnect");
|
|
@@ -9725,9 +9725,9 @@ var require_agent = __commonJS({
|
|
|
9725
9725
|
var kOptions = /* @__PURE__ */ Symbol("options");
|
|
9726
9726
|
var kOrigins = /* @__PURE__ */ Symbol("origins");
|
|
9727
9727
|
function defaultFactory(origin, opts) {
|
|
9728
|
-
return opts && opts.connections === 1 ? new Client(origin, opts) : new
|
|
9728
|
+
return opts && opts.connections === 1 ? new Client(origin, opts) : new Pool2(origin, opts);
|
|
9729
9729
|
}
|
|
9730
|
-
var
|
|
9730
|
+
var Agent2 = class extends DispatcherBase {
|
|
9731
9731
|
constructor({ factory = defaultFactory, maxOrigins = Infinity, connect, ...options } = {}) {
|
|
9732
9732
|
if (typeof factory !== "function") {
|
|
9733
9733
|
throw new InvalidArgumentError("factory must be a function.");
|
|
@@ -9836,7 +9836,7 @@ var require_agent = __commonJS({
|
|
|
9836
9836
|
return allClientStats;
|
|
9837
9837
|
}
|
|
9838
9838
|
};
|
|
9839
|
-
module.exports =
|
|
9839
|
+
module.exports = Agent2;
|
|
9840
9840
|
}
|
|
9841
9841
|
});
|
|
9842
9842
|
|
|
@@ -10344,7 +10344,7 @@ var require_socks5_proxy_agent = __commonJS({
|
|
|
10344
10344
|
var { InvalidArgumentError } = require_errors();
|
|
10345
10345
|
var { Socks5Client, STATES } = require_socks5_client();
|
|
10346
10346
|
var { kDispatch, kClose, kDestroy } = require_symbols();
|
|
10347
|
-
var
|
|
10347
|
+
var Pool2 = require_pool();
|
|
10348
10348
|
var buildConnector = require_connect();
|
|
10349
10349
|
var { debuglog } = __require("util");
|
|
10350
10350
|
var debug = debuglog("undici:socks5-proxy");
|
|
@@ -10467,7 +10467,7 @@ var require_socks5_proxy_agent = __commonJS({
|
|
|
10467
10467
|
const originKey = String(origin);
|
|
10468
10468
|
let pool = this[kPools].get(originKey);
|
|
10469
10469
|
if (!pool || pool.destroyed || pool.closed) {
|
|
10470
|
-
pool = new
|
|
10470
|
+
pool = new Pool2(origin, {
|
|
10471
10471
|
pipelining: opts.pipelining,
|
|
10472
10472
|
connections: opts.connections,
|
|
10473
10473
|
connect: async (connectOpts, callback) => {
|
|
@@ -10542,8 +10542,8 @@ var require_proxy_agent = __commonJS({
|
|
|
10542
10542
|
"node_modules/undici/lib/dispatcher/proxy-agent.js"(exports, module) {
|
|
10543
10543
|
"use strict";
|
|
10544
10544
|
var { kProxy, kClose, kDestroy, kDispatch } = require_symbols();
|
|
10545
|
-
var
|
|
10546
|
-
var
|
|
10545
|
+
var Agent2 = require_agent();
|
|
10546
|
+
var Pool2 = require_pool();
|
|
10547
10547
|
var DispatcherBase = require_dispatcher_base();
|
|
10548
10548
|
var { InvalidArgumentError, RequestAbortedError, SecureProxyConnectionError } = require_errors();
|
|
10549
10549
|
var buildConnector = require_connect();
|
|
@@ -10561,7 +10561,7 @@ var require_proxy_agent = __commonJS({
|
|
|
10561
10561
|
return protocol === "https:" ? 443 : 80;
|
|
10562
10562
|
}
|
|
10563
10563
|
function defaultFactory(origin, opts) {
|
|
10564
|
-
return new
|
|
10564
|
+
return new Pool2(origin, opts);
|
|
10565
10565
|
}
|
|
10566
10566
|
var noop = () => {
|
|
10567
10567
|
};
|
|
@@ -10569,7 +10569,7 @@ var require_proxy_agent = __commonJS({
|
|
|
10569
10569
|
if (opts.connections === 1) {
|
|
10570
10570
|
return new Client(origin, opts);
|
|
10571
10571
|
}
|
|
10572
|
-
return new
|
|
10572
|
+
return new Pool2(origin, opts);
|
|
10573
10573
|
}
|
|
10574
10574
|
var Http1ProxyWrapper = class extends DispatcherBase {
|
|
10575
10575
|
#client;
|
|
@@ -10673,7 +10673,7 @@ var require_proxy_agent = __commonJS({
|
|
|
10673
10673
|
} else {
|
|
10674
10674
|
this[kClient] = clientFactory(url, { connect });
|
|
10675
10675
|
}
|
|
10676
|
-
this[kAgent] = new
|
|
10676
|
+
this[kAgent] = new Agent2({
|
|
10677
10677
|
...opts,
|
|
10678
10678
|
factory,
|
|
10679
10679
|
connect: async (opts2, callback) => {
|
|
@@ -10801,7 +10801,7 @@ var require_env_http_proxy_agent = __commonJS({
|
|
|
10801
10801
|
var DispatcherBase = require_dispatcher_base();
|
|
10802
10802
|
var { kClose, kDestroy, kClosed, kDestroyed, kDispatch, kNoProxyAgent, kHttpProxyAgent, kHttpsProxyAgent } = require_symbols();
|
|
10803
10803
|
var ProxyAgent2 = require_proxy_agent();
|
|
10804
|
-
var
|
|
10804
|
+
var Agent2 = require_agent();
|
|
10805
10805
|
var DEFAULT_PORTS = {
|
|
10806
10806
|
"http:": 80,
|
|
10807
10807
|
"https:": 443
|
|
@@ -10814,7 +10814,7 @@ var require_env_http_proxy_agent = __commonJS({
|
|
|
10814
10814
|
super();
|
|
10815
10815
|
this.#opts = opts;
|
|
10816
10816
|
const { httpProxy, httpsProxy, noProxy, ...agentOpts } = opts;
|
|
10817
|
-
this[kNoProxyAgent] = new
|
|
10817
|
+
this[kNoProxyAgent] = new Agent2(agentOpts);
|
|
10818
10818
|
const HTTP_PROXY = httpProxy ?? process.env.http_proxy ?? process.env.HTTP_PROXY;
|
|
10819
10819
|
if (HTTP_PROXY) {
|
|
10820
10820
|
this[kHttpProxyAgent] = new ProxyAgent2({ ...agentOpts, uri: HTTP_PROXY });
|
|
@@ -13384,7 +13384,7 @@ var require_mock_pool = __commonJS({
|
|
|
13384
13384
|
"node_modules/undici/lib/mock/mock-pool.js"(exports, module) {
|
|
13385
13385
|
"use strict";
|
|
13386
13386
|
var { promisify: promisify2 } = __require("util");
|
|
13387
|
-
var
|
|
13387
|
+
var Pool2 = require_pool();
|
|
13388
13388
|
var { buildMockDispatch } = require_mock_utils();
|
|
13389
13389
|
var {
|
|
13390
13390
|
kDispatches,
|
|
@@ -13399,7 +13399,7 @@ var require_mock_pool = __commonJS({
|
|
|
13399
13399
|
var { MockInterceptor } = require_mock_interceptor();
|
|
13400
13400
|
var Symbols = require_symbols();
|
|
13401
13401
|
var { InvalidArgumentError } = require_errors();
|
|
13402
|
-
var MockPool = class extends
|
|
13402
|
+
var MockPool = class extends Pool2 {
|
|
13403
13403
|
constructor(origin, opts) {
|
|
13404
13404
|
if (!opts || !opts.agent || typeof opts.agent.dispatch !== "function") {
|
|
13405
13405
|
throw new InvalidArgumentError("Argument opts.agent must implement Agent");
|
|
@@ -13486,7 +13486,7 @@ var require_mock_agent = __commonJS({
|
|
|
13486
13486
|
"node_modules/undici/lib/mock/mock-agent.js"(exports, module) {
|
|
13487
13487
|
"use strict";
|
|
13488
13488
|
var { kClients } = require_symbols();
|
|
13489
|
-
var
|
|
13489
|
+
var Agent2 = require_agent();
|
|
13490
13490
|
var {
|
|
13491
13491
|
kAgent,
|
|
13492
13492
|
kMockAgentSet,
|
|
@@ -13524,7 +13524,7 @@ var require_mock_agent = __commonJS({
|
|
|
13524
13524
|
if (opts?.agent && typeof opts.agent.dispatch !== "function") {
|
|
13525
13525
|
throw new InvalidArgumentError("Argument opts.agent must implement Agent");
|
|
13526
13526
|
}
|
|
13527
|
-
const agent = opts?.agent ? opts.agent : new
|
|
13527
|
+
const agent = opts?.agent ? opts.agent : new Agent2(opts);
|
|
13528
13528
|
this[kAgent] = agent;
|
|
13529
13529
|
this[kClients] = agent[kClients];
|
|
13530
13530
|
this[kOptions] = mockOptions;
|
|
@@ -14131,7 +14131,7 @@ var require_snapshot_recorder = __commonJS({
|
|
|
14131
14131
|
var require_snapshot_agent = __commonJS({
|
|
14132
14132
|
"node_modules/undici/lib/mock/snapshot-agent.js"(exports, module) {
|
|
14133
14133
|
"use strict";
|
|
14134
|
-
var
|
|
14134
|
+
var Agent2 = require_agent();
|
|
14135
14135
|
var MockAgent = require_mock_agent();
|
|
14136
14136
|
var { SnapshotRecorder } = require_snapshot_recorder();
|
|
14137
14137
|
var WrapHandler = require_wrap_handler();
|
|
@@ -14182,7 +14182,7 @@ var require_snapshot_agent = __commonJS({
|
|
|
14182
14182
|
});
|
|
14183
14183
|
this[kSnapshotLoaded] = false;
|
|
14184
14184
|
if (this[kSnapshotMode] === "record" || this[kSnapshotMode] === "update" || this[kSnapshotMode] === "playback" && opts.excludeUrls && opts.excludeUrls.length > 0) {
|
|
14185
|
-
this[kRealAgent] = new
|
|
14185
|
+
this[kRealAgent] = new Agent2(opts);
|
|
14186
14186
|
}
|
|
14187
14187
|
if ((this[kSnapshotMode] === "playback" || this[kSnapshotMode] === "update") && this[kSnapshotPath]) {
|
|
14188
14188
|
this.loadSnapshots().catch(() => {
|
|
@@ -14421,9 +14421,9 @@ var require_global2 = __commonJS({
|
|
|
14421
14421
|
var globalDispatcher = /* @__PURE__ */ Symbol.for("undici.globalDispatcher.2");
|
|
14422
14422
|
var legacyGlobalDispatcher = /* @__PURE__ */ Symbol.for("undici.globalDispatcher.1");
|
|
14423
14423
|
var { InvalidArgumentError } = require_errors();
|
|
14424
|
-
var
|
|
14424
|
+
var Agent2 = require_agent();
|
|
14425
14425
|
if (getGlobalDispatcher() === void 0) {
|
|
14426
|
-
setGlobalDispatcher(new
|
|
14426
|
+
setGlobalDispatcher(new Agent2());
|
|
14427
14427
|
}
|
|
14428
14428
|
function setGlobalDispatcher(agent) {
|
|
14429
14429
|
if (!agent || typeof agent.dispatch !== "function") {
|
|
@@ -18828,7 +18828,7 @@ var require_headers = __commonJS({
|
|
|
18828
18828
|
}
|
|
18829
18829
|
}
|
|
18830
18830
|
};
|
|
18831
|
-
var
|
|
18831
|
+
var Headers2 = class _Headers {
|
|
18832
18832
|
#guard;
|
|
18833
18833
|
/**
|
|
18834
18834
|
* @type {HeadersList}
|
|
@@ -18969,13 +18969,13 @@ var require_headers = __commonJS({
|
|
|
18969
18969
|
target.#headersList = list;
|
|
18970
18970
|
}
|
|
18971
18971
|
};
|
|
18972
|
-
var { getHeadersGuard, setHeadersGuard, getHeadersList, setHeadersList } =
|
|
18973
|
-
Reflect.deleteProperty(
|
|
18974
|
-
Reflect.deleteProperty(
|
|
18975
|
-
Reflect.deleteProperty(
|
|
18976
|
-
Reflect.deleteProperty(
|
|
18977
|
-
iteratorMixin("Headers",
|
|
18978
|
-
Object.defineProperties(
|
|
18972
|
+
var { getHeadersGuard, setHeadersGuard, getHeadersList, setHeadersList } = Headers2;
|
|
18973
|
+
Reflect.deleteProperty(Headers2, "getHeadersGuard");
|
|
18974
|
+
Reflect.deleteProperty(Headers2, "setHeadersGuard");
|
|
18975
|
+
Reflect.deleteProperty(Headers2, "getHeadersList");
|
|
18976
|
+
Reflect.deleteProperty(Headers2, "setHeadersList");
|
|
18977
|
+
iteratorMixin("Headers", Headers2, headersListSortAndCombine, 0, 1);
|
|
18978
|
+
Object.defineProperties(Headers2.prototype, {
|
|
18979
18979
|
append: kEnumerableProperty,
|
|
18980
18980
|
delete: kEnumerableProperty,
|
|
18981
18981
|
get: kEnumerableProperty,
|
|
@@ -18993,7 +18993,7 @@ var require_headers = __commonJS({
|
|
|
18993
18993
|
webidl.converters.HeadersInit = function(V2, prefix, argument) {
|
|
18994
18994
|
if (webidl.util.Type(V2) === webidl.util.Types.OBJECT) {
|
|
18995
18995
|
const iterator = Reflect.get(V2, Symbol.iterator);
|
|
18996
|
-
if (!util.types.isProxy(V2) && iterator ===
|
|
18996
|
+
if (!util.types.isProxy(V2) && iterator === Headers2.prototype.entries) {
|
|
18997
18997
|
try {
|
|
18998
18998
|
return getHeadersList(V2).entriesList;
|
|
18999
18999
|
} catch {
|
|
@@ -19014,7 +19014,7 @@ var require_headers = __commonJS({
|
|
|
19014
19014
|
fill: fill2,
|
|
19015
19015
|
// for test.
|
|
19016
19016
|
compareHeaderName,
|
|
19017
|
-
Headers,
|
|
19017
|
+
Headers: Headers2,
|
|
19018
19018
|
HeadersList,
|
|
19019
19019
|
getHeadersGuard,
|
|
19020
19020
|
setHeadersGuard,
|
|
@@ -19028,7 +19028,7 @@ var require_headers = __commonJS({
|
|
|
19028
19028
|
var require_response = __commonJS({
|
|
19029
19029
|
"node_modules/undici/lib/web/fetch/response.js"(exports, module) {
|
|
19030
19030
|
"use strict";
|
|
19031
|
-
var { Headers, HeadersList, fill: fill2, getHeadersGuard, setHeadersGuard, setHeadersList } = require_headers();
|
|
19031
|
+
var { Headers: Headers2, HeadersList, fill: fill2, getHeadersGuard, setHeadersGuard, setHeadersList } = require_headers();
|
|
19032
19032
|
var { extractBody, cloneBody, mixinBody, streamRegistry, bodyUnusable } = require_body();
|
|
19033
19033
|
var util = require_util();
|
|
19034
19034
|
var nodeUtil = __require("util");
|
|
@@ -19104,7 +19104,7 @@ var require_response = __commonJS({
|
|
|
19104
19104
|
}
|
|
19105
19105
|
init = webidl.converters.ResponseInit(init);
|
|
19106
19106
|
this.#state = makeResponse({});
|
|
19107
|
-
this.#headers = new
|
|
19107
|
+
this.#headers = new Headers2(kConstruct);
|
|
19108
19108
|
setHeadersGuard(this.#headers, "response");
|
|
19109
19109
|
setHeadersList(this.#headers, this.#state.headersList);
|
|
19110
19110
|
let bodyWithType = null;
|
|
@@ -19379,7 +19379,7 @@ var require_response = __commonJS({
|
|
|
19379
19379
|
function fromInnerResponse(innerResponse, guard) {
|
|
19380
19380
|
const response = new Response2(kConstruct);
|
|
19381
19381
|
setResponseState(response, innerResponse);
|
|
19382
|
-
const headers = new
|
|
19382
|
+
const headers = new Headers2(kConstruct);
|
|
19383
19383
|
setResponseHeaders(response, headers);
|
|
19384
19384
|
setHeadersList(headers, innerResponse.headersList);
|
|
19385
19385
|
setHeadersGuard(headers, guard);
|
|
@@ -19451,7 +19451,7 @@ var require_request2 = __commonJS({
|
|
|
19451
19451
|
"node_modules/undici/lib/web/fetch/request.js"(exports, module) {
|
|
19452
19452
|
"use strict";
|
|
19453
19453
|
var { extractBody, mixinBody, cloneBody, bodyUnusable } = require_body();
|
|
19454
|
-
var { Headers, fill: fillHeaders, HeadersList, setHeadersGuard, getHeadersGuard, setHeadersList, getHeadersList } = require_headers();
|
|
19454
|
+
var { Headers: Headers2, fill: fillHeaders, HeadersList, setHeadersGuard, getHeadersGuard, setHeadersList, getHeadersList } = require_headers();
|
|
19455
19455
|
var util = require_util();
|
|
19456
19456
|
var nodeUtil = __require("util");
|
|
19457
19457
|
var {
|
|
@@ -19720,7 +19720,7 @@ var require_request2 = __commonJS({
|
|
|
19720
19720
|
requestFinalizer.register(ac, { signal, abort }, abort);
|
|
19721
19721
|
}
|
|
19722
19722
|
}
|
|
19723
|
-
this.#headers = new
|
|
19723
|
+
this.#headers = new Headers2(kConstruct);
|
|
19724
19724
|
setHeadersList(this.#headers, request.headersList);
|
|
19725
19725
|
setHeadersGuard(this.#headers, "request");
|
|
19726
19726
|
if (mode === "no-cors") {
|
|
@@ -20061,7 +20061,7 @@ var require_request2 = __commonJS({
|
|
|
20061
20061
|
setRequestState(request, innerRequest);
|
|
20062
20062
|
setRequestDispatcher(request, dispatcher);
|
|
20063
20063
|
setRequestSignal(request, signal);
|
|
20064
|
-
const headers = new
|
|
20064
|
+
const headers = new Headers2(kConstruct);
|
|
20065
20065
|
setRequestHeaders(request, headers);
|
|
20066
20066
|
setHeadersList(headers, innerRequest.headersList);
|
|
20067
20067
|
setHeadersGuard(headers, guard);
|
|
@@ -22538,8 +22538,8 @@ var require_cookies = __commonJS({
|
|
|
22538
22538
|
var { parseSetCookie } = require_parse();
|
|
22539
22539
|
var { stringify } = require_util4();
|
|
22540
22540
|
var { webidl } = require_webidl();
|
|
22541
|
-
var { Headers } = require_headers();
|
|
22542
|
-
var brandChecks = webidl.brandCheckMultiple([
|
|
22541
|
+
var { Headers: Headers2 } = require_headers();
|
|
22542
|
+
var brandChecks = webidl.brandCheckMultiple([Headers2, globalThis.Headers].filter(Boolean));
|
|
22543
22543
|
function getCookies(headers) {
|
|
22544
22544
|
webidl.argumentLengthCheck(arguments, 1, "getCookies");
|
|
22545
22545
|
brandChecks(headers);
|
|
@@ -23280,7 +23280,7 @@ var require_connection = __commonJS({
|
|
|
23280
23280
|
var { parseExtensions, isClosed, isClosing, isEstablished, isConnecting, validateCloseCodeAndReason } = require_util5();
|
|
23281
23281
|
var { makeRequest } = require_request2();
|
|
23282
23282
|
var { fetching } = require_fetch();
|
|
23283
|
-
var { Headers, getHeadersList } = require_headers();
|
|
23283
|
+
var { Headers: Headers2, getHeadersList } = require_headers();
|
|
23284
23284
|
var { getDecodeSplit } = require_util2();
|
|
23285
23285
|
var { WebsocketFrameSend } = require_frame();
|
|
23286
23286
|
var assert = __require("assert");
|
|
@@ -23302,7 +23302,7 @@ var require_connection = __commonJS({
|
|
|
23302
23302
|
useURLCredentials: true
|
|
23303
23303
|
});
|
|
23304
23304
|
if (options.headers) {
|
|
23305
|
-
const headersList = getHeadersList(new
|
|
23305
|
+
const headersList = getHeadersList(new Headers2(options.headers));
|
|
23306
23306
|
request.headersList = headersList;
|
|
23307
23307
|
}
|
|
23308
23308
|
const keyValue = crypto2.randomBytes(16).toString("base64");
|
|
@@ -25411,10 +25411,10 @@ var require_undici = __commonJS({
|
|
|
25411
25411
|
"use strict";
|
|
25412
25412
|
var Client = require_client();
|
|
25413
25413
|
var Dispatcher = require_dispatcher();
|
|
25414
|
-
var
|
|
25414
|
+
var Pool2 = require_pool();
|
|
25415
25415
|
var BalancedPool = require_balanced_pool();
|
|
25416
25416
|
var RoundRobinPool = require_round_robin_pool();
|
|
25417
|
-
var
|
|
25417
|
+
var Agent2 = require_agent();
|
|
25418
25418
|
var ProxyAgent2 = require_proxy_agent();
|
|
25419
25419
|
var Socks5ProxyAgent = require_socks5_proxy_agent();
|
|
25420
25420
|
var EnvHttpProxyAgent = require_env_http_proxy_agent();
|
|
@@ -25438,10 +25438,10 @@ var require_undici = __commonJS({
|
|
|
25438
25438
|
Object.assign(Dispatcher.prototype, api);
|
|
25439
25439
|
module.exports.Dispatcher = Dispatcher;
|
|
25440
25440
|
module.exports.Client = Client;
|
|
25441
|
-
module.exports.Pool =
|
|
25441
|
+
module.exports.Pool = Pool2;
|
|
25442
25442
|
module.exports.BalancedPool = BalancedPool;
|
|
25443
25443
|
module.exports.RoundRobinPool = RoundRobinPool;
|
|
25444
|
-
module.exports.Agent =
|
|
25444
|
+
module.exports.Agent = Agent2;
|
|
25445
25445
|
module.exports.ProxyAgent = ProxyAgent2;
|
|
25446
25446
|
module.exports.Socks5ProxyAgent = Socks5ProxyAgent;
|
|
25447
25447
|
module.exports.EnvHttpProxyAgent = EnvHttpProxyAgent;
|
|
@@ -47245,7 +47245,7 @@ function closeLogger() {
|
|
|
47245
47245
|
}
|
|
47246
47246
|
|
|
47247
47247
|
// src/upstream-proxy.ts
|
|
47248
|
-
var
|
|
47248
|
+
var import_undici2 = __toESM(require_undici(), 1);
|
|
47249
47249
|
import { execFileSync } from "child_process";
|
|
47250
47250
|
import net from "net";
|
|
47251
47251
|
import tls from "tls";
|
|
@@ -47323,6 +47323,179 @@ function maskHostInText(text, host) {
|
|
|
47323
47323
|
return text;
|
|
47324
47324
|
}
|
|
47325
47325
|
|
|
47326
|
+
// src/fetch-util.ts
|
|
47327
|
+
var import_undici = __toESM(require_undici(), 1);
|
|
47328
|
+
var MAX_REQUEST_BYTES = 100 * 1024 * 1024;
|
|
47329
|
+
var UPSTREAM_TIMEOUT_MS = 12 * 60 * 1e3;
|
|
47330
|
+
var liveUpstreamTimers = /* @__PURE__ */ new Set();
|
|
47331
|
+
function upstreamTimeoutMs() {
|
|
47332
|
+
const raw = Number(process.env.BILI_UPSTREAM_TIMEOUT_MS);
|
|
47333
|
+
return Number.isInteger(raw) && raw > 0 ? raw : UPSTREAM_TIMEOUT_MS;
|
|
47334
|
+
}
|
|
47335
|
+
var directDispatchers = /* @__PURE__ */ new Map();
|
|
47336
|
+
function directDispatcher(timeoutMs) {
|
|
47337
|
+
let agent = directDispatchers.get(timeoutMs);
|
|
47338
|
+
if (!agent) {
|
|
47339
|
+
agent = new import_undici.Agent({ headersTimeout: timeoutMs, bodyTimeout: timeoutMs });
|
|
47340
|
+
directDispatchers.set(timeoutMs, agent);
|
|
47341
|
+
}
|
|
47342
|
+
return agent;
|
|
47343
|
+
}
|
|
47344
|
+
async function fetchWithTimeout(url, opts, timeoutMs, externalSignal) {
|
|
47345
|
+
const effective = timeoutMs ?? upstreamTimeoutMs();
|
|
47346
|
+
const controller = new AbortController();
|
|
47347
|
+
let cleared = false;
|
|
47348
|
+
const armTimer = () => {
|
|
47349
|
+
const t = setTimeout(() => {
|
|
47350
|
+
liveUpstreamTimers.delete(t);
|
|
47351
|
+
controller.abort();
|
|
47352
|
+
}, effective);
|
|
47353
|
+
liveUpstreamTimers.add(t);
|
|
47354
|
+
return t;
|
|
47355
|
+
};
|
|
47356
|
+
let timer3 = armTimer();
|
|
47357
|
+
const rearm = () => {
|
|
47358
|
+
if (cleared) return;
|
|
47359
|
+
clearTimeout(timer3);
|
|
47360
|
+
liveUpstreamTimers.delete(timer3);
|
|
47361
|
+
timer3 = armTimer();
|
|
47362
|
+
};
|
|
47363
|
+
let onExternalAbort = null;
|
|
47364
|
+
if (externalSignal) {
|
|
47365
|
+
if (externalSignal.aborted) controller.abort();
|
|
47366
|
+
else {
|
|
47367
|
+
onExternalAbort = () => controller.abort();
|
|
47368
|
+
externalSignal.addEventListener("abort", onExternalAbort, { once: true });
|
|
47369
|
+
}
|
|
47370
|
+
}
|
|
47371
|
+
const cleanup = () => {
|
|
47372
|
+
cleared = true;
|
|
47373
|
+
clearTimeout(timer3);
|
|
47374
|
+
liveUpstreamTimers.delete(timer3);
|
|
47375
|
+
if (onExternalAbort && externalSignal) externalSignal.removeEventListener("abort", onExternalAbort);
|
|
47376
|
+
};
|
|
47377
|
+
try {
|
|
47378
|
+
const finalOpts = {
|
|
47379
|
+
...opts,
|
|
47380
|
+
signal: controller.signal,
|
|
47381
|
+
dispatcher: opts.dispatcher ?? directDispatcher(effective)
|
|
47382
|
+
};
|
|
47383
|
+
const raw = await fetch(url, finalOpts);
|
|
47384
|
+
if (raw.body) {
|
|
47385
|
+
const wrapped = armIdleBody(raw.body, rearm);
|
|
47386
|
+
return {
|
|
47387
|
+
response: new Response(wrapped, {
|
|
47388
|
+
status: raw.status,
|
|
47389
|
+
statusText: raw.statusText,
|
|
47390
|
+
headers: raw.headers
|
|
47391
|
+
}),
|
|
47392
|
+
clearTimer: cleanup
|
|
47393
|
+
};
|
|
47394
|
+
}
|
|
47395
|
+
return { response: raw, clearTimer: cleanup };
|
|
47396
|
+
} catch (e) {
|
|
47397
|
+
cleanup();
|
|
47398
|
+
throw e;
|
|
47399
|
+
}
|
|
47400
|
+
}
|
|
47401
|
+
function armIdleBody(body, rearm) {
|
|
47402
|
+
const reader = body.getReader();
|
|
47403
|
+
return new ReadableStream({
|
|
47404
|
+
async pull(controller) {
|
|
47405
|
+
try {
|
|
47406
|
+
const result = await reader.read();
|
|
47407
|
+
if (result.done) {
|
|
47408
|
+
controller.close();
|
|
47409
|
+
return;
|
|
47410
|
+
}
|
|
47411
|
+
rearm();
|
|
47412
|
+
controller.enqueue(result.value);
|
|
47413
|
+
} catch (e) {
|
|
47414
|
+
controller.error(e);
|
|
47415
|
+
}
|
|
47416
|
+
},
|
|
47417
|
+
async cancel(reason) {
|
|
47418
|
+
try {
|
|
47419
|
+
await reader.cancel(reason);
|
|
47420
|
+
} catch {
|
|
47421
|
+
}
|
|
47422
|
+
}
|
|
47423
|
+
});
|
|
47424
|
+
}
|
|
47425
|
+
var UpstreamHttpError = class extends Error {
|
|
47426
|
+
status;
|
|
47427
|
+
body;
|
|
47428
|
+
attempts;
|
|
47429
|
+
constructor(status, body, attempts) {
|
|
47430
|
+
super(`upstream error ${status}`);
|
|
47431
|
+
this.name = "UpstreamHttpError";
|
|
47432
|
+
this.status = status;
|
|
47433
|
+
this.body = body;
|
|
47434
|
+
this.attempts = attempts;
|
|
47435
|
+
}
|
|
47436
|
+
};
|
|
47437
|
+
var TRANSIENT_BODY_MARKERS = [
|
|
47438
|
+
"captcha",
|
|
47439
|
+
"verify failed",
|
|
47440
|
+
"risk control",
|
|
47441
|
+
"\u98CE\u63A7",
|
|
47442
|
+
"rate limit",
|
|
47443
|
+
"too many requests",
|
|
47444
|
+
"try again"
|
|
47445
|
+
];
|
|
47446
|
+
function isTransientUpstreamError(status, body) {
|
|
47447
|
+
if (status === 429 || status >= 500) return true;
|
|
47448
|
+
if (status < 400) return false;
|
|
47449
|
+
const lower = body.toLowerCase();
|
|
47450
|
+
return TRANSIENT_BODY_MARKERS.some((marker) => lower.includes(marker));
|
|
47451
|
+
}
|
|
47452
|
+
var REPLAY_MAX_ATTEMPTS = 3;
|
|
47453
|
+
function replayMaxAttempts() {
|
|
47454
|
+
const raw = Number(process.env.BILI_REPLAY_RETRY_MAX);
|
|
47455
|
+
return Number.isInteger(raw) && raw >= 1 ? raw : REPLAY_MAX_ATTEMPTS;
|
|
47456
|
+
}
|
|
47457
|
+
function replayBaseDelayMs() {
|
|
47458
|
+
const raw = Number(process.env.BILI_REPLAY_RETRY_BASE_MS);
|
|
47459
|
+
return Number.isFinite(raw) && raw >= 0 ? raw : 1500;
|
|
47460
|
+
}
|
|
47461
|
+
function maxShrinkPerCompress() {
|
|
47462
|
+
const raw = Number(process.env.BILI_MAX_SHRINK_PER_COMPRESS);
|
|
47463
|
+
return Number.isFinite(raw) && raw > 0 && raw <= 1 ? raw : void 0;
|
|
47464
|
+
}
|
|
47465
|
+
function replayBackoffMs(attempt) {
|
|
47466
|
+
return replayBaseDelayMs() * 2 ** (attempt - 1);
|
|
47467
|
+
}
|
|
47468
|
+
function sleep(ms2, signal) {
|
|
47469
|
+
if (ms2 <= 0 || signal?.aborted) return Promise.resolve();
|
|
47470
|
+
return new Promise((resolve) => {
|
|
47471
|
+
let timer3 = null;
|
|
47472
|
+
const finish2 = () => {
|
|
47473
|
+
if (timer3) clearTimeout(timer3);
|
|
47474
|
+
if (signal) signal.removeEventListener("abort", finish2);
|
|
47475
|
+
resolve();
|
|
47476
|
+
};
|
|
47477
|
+
timer3 = setTimeout(finish2, ms2);
|
|
47478
|
+
if (signal) signal.addEventListener("abort", finish2, { once: true });
|
|
47479
|
+
});
|
|
47480
|
+
}
|
|
47481
|
+
async function fetchWithRetry(url, opts, timeoutMs, externalSignal, onRetry) {
|
|
47482
|
+
const maxAttempts = replayMaxAttempts();
|
|
47483
|
+
for (let attempt = 1; ; attempt++) {
|
|
47484
|
+
const result = await fetchWithTimeout(url, opts, timeoutMs, externalSignal);
|
|
47485
|
+
if (result.response.ok) return result;
|
|
47486
|
+
const errText2 = await result.response.text().catch(() => "upstream error");
|
|
47487
|
+
result.clearTimer();
|
|
47488
|
+
const lastAttempt = attempt >= maxAttempts;
|
|
47489
|
+
if (!lastAttempt && isTransientUpstreamError(result.response.status, errText2)) {
|
|
47490
|
+
const delayMs = replayBackoffMs(attempt);
|
|
47491
|
+
onRetry?.({ attempt, status: result.response.status, detail: errText2, delayMs, maxAttempts });
|
|
47492
|
+
await sleep(delayMs, externalSignal);
|
|
47493
|
+
continue;
|
|
47494
|
+
}
|
|
47495
|
+
throw new UpstreamHttpError(result.response.status, errText2, attempt);
|
|
47496
|
+
}
|
|
47497
|
+
}
|
|
47498
|
+
|
|
47326
47499
|
// src/upstream-proxy.ts
|
|
47327
47500
|
var dispatcherCache = /* @__PURE__ */ new Map();
|
|
47328
47501
|
var lastConnection = {};
|
|
@@ -47519,12 +47692,21 @@ function resolveProxyDecision(routes, globalProxy, upstreamUrl, fallback = {}) {
|
|
|
47519
47692
|
function resolveProxy(routes, globalProxy, upstreamUrl, fallback = {}) {
|
|
47520
47693
|
return resolveProxyDecision(routes, globalProxy, upstreamUrl, fallback).proxy;
|
|
47521
47694
|
}
|
|
47522
|
-
function proxyDispatcher(proxyUrl) {
|
|
47695
|
+
function proxyDispatcher(proxyUrl, timeoutMs) {
|
|
47523
47696
|
if (!proxyUrl) return void 0;
|
|
47524
|
-
|
|
47697
|
+
const t = timeoutMs ?? upstreamTimeoutMs();
|
|
47698
|
+
const key = `${proxyUrl}\0${t}`;
|
|
47699
|
+
let agent = dispatcherCache.get(key);
|
|
47525
47700
|
if (!agent) {
|
|
47526
|
-
|
|
47527
|
-
|
|
47701
|
+
const withTimeouts = (options) => ({ ...options, headersTimeout: t, bodyTimeout: t });
|
|
47702
|
+
agent = new import_undici2.ProxyAgent({
|
|
47703
|
+
uri: proxyUrl,
|
|
47704
|
+
headersTimeout: t,
|
|
47705
|
+
bodyTimeout: t,
|
|
47706
|
+
factory: (origin, options) => new import_undici2.Pool(origin, withTimeouts(options)),
|
|
47707
|
+
clientFactory: (origin, options) => new import_undici2.Pool(origin, withTimeouts(options))
|
|
47708
|
+
});
|
|
47709
|
+
dispatcherCache.set(key, agent);
|
|
47528
47710
|
}
|
|
47529
47711
|
return agent;
|
|
47530
47712
|
}
|
|
@@ -47677,6 +47859,95 @@ function getUpstreamConnectionStatus() {
|
|
|
47677
47859
|
return { ...lastConnection };
|
|
47678
47860
|
}
|
|
47679
47861
|
|
|
47862
|
+
// src/compat-roles.ts
|
|
47863
|
+
function parseCompatRoles(v2) {
|
|
47864
|
+
if (!v2 || typeof v2 !== "object" || Array.isArray(v2)) return void 0;
|
|
47865
|
+
const obj = v2;
|
|
47866
|
+
let out;
|
|
47867
|
+
for (const [k2, val] of Object.entries(obj)) {
|
|
47868
|
+
if (typeof val !== "string" || val.length === 0) continue;
|
|
47869
|
+
out ??= {};
|
|
47870
|
+
out[k2] = val;
|
|
47871
|
+
}
|
|
47872
|
+
return out;
|
|
47873
|
+
}
|
|
47874
|
+
function resolveCompatRoles(routes, upstreamUrl, globalRoles) {
|
|
47875
|
+
const providerRoles = findRoute(routes, upstreamUrl)?.compat?.roles;
|
|
47876
|
+
if (!globalRoles && !providerRoles) return {};
|
|
47877
|
+
return { ...globalRoles, ...providerRoles };
|
|
47878
|
+
}
|
|
47879
|
+
function applyCompatRolesJson(parsed, protocol, roles) {
|
|
47880
|
+
if (Object.keys(roles).length === 0) return 0;
|
|
47881
|
+
const items = protocol === "openai" ? parsed.messages : parsed.input;
|
|
47882
|
+
if (!Array.isArray(items)) return 0;
|
|
47883
|
+
let rewritten = 0;
|
|
47884
|
+
for (const item of items) {
|
|
47885
|
+
if (!item || typeof item !== "object" || Array.isArray(item)) continue;
|
|
47886
|
+
const msg = item;
|
|
47887
|
+
if (protocol === "responses" && msg.type !== void 0 && msg.type !== "message") continue;
|
|
47888
|
+
const role = msg.role;
|
|
47889
|
+
if (typeof role !== "string") continue;
|
|
47890
|
+
const mapped = roles[role];
|
|
47891
|
+
if (mapped === void 0 || mapped === role) continue;
|
|
47892
|
+
msg.role = mapped;
|
|
47893
|
+
rewritten++;
|
|
47894
|
+
}
|
|
47895
|
+
return rewritten;
|
|
47896
|
+
}
|
|
47897
|
+
var ROLE_CAPTURE_STOPWORDS = /* @__PURE__ */ new Set([
|
|
47898
|
+
"must",
|
|
47899
|
+
"should",
|
|
47900
|
+
"be",
|
|
47901
|
+
"is",
|
|
47902
|
+
"one",
|
|
47903
|
+
"of",
|
|
47904
|
+
"the",
|
|
47905
|
+
"a",
|
|
47906
|
+
"an",
|
|
47907
|
+
"in",
|
|
47908
|
+
"for",
|
|
47909
|
+
"not",
|
|
47910
|
+
"was",
|
|
47911
|
+
"and",
|
|
47912
|
+
"or",
|
|
47913
|
+
"to",
|
|
47914
|
+
"only",
|
|
47915
|
+
"allowed",
|
|
47916
|
+
"supported",
|
|
47917
|
+
"valid",
|
|
47918
|
+
"value",
|
|
47919
|
+
"message",
|
|
47920
|
+
"messages"
|
|
47921
|
+
]);
|
|
47922
|
+
var ROLE_REJECTION_PATTERNS = [
|
|
47923
|
+
/(?:invalid|unknown|unsupported|unrecognized|unexpected)[a-z ]{0,24}?role\s*[:=]?\s*["'`]?([a-z][a-z0-9_-]{1,31})["'`]?/i,
|
|
47924
|
+
/role\s*[:=]?\s*["'`]?([a-z][a-z0-9_-]{1,31})["'`]?\s+(?:is\s+)?not\s+(?:supported|allowed|recognized|valid|accepted)/i
|
|
47925
|
+
];
|
|
47926
|
+
function detectRoleRejection(status, bodyText) {
|
|
47927
|
+
if (status !== 400) return null;
|
|
47928
|
+
const head = bodyText.slice(0, 4096);
|
|
47929
|
+
for (const re2 of ROLE_REJECTION_PATTERNS) {
|
|
47930
|
+
const m2 = re2.exec(head);
|
|
47931
|
+
if (!m2) continue;
|
|
47932
|
+
const role = m2[1].toLowerCase();
|
|
47933
|
+
if (ROLE_CAPTURE_STOPWORDS.has(role)) continue;
|
|
47934
|
+
return { role };
|
|
47935
|
+
}
|
|
47936
|
+
return null;
|
|
47937
|
+
}
|
|
47938
|
+
function applyCompatRoles(body, protocol, roles) {
|
|
47939
|
+
if (Object.keys(roles).length === 0) return { body, rewritten: 0 };
|
|
47940
|
+
let parsed;
|
|
47941
|
+
try {
|
|
47942
|
+
parsed = JSON.parse(body);
|
|
47943
|
+
} catch {
|
|
47944
|
+
return { body, rewritten: 0 };
|
|
47945
|
+
}
|
|
47946
|
+
const rewritten = applyCompatRolesJson(parsed, protocol, roles);
|
|
47947
|
+
if (rewritten === 0) return { body, rewritten: 0 };
|
|
47948
|
+
return { body: JSON.stringify(parsed), rewritten };
|
|
47949
|
+
}
|
|
47950
|
+
|
|
47680
47951
|
// src/config.ts
|
|
47681
47952
|
function safeReadJson(path18) {
|
|
47682
47953
|
try {
|
|
@@ -47823,6 +48094,7 @@ function loadOptions(env = process.env) {
|
|
|
47823
48094
|
promptCache: {
|
|
47824
48095
|
routing: parsePromptCacheRouting(env.ACP_PROMPT_CACHE_ROUTING ?? fileConfig.promptCache?.routing)
|
|
47825
48096
|
},
|
|
48097
|
+
compat: { roles: parseCompatRoles(fileConfig.compat?.roles) ?? {} },
|
|
47826
48098
|
sessionHeader: env.ACP_SESSION_HEADER ?? fileConfig.sessionHeader ?? "x-acp-session",
|
|
47827
48099
|
log: env.ACP_LOG !== "0" && fileConfig.log !== false,
|
|
47828
48100
|
debug: (env.ACP_DEBUG ?? (fileConfig.debug ? "1" : "0")) === "1",
|
|
@@ -47892,6 +48164,8 @@ function parseRouteEntry(v2) {
|
|
|
47892
48164
|
if (typeof obj.proxy === "string") route.proxy = obj.proxy;
|
|
47893
48165
|
if (obj.compressProtocol === "marker" || obj.compressProtocol === "tools") route.compressProtocol = obj.compressProtocol;
|
|
47894
48166
|
if (obj.compress) route.compress = obj.compress;
|
|
48167
|
+
const compatRoles = parseCompatRoles(obj.compat?.roles);
|
|
48168
|
+
if (compatRoles) route.compat = { roles: compatRoles };
|
|
47895
48169
|
return route;
|
|
47896
48170
|
}
|
|
47897
48171
|
if (v2 === null) return {};
|
|
@@ -48072,157 +48346,6 @@ import { readFile, writeFile, mkdir } from "fs/promises";
|
|
|
48072
48346
|
import { existsSync as existsSync2, statSync as statSync2 } from "fs";
|
|
48073
48347
|
import path3 from "path";
|
|
48074
48348
|
|
|
48075
|
-
// src/fetch-util.ts
|
|
48076
|
-
var MAX_REQUEST_BYTES = 100 * 1024 * 1024;
|
|
48077
|
-
var UPSTREAM_TIMEOUT_MS = 10 * 60 * 1e3;
|
|
48078
|
-
var liveUpstreamTimers = /* @__PURE__ */ new Set();
|
|
48079
|
-
async function fetchWithTimeout(url, opts, timeoutMs = UPSTREAM_TIMEOUT_MS, externalSignal) {
|
|
48080
|
-
const controller = new AbortController();
|
|
48081
|
-
const armTimer = () => {
|
|
48082
|
-
const t = setTimeout(() => {
|
|
48083
|
-
liveUpstreamTimers.delete(t);
|
|
48084
|
-
controller.abort();
|
|
48085
|
-
}, timeoutMs);
|
|
48086
|
-
liveUpstreamTimers.add(t);
|
|
48087
|
-
return t;
|
|
48088
|
-
};
|
|
48089
|
-
let timer3 = armTimer();
|
|
48090
|
-
const rearm = () => {
|
|
48091
|
-
clearTimeout(timer3);
|
|
48092
|
-
liveUpstreamTimers.delete(timer3);
|
|
48093
|
-
timer3 = armTimer();
|
|
48094
|
-
};
|
|
48095
|
-
let onExternalAbort = null;
|
|
48096
|
-
if (externalSignal) {
|
|
48097
|
-
if (externalSignal.aborted) controller.abort();
|
|
48098
|
-
else {
|
|
48099
|
-
onExternalAbort = () => controller.abort();
|
|
48100
|
-
externalSignal.addEventListener("abort", onExternalAbort, { once: true });
|
|
48101
|
-
}
|
|
48102
|
-
}
|
|
48103
|
-
const cleanup = () => {
|
|
48104
|
-
clearTimeout(timer3);
|
|
48105
|
-
liveUpstreamTimers.delete(timer3);
|
|
48106
|
-
if (onExternalAbort && externalSignal) externalSignal.removeEventListener("abort", onExternalAbort);
|
|
48107
|
-
};
|
|
48108
|
-
try {
|
|
48109
|
-
const finalOpts = { ...opts, signal: controller.signal };
|
|
48110
|
-
const raw = await fetch(url, finalOpts);
|
|
48111
|
-
if (raw.body) {
|
|
48112
|
-
const wrapped = armIdleBody(raw.body, rearm);
|
|
48113
|
-
return {
|
|
48114
|
-
response: new Response(wrapped, {
|
|
48115
|
-
status: raw.status,
|
|
48116
|
-
statusText: raw.statusText,
|
|
48117
|
-
headers: raw.headers
|
|
48118
|
-
}),
|
|
48119
|
-
clearTimer: cleanup
|
|
48120
|
-
};
|
|
48121
|
-
}
|
|
48122
|
-
return { response: raw, clearTimer: cleanup };
|
|
48123
|
-
} catch (e) {
|
|
48124
|
-
cleanup();
|
|
48125
|
-
throw e;
|
|
48126
|
-
}
|
|
48127
|
-
}
|
|
48128
|
-
function armIdleBody(body, rearm) {
|
|
48129
|
-
const reader = body.getReader();
|
|
48130
|
-
return new ReadableStream({
|
|
48131
|
-
async pull(controller) {
|
|
48132
|
-
try {
|
|
48133
|
-
const result = await reader.read();
|
|
48134
|
-
if (result.done) {
|
|
48135
|
-
controller.close();
|
|
48136
|
-
return;
|
|
48137
|
-
}
|
|
48138
|
-
rearm();
|
|
48139
|
-
controller.enqueue(result.value);
|
|
48140
|
-
} catch (e) {
|
|
48141
|
-
controller.error(e);
|
|
48142
|
-
}
|
|
48143
|
-
},
|
|
48144
|
-
async cancel(reason) {
|
|
48145
|
-
try {
|
|
48146
|
-
await reader.cancel(reason);
|
|
48147
|
-
} catch {
|
|
48148
|
-
}
|
|
48149
|
-
}
|
|
48150
|
-
});
|
|
48151
|
-
}
|
|
48152
|
-
var UpstreamHttpError = class extends Error {
|
|
48153
|
-
status;
|
|
48154
|
-
body;
|
|
48155
|
-
attempts;
|
|
48156
|
-
constructor(status, body, attempts) {
|
|
48157
|
-
super(`upstream error ${status}`);
|
|
48158
|
-
this.name = "UpstreamHttpError";
|
|
48159
|
-
this.status = status;
|
|
48160
|
-
this.body = body;
|
|
48161
|
-
this.attempts = attempts;
|
|
48162
|
-
}
|
|
48163
|
-
};
|
|
48164
|
-
var TRANSIENT_BODY_MARKERS = [
|
|
48165
|
-
"captcha",
|
|
48166
|
-
"verify failed",
|
|
48167
|
-
"risk control",
|
|
48168
|
-
"\u98CE\u63A7",
|
|
48169
|
-
"rate limit",
|
|
48170
|
-
"too many requests",
|
|
48171
|
-
"try again"
|
|
48172
|
-
];
|
|
48173
|
-
function isTransientUpstreamError(status, body) {
|
|
48174
|
-
if (status === 429 || status >= 500) return true;
|
|
48175
|
-
if (status < 400) return false;
|
|
48176
|
-
const lower = body.toLowerCase();
|
|
48177
|
-
return TRANSIENT_BODY_MARKERS.some((marker) => lower.includes(marker));
|
|
48178
|
-
}
|
|
48179
|
-
var REPLAY_MAX_ATTEMPTS = 3;
|
|
48180
|
-
function replayMaxAttempts() {
|
|
48181
|
-
const raw = Number(process.env.BILI_REPLAY_RETRY_MAX);
|
|
48182
|
-
return Number.isInteger(raw) && raw >= 1 ? raw : REPLAY_MAX_ATTEMPTS;
|
|
48183
|
-
}
|
|
48184
|
-
function replayBaseDelayMs() {
|
|
48185
|
-
const raw = Number(process.env.BILI_REPLAY_RETRY_BASE_MS);
|
|
48186
|
-
return Number.isFinite(raw) && raw >= 0 ? raw : 1500;
|
|
48187
|
-
}
|
|
48188
|
-
function maxShrinkPerCompress() {
|
|
48189
|
-
const raw = Number(process.env.BILI_MAX_SHRINK_PER_COMPRESS);
|
|
48190
|
-
return Number.isFinite(raw) && raw > 0 && raw <= 1 ? raw : void 0;
|
|
48191
|
-
}
|
|
48192
|
-
function replayBackoffMs(attempt) {
|
|
48193
|
-
return replayBaseDelayMs() * 2 ** (attempt - 1);
|
|
48194
|
-
}
|
|
48195
|
-
function sleep(ms2, signal) {
|
|
48196
|
-
if (ms2 <= 0 || signal?.aborted) return Promise.resolve();
|
|
48197
|
-
return new Promise((resolve) => {
|
|
48198
|
-
let timer3 = null;
|
|
48199
|
-
const finish2 = () => {
|
|
48200
|
-
if (timer3) clearTimeout(timer3);
|
|
48201
|
-
if (signal) signal.removeEventListener("abort", finish2);
|
|
48202
|
-
resolve();
|
|
48203
|
-
};
|
|
48204
|
-
timer3 = setTimeout(finish2, ms2);
|
|
48205
|
-
if (signal) signal.addEventListener("abort", finish2, { once: true });
|
|
48206
|
-
});
|
|
48207
|
-
}
|
|
48208
|
-
async function fetchWithRetry(url, opts, timeoutMs, externalSignal, onRetry) {
|
|
48209
|
-
const maxAttempts = replayMaxAttempts();
|
|
48210
|
-
for (let attempt = 1; ; attempt++) {
|
|
48211
|
-
const result = await fetchWithTimeout(url, opts, timeoutMs, externalSignal);
|
|
48212
|
-
if (result.response.ok) return result;
|
|
48213
|
-
const errText2 = await result.response.text().catch(() => "upstream error");
|
|
48214
|
-
result.clearTimer();
|
|
48215
|
-
const lastAttempt = attempt >= maxAttempts;
|
|
48216
|
-
if (!lastAttempt && isTransientUpstreamError(result.response.status, errText2)) {
|
|
48217
|
-
const delayMs = replayBackoffMs(attempt);
|
|
48218
|
-
onRetry?.({ attempt, status: result.response.status, detail: errText2, delayMs, maxAttempts });
|
|
48219
|
-
await sleep(delayMs, externalSignal);
|
|
48220
|
-
continue;
|
|
48221
|
-
}
|
|
48222
|
-
throw new UpstreamHttpError(result.response.status, errText2, attempt);
|
|
48223
|
-
}
|
|
48224
|
-
}
|
|
48225
|
-
|
|
48226
48349
|
// src/registry-snapshot.json
|
|
48227
48350
|
var registry_snapshot_default = { fetchedAt: "2026-08-24T10:39:50.602Z", count: 355, models: { "tencent/hy3-preview": { id: "tencent/hy3-preview", name: "Hy3 preview", description: "Tencent Hy reasoning model for coding, instruction following, and agent tasks", family: "Hy", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-04-20", last_updated: "2026-04-20", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 64e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/tencent/Hy3-preview" }], benchmarks: [{ name: "SWE-Bench Verified", score: 74.4, metric: "resolved", source: "https://huggingface.co/tencent/Hy3-preview" }] }, "tencent/hy3": { id: "tencent/hy3", name: "Hy3", description: "Tencent Hy reasoning model for coding, instruction following, and agent tasks", family: "Hy", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-07-06", last_updated: "2026-07-06", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 64e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/tencent/Hy3" }], benchmarks: [{ name: "SWE-Bench Verified", score: 78, metric: "resolved", source: "https://huggingface.co/tencent/Hy3" }] }, "nvidia/llama-3.3-nemotron-super-49b-v1": { id: "nvidia/llama-3.3-nemotron-super-49b-v1", name: "Llama 3.3 Nemotron Super 49B v1", description: "Nemotron model for efficient reasoning, coding, and specialized AI agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-04-07", last_updated: "2025-04-07", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 } }, "nvidia/nemotron-3-nano-30b-a3b": { id: "nvidia/nemotron-3-nano-30b-a3b", name: "Nemotron 3 Nano 30B A3B", description: "Small Nemotron 3 MoE for efficient coding, math, and long-context agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-15", last_updated: "2025-12-15", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 } }, "nvidia/nemotron-3.5-content-safety": { id: "nvidia/nemotron-3.5-content-safety", name: "Nemotron 3.5 Content Safety", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "nemotron", attachment: true, reasoning: true, tool_call: false, temperature: true, release_date: "2026-06-04", last_updated: "2026-06-04", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/nemotron-cascade-2-30b-a3b": { id: "nvidia/nemotron-cascade-2-30b-a3b", name: "Nemotron Cascade 2 30B A3B", description: "Nemotron model for efficient reasoning, coding, and specialized AI agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-03-24", last_updated: "2026-04-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 32768 } }, "nvidia/nemotron-3-content-safety": { id: "nvidia/nemotron-3-content-safety", name: "Nemotron 3 Content Safety", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "nemotron", attachment: false, reasoning: false, tool_call: false, temperature: false, release_date: "2026-04-16", last_updated: "2026-04-16", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 } }, "nvidia/llama-3.1-nemotron-70b-instruct": { id: "nvidia/llama-3.1-nemotron-70b-instruct", name: "Llama 3.1 Nemotron 70B Instruct", description: "Nemotron model for efficient reasoning, coding, and specialized AI agents", family: "nemotron", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-04-15", last_updated: "2025-04-15", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/nemotron-nano-12b-v2-vl": { id: "nvidia/nemotron-nano-12b-v2-vl", name: "Nemotron Nano 12B v2 VL", description: "Nemotron multimodal model for visual reasoning and agentic AI workflows", family: "nemotron", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2025-10-28", last_updated: "2025-10-28", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 } }, "nvidia/nemotron-voicechat": { id: "nvidia/nemotron-voicechat", name: "Nemotron VoiceChat", description: "Nemotron multimodal model for visual reasoning and agentic AI workflows", family: "nemotron", attachment: true, reasoning: false, tool_call: true, temperature: true, release_date: "2026-03-16", last_updated: "2026-03-16", modalities: { input: ["text", "audio"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/llama-nemotron-rerank-vl-1b-v2": { id: "nvidia/llama-nemotron-rerank-vl-1b-v2", name: "Llama Nemotron Rerank VL 1B v2", description: "Reranking model for improving retrieval quality in search and recommendation systems", family: "nemotron", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-03-31", last_updated: "2026-03-31", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 } }, "nvidia/llama-3.3-nemotron-super-49b-v1.5": { id: "nvidia/llama-3.3-nemotron-super-49b-v1.5", name: "Llama 3.3 Nemotron Super 49B v1.5", description: "Nemotron model for efficient reasoning, coding, and specialized AI agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-07-25", last_updated: "2025-07-25", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 } }, "nvidia/nemotron-3-super-120b-a12b": { id: "nvidia/nemotron-3-super-120b-a12b", name: "Nemotron 3 Super 120B A12B", description: "Nemotron middle tier for collaborative agents and high-volume reasoning workloads", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-03-11", last_updated: "2026-03-11", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 } }, "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning": { id: "nvidia/nemotron-3-nano-omni-30b-a3b-reasoning", name: "Nemotron 3 Nano Omni 30B A3B Reasoning", description: "Open Nemotron omni model combining reasoning with text, vision, and audio", family: "nemotron", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-04-28", last_updated: "2026-04-28", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 65536 } }, "nvidia/nemotron-3.5-lightning": { id: "nvidia/nemotron-3.5-lightning", name: "Nemotron 3.5 Lightning 30B A3B", description: "Fast NVIDIA Nemotron MoE for reliable agentic tasks across enterprise workloads", family: "nemotron", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-11", last_updated: "2026-08-11", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 } }, "nvidia/nemotron-content-safety-reasoning-4b": { id: "nvidia/nemotron-content-safety-reasoning-4b", name: "Nemotron Content Safety Reasoning 4B", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "nemotron", attachment: false, reasoning: true, tool_call: false, temperature: false, release_date: "2026-01-22", last_updated: "2026-01-22", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 } }, "nvidia/nemotron-3-ultra-550b-a55b": { id: "nvidia/nemotron-3-ultra-550b-a55b", name: "Nemotron 3 Ultra 550B A55B", description: "Largest Nemotron 3 model for maximum open-weight reasoning and agent accuracy", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-04", last_updated: "2026-06-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 70.7, metric: "resolved", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "SWE-Bench Multilingual", score: 67.7, metric: "resolve rate", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "Terminal-Bench", score: 56.4, metric: "success rate", version: "2.1", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "GPQA", score: 87, metric: "accuracy", variant: "no tools", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "Humanity's Last Exam", score: 26.7, metric: "accuracy", variant: "no tools", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "Humanity's Last Exam", score: 37.4, metric: "accuracy", variant: "with tools", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "LiveCodeBench", score: 89, metric: "pass@1", version: "v6", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "MMLU-Pro", score: 86.8, metric: "accuracy", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "BrowseComp", score: 44.4, metric: "accuracy", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "IFBench", score: 81.7, metric: "accuracy", variant: "prompt loose", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }, { name: "GDPval", score: 46.7, metric: "wins or ties", source: "https://huggingface.co/nvidia/NVIDIA-Nemotron-3-Ultra-550B-A55B-BF16", date: "2026-06-04" }] }, "nvidia/mistral-nemotron": { id: "nvidia/mistral-nemotron", name: "Mistral Nemotron", description: "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", family: "nemotron", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-06-11", last_updated: "2025-06-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/llama-3.1-nemotron-safety-guard-8b-v3": { id: "nvidia/llama-3.1-nemotron-safety-guard-8b-v3", name: "Llama 3.1 Nemotron Safety Guard 8B v3", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "nemotron", attachment: false, reasoning: false, tool_call: false, temperature: false, release_date: "2025-10-28", last_updated: "2025-10-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 } }, "nvidia/nemotron-nano-9b-v2": { id: "nvidia/nemotron-nano-9b-v2", name: "Nemotron Nano 9B v2", description: "Compact Nemotron model for efficient reasoning and deployable AI agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-08-18", last_updated: "2025-08-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 } }, "nvidia/llama-3.1-nemotron-ultra-253b": { id: "nvidia/llama-3.1-nemotron-ultra-253b", name: "Llama 3.1 Nemotron Ultra 253B", description: "Flagship Nemotron model for high-throughput reasoning and complex agents", family: "nemotron", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-04-07", last_updated: "2025-04-07", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/nemotron-mini-4b-instruct": { id: "nvidia/nemotron-mini-4b-instruct", name: "Nemotron Mini 4B Instruct", description: "Compact Nemotron model for efficient reasoning and deployable AI agents", family: "nemotron", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2024-08-21", last_updated: "2024-08-26", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8192 } }, "nvidia/llama-nemotron-embed-vl-1b-v2": { id: "nvidia/llama-nemotron-embed-vl-1b-v2", name: "Llama Nemotron Embed VL 1B v2", description: "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", family: "nemotron", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-02-10", last_updated: "2026-02-10", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 32768, output: 2048 } }, "aisingapore/gemma-sea-lion-v4-27b-it": { id: "aisingapore/gemma-sea-lion-v4-27b-it", name: "Gemma-SEA-LION-v4-27B-IT", description: "Gemma 3 27B tuned by AI Singapore for Southeast Asian languages and instruction following", family: "gemma", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/aisingapore/Gemma-SEA-LION-v4-27B-IT" }] }, "microsoft/phi-4-mini": { id: "microsoft/phi-4-mini", name: "Phi-4-mini", description: "Compact Microsoft instruction model tuned for efficient coding assistance, reasoning, and low-latency agent tasks", family: "phi", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-10", release_date: "2024-12-11", last_updated: "2024-12-11", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, links: [{ label: "Weights", url: "https://huggingface.co/microsoft/Phi-4-mini-instruct", type: "weights" }], benchmarks: [{ name: "MMLU", score: 67.3, metric: "accuracy", source: "https://huggingface.co/microsoft/Phi-4-mini-instruct/resolve/main/README.md" }] }, "microsoft/mai-code-1-flash": { id: "microsoft/mai-code-1-flash", name: "MAI-Code-1-Flash", description: "Microsoft coding model built for fast, efficient assistance in everyday developer workflows", family: "mai", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-12", release_date: "2026-06-02", last_updated: "2026-06-08", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 128e3 }, links: [{ label: "Model card", url: "https://microsoft.ai/pdf/MAI-Code-1-Flash-Model-Card.PDF", type: "model_card" }, { label: "Announcement", url: "https://microsoft.ai/news/introducingmai-code-1-flash/", type: "announcement" }], benchmarks: [{ name: "SWE-Bench Pro", score: 51.2, metric: "resolve rate", harness: "GitHub Copilot", source: "https://microsoft.ai/news/introducingmai-code-1-flash/", date: "2026-06-02" }, { name: "SWE-Bench Verified", score: 71.6, metric: "resolved", source: "https://llm-stats.com/benchmarks/swe-bench-verified" }, { name: "Terminal-Bench", score: 54.8, metric: "success rate", version: "2.0", source: "https://llm-stats.com/benchmarks/terminal-bench-2" }, { name: "GPQA Diamond", score: 84.6, metric: "accuracy", source: "https://llm-stats.com/benchmarks/gpqa" }] }, "microsoft/mai-code-1.1-flash": { id: "microsoft/mai-code-1.1-flash", name: "MAI-Code-1.1-Flash", description: "Microsoft coding model with native vision support, optimized for fast and efficient software development", family: "mai", attachment: true, reasoning: true, tool_call: true, structured_output: true, release_date: "2026-08-11", last_updated: "2026-08-11", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 128e3 }, links: [{ label: "Announcement", url: "https://microsoft.ai/news/mai-code-1-1-flash-br-better-faster-at-a-quarter-of-the-cost/", type: "announcement" }] }, "deepseek/deepseek-r1-distill-qwen-32b": { id: "deepseek/deepseek-r1-distill-qwen-32b", name: "DeepSeek-R1-Distill-Qwen-32B", description: "R1 reasoning distilled into Qwen 2.5 32B for efficient open-weight step-by-step problem solving", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: false, temperature: true, release_date: "2025-01-20", last_updated: "2025-01-20", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-R1-Distill-Qwen-32B" }] }, "deepseek/deepseek-r1": { id: "deepseek/deepseek-r1", name: "DeepSeek-R1", description: "Classic open reasoning model for transparent math, coding, and deliberate problem solving", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-07", release_date: "2025-01-20", last_updated: "2025-05-29", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-R1" }], benchmarks: [{ name: "Aider Polyglot", score: 56.9, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-20" }, { name: "Artificial Analysis Coding Index", score: 15.9, metric: "index", source: "https://openrouter.ai/deepseek/deepseek-r1/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 35.7, metric: "percent correct", source: "https://openrouter.ai/deepseek/deepseek-r1/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 6.1, metric: "success rate", source: "https://openrouter.ai/deepseek/deepseek-r1/benchmarks", date: "2026-03-11" }] }, "deepseek/deepseek-v3-0324": { id: "deepseek/deepseek-v3-0324", name: "DeepSeek V3 0324", description: "March 2025 checkpoint of DeepSeek-V3 with improved reasoning and coding", family: "deepseek", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-03-24", last_updated: "2025-03-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 163840, output: 163840 }, weights: [{ label: "Model weights", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3-0324", format: "safetensors" }] }, "deepseek/deepseek-v4-flash": { id: "deepseek/deepseek-v4-flash", name: "DeepSeek V4 Flash", description: "Fast DeepSeek V4 lane for economical reasoning, coding, and long-context work", family: "deepseek-flash", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-05", release_date: "2026-04-24", last_updated: "2026-04-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash" }], benchmarks: [{ name: "SWE-Bench Verified", score: 79, metric: "resolved", source: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash" }] }, "deepseek/deepseek-v4-pro-0813": { id: "deepseek/deepseek-v4-pro-0813", name: "DeepSeek V4 Pro 0813", description: "DeepSeek V4 Pro snapshot with million-token context and support for thinking and non-thinking modes", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-12", last_updated: "2026-08-22", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, license: "MIT", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro-0813" }] }, "deepseek/deepseek-v3.1": { id: "deepseek/deepseek-v3.1", name: "DeepSeek-V3.1", description: "Hybrid-reasoning DeepSeek model with thinking and non-thinking modes", family: "deepseek", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-08-21", last_updated: "2025-08-21", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "MIT License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3.1" }] }, "deepseek/deepseek-v4-pro": { id: "deepseek/deepseek-v4-pro", name: "DeepSeek V4 Pro", description: "Open MoE flagship with million-token context for coding and long agent runs", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-05", release_date: "2026-04-24", last_updated: "2026-04-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro" }], benchmarks: [{ name: "SWE-Bench Verified", score: 80.6, metric: "resolved", source: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Pro" }, { name: "Artificial Analysis Coding Agent Index", score: 50.1, metric: "average pass@1", harness: "Claude Code", variant: "high", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 67.8, metric: "pass@1", harness: "Claude Code", variant: "high", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 18, metric: "pass@1", harness: "Claude Code", variant: "high", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 64.7, metric: "pass@1", harness: "Claude Code", variant: "high", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }] }, "deepseek/deepseek-v3": { id: "deepseek/deepseek-v3", name: "DeepSeek-V3", description: "Open DeepSeek MoE chat model for coding, math, and general reasoning", family: "deepseek", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2024-12-26", last_updated: "2024-12-26", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "DeepSeek Model License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3" }] }, "deepseek/deepseek-v3.2": { id: "deepseek/deepseek-v3.2", name: "DeepSeek V3.2", description: "Hybrid-reasoning DeepSeek model with thinking and non-thinking modes, sparse attention, and tool-use", family: "deepseek", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-07", release_date: "2025-12-01", last_updated: "2025-12-01", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 64e3 }, license: "MIT License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3.2" }] }, "deepseek/deepseek-chat": { id: "deepseek/deepseek-chat", name: "DeepSeek Chat", description: "DeepSeek chat model for instruction following, coding, and analysis", family: "deepseek", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-09", release_date: "2025-12-01", last_updated: "2026-02-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3.2" }], benchmarks: [{ name: "Aider Polyglot", score: 70.2, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-10-03" }] }, "deepseek/deepseek-reasoner": { id: "deepseek/deepseek-reasoner", name: "DeepSeek Reasoner", description: "DeepSeek reasoning model for multi-step analysis, math, coding, and tools", family: "deepseek-thinking", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-09", release_date: "2025-12-01", last_updated: "2026-02-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V3.2" }], benchmarks: [{ name: "Aider Polyglot", score: 74.2, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-10-03" }] }, "deepseek/deepseek-ocr-2": { id: "deepseek/deepseek-ocr-2", name: "DeepSeek OCR 2", description: "High-accuracy OCR model for extracting text from documents, screenshots, receipts, and natural scenes", attachment: true, reasoning: false, tool_call: false, release_date: "2026-01-27", last_updated: "2026-01-27", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 8192, output: 8192 } }, "deepseek/deepseek-v4-flash-0731": { id: "deepseek/deepseek-v4-flash-0731", name: "DeepSeek V4 Flash 0731", description: "Official DeepSeek V4 Flash release with enhanced agentic capabilities and integrated DSpark speculative decoding", family: "deepseek-flash", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-05", release_date: "2026-07-31", last_updated: "2026-07-31", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 }, license: "MIT", weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731" }], benchmarks: [{ name: "Terminal-Bench", score: 82.7, metric: "pass@1", variant: "max", version: "2.1", source: "https://huggingface.co/deepseek-ai/DeepSeek-V4-Flash-0731" }, { name: "NL2Repo", score: 54.2, metric: "resolve rate", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "CyberGym", score: 76.7, metric: "score", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "DeepSWE", score: 54.4, metric: "resolve rate", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "Toolathlon-Verified", score: 70.3, metric: "score", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "Agents' Last Exam", score: 25.2, metric: "score", harness: "DeepSeek Harness minimal mode", variant: "max effort", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "AutomationBench", score: 25.1, metric: "success rate", variant: "max effort", dataset: "public", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "DSBench-FullStack", score: 68.7, metric: "score", variant: "max effort", dataset: "internal", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }, { name: "DSBench-Hard", score: 59.6, metric: "score", variant: "max effort", dataset: "internal", source: "https://api-docs.deepseek.com/updates/", date: "2026-07-31" }] }, "deepseek/deepseek-v4-pro-0423": { id: "deepseek/deepseek-v4-pro-0423", name: "DeepSeek V4 Pro 0423", description: "DeepSeek V4 Pro initial snapshot with million-token context and support for thinking and non-thinking modes", family: "deepseek-thinking", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-05", release_date: "2026-04-23", last_updated: "2026-04-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 384e3 } }, "deepseek/deepseek-v4-flash-vision-exp": { id: "deepseek/deepseek-v4-flash-vision-exp", name: "DeepSeek V4 Flash Vision Exp", description: "Experimental multimodal DeepSeek V4 Flash model for image understanding, coding, and agentic work", family: "deepseek-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-21", last_updated: "2026-08-21", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 384e3 } }, "arcee-ai/trinity-large-preview": { id: "arcee-ai/trinity-large-preview", name: "Trinity Large Preview", description: "Lightly post-trained 398B MoE chat model for creative work, long-context prompts, and tool-using agents", family: "trinity", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2026-01-27", last_updated: "2026-05-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 524288, output: 262144 }, license: "OpenMDW-1.1", links: [{ label: "Model card", url: "https://huggingface.co/arcee-ai/Trinity-Large-Preview", type: "model_card" }, { label: "Announcement", url: "https://www.arcee.ai/blog/trinity-large", type: "announcement" }, { label: "License", url: "https://huggingface.co/arcee-ai/Trinity-Large-Preview/blob/main/LICENSE", type: "license" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/arcee-ai/Trinity-Large-Preview", format: "safetensors" }] }, "arcee-ai/trinity-mini": { id: "arcee-ai/trinity-mini", name: "Trinity Mini", description: "Reasoning-tuned 26B MoE model with 3B active parameters for agents, tools, and multi-step workloads", family: "trinity", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-01", last_updated: "2026-05-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, license: "OpenMDW-1.1", links: [{ label: "Model card", url: "https://huggingface.co/arcee-ai/Trinity-Mini", type: "model_card" }, { label: "Announcement", url: "https://www.arcee.ai/blog/the-trinity-manifesto", type: "announcement" }, { label: "License", url: "https://huggingface.co/arcee-ai/Trinity-Mini/blob/main/LICENSE", type: "license" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/arcee-ai/Trinity-Mini", format: "safetensors" }] }, "arcee-ai/trinity-large-thinking": { id: "arcee-ai/trinity-large-thinking", name: "Trinity Large Thinking", description: "Reasoning-optimized 398B MoE agent model with extended thinking for long-horizon and multi-turn tool use", family: "trinity", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-04-01", last_updated: "2026-05-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 524288, output: 262144 }, license: "OpenMDW-1.1", links: [{ label: "Model card", url: "https://huggingface.co/arcee-ai/Trinity-Large-Thinking", type: "model_card" }, { label: "Announcement", url: "https://www.arcee.ai/blog/trinity-large-thinking", type: "announcement" }, { label: "License", url: "https://huggingface.co/arcee-ai/Trinity-Large-Thinking/blob/main/LICENSE", type: "license" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/arcee-ai/Trinity-Large-Thinking", format: "safetensors" }] }, "arcee-ai/trinity-nano-preview": { id: "arcee-ai/trinity-nano-preview", name: "Trinity Nano Preview", description: "Experimental chat-tuned 6B MoE model with 1B active parameters for low-resource chat and instruction following", family: "trinity", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-12-01", last_updated: "2026-05-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, license: "OpenMDW-1.1", links: [{ label: "Model card", url: "https://huggingface.co/arcee-ai/Trinity-Nano-Preview", type: "model_card" }, { label: "Announcement", url: "https://www.arcee.ai/blog/the-trinity-manifesto", type: "announcement" }, { label: "License", url: "https://huggingface.co/arcee-ai/Trinity-Nano-Preview/blob/main/LICENSE", type: "license" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/arcee-ai/Trinity-Nano-Preview", format: "safetensors" }] }, "google/gemini-3.1-flash-tts-preview": { id: "google/gemini-3.1-flash-tts-preview", name: "Gemini 3.1 Flash TTS Preview", description: "Low-latency speech generation with steerable prompts and expressive audio tags", family: "gemini-flash", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2026-04-15", last_updated: "2026-04-15", modalities: { input: ["text"], output: ["audio"] }, open_weights: false, limit: { context: 8192, output: 16384 } }, "google/gemma-4-26b-a4b-it": { id: "google/gemma-4-26b-a4b-it", name: "Gemma 4 26B A4B IT", description: "Open Gemma instruction model for efficient chat and self-hosted deployments", family: "gemma", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/google/gemma-4-26B-A4B-it" }] }, "google/gemini-3-pro-image": { id: "google/gemini-3-pro-image", name: "Nano Banana Pro", description: "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", family: "gemini-pro", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2026-05-28", last_updated: "2026-05-28", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 65536, output: 32768 } }, "google/gemini-3-pro-image-preview": { id: "google/gemini-3-pro-image-preview", name: "Nano Banana Pro", description: "Nano Banana Pro for higher-fidelity image generation and design-heavy edits", family: "gemini-pro", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2025-11-20", last_updated: "2025-11-20", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 65536, output: 32768 } }, "google/gemini-flash-lite-latest": { id: "google/gemini-flash-lite-latest", name: "Gemini Flash-Lite Latest", description: "Fast Gemini model balancing multimodal reasoning, tool use, and cost", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-07-21", last_updated: "2026-07-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-2.0-flash": { id: "google/gemini-2.0-flash", name: "Gemini 2.0 Flash", description: "Earlier Gemini Flash workhorse for responsive multimodal apps and tool use", family: "gemini-flash", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-06", release_date: "2024-12-11", last_updated: "2024-12-11", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 8192 } }, "google/gemini-2.5-flash-tts": { id: "google/gemini-2.5-flash-tts", name: "Gemini 2.5 Flash TTS", description: "Speech generation model for controllable voice, narration, and audio delivery", family: "gemini-flash", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2025-09-30", last_updated: "2025-12-10", modalities: { input: ["text"], output: ["audio"] }, open_weights: false, limit: { context: 32768, output: 16384 } }, "google/gemini-3.5-flash-lite": { id: "google/gemini-3.5-flash-lite", name: "Gemini 3.5 Flash Lite", description: "Fast Gemini model balancing multimodal reasoning, tool use, and cost", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-07-21", last_updated: "2026-07-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 54.2, metric: "resolve rate", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "Terminal-Bench", score: 54, metric: "accuracy", harness: "Terminus 2", version: "2.1", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "MLE-Bench", score: 39.2, metric: "average position score", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "GDPval-AA", score: 1140, metric: "Elo", version: "v2", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "OSWorld-Verified", score: 74, metric: "success rate", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "CharXiv Reasoning", score: 74.5, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "CharXiv Reasoning", score: 76.5, metric: "accuracy", variant: "with tools", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "GDM-MRCR", score: 72.2, metric: "accuracy", variant: "128k average, 8-needle", version: "v2", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }, { name: "GDM-MRCR", score: 21.3, metric: "accuracy", variant: "1M pointwise, 8-needle", version: "v2", source: "https://deepmind.google/models/model-cards/gemini-3-5-flash-lite/", date: "2026-07-21" }] }, "google/gemini-3.1-flash-lite-image": { id: "google/gemini-3.1-flash-lite-image", name: "Nano Banana 2 Lite", description: "Fastest, most cost-efficient Gemini image model for high-volume 1K generation and editing", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: true, knowledge: "2025-01", release_date: "2026-06-30", last_updated: "2026-06-30", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 65536, output: 4096 } }, "google/gemini-3.1-pro-preview": { id: "google/gemini-3.1-pro-preview", name: "Gemini 3.1 Pro Preview", description: "Reasoning-first Gemini preview for agentic coding and complex problem solving", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-02-19", last_updated: "2026-02-19", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 54.2, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "Terminal-Bench", score: 70.3, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "SWE-Bench Pro", score: 46.1, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }, { name: "SWE-Atlas Codebase QnA", score: 13.5, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 33.81, metric: "score", harness: "Gemini CLI", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 29.84, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 43, metric: "average pass@1", harness: "Gemini CLI", variant: "high", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 45.6, metric: "pass@1", harness: "Gemini CLI", variant: "high", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 15.1, metric: "pass@1", harness: "Gemini CLI", variant: "high", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 68.3, metric: "pass@1", harness: "Gemini CLI", variant: "high", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "GPQA Diamond", score: 94.3, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 44.4, metric: "accuracy", dataset: "full set, text + MM", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "ARC-AGI-2", score: 77.1, metric: "accuracy", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "MMMU Pro", score: 80.5, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "MCP Atlas", score: 78.2, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "OSWorld-Verified", score: 76.2, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "CharXiv Reasoning", score: 83.3, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "GDPval-AA", score: 1314, metric: "Elo", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }] }, "google/gemini-3.1-pro-preview-customtools": { id: "google/gemini-3.1-pro-preview-customtools", name: "Gemini 3.1 Pro Preview Custom Tools", description: "Advanced Gemini model for complex reasoning, coding, and multimodal analysis", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-02-19", last_updated: "2026-02-19", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/deep-research-preview-04-2026": { id: "google/deep-research-preview-04-2026", name: "Gemini Deep Research Preview", description: "Agentic model for autonomous multi-step research, synthesis, and cited reports", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2025-01", release_date: "2026-04-21", last_updated: "2026-04-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text", "image"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-2.5-flash-lite": { id: "google/gemini-2.5-flash-lite", name: "Gemini 2.5 Flash-Lite", description: "Lean Gemini 2.5 lane for cheap multimodal traffic and quick agents", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-06-17", last_updated: "2025-06-17", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 9.5, metric: "index", source: "https://openrouter.ai/google/gemini-2.5-flash-lite/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 19.3, metric: "percent correct", source: "https://openrouter.ai/google/gemini-2.5-flash-lite/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 4.5, metric: "success rate", source: "https://openrouter.ai/google/gemini-2.5-flash-lite/benchmarks", date: "2026-03-11" }] }, "google/gemini-robotics-er-1.6-preview": { id: "google/gemini-robotics-er-1.6-preview", name: "Gemini Robotics-ER 1.6 Preview", description: "Vision-language model for embodied reasoning: spatial understanding, task planning, and physical-world agentic robotics", family: "gemini", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-04-14", last_updated: "2026-04-14", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 65536 } }, "google/gemma-4-31b-it": { id: "google/gemma-4-31b-it", name: "Gemma 4 31B IT", description: "Largest Gemma 4 instruction model for open, self-hosted chat and reasoning", family: "gemma", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/google/gemma-4-31B-it" }] }, "google/gemini-2.5-computer-use-preview-10-2025": { id: "google/gemini-2.5-computer-use-preview-10-2025", name: "Gemini 2.5 Computer Use Preview", description: "Specialized Gemini 2.5 model for browser-control agents that automate UI tasks", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-01", release_date: "2025-10-07", last_updated: "2025-10-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 64e3 } }, "google/lyria-3-clip-preview": { id: "google/lyria-3-clip-preview", name: "Lyria 3 Clip Preview", description: "Music generation model for short 30-second clips, loops, and previews from text or image prompts", family: "lyria", attachment: true, reasoning: false, tool_call: false, structured_output: false, temperature: true, release_date: "2026-03-25", last_updated: "2026-03-25", modalities: { input: ["text", "image"], output: ["text", "audio"] }, open_weights: false, limit: { context: 131072, output: 65536 } }, "google/veo-3.1-fast-generate-preview": { id: "google/veo-3.1-fast-generate-preview", name: "Veo 3.1 Fast Preview", description: "Video model for prompt-guided generation, editing, and motion workflows", family: "veo", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2025-10-15", last_updated: "2026-01-01", modalities: { input: ["text", "image", "video"], output: ["video"] }, open_weights: false, limit: { context: 1024, output: 0 } }, "google/gemini-embedding-001": { id: "google/gemini-embedding-001", name: "Gemini Embedding 001", description: "Embedding model for semantic search, retrieval, clustering, and ranking pipelines", family: "gemini", attachment: false, reasoning: false, tool_call: false, temperature: false, knowledge: "2025-05", release_date: "2025-05-20", last_updated: "2025-05-20", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 2048, output: 1 } }, "google/gemini-flash-latest": { id: "google/gemini-flash-latest", name: "Gemini Flash Latest", description: "High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-08-13", last_updated: "2026-08-13", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-3.5-flash": { id: "google/gemini-3.5-flash", name: "Gemini 3.5 Flash", description: "Fast Gemini model balancing multimodal reasoning, tool use, and cost", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-05-19", last_updated: "2026-05-19", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "Terminal-Bench", score: 76.2, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "SWE-Bench Pro", score: 55.1, metric: "resolve rate", variant: "single attempt", dataset: "public", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "MCP Atlas", score: 83.6, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "Toolathlon", score: 56.5, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "OSWorld-Verified", score: 78.4, metric: "success rate", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "MMMU Pro", score: 83.6, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "CharXiv Reasoning", score: 84.2, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "Humanity's Last Exam", score: 40.2, metric: "accuracy", dataset: "full set, text + MM", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "ARC-AGI-2", score: 72.1, metric: "accuracy", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }, { name: "GDPval-AA", score: 1656, metric: "Elo", source: "https://deepmind.google/models/gemini/flash/", date: "2026-05-19" }] }, "google/gemini-3.5-live-translate-preview": { id: "google/gemini-3.5-live-translate-preview", name: "Gemini 3.5 Live Translate Preview", description: "Low-latency audio-to-audio model for real-time speech translation across 70+ languages", family: "gemini-pro", attachment: false, reasoning: false, tool_call: false, temperature: false, knowledge: "2025-01", release_date: "2026-06-09", last_updated: "2026-06-09", modalities: { input: ["audio"], output: ["audio", "text"] }, open_weights: false, limit: { context: 131072, output: 65536 } }, "google/veo-3.1-lite-generate-preview": { id: "google/veo-3.1-lite-generate-preview", name: "Veo 3.1 Lite Preview", description: "Video model for prompt-guided generation, editing, and motion workflows", family: "veo", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-03-31", last_updated: "2026-03-31", modalities: { input: ["text", "image"], output: ["video"] }, open_weights: false, limit: { context: 1024, output: 0 } }, "google/gemini-omni-flash-preview": { id: "google/gemini-omni-flash-preview", name: "Gemini Omni Flash Preview", description: "Video generation and editing model for fast, conversational text- and image-to-video workflows", family: "gemini", attachment: true, reasoning: true, tool_call: false, release_date: "2026-06-30", last_updated: "2026-06-30", modalities: { input: ["text", "image", "video"], output: ["video"] }, open_weights: false, limit: { context: 1048576, output: 57920 }, benchmarks: [{ name: "LMArena Text-to-Video Arena", score: 1527, metric: "Elo", source: "https://venturebeat.com/technology/googles-gemini-omni-flash-hits-the-api-turning-enterprise-video-production-into-a-conversation", date: "2026-06-30" }] }, "google/gemini-3.1-flash-lite": { id: "google/gemini-3.1-flash-lite", name: "Gemini 3.1 Flash Lite", description: "Low-latency Gemini model for high-volume multimodal and agent workloads", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-05-07", last_updated: "2026-05-07", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-2.5-flash": { id: "google/gemini-2.5-flash", name: "Gemini 2.5 Flash", description: "Fast Gemini workhorse for multimodal apps where latency and price matter", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-06-17", last_updated: "2025-06-17", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "Aider Polyglot", score: 55.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-25" }, { name: "Artificial Analysis Coding Index", score: 22.2, metric: "index", source: "https://openrouter.ai/google/gemini-2.5-flash/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 39.4, metric: "percent correct", source: "https://openrouter.ai/google/gemini-2.5-flash/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 13.6, metric: "success rate", source: "https://openrouter.ai/google/gemini-2.5-flash/benchmarks", date: "2026-06-02" }] }, "google/gemini-3-pro-preview": { id: "google/gemini-3-pro-preview", name: "Gemini 3 Pro Preview", description: "Preview Gemini flagship for complex reasoning, coding, and rich multimodal prompts", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-11-18", last_updated: "2025-11-18", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 43.3, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "google/veo-3.1-generate-preview": { id: "google/veo-3.1-generate-preview", name: "Veo 3.1 Preview", description: "Video model for prompt-guided generation, editing, and motion workflows", family: "veo", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2025-10-15", last_updated: "2026-01", modalities: { input: ["text", "image"], output: ["video"] }, open_weights: false, limit: { context: 1024, output: 1 } }, "google/gemini-2.5-pro-tts": { id: "google/gemini-2.5-pro-tts", name: "Gemini 2.5 Pro TTS", description: "Speech generation model for controllable voice, narration, and audio delivery", family: "gemini-pro", attachment: false, reasoning: false, tool_call: false, temperature: false, knowledge: "2025-01", release_date: "2025-09-30", last_updated: "2025-12-10", modalities: { input: ["text"], output: ["audio"] }, open_weights: false, limit: { context: 32768, output: 16384 } }, "google/gemini-embedding-2": { id: "google/gemini-embedding-2", name: "Gemini Embedding 2", description: "Multimodal embedding model mapping text, images, video, audio, and PDFs into a unified embedding space", family: "gemini", attachment: true, reasoning: false, tool_call: false, temperature: false, knowledge: "2025-11", release_date: "2026-04-22", last_updated: "2026-04-22", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 8192, output: 3072 } }, "google/gemma-4-E4B-it": { id: "google/gemma-4-E4B-it", name: "Gemma 4 E4B IT", description: "Open Gemma instruction model for efficient chat and self-hosted deployments", family: "gemma", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image", "audio"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/google/gemma-4-E4B-it" }] }, "google/gemma-4-E2B-it": { id: "google/gemma-4-E2B-it", name: "Gemma 4 E2B IT", description: "Open Gemma instruction model for efficient chat and self-hosted deployments", family: "gemma", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image", "audio"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/google/gemma-4-E2B-it" }] }, "google/gemini-3.6-flash": { id: "google/gemini-3.6-flash", name: "Gemini 3.6 Flash", description: "Fast Gemini model balancing multimodal reasoning, tool use, and cost", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-07-21", last_updated: "2026-07-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 58.7, metric: "resolve rate", harness: "Antigravity", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "DeepSWE", score: 49, metric: "resolve rate", variant: "high reasoning", version: "1.1", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "Terminal-Bench", score: 78, metric: "accuracy", harness: "Terminus 2", version: "2.1", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "MLE-Bench", score: 63.9, metric: "average position score", dataset: "Partial 30", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "GDPval-AA", score: 1421, metric: "Elo", version: "v2", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "OSWorld-Verified", score: 83, metric: "success rate", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "CharXiv Reasoning", score: 85.2, metric: "accuracy", variant: "no tools", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "CharXiv Reasoning", score: 89.4, metric: "accuracy", variant: "with tools", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "GDM-MRCR", score: 91.8, metric: "accuracy", variant: "128k average, 8-needle", version: "v2", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }, { name: "GDM-MRCR", score: 54, metric: "accuracy", variant: "1M pointwise, 8-needle", version: "v2", source: "https://deepmind.google/models/evals-methodology/gemini-3-6-flash/", date: "2026-07-21" }] }, "google/gemini-3.1-flash-image": { id: "google/gemini-3.1-flash-image", name: "Nano Banana 2", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "gemini-flash", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2026-05-28", last_updated: "2026-05-28", modalities: { input: ["text", "image", "video", "pdf"], output: ["text", "image"] }, open_weights: false, limit: { context: 131072, output: 32768 } }, "google/gemini-3.1-flash-image-preview": { id: "google/gemini-3.1-flash-image-preview", name: "Nano Banana 2", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "gemini-flash", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-01", release_date: "2026-02-26", last_updated: "2026-02-26", modalities: { input: ["text", "image", "pdf"], output: ["text", "image"] }, open_weights: false, limit: { context: 65536, output: 65536 } }, "google/gemini-2.0-flash-lite": { id: "google/gemini-2.0-flash-lite", name: "Gemini 2.0 Flash-Lite", description: "Low-latency Gemini model for high-volume multimodal and agent workloads", family: "gemini-flash-lite", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-06", release_date: "2024-12-11", last_updated: "2024-12-11", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 8192 } }, "google/lyria-3-pro-preview": { id: "google/lyria-3-pro-preview", name: "Lyria 3 Pro Preview", description: "Music generation model for full-length songs from text or images with vocals and structure", family: "lyria", attachment: true, reasoning: false, tool_call: false, structured_output: false, temperature: true, release_date: "2026-03-25", last_updated: "2026-03-25", modalities: { input: ["text", "image"], output: ["text", "audio"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "google/gemini-3.1-flash-lite-preview": { id: "google/gemini-3.1-flash-lite-preview", name: "Gemini 3.1 Flash Lite Preview", description: "Low-latency Gemini model for high-volume multimodal and agent workloads", family: "gemini-flash-lite", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-03-03", last_updated: "2026-03-03", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-3-flash-preview": { id: "google/gemini-3-flash-preview", name: "Gemini 3 Flash Preview", description: "New Gemini flash lane bringing frontier-style multimodal reasoning to cheaper runs", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-12-17", last_updated: "2025-12-17", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "SWE-Bench Pro", score: 34.63, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }, { name: "SWE-Atlas Codebase QnA", score: 8.2, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 10, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 30.3, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] }, "google/gemini-2.5-flash-image": { id: "google/gemini-2.5-flash-image", name: "Nano Banana", description: "Nano Banana image model for fast generation, edits, and character-consistent assets", family: "gemini-flash", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2024-06", release_date: "2025-08-26", last_updated: "2025-08-26", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 32768, output: 32768 } }, "google/gemini-2.5-pro": { id: "google/gemini-2.5-pro", name: "Gemini 2.5 Pro", description: "Google's proven reasoning model for coding, math, and multimodal analysis", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2025-06-17", last_updated: "2025-06-17", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "Aider Polyglot", score: 83.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-06-06" }, { name: "Artificial Analysis Coding Index", score: 32, metric: "index", source: "https://openrouter.ai/google/gemini-2.5-pro/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 42.8, metric: "percent correct", source: "https://openrouter.ai/google/gemini-2.5-pro/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 26.5, metric: "success rate", source: "https://openrouter.ai/google/gemini-2.5-pro/benchmarks", date: "2026-06-02" }] }, "google/deep-research-max-preview-04-2026": { id: "google/deep-research-max-preview-04-2026", name: "Deep Research Max Preview", description: "Maximum-comprehensiveness agentic researcher for multi-step investigation, synthesis, and cited reports", family: "gemini-pro", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2025-01", release_date: "2026-04-21", last_updated: "2026-04-21", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text", "image"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "google/gemini-3.1-flash-live-preview": { id: "google/gemini-3.1-flash-live-preview", name: "Gemini 3.1 Flash Live Preview", description: "High-quality, low-latency Live API model for real-time dialogue and voice-first AI applications", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: true, knowledge: "2025-01", release_date: "2026-03-26", last_updated: "2026-03-26", modalities: { input: ["text", "image", "video", "audio"], output: ["text", "audio"] }, open_weights: false, limit: { context: 131072, output: 65536 } }, "google/gemini-3.7-flash": { id: "google/gemini-3.7-flash", name: "Gemini 3.7 Flash", description: "High-efficiency Gemini model for agentic workflows, coding, and multimodal reasoning", family: "gemini-flash", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-03", release_date: "2026-08-13", last_updated: "2026-08-13", modalities: { input: ["text", "image", "video", "audio", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 }, benchmarks: [{ name: "FrontierCode", score: 43.6, metric: "score", version: "1.1 Main", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "DeepSWE", score: 65.3, metric: "resolve rate", version: "1.1", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "Terminal-Bench", score: 85.8, metric: "accuracy", version: "2.1", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "AutomationBench", score: 30.4, metric: "accuracy", dataset: "private set", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "GDP.pdf", score: 34, metric: "accuracy", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }, { name: "GDM-MRCR", score: 97, metric: "accuracy", variant: "128k average, 8-needle", version: "v2", source: "https://deepmind.google/models/model-cards/gemini-3-7-flash/", date: "2026-08-13" }] }, "meta/llama-guard-3-8b": { id: "meta/llama-guard-3-8b", name: "Llama-Guard-3-8B", description: "Llama 3.1-based safety classifier for moderating prompts and model responses", family: "llama", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2023-12", release_date: "2024-07-23", last_updated: "2024-07-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-Guard-3-8B" }] }, "meta/llama-3.3-70b-instruct": { id: "meta/llama-3.3-70b-instruct", name: "Llama-3.3-70B-Instruct", description: "Popular open Llama workhorse for multilingual chat, coding, and self-hosting", family: "llama", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-12", release_date: "2024-12-06", last_updated: "2024-12-06", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.3-70B-Instruct" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 10.7, metric: "index", source: "https://openrouter.ai/meta-llama/llama-3.3-70b-instruct/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 26, metric: "percent correct", source: "https://openrouter.ai/meta-llama/llama-3.3-70b-instruct/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 3, metric: "success rate", source: "https://openrouter.ai/meta-llama/llama-3.3-70b-instruct/benchmarks", date: "2026-03-11" }] }, "meta/llama-4-scout-17b-instruct": { id: "meta/llama-4-scout-17b-instruct", name: "Llama 4 Scout 17B Instruct", description: "Open Llama with long-context vision for efficient multimodal agents", family: "llama", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-08", release_date: "2025-04-05", last_updated: "2025-04-05", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 35e5, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-4-Scout-17B-16E-Instruct" }] }, "meta/muse-glimmer-30b": { id: "meta/muse-glimmer-30b", name: "Muse Glimmer 30B", description: "Muse Glimmer is a 30-billion-parameter open-weight multimodal model from Meta Superintelligence Labs, distilled from Muse Spark for always-on local agents, tool use, coding, and image understanding.", family: "muse", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-01-04", release_date: "2026-08-10", last_updated: "2026-08-10", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, license: "Apache 2.0", links: [{ label: "Announcement", url: "https://research.meta.ai/blog/introducing-muse-glimmer-open-agentic-model", type: "announcement" }, { label: "Model card", url: "https://huggingface.co/meta-models/Muse-Glimmer-30B", type: "model_card" }, { label: "Developer docs", url: "https://developer.meta.com/ai/models/muse-glimmer/", type: "docs" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-models/Muse-Glimmer-30B" }], benchmarks: [{ name: "MCP Atlas", score: 75.5, metric: "success rate", variant: "public", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "DeepSearch QA", score: 74.6, source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "SWE-Bench Pro", score: 51.2, metric: "resolve rate", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "SWE-Bench Verified", score: 76, metric: "resolve rate", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "Terminal-Bench", score: 51.7, metric: "success rate", variant: "with terminus2", version: "2.1", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "OSWorld-Verified", score: 65.9, metric: "success rate", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "AIME 2026", score: 94.7, metric: "accuracy", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "GPQA Diamond", score: 83.5, metric: "accuracy", variant: "AA", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }, { name: "CharXiv Reasoning", score: 78.8, metric: "accuracy", source: "https://huggingface.co/meta-models/Muse-Glimmer-30B", date: "2026-08-10" }] }, "meta/llama-3.1-8b-instruct": { id: "meta/llama-3.1-8b-instruct", name: "Llama-3.1-8B-Instruct", description: "Compact open Llama model for lightweight chat, drafting, and self-hosting", family: "llama", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-12", release_date: "2024-07-23", last_updated: "2024-07-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.1-8B-Instruct" }] }, "meta/muse-spark-1.2": { id: "meta/muse-spark-1.2", name: "Muse Spark 1.2", description: "Muse Spark 1.2 is a coding-focused update to Muse Spark 1.1 with improvements in code generation, complex debugging, codebase understanding, and end-to-end developer workflows.", family: "muse", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-05", last_updated: "2026-08-05", modalities: { input: ["text", "image", "video", "pdf", "audio"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 131072 } }, "meta/muse-spark-1.1": { id: "meta/muse-spark-1.1", name: "Muse Spark 1.1", description: "Muse Spark is a natively multimodal reasoning model with support for tool-use, visual chain of thought, and multi-agent orchestration.", family: "muse", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-08", last_updated: "2026-07-09", modalities: { input: ["text", "image", "pdf", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 32e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 61.5, metric: "resolve rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "Terminal-Bench", score: 80, metric: "success rate", version: "2.1", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "DeepSWE", score: 53.3, metric: "resolve rate", version: "1.1", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "MCP Atlas", score: 88.1, metric: "success rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "JobBench", score: 54.7, metric: "success rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "Toolathlon-Verified", score: 75.6, metric: "success rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "Humanity's Last Exam", score: 62.1, metric: "accuracy", variant: "with tools", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "OSWorld-Verified", score: 80.8, metric: "success rate", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "Finance Agent", score: 57.2, metric: "accuracy", version: "v2", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "CharXiv Reasoning", score: 88.4, metric: "accuracy", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }, { name: "BabyVision", score: 76.3, metric: "accuracy", source: "https://ai.meta.com/blog/introducing-muse-spark-meta-model-api/", date: "2026-07-09" }] }, "meta/llama-3.2-1b": { id: "meta/llama-3.2-1b", name: "Llama-3.2-1B", description: "Compact open Llama base model for lightweight and on-device use", family: "llama", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2023-12", release_date: "2024-09-25", last_updated: "2024-09-25", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "Llama 3.2 Community License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.2-1B" }] }, "meta/llama-4-maverick-17b-instruct": { id: "meta/llama-4-maverick-17b-instruct", name: "Llama 4 Maverick 17B Instruct", description: "Open multimodal Llama for strong reasoning with efficient everyday serving", family: "llama", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-08", release_date: "2025-04-05", last_updated: "2025-04-05", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-4-Maverick-17B-128E-Instruct" }], benchmarks: [{ name: "Aider Polyglot", score: 15.6, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-06" }, { name: "SWE-Bench Pro", score: 5.24, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "meta/llama-3.2-3b": { id: "meta/llama-3.2-3b", name: "Llama-3.2-3B", description: "Small open Llama base model for lightweight text generation and self-hosting", family: "llama", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2023-12", release_date: "2024-09-25", last_updated: "2024-09-25", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "Llama 3.2 Community License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.2-3B" }] }, "meta/llama-3.2-11b-vision-instruct": { id: "meta/llama-3.2-11b-vision-instruct", name: "Llama-3.2-11B-Vision-Instruct", description: "Open multimodal Llama model for image understanding, captioning, and visual QA", family: "llama", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-12", release_date: "2024-09-25", last_updated: "2024-09-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/meta-llama/Llama-3.2-11B-Vision-Instruct" }] }, "sdaia/allam-2-7b": { id: "sdaia/allam-2-7b", name: "ALLaM-2-7b", description: "ALLaM-2-7b instruction tuned model by SDAIA", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2025-01-23", last_updated: "2025-01-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 4096, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/ALLaM-AI/ALLaM-2.0-7B-Instruct" }] }, "poolside/laguna-m.1": { id: "poolside/laguna-m.1", name: "Laguna M.1", description: "Poolside's open-weight model for agentic coding and long-horizon work", family: "laguna", attachment: false, reasoning: true, tool_call: true, structured_output: false, temperature: true, release_date: "2026-04-28", last_updated: "2026-06-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 } }, "poolside/laguna-s-2.1": { id: "poolside/laguna-s-2.1", name: "Laguna S 2.1", description: "Agentic coding model from Poolside in the XS size class for local deployment", family: "laguna", attachment: false, reasoning: true, tool_call: true, structured_output: false, temperature: true, release_date: "2026-07-21", last_updated: "2026-07-21", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 32768 } }, "poolside/laguna-xs-2.1": { id: "poolside/laguna-xs-2.1", name: "Laguna XS 2.1", description: "Agentic coding model from Poolside in the XS size class for local deployment", family: "laguna", attachment: false, reasoning: true, tool_call: true, structured_output: false, temperature: true, release_date: "2026-07-02", last_updated: "2026-07-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 }, benchmarks: [{ name: "SWE-Bench Verified", score: 70.9, metric: "resolved", harness: "Harbor", source: "https://poolside.ai/blog/introducing-laguna-xs-2-1", date: "2026-07-02" }, { name: "SWE-Bench Multilingual", score: 63.1, metric: "resolve rate", harness: "Harbor", source: "https://poolside.ai/blog/introducing-laguna-xs-2-1", date: "2026-07-02" }, { name: "SWE-Bench Pro", score: 47.6, metric: "resolve rate", harness: "Harbor", source: "https://poolside.ai/blog/introducing-laguna-xs-2-1", date: "2026-07-02" }, { name: "Terminal-Bench", score: 37.5, metric: "success rate", harness: "Harbor", version: "2.0", source: "https://poolside.ai/blog/introducing-laguna-xs-2-1", date: "2026-07-02" }] }, "poolside/laguna-xs.2": { id: "poolside/laguna-xs.2", name: "Laguna XS.2", description: "Agentic coding model from Poolside in the XS size class for local deployment", family: "laguna", attachment: false, reasoning: true, tool_call: true, structured_output: false, temperature: true, release_date: "2026-04-28", last_updated: "2026-06-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 } }, "bytedance-seed/seed-2.0-code": { id: "bytedance-seed/seed-2.0-code", name: "Seed 2.0 Code", description: "ByteDance Seed coding model for multimodal software engineering and long-running agents", family: "seed", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-14", last_updated: "2026-02-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 131072 } }, "bytedance-seed/seed-2.1-turbo": { id: "bytedance-seed/seed-2.1-turbo", name: "Seed 2.1 Turbo", description: "Faster ByteDance Seed 2.1 model for multimodal reasoning and latency-sensitive agent workflows", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-23", last_updated: "2026-06-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "bytedance-seed/seed-1-6": { id: "bytedance-seed/seed-1-6", name: "Seed 1.6", description: "ByteDance Seed model for long-context reasoning, instruction following, and tool-assisted tasks", family: "seed", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-10-15", last_updated: "2025-10-15", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 64e3 } }, "bytedance-seed/seed-character": { id: "bytedance-seed/seed-character", name: "Seed Character", description: "ByteDance Seed model optimized for character-driven dialogue and consistent conversational behavior", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-23", last_updated: "2026-06-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "bytedance-seed/seed-1-6-vision": { id: "bytedance-seed/seed-1-6-vision", name: "Seed 1.6 Vision", description: "ByteDance Seed multimodal model for image understanding, visual reasoning, and tool-assisted tasks", family: "seed", attachment: true, reasoning: false, tool_call: true, temperature: true, release_date: "2025-08-15", last_updated: "2025-08-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 32e3 } }, "bytedance-seed/seed-2.0-pro": { id: "bytedance-seed/seed-2.0-pro", name: "Seed 2.0 Pro", description: "Flagship ByteDance Seed 2.0 model for complex multimodal reasoning and long-horizon agent workflows", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-14", last_updated: "2026-02-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 128e3 } }, "bytedance-seed/seed-evolving": { id: "bytedance-seed/seed-evolving", name: "Seed Evolving", description: "Rolling ByteDance Seed model for rapidly updated reasoning, coding, and agent capabilities", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-23", last_updated: "2026-06-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "bytedance-seed/seed-2.1-pro": { id: "bytedance-seed/seed-2.1-pro", name: "Seed 2.1 Pro", description: "Flagship ByteDance Seed 2.1 model for complex multimodal reasoning, coding, and agents", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-23", last_updated: "2026-06-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "bytedance-seed/seed-1-6-flash": { id: "bytedance-seed/seed-1-6-flash", name: "Seed 1.6 Flash", description: "Low-latency ByteDance Seed model for high-throughput chat, extraction, and lightweight tool use", family: "seed", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-08-28", last_updated: "2025-08-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 32e3 } }, "bytedance-seed/seed-2.0-mini": { id: "bytedance-seed/seed-2.0-mini", name: "Seed 2.0 Mini", description: "Lightweight ByteDance Seed 2.0 model for low-latency multimodal reasoning and high-volume tasks", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-14", last_updated: "2026-02-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 32e3 } }, "bytedance-seed/seed-1-8": { id: "bytedance-seed/seed-1-8", name: "Seed 1.8", description: "ByteDance Seed model for multimodal reasoning, long-context analysis, and agent workflows", family: "seed", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-28", last_updated: "2025-12-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 64e3 } }, "bytedance-seed/seed-2.0-lite": { id: "bytedance-seed/seed-2.0-lite", name: "Seed 2.0 Lite", description: "Cost-efficient ByteDance Seed 2.0 model for production chat, analysis, and structured generation", family: "seed", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-14", last_updated: "2026-02-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 32e3 } }, "moonshotai/kimi-k2.7-code": { id: "moonshotai/kimi-k2.7-code", name: "Kimi K2.7 Code", description: "Coding-focused Kimi model, stronger on long-horizon repo work with less overthinking", family: "kimi-k2", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-01", release_date: "2026-06-12", last_updated: "2026-06-12", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2.7-Code" }], benchmarks: [{ name: "Kimi Code Bench", score: 62, harness: "Kimi Code CLI", version: "v2", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "Program Bench", score: 53.6, harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "MLS Bench Lite", score: 35.1, harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "MCP Atlas", score: 76, metric: "success rate", harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "MCP Mark Verified", score: 81.1, metric: "success rate", harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }, { name: "Kimi Claw 24/7 Bench", score: 46.9, harness: "Kimi Code CLI", source: "https://huggingface.co/moonshotai/Kimi-K2.7-Code", date: "2026-06-12" }] }, "moonshotai/kimi-k3": { id: "moonshotai/kimi-k3", name: "Kimi K3", description: "Multimodal Kimi model with 1M context and toggleable max-effort thinking for long-horizon agent work", family: "kimi-k3", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, release_date: "2026-07-16", last_updated: "2026-07-16", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 131072 }, benchmarks: [{ name: "DeepSWE", score: 67.5, metric: "resolve rate", harness: "Kimi Code", variant: "max effort", version: "1.1", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "Terminal-Bench", score: 88.3, metric: "accuracy", harness: "Kimi Code", variant: "max effort", version: "2.1", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "FrontierSWE", score: 81.2, metric: "dominance score", harness: "Kimi Code", variant: "max effort", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "Program Bench", score: 77.8, metric: "score", harness: "Kimi Code", variant: "max effort", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "SWE Marathon", score: 42, metric: "resolve rate", harness: "Claude Code", variant: "max effort", version: "1.1", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "GDPval-AA", score: 1668, metric: "Elo", variant: "max effort", version: "v2", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "AA-Briefcase", score: 1548, metric: "Elo", variant: "max effort", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "AutomationBench", score: 30.8, metric: "success rate", variant: "max effort", dataset: "600-task public subset", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "JobBench", score: 52.9, metric: "score", variant: "max effort", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "SpreadsheetBench", score: 34.8, metric: "score", harness: "Claude Code", variant: "max effort", version: "2", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "BrowseComp", score: 91.2, metric: "accuracy", variant: "max effort, context compaction", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "CharXiv Reasoning", score: 91.3, metric: "accuracy", variant: "max effort, with tools", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }, { name: "ZeroBench", score: 41, metric: "pass@5", variant: "max effort, with tools", source: "https://www.kimi.com/blog/kimi-k3", date: "2026-07-16" }] }, "moonshotai/kimi-k2-thinking-turbo": { id: "moonshotai/kimi-k2-thinking-turbo", name: "Kimi K2 Thinking Turbo", description: "Kimi reasoning model for long-horizon research, planning, and tool use", family: "kimi-thinking", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-08", release_date: "2025-11-06", last_updated: "2025-11-06", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2-Thinking" }] }, "moonshotai/kimi-k2.5": { id: "moonshotai/kimi-k2.5", name: "Kimi K2.5", description: "Earlier Kimi frontier model for long-context agents, coding, and multimodal work", family: "kimi-k2", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-01", release_date: "2026-01", last_updated: "2026-01", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2.5" }], benchmarks: [{ name: "SWE-Bench Verified", score: 70.8, metric: "resolved", source: "https://www.swebench.com/" }, { name: "SWE-Atlas Codebase QnA", score: 13.1, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 20.95, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 25.77, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] }, "moonshotai/kimi-k2.7-code-highspeed": { id: "moonshotai/kimi-k2.7-code-highspeed", name: "Kimi K2.7 Code Highspeed", description: "Lower-latency Kimi Code variant for interactive edits and coding-agent loops", family: "kimi-k2", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-01", release_date: "2026-06-12", last_updated: "2026-06-12", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2.7-Code" }] }, "moonshotai/kimi-k2-thinking": { id: "moonshotai/kimi-k2-thinking", name: "Kimi K2 Thinking", description: "Thinking Kimi model for slower research passes, planning, and hard technical questions", family: "kimi-thinking", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-08", release_date: "2025-11-06", last_updated: "2025-11-06", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2-Thinking" }], benchmarks: [{ name: "SWE-Bench Verified", score: 71.3, metric: "resolved", source: "https://huggingface.co/moonshotai/Kimi-K2-Thinking" }] }, "moonshotai/kimi-k2.6": { id: "moonshotai/kimi-k2.6", name: "Kimi K2.6", description: "Multimodal Kimi workhorse for agent loops, coding tasks, and visual context", family: "kimi-k2", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-01", release_date: "2026-04-21", last_updated: "2026-04-21", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/moonshotai/Kimi-K2.6" }], benchmarks: [{ name: "SWE-Bench Verified", score: 80.2, metric: "resolved", source: "https://huggingface.co/moonshotai/Kimi-K2.6" }, { name: "Artificial Analysis Coding Agent Index", score: 50.5, metric: "average pass@1", harness: "Claude Code", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 59.8, metric: "pass@1", harness: "Claude Code", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 27.3, metric: "pass@1", harness: "Claude Code", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 64.3, metric: "pass@1", harness: "Claude Code", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }] }, "anthropic/claude-opus-4-20250514": { id: "anthropic/claude-opus-4-20250514", name: "Claude Opus 4", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-05-22", last_updated: "2025-05-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 32e3 }, benchmarks: [{ name: "Aider Polyglot", score: 72, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-25" }] }, "anthropic/claude-opus-4-7": { id: "anthropic/claude-opus-4-7", name: "Claude Opus 4.7", description: "Stronger Opus tier for advanced software work and high-stakes reasoning", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-01-31", release_date: "2026-04-16", last_updated: "2026-04-16", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 64.3, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "Terminal-Bench", score: 66.1, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "SWE-Atlas Refactoring", score: 48.57, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "Artificial Analysis Coding Agent Index", score: 66.6, metric: "average pass@1", harness: "Claude Code", variant: "max", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 81, metric: "pass@1", harness: "Claude Code", variant: "max", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 44.9, metric: "pass@1", harness: "Claude Code", variant: "max", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 73.8, metric: "pass@1", harness: "Claude Code", variant: "max", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 61.2, metric: "average pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 78.4, metric: "pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 34.4, metric: "pass@1", harness: "Cursor CLI", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 70.6, metric: "pass@1", harness: "Cursor CLI", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 59.9, metric: "average pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 71.7, metric: "pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 36.4, metric: "pass@1", harness: "Claude Code", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 71.4, metric: "pass@1", harness: "Claude Code", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "GPQA Diamond", score: 94.2, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 46.9, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 54.7, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "OSWorld-Verified", score: 78, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "anthropic/claude-mythos-5": { id: "anthropic/claude-mythos-5", name: "Claude Mythos 5", description: "Restricted Claude model for advanced cybersecurity and biology research workflows", family: "claude-mythos", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2026-01-31", release_date: "2026-06-09", last_updated: "2026-06-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 } }, "anthropic/claude-opus-4-1-20250805": { id: "anthropic/claude-opus-4-1-20250805", name: "Claude Opus 4.1", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-08-05", last_updated: "2025-08-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 32e3 } }, "anthropic/claude-3-5-sonnet-20241022": { id: "anthropic/claude-3-5-sonnet-20241022", name: "Claude Sonnet 3.5 v2", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04-30", release_date: "2024-10-22", last_updated: "2024-10-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 8192 }, benchmarks: [{ name: "Aider Polyglot", score: 51.6, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-17" }] }, "anthropic/claude-opus-4-8": { id: "anthropic/claude-opus-4-8", name: "Claude Opus 4.8", description: "Top Claude Opus tier for the hardest reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-01", release_date: "2026-05-28", last_updated: "2026-05-28", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 69.2, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "Terminal-Bench", score: 74.6, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "SWE-Bench Verified", score: 88.6, metric: "resolved", source: "https://benchlm.ai/benchmarks/sweVerified" }, { name: "Humanity's Last Exam", score: 49.8, metric: "accuracy", variant: "no tools", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "Humanity's Last Exam", score: 57.9, metric: "accuracy", variant: "with tools", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "OSWorld-Verified", score: 83.4, metric: "success rate", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "FrontierCode", score: 13.4, metric: "pass rate", variant: "high effort", dataset: "Diamond", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }] }, "anthropic/claude-3-5-haiku-20241022": { id: "anthropic/claude-3-5-haiku-20241022", name: "Claude Haiku 3.5", description: "Fast Claude model for responsive assistance, classification, and lightweight agents", family: "claude-haiku", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-07-31", release_date: "2024-10-22", last_updated: "2024-10-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 8192 }, benchmarks: [{ name: "Aider Polyglot", score: 28, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-21" }] }, "anthropic/claude-opus-4-1": { id: "anthropic/claude-opus-4-1", name: "Claude Opus 4.1 (latest)", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-08-05", last_updated: "2025-08-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 32e3 } }, "anthropic/claude-sonnet-5": { id: "anthropic/claude-sonnet-5", name: "Claude Sonnet 5", description: "Everyday Claude agent model for coding, planning, browsing, and general work", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-01-31", release_date: "2026-06-30", last_updated: "2026-06-30", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 85.2, metric: "resolved", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "SWE-Bench Pro", score: 63.2, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "SWE-Bench Multilingual", score: 78.3, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "Terminal-Bench", score: 80.4, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "OSWorld-Verified", score: 81.2, metric: "success rate", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "BrowseComp", score: 84.7, metric: "accuracy", variant: "single agent", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "FrontierCode", score: 38.8, metric: "pass rate", version: "v1", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }] }, "anthropic/claude-3-7-sonnet-20250219": { id: "anthropic/claude-3-7-sonnet-20250219", name: "Claude Sonnet 3.7", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-10-31", release_date: "2025-02-19", last_updated: "2025-02-19", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "Aider Polyglot", score: 64.9, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-02-24" }] }, "anthropic/claude-sonnet-4-5-20250929": { id: "anthropic/claude-sonnet-4-5-20250929", name: "Claude Sonnet 4.5", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-07-31", release_date: "2025-09-29", last_updated: "2025-09-29", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 } }, "anthropic/claude-opus-4-5": { id: "anthropic/claude-opus-4-5", name: "Claude Opus 4.5 (latest)", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-11-24", last_updated: "2025-11-24", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 } }, "anthropic/claude-sonnet-4-6": { id: "anthropic/claude-sonnet-4-6", name: "Claude Sonnet 4.6", description: "Claude workhorse for coding agents, careful analysis, and production cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-08-31", release_date: "2026-02-17", last_updated: "2026-03-13", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 64e3 }, benchmarks: [{ name: "SWE-Atlas Codebase QnA", score: 31.2, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 32.21, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 31.76, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 49.4, metric: "average pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 70.3, metric: "pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 14.9, metric: "pass@1", harness: "Claude Code", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 63.1, metric: "pass@1", harness: "Claude Code", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 67, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "Humanity's Last Exam", score: 34.6, metric: "accuracy", variant: "no tools", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "Humanity's Last Exam", score: 46.8, metric: "accuracy", variant: "with tools", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }, { name: "OSWorld-Verified", score: 78.5, metric: "success rate", source: "https://www.anthropic.com/news/claude-sonnet-5", date: "2026-06-30" }] }, "anthropic/claude-opus-4-0": { id: "anthropic/claude-opus-4-0", name: "Claude Opus 4 (latest)", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-05-22", last_updated: "2025-05-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 32e3 }, benchmarks: [{ name: "Aider Polyglot", score: 72, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-25" }] }, "anthropic/claude-haiku-4-5-20251001": { id: "anthropic/claude-haiku-4-5-20251001", name: "Claude Haiku 4.5", description: "Fast Claude model for responsive assistance, classification, and lightweight agents", family: "claude-haiku", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-02-28", release_date: "2025-10-15", last_updated: "2025-10-15", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 } }, "anthropic/claude-sonnet-4-5": { id: "anthropic/claude-sonnet-4-5", name: "Claude Sonnet 4.5 (latest)", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-07-31", release_date: "2025-09-29", last_updated: "2025-09-29", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 43.6, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "anthropic/claude-opus-4-5-20251101": { id: "anthropic/claude-opus-4-5-20251101", name: "Claude Opus 4.5", description: "Flagship Claude model for deep reasoning, coding, and long-horizon agents", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-11-01", last_updated: "2025-11-01", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 45.89, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "anthropic/claude-fable-5": { id: "anthropic/claude-fable-5", name: "Claude Fable 5", description: "Claude model for creative writing, analysis, and controlled agent workflows", family: "claude-fable", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-01-31", release_date: "2026-06-09", last_updated: "2026-06-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 80.3, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "SWE-Bench Verified", score: 95, metric: "resolved", source: "https://benchlm.ai/benchmarks/sweVerified" }, { name: "Terminal-Bench", score: 88, metric: "success rate", version: "2.1", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "Humanity's Last Exam", score: 59, metric: "accuracy", variant: "no tools", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "Humanity's Last Exam", score: 64.5, metric: "accuracy", variant: "with tools", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "OSWorld-Verified", score: 85, metric: "success rate", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "FrontierCode", score: 29.3, metric: "pass rate", variant: "high effort", dataset: "Diamond", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "GDPval-AA", score: 1932, metric: "Elo", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }, { name: "AutomationBench", score: 17.4, metric: "success rate", source: "https://www.anthropic.com/news/claude-fable-5-mythos-5", date: "2026-06-09" }] }, "anthropic/claude-sonnet-4-20250514": { id: "anthropic/claude-sonnet-4-20250514", name: "Claude Sonnet 4", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-05-22", last_updated: "2025-05-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "Aider Polyglot", score: 61.3, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-24" }] }, "anthropic/claude-sonnet-4-0": { id: "anthropic/claude-sonnet-4-0", name: "Claude Sonnet 4 (latest)", description: "Balanced Claude model for coding, analysis, agent workflows, and cost control", family: "claude-sonnet", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-05-22", last_updated: "2025-05-22", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "Aider Polyglot", score: 61.3, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-24" }, { name: "SWE-Bench Pro", score: 42.7, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "anthropic/claude-haiku-4-5": { id: "anthropic/claude-haiku-4-5", name: "Claude Haiku 4.5 (latest)", description: "Fast Claude lane for lightweight agents, office tasks, and responsive chat", family: "claude-haiku", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-02-28", release_date: "2025-10-15", last_updated: "2025-10-15", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 64e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 39.45, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "anthropic/claude-opus-4-6": { id: "anthropic/claude-opus-4-6", name: "Claude Opus 4.6", description: "High-end Claude for difficult coding, planning, and slower expert reasoning", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-05-31", release_date: "2026-02-05", last_updated: "2026-03-13", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 51.9, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }, { name: "SWE-Atlas Codebase QnA", score: 33.3, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Codebase QnA", score: 30, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 35.58, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 36.67, metric: "score", harness: "Claude Code", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "SWE-Atlas Test Writing", score: 36.08, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 51.3, metric: "average pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 71.9, metric: "pass@1", harness: "Claude Code", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 11.8, metric: "pass@1", harness: "Claude Code", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 70.2, metric: "pass@1", harness: "Claude Code", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }] }, "anthropic/claude-3-haiku-20240307": { id: "anthropic/claude-3-haiku-20240307", name: "Claude Haiku 3", description: "Legacy model retained for compatibility with older integrations", family: "claude-haiku", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2023-08-31", release_date: "2024-03-13", last_updated: "2024-03-13", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 4096 } }, "anthropic/claude-opus-5": { id: "anthropic/claude-opus-5", name: "Claude Opus 5", description: "Strongest Claude Opus model for coding, agents, and professional work", family: "claude-opus", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2026-05", release_date: "2026-07-24", last_updated: "2026-07-24", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 96, metric: "resolved", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "SWE-Bench Pro", score: 79.2, metric: "resolve rate", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "SWE-Bench Multilingual", score: 89.5, metric: "resolve rate", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "SWE-Bench Multimodal", score: 59.4, metric: "resolve rate", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "DeepSWE", score: 68.8, metric: "resolve rate", version: "1.1", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "FrontierCode", score: 53.4, metric: "mean@5", variant: "medium effort", dataset: "Main", version: "1.1", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "Frontier-Bench", score: 43.3, metric: "mean reward", harness: "mini-SWE-agent", variant: "max effort", version: "v0.1", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "BrowseComp", score: 90.8, metric: "accuracy", variant: "single agent", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "Humanity's Last Exam", score: 56.3, metric: "accuracy", variant: "no tools", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "Humanity's Last Exam", score: 64.7, metric: "accuracy", variant: "with tools", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "DeepSearchQA", score: 95, metric: "F1", variant: "max effort", source: "https://www.anthropic.com/news/claude-opus-5", date: "2026-07-24" }, { name: "OSWorld", score: 70.6, metric: "success rate", version: "2.0", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "GDPval-AA", score: 1861, metric: "Elo", variant: "max effort", version: "v2", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "AA-Briefcase", score: 1720, metric: "Elo", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "AutomationBench", score: 26, metric: "success rate", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "ARC-AGI-1", score: 97.5, metric: "accuracy", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "ARC-AGI-2", score: 90.4, metric: "accuracy", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "ARC-AGI-3", score: 30.2, metric: "RHAE", variant: "high effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }, { name: "HealthBench Professional", score: 59.8, metric: "score", variant: "max effort", source: "https://www.anthropic.com/claude-opus-5-system-card", date: "2026-07-24" }] }, "cohere/c4ai-aya-expanse-32b": { id: "cohere/c4ai-aya-expanse-32b", name: "Aya Expanse 32B", description: "Open multilingual model optimized for generation across 23 languages", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2024-10-24", last_updated: "2024-10-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, license: "CC-BY-NC-4.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/aya-expanse-32b" }] }, "cohere/command-a-reasoning-08-2025": { id: "cohere/command-a-reasoning-08-2025", name: "Command A Reasoning", description: "Cohere reasoning model for multilingual enterprise agents, tools, and complex workflows", family: "command-a", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2025-08-21", last_updated: "2025-08-21", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 32e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-a-reasoning-08-2025" }] }, "cohere/c4ai-aya-vision-32b": { id: "cohere/c4ai-aya-vision-32b", name: "Aya Vision 32B", description: "Open multilingual vision model for OCR, visual reasoning, and image question answering", attachment: true, reasoning: false, tool_call: false, temperature: true, release_date: "2025-03-04", last_updated: "2025-05-14", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 16e3, output: 4e3 }, license: "CC-BY-NC-4.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/aya-vision-32b" }] }, "cohere/command-r-plus-08-2024": { id: "cohere/command-r-plus-08-2024", name: "Command R+", description: "Cohere's RAG workhorse for long-context enterprise search and tool use", family: "command-r", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2024-08-30", last_updated: "2024-08-30", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-r-plus-08-2024" }] }, "cohere/command-a-translate-08-2025": { id: "cohere/command-a-translate-08-2025", name: "Command A Translate", description: "Translation model for multilingual conversion, localization, and cross-language workflows", family: "command-a", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2025-08-28", last_updated: "2025-08-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 8e3, output: 8e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-a-translate-08-2025" }] }, "cohere/command-r7b-arabic-02-2025": { id: "cohere/command-r7b-arabic-02-2025", name: "Command R7B Arabic", description: "Open Command R model optimized for Arabic enterprise chat, RAG, and cultural knowledge", family: "command-r", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2025-02-27", last_updated: "2025-02-27", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-r7b-arabic-02-2025" }] }, "cohere/c4ai-aya-expanse-8b": { id: "cohere/c4ai-aya-expanse-8b", name: "Aya Expanse 8B", description: "Compact open multilingual model optimized for generation across 23 languages", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2024-10-24", last_updated: "2024-10-24", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 8e3, output: 4e3 }, license: "CC-BY-NC-4.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/aya-expanse-8b" }] }, "cohere/command-a-vision-07-2025": { id: "cohere/command-a-vision-07-2025", name: "Command A Vision", description: "Cohere vision model for multilingual document analysis, OCR, and image understanding", family: "command-a", attachment: true, reasoning: false, tool_call: false, temperature: true, knowledge: "2024-06-01", release_date: "2025-07-31", last_updated: "2025-07-31", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 8e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-a-vision-07-2025" }] }, "cohere/command-a-plus-05-2026": { id: "cohere/command-a-plus-05-2026", name: "Command A Plus", description: "Cohere's stronger command model for multilingual agents and enterprise workflows", family: "command-a", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-04-01", release_date: "2026-05-20", last_updated: "2026-06-09", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 64e3 } }, "cohere/command-a-03-2025": { id: "cohere/command-a-03-2025", name: "Command A", description: "Cohere command model for multilingual enterprise agents, tools, and chat", family: "command-a", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2025-03-13", last_updated: "2025-03-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 8e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-a-03-2025" }], benchmarks: [{ name: "Aider Polyglot", score: 12, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-03-14" }] }, "cohere/command-r7b-12-2024": { id: "cohere/command-r7b-12-2024", name: "Command R7B", description: "Cohere retrieval model for long-context chat and enterprise RAG workflows", family: "command-r", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2024-12-02", last_updated: "2024-12-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-r7b-12-2024" }] }, "cohere/command-r-08-2024": { id: "cohere/command-r-08-2024", name: "Command R", description: "Cohere retrieval model for long-context chat and enterprise RAG workflows", family: "command-r", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-06-01", release_date: "2024-08-30", last_updated: "2024-08-30", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 4e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/c4ai-command-r-08-2024" }] }, "cohere/north-mini-code-1-0": { id: "cohere/north-mini-code-1-0", name: "North Mini Code", description: "Cohere coding model for practical software engineering and agentic edits", family: "north", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-09-23", release_date: "2026-06-09", last_updated: "2026-06-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 64e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 67.6, metric: "resolved", harness: "SWE-agent", source: "https://huggingface.co/CohereLabs/North-Mini-Code-1.0", date: "2026-06-09" }, { name: "SWE-Bench Pro", score: 40.2, metric: "resolve rate", harness: "SWE-agent", source: "https://huggingface.co/CohereLabs/North-Mini-Code-1.0", date: "2026-06-09" }, { name: "Artificial Analysis Intelligence Index", score: 27.6, metric: "index score", source: "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model", date: "2026-06-09" }, { name: "Artificial Analysis Coding Index", score: 33.4, metric: "index score", source: "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model", date: "2026-06-09" }, { name: "GDPval-AA", score: 14, metric: "win rate", source: "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model", date: "2026-06-09" }, { name: "\u03C4\xB2-Bench Telecom", score: 37, metric: "success rate", source: "https://artificialanalysis.ai/articles/north-mini-code-cohere-s-small-coding-focused-moe-model", date: "2026-06-09" }] }, "cohere/c4ai-aya-vision-8b": { id: "cohere/c4ai-aya-vision-8b", name: "Aya Vision 8B", description: "Compact open multilingual vision model for OCR and visual question answering", attachment: true, reasoning: false, tool_call: false, temperature: true, release_date: "2025-03-04", last_updated: "2025-05-14", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 16e3, output: 4e3 }, license: "CC-BY-NC-4.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/CohereLabs/aya-vision-8b" }] }, "openai/gpt-4o": { id: "openai/gpt-4o", name: "GPT-4o", description: "Omni-era GPT for multimodal chat, practical coding, and general assistants", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-05-13", last_updated: "2024-08-06", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 }, benchmarks: [{ name: "Aider Polyglot", score: 23.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-30" }] }, "openai/gpt-image-1.5": { id: "openai/gpt-image-1.5", name: "GPT-Image-1.5", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "gpt-image", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2025-11-25", last_updated: "2025-11-25", modalities: { input: ["text", "image"], output: ["text", "image"] }, open_weights: false, limit: { context: 0, output: 0 } }, "openai/gpt-5.3-chat-latest": { id: "openai/gpt-5.3-chat-latest", name: "GPT-5.3 Chat (latest)", description: "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-08-31", release_date: "2026-03-03", last_updated: "2026-03-03", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 } }, "openai/gpt-5-nano": { id: "openai/gpt-5-nano", name: "GPT-5 Nano", description: "Tiny GPT-5 lane for routing, extraction, classification, and bulk jobs", family: "gpt-nano", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05-30", release_date: "2025-08-07", last_updated: "2025-08-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-4o-mini": { id: "openai/gpt-4o-mini", name: "GPT-4o mini", description: "Small omni GPT for cheap multimodal assistance and production-scale traffic", family: "gpt-mini", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-07-18", last_updated: "2024-07-18", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 }, benchmarks: [{ name: "Aider Polyglot", score: 3.6, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-21" }, { name: "SciCode", score: 22.9, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4o-mini/benchmarks", date: "2026-03-11" }] }, "openai/gpt-3.5-turbo": { id: "openai/gpt-3.5-turbo", name: "GPT-3.5-turbo", description: "Compact GPT model for low-latency assistance and high-volume workloads", family: "gpt", attachment: false, reasoning: false, tool_call: false, structured_output: false, temperature: true, knowledge: "2021-09-01", release_date: "2023-03-01", last_updated: "2023-11-06", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 16385, output: 4096 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 10.7, metric: "index", source: "https://openrouter.ai/openai/gpt-3.5-turbo/benchmarks", date: "2026-03-11" }] }, "openai/o1-pro": { id: "openai/o1-pro", name: "o1-pro", description: "O-series reasoning model for hard analysis, math, coding, and planning", family: "o-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2023-09", release_date: "2025-03-19", last_updated: "2025-03-19", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 } }, "openai/whisper-large-v3": { id: "openai/whisper-large-v3", name: "Whisper 3 Large", description: "Open Whisper checkpoint for robust multilingual transcription and captioning", family: "whisper", attachment: false, reasoning: false, tool_call: false, release_date: "2024-10-01", last_updated: "2024-10-01", modalities: { input: ["audio"], output: ["text"] }, open_weights: true, limit: { context: 448, output: 4096 } }, "openai/gpt-5.5-pro": { id: "openai/gpt-5.5-pro", name: "GPT-5.5 Pro", description: "Highest-accuracy GPT-5.5 tier for slower, precision-heavy reasoning and coding", family: "gpt-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-12-01", release_date: "2026-04-23", last_updated: "2026-04-23", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "BrowseComp", score: 90.1, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 43.1, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 57.2, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 52.4, metric: "accuracy", dataset: "Tier 1-3", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 39.6, metric: "accuracy", dataset: "Tier 4", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GDPval", score: 82.3, metric: "wins or ties", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GeneBench", score: 33.2, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "openai/gpt-4-turbo": { id: "openai/gpt-4-turbo", name: "GPT-4 Turbo", description: "Compact GPT model for low-latency assistance and high-volume workloads", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: false, temperature: true, knowledge: "2023-12", release_date: "2023-11-06", last_updated: "2024-04-09", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 4096 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 21.5, metric: "index", source: "https://openrouter.ai/openai/gpt-4-turbo/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 31.9, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4-turbo/benchmarks", date: "2026-03-11" }] }, "openai/gpt-4": { id: "openai/gpt-4", name: "GPT-4", description: "GPT model for general reasoning, writing, coding, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: false, temperature: true, knowledge: "2023-11", release_date: "2023-11-06", last_updated: "2024-04-09", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 8192, output: 8192 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 13.1, metric: "index", source: "https://openrouter.ai/openai/gpt-4/benchmarks", date: "2026-03-11" }] }, "openai/gpt-5-chat-latest": { id: "openai/gpt-5-chat-latest", name: "GPT-5 Chat (latest)", description: "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", family: "gpt-codex", attachment: true, reasoning: true, tool_call: false, structured_output: true, temperature: true, knowledge: "2024-09-30", release_date: "2025-08-07", last_updated: "2025-08-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5.6-sol": { id: "openai/gpt-5.6-sol", name: "GPT-5.6 Sol", description: "Frontier GPT-5.6 model for complex professional work, coding, and agentic workflows", family: "gpt-sol", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2026-02-16", release_date: "2026-07-09", last_updated: "2026-07-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 64.6, metric: "resolve rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Terminal-Bench", score: 88.8, metric: "success rate", version: "2.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "DeepSWE", score: 72.7, metric: "resolve rate", version: "1.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "GPQA Diamond", score: 94.6, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "FrontierMath", score: 89, metric: "accuracy", dataset: "Tier 1-3", version: "v2", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "BrowseComp", score: 90.4, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "OSWorld", score: 62.6, metric: "success rate", version: "2.0", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "MMMU Pro", score: 83, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Agents' Last Exam", score: 52.7, source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Toolathlon", score: 58, metric: "success rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Artificial Analysis Intelligence Index", score: 58.9, metric: "index score", variant: "max", version: "4.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }, { name: "Artificial Analysis Coding Agent Index", score: 80, metric: "index score", harness: "Codex", variant: "max", version: "1.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }] }, "openai/o3-pro": { id: "openai/o3-pro", name: "o3-pro", description: "High-effort o3 tier for difficult technical reasoning and careful answers", family: "o-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05", release_date: "2025-06-10", last_updated: "2025-06-10", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 84.9, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-06-28" }] }, "openai/o3-deep-research": { id: "openai/o3-deep-research", name: "o3-deep-research", description: "Research model for long-horizon investigation, synthesis, and analytical reports", family: "o", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2024-05", release_date: "2024-06-26", last_updated: "2024-06-26", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 } }, "openai/gpt-5.4-pro": { id: "openai/gpt-5.4-pro", name: "GPT-5.4 Pro", description: "More exact GPT-5.4 tier for demanding professional reasoning and agent tasks", family: "gpt-pro", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: false, knowledge: "2025-08-31", release_date: "2026-03-05", last_updated: "2026-03-05", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "GPQA Diamond", score: 94.4, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 42.7, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 58.7, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "BrowseComp", score: 89.3, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GDPval", score: 82, metric: "wins or ties", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 50, metric: "accuracy", dataset: "Tier 1-3", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 38, metric: "accuracy", dataset: "Tier 4", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "ARC-AGI-1", score: 94.5, metric: "accuracy", variant: "Verified", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "ARC-AGI-2", score: 83.3, metric: "accuracy", variant: "Verified", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FinanceAgent", score: 61.5, metric: "accuracy", version: "1.1", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GeneBench", score: 25.6, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "openai/gpt-5": { id: "openai/gpt-5", name: "GPT-5", description: "Original GPT-5 workhorse for reasoning, coding, writing, and tool workflows", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-08-07", last_updated: "2025-08-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "Aider Polyglot", score: 88, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-08-23" }, { name: "SWE-Bench Pro", score: 41.78, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "openai/gpt-5.1-codex-mini": { id: "openai/gpt-5.1-codex-mini", name: "GPT-5.1 Codex mini", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5-codex": { id: "openai/gpt-5-codex", name: "GPT-5-Codex", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-09-15", last_updated: "2025-09-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 38.9, metric: "index", source: "https://openrouter.ai/openai/gpt-5-codex/benchmarks", date: "2026-06-01" }, { name: "SciCode", score: 40.9, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-5-codex/benchmarks", date: "2026-06-01" }, { name: "Terminal-Bench Hard", score: 37.9, metric: "success rate", source: "https://openrouter.ai/openai/gpt-5-codex/benchmarks", date: "2026-06-01" }] }, "openai/gpt-5-mini": { id: "openai/gpt-5-mini", name: "GPT-5 Mini", description: "Small GPT-5 for responsive agents, coding help, and everyday automation", family: "gpt-mini", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05-30", release_date: "2025-08-07", last_updated: "2025-08-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5.6-luna": { id: "openai/gpt-5.6-luna", name: "GPT-5.6 Luna", description: "Cost-efficient GPT-5.6 model for fast, high-volume workloads", family: "gpt-luna", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2026-02-16", release_date: "2026-07-09", last_updated: "2026-07-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 62.7, metric: "resolve rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Terminal-Bench", score: 84.7, metric: "success rate", version: "2.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "DeepSWE", score: 67.2, metric: "resolve rate", version: "1.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "GPQA Diamond", score: 92.3, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "FrontierMath", score: 78.6, metric: "accuracy", dataset: "Tier 1-3", version: "v2", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "BrowseComp", score: 83.3, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "OSWorld", score: 45.6, metric: "success rate", version: "2.0", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "MMMU Pro", score: 78.4, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Agents' Last Exam", score: 50.3, source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Toolathlon", score: 53.4, metric: "success rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Artificial Analysis Intelligence Index", score: 51.2, metric: "index score", variant: "max", version: "4.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }, { name: "Artificial Analysis Coding Agent Index", score: 74.6, metric: "index score", harness: "Codex", variant: "max", version: "1.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }] }, "openai/gpt-5.3-codex": { id: "openai/gpt-5.3-codex", name: "GPT-5.3 Codex", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-02-05", last_updated: "2026-02-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Atlas Codebase QnA", score: 32.6, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 42.38, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 38.98, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] }, "openai/gpt-oss-120b": { id: "openai/gpt-oss-120b", name: "GPT OSS 120B", description: "Open GPT reasoning model for self-hosted agents and controllable deployments", family: "gpt-oss", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2025-08-05", last_updated: "2025-08-05", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/openai/gpt-oss-120b" }] }, "openai/gpt-5.3-codex-spark": { id: "openai/gpt-5.3-codex-spark", name: "GPT-5.3 Codex Spark", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex-spark", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-02-05", last_updated: "2026-02-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 128e3, input: 1e5, output: 32e3 } }, "openai/gpt-4.1-nano": { id: "openai/gpt-4.1-nano", name: "GPT-4.1 nano", description: "Tiny GPT-4.1 option for classification, routing, and very high-volume tasks", family: "gpt-nano", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-04", release_date: "2025-04-14", last_updated: "2025-04-14", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 1047576, output: 32768 }, benchmarks: [{ name: "Aider Polyglot", score: 8.9, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-14" }] }, "openai/o1": { id: "openai/o1", name: "o1", description: "O-series reasoning model for hard analysis, math, coding, and planning", family: "o", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2023-09", release_date: "2024-12-05", last_updated: "2024-12-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 61.7, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-21" }] }, "openai/gpt-5.2": { id: "openai/gpt-5.2", name: "GPT-5.2", description: "Reliable GPT generation for broad coding, writing, and tool-assisted product work", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2025-12-11", last_updated: "2025-12-11", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 29.94, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "openai/gpt-image-1": { id: "openai/gpt-image-1", name: "GPT-Image-1", description: "OpenAI image model for production generation, edits, and brand-safe visual workflows", family: "gpt-image", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2025-04-24", last_updated: "2025-04-24", modalities: { input: ["text", "image"], output: ["image"] }, open_weights: false, limit: { context: 0, output: 0 } }, "openai/gpt-realtime-2.1": { id: "openai/gpt-realtime-2.1", name: "GPT-Realtime-2.1", description: "Realtime speech-to-speech model with configurable reasoning, tool use, and robust voice-agent behavior", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: false, knowledge: "2024-09-30", release_date: "2026-07-06", last_updated: "2026-07-06", modalities: { input: ["text", "audio", "image"], output: ["text", "audio"] }, open_weights: false, limit: { context: 128e3, input: 96e3, output: 32e3 } }, "openai/gpt-5.4-mini": { id: "openai/gpt-5.4-mini", name: "GPT-5.4 mini", description: "Strong small GPT for coding subagents, quick tool use, and high-volume work", family: "gpt-mini", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-03-17", last_updated: "2026-03-17", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 54.4, metric: "resolve rate", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Terminal-Bench", score: 60, metric: "accuracy", variant: "reasoning effort xhigh", version: "2.0", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MCP Atlas", score: 57.7, metric: "score", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Toolathlon", score: 42.9, metric: "score", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "\u03C4\xB2-Bench Telecom", score: 93.4, metric: "accuracy", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "GPQA Diamond", score: 88, metric: "accuracy", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Humanity's Last Exam", score: 41.5, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Humanity's Last Exam", score: 28.2, metric: "accuracy", variant: "without tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OSWorld-Verified", score: 72.1, metric: "success rate", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MMMU Pro", score: 78, metric: "accuracy", variant: "with Python", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MMMU Pro", score: 76.6, metric: "accuracy", variant: "without tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OmniDocBench", score: 0.1263, metric: "overall edit distance", variant: "reasoning effort none", version: "1.5", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OpenAI MRCR", score: 47.7, metric: "accuracy", variant: "8-needle, 64K-128K", version: "v2", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OpenAI MRCR", score: 33.6, metric: "accuracy", variant: "8-needle, 128K-256K", version: "v2", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Graphwalks", score: 76.3, metric: "accuracy", variant: "BFS, 0-128K", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Graphwalks", score: 71.5, metric: "accuracy", variant: "parents, 0-128K", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }] }, "openai/gpt-4o-2024-05-13": { id: "openai/gpt-4o-2024-05-13", name: "GPT-4o (2024-05-13)", description: "GPT model for general reasoning, writing, coding, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-05-13", last_updated: "2024-05-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 4096 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 24.2, metric: "index", source: "https://openrouter.ai/openai/gpt-4o-2024-05-13/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 30.9, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4o-2024-05-13/benchmarks", date: "2026-03-11" }] }, "openai/o3": { id: "openai/o3", name: "o3", description: "Deliberate o-series reasoner for hard math, coding, and multi-step analysis", family: "o", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05", release_date: "2025-04-16", last_updated: "2025-04-16", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 81.3, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-06-25" }] }, "openai/o4-mini-deep-research": { id: "openai/o4-mini-deep-research", name: "o4-mini-deep-research", description: "Research model for long-horizon investigation, synthesis, and analytical reports", family: "o-mini", attachment: true, reasoning: true, tool_call: true, temperature: false, knowledge: "2024-05", release_date: "2024-06-26", last_updated: "2024-06-26", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 } }, "openai/gpt-5-pro": { id: "openai/gpt-5-pro", name: "GPT-5 Pro", description: "Higher-accuracy GPT-5 tier for tough analysis, coding reviews, and planning", family: "gpt-pro", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-10-06", last_updated: "2025-10-06", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 272e3 } }, "openai/gpt-5.5": { id: "openai/gpt-5.5", name: "GPT-5.5", description: "Default frontier GPT for coding, computer use, research, and knowledge work", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-12-01", release_date: "2026-04-23", last_updated: "2026-04-23", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 58.6, metric: "resolve rate", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "Terminal-Bench", score: 78.2, metric: "success rate", harness: "Terminus-2", version: "2.1", source: "https://www.anthropic.com/news/claude-opus-4-8", date: "2026-05-28" }, { name: "SWE-Atlas Codebase QnA", score: 45.43, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 44.79, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 42.59, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 65.3, metric: "average pass@1", harness: "Codex", variant: "xhigh", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 80.8, metric: "pass@1", harness: "Codex", variant: "xhigh", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 30.9, metric: "pass@1", harness: "Codex", variant: "xhigh", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 84.1, metric: "pass@1", harness: "Codex", variant: "xhigh", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 60.4, metric: "average pass@1", harness: "Codex", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 79.1, metric: "pass@1", harness: "Codex", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 26.2, metric: "pass@1", harness: "Codex", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 75.8, metric: "pass@1", harness: "Codex", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 57.8, metric: "average pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 75, metric: "pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 24.9, metric: "pass@1", harness: "Cursor CLI", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 73.4, metric: "pass@1", harness: "Cursor CLI", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 82.7, metric: "success rate", version: "2.0", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GPQA Diamond", score: 93.6, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 41.4, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 52.2, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "OSWorld-Verified", score: 78.7, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "BrowseComp", score: 84.4, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "MMMU Pro", score: 81.2, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "ARC-AGI-2", score: 85, metric: "accuracy", variant: "Verified", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 51.7, metric: "accuracy", dataset: "Tier 1-3", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 35.4, metric: "accuracy", dataset: "Tier 4", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GDPval", score: 84.9, metric: "wins or ties", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "MCP Atlas", score: 75.3, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Toolathlon", score: 55.6, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "\u03C4\xB2-Bench Telecom", score: 98, metric: "success rate", variant: "original prompts", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "openai/gpt-5.5-instant": { id: "openai/gpt-5.5-instant", name: "GPT-5.5 Instant", description: "Compact GPT model for low-latency assistance and high-volume workloads", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-12-01", release_date: "2026-05-05", last_updated: "2026-05-28", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 4e5, output: 128e3 } }, "openai/gpt-5.2-codex": { id: "openai/gpt-5.2-codex", name: "GPT-5.2 Codex", description: "Code-specialist GPT for repository edits, reviews, and long-running software agents", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2025-12-11", last_updated: "2025-12-11", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 41.04, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "openai/gpt-4.1": { id: "openai/gpt-4.1", name: "GPT-4.1", description: "Long-lived GPT workhorse for coding, instruction following, and production apps", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-04", release_date: "2025-04-14", last_updated: "2025-04-14", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1047576, output: 32768 }, benchmarks: [{ name: "Aider Polyglot", score: 52.4, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-14" }] }, "openai/gpt-4o-2024-08-06": { id: "openai/gpt-4o-2024-08-06", name: "GPT-4o (2024-08-06)", description: "GPT model for general reasoning, writing, coding, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-08-06", last_updated: "2024-08-06", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 }, benchmarks: [{ name: "Aider Polyglot", score: 23.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-30" }, { name: "Artificial Analysis Coding Index", score: 16.6, metric: "index", source: "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 33.1, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 8.3, metric: "success rate", source: "https://openrouter.ai/openai/gpt-4o-2024-08-06/benchmarks", date: "2026-03-11" }] }, "openai/gpt-5.6-terra": { id: "openai/gpt-5.6-terra", name: "GPT-5.6 Terra", description: "Balanced GPT-5.6 model for capable, cost-efficient everyday work", family: "gpt-terra", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2026-02-16", release_date: "2026-07-09", last_updated: "2026-07-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 63.4, metric: "resolve rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Terminal-Bench", score: 87.4, metric: "success rate", version: "2.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "DeepSWE", score: 69.6, metric: "resolve rate", version: "1.1", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "GPQA Diamond", score: 92.9, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "FrontierMath", score: 84.9, metric: "accuracy", dataset: "Tier 1-3", version: "v2", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "BrowseComp", score: 87.5, metric: "accuracy", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "OSWorld", score: 50.2, metric: "success rate", version: "2.0", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "MMMU Pro", score: 80.7, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Agents' Last Exam", score: 50.4, source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Toolathlon", score: 53.1, metric: "success rate", source: "https://openai.com/index/gpt-5-6/", date: "2026-07-09" }, { name: "Artificial Analysis Intelligence Index", score: 55, metric: "index score", variant: "max", version: "4.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }, { name: "Artificial Analysis Coding Agent Index", score: 77.4, metric: "index score", harness: "Codex", variant: "max", version: "1.1", source: "https://artificialanalysis.ai/articles/gpt-5-6-has-landed", date: "2026-07-09" }] }, "openai/o4-mini": { id: "openai/o4-mini", name: "o4-mini", description: "Fast o-series model for compact reasoning, coding, and tool use", family: "o-mini", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05", release_date: "2025-04-16", last_updated: "2025-04-16", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 72, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-16" }] }, "openai/gpt-5.4": { id: "openai/gpt-5.4", name: "GPT-5.4", description: "Agent-ready GPT for coding and computer-use workflows at a lower cost", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-03-05", last_updated: "2026-03-05", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 105e4, input: 922e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 59.1, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }, { name: "SWE-Atlas Codebase QnA", score: 40.8, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Codebase QnA", score: 36.3, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 44.29, metric: "score", harness: "Codex", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 44.36, metric: "score", harness: "Codex CLI", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "SWE-Atlas Test Writing", score: 40, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }, { name: "Artificial Analysis Coding Agent Index", score: 53.6, metric: "average pass@1", harness: "Codex", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 72.4, metric: "pass@1", harness: "Codex", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 18.4, metric: "pass@1", harness: "Codex", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 69.8, metric: "pass@1", harness: "Codex", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Artificial Analysis Coding Agent Index", score: 52.2, metric: "average pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 72.9, metric: "pass@1", harness: "Cursor CLI", variant: "medium", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 18.9, metric: "pass@1", harness: "Cursor CLI", variant: "medium", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 64.7, metric: "pass@1", harness: "Cursor CLI", variant: "medium", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 75.1, metric: "success rate", version: "2.0", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GPQA Diamond", score: 92.8, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 39.8, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "Humanity's Last Exam", score: 52.1, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "OSWorld-Verified", score: 75, metric: "success rate", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "BrowseComp", score: 82.7, metric: "accuracy", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "GDPval", score: 83, metric: "wins or ties", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "ARC-AGI-2", score: 73.3, metric: "accuracy", variant: "Verified", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 47.6, metric: "accuracy", dataset: "Tier 1-3", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "FrontierMath", score: 27.1, metric: "accuracy", dataset: "Tier 4", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }, { name: "MMMU Pro", score: 81.2, metric: "accuracy", variant: "no tools", source: "https://openai.com/index/introducing-gpt-5-5/", date: "2026-04-23" }] }, "openai/o3-mini": { id: "openai/o3-mini", name: "o3-mini", description: "Smaller o-series reasoner for economical coding, math, and planning tasks", family: "o-mini", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-05", release_date: "2024-12-20", last_updated: "2025-01-29", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 1e5 }, benchmarks: [{ name: "Aider Polyglot", score: 60.4, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-31" }] }, "openai/gpt-4o-2024-11-20": { id: "openai/gpt-4o-2024-11-20", name: "GPT-4o (2024-11-20)", description: "GPT model for general reasoning, writing, coding, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2023-09", release_date: "2024-11-20", last_updated: "2024-11-20", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 }, benchmarks: [{ name: "Aider Polyglot", score: 18.2, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2024-12-30" }, { name: "Artificial Analysis Coding Index", score: 16.7, metric: "index", source: "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 33.3, metric: "percent correct", source: "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 8.3, metric: "success rate", source: "https://openrouter.ai/openai/gpt-4o-2024-11-20/benchmarks", date: "2026-03-11" }] }, "openai/gpt-oss-safeguard-120b": { id: "openai/gpt-oss-safeguard-120b", name: "GPT OSS Safeguard 120B", description: "Safety model for policy screening, moderation, and risk-aware routing workflows", family: "gpt-oss", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2025-10-29", last_updated: "2025-10-29", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/openai/gpt-oss-safeguard-120b" }] }, "openai/gpt-realtime-whisper": { id: "openai/gpt-realtime-whisper", name: "GPT Realtime Whisper", description: "Streaming speech-to-text model for low-latency transcript deltas from live audio", family: "whisper", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2026-05-07", last_updated: "2026-05-07", modalities: { input: ["audio"], output: ["text"] }, open_weights: false, limit: { context: 0, output: 0 } }, "openai/gpt-5.2-chat-latest": { id: "openai/gpt-5.2-chat-latest", name: "GPT-5.2 Chat", description: "Chat-tuned GPT model for conversational assistance, writing, and tool workflows", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2025-12-11", last_updated: "2025-12-11", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 } }, "openai/gpt-5.2-pro": { id: "openai/gpt-5.2-pro", name: "GPT-5.2 Pro", description: "Higher-accuracy GPT-5.2 variant for tougher reasoning and review workflows", family: "gpt-pro", attachment: true, reasoning: true, tool_call: true, structured_output: false, temperature: false, knowledge: "2025-08-31", release_date: "2025-12-11", last_updated: "2025-12-11", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5.1-chat-latest": { id: "openai/gpt-5.1-chat-latest", name: "GPT-5.1 Chat", description: "Chat-tuned GPT-5.1 for polished assistants, writing, and product conversations", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 } }, "openai/gpt-image-2": { id: "openai/gpt-image-2", name: "GPT-Image-2", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "gpt-image", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-04-21", last_updated: "2026-04-21", modalities: { input: ["text", "image"], output: ["image"] }, open_weights: false, limit: { context: 0, output: 0 } }, "openai/gpt-5.1-codex-max": { id: "openai/gpt-5.1-codex-max", name: "GPT-5.1 Codex Max", description: "Coding-optimized GPT model for repository edits, reviews, and agentic software work", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-5.1-codex": { id: "openai/gpt-5.1-codex", name: "GPT-5.1 Codex", description: "Codex GPT for repository edits, code review, and practical software agents", family: "gpt-codex", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "openai/gpt-4.1-mini": { id: "openai/gpt-4.1-mini", name: "GPT-4.1 mini", description: "Affordable GPT-4.1 lane for fast coding help and structured extraction", family: "gpt-mini", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-04", release_date: "2025-04-14", last_updated: "2025-04-14", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1047576, output: 32768 }, benchmarks: [{ name: "Aider Polyglot", score: 32.4, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-04-14" }] }, "openai/gpt-oss-20b": { id: "openai/gpt-oss-20b", name: "GPT OSS 20B", description: "Open GPT reasoning model for self-hosted agents and controllable deployments", family: "gpt-oss", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2025-08-05", last_updated: "2025-08-05", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/openai/gpt-oss-20b" }] }, "openai/whisper-large-v3-turbo": { id: "openai/whisper-large-v3-turbo", name: "Whisper Large v3 Turbo", description: "Speech transcription model for accurate audio-to-text and captioning workflows", family: "whisper", attachment: false, reasoning: false, tool_call: false, release_date: "2024-10-01", last_updated: "2024-10-01", modalities: { input: ["audio"], output: ["text"] }, open_weights: true, limit: { context: 448, output: 448 } }, "openai/gpt-5.4-nano": { id: "openai/gpt-5.4-nano", name: "GPT-5.4 nano", description: "Cheapest GPT-5.4 lane for simple routing, extraction, and bulk automation", family: "gpt-nano", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2025-08-31", release_date: "2026-03-17", last_updated: "2026-03-17", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Pro", score: 52.4, metric: "resolve rate", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Terminal-Bench", score: 46.3, metric: "accuracy", variant: "reasoning effort xhigh", version: "2.0", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MCP Atlas", score: 56.1, metric: "score", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Toolathlon", score: 35.5, metric: "score", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "\u03C4\xB2-Bench Telecom", score: 92.5, metric: "accuracy", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "GPQA Diamond", score: 82.8, metric: "accuracy", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Humanity's Last Exam", score: 37.7, metric: "accuracy", variant: "with tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Humanity's Last Exam", score: 24.3, metric: "accuracy", variant: "without tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OSWorld-Verified", score: 39, metric: "success rate", variant: "reasoning effort xhigh", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MMMU Pro", score: 69.5, metric: "accuracy", variant: "with Python", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "MMMU Pro", score: 66.1, metric: "accuracy", variant: "without tools", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OmniDocBench", score: 0.2419, metric: "overall edit distance", variant: "reasoning effort none", version: "1.5", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OpenAI MRCR", score: 44.2, metric: "accuracy", variant: "8-needle, 64K-128K", version: "v2", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "OpenAI MRCR", score: 33.1, metric: "accuracy", variant: "8-needle, 128K-256K", version: "v2", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Graphwalks", score: 73.4, metric: "accuracy", variant: "BFS, 0-128K", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }, { name: "Graphwalks", score: 50.8, metric: "accuracy", variant: "parents, 0-128K", source: "https://openai.com/index/introducing-gpt-5-4-mini-and-nano/", date: "2026-03-17" }] }, "openai/gpt-5.1": { id: "openai/gpt-5.1", name: "GPT-5.1", description: "Sharper GPT-5 generation for coding, product work, and tool-assisted tasks", family: "gpt", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, knowledge: "2024-09-30", release_date: "2025-11-13", last_updated: "2025-11-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 4e5, input: 272e3, output: 128e3 } }, "xai/grok-4.6": { id: "xai/grok-4.6", name: "Grok 4.6", description: "xAI's frontier model for long-running agents, coding, knowledge work, and visual projects", family: "grok", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-02-01", release_date: "2026-08-12", last_updated: "2026-08-12", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 5e5, output: 5e5 } }, "xai/grok-4.5": { id: "xai/grok-4.5", name: "Grok 4.5", description: "xAI's Grok model for chat, coding, agentic tools, and lower hallucination risk", family: "grok", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-07-08", last_updated: "2026-07-08", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 5e5, output: 5e5 }, benchmarks: [{ name: "SWE-Bench Pro", score: 64.7, metric: "resolve rate", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "SWE-Bench Multilingual", score: 78, metric: "resolve rate", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "Terminal-Bench", score: 83.3, metric: "success rate", version: "2.1", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "DeepSWE", score: 62, metric: "resolve rate", version: "1.0", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "DeepSWE", score: 53, metric: "resolve rate", harness: "mini-swe-agent", version: "1.1", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }, { name: "SWE Marathon", score: 29, metric: "pass@1", source: "https://x.ai/news/grok-4-5", date: "2026-07-08" }] }, "xai/grok-4.20-0309-non-reasoning": { id: "xai/grok-4.20-0309-non-reasoning", name: "Grok 4.20 (Non-Reasoning)", description: "Grok model for agentic tool use, reasoning, coding, and live assistance", family: "grok", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, release_date: "2026-03-09", last_updated: "2026-03-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 3e4 } }, "xai/grok-imagine-video-1.5": { id: "xai/grok-imagine-video-1.5", name: "Grok Imagine Video 1.5", description: "Video model for image-to-video generation, editing, and extension workflows", family: "grok", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-05-30", last_updated: "2026-05-30", modalities: { input: ["text", "image", "video"], output: ["video"] }, open_weights: false, limit: { context: 1024, output: 0 } }, "xai/grok-4.1-fast": { id: "xai/grok-4.1-fast", name: "Grok 4.1 Fast", description: "xAI's fast agentic tool-calling model with a 2M context window; non-reasoning variant for low-latency responses", family: "grok", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, release_date: "2025-11-19", last_updated: "2025-11-19", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e6, output: 3e4 } }, "xai/grok-4.20-0309-reasoning": { id: "xai/grok-4.20-0309-reasoning", name: "Grok 4.20 (Reasoning)", description: "Reasoning Grok for document-heavy analysis and long-horizon tool use", family: "grok", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-03-09", last_updated: "2026-03-09", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 3e4 } }, "xai/grok-imagine-image-2.0": { id: "xai/grok-imagine-image-2.0", name: "Grok Imagine Image 2.0", description: "Image model for prompt-driven generation, editing, and visual design workflows", family: "grok", attachment: true, reasoning: false, tool_call: false, temperature: false, release_date: "2026-08-07", last_updated: "2026-08-07", modalities: { input: ["text", "image"], output: ["image"] }, open_weights: false, limit: { context: 8e3, output: 0 } }, "xai/grok-build-0.1": { id: "xai/grok-build-0.1", name: "Grok Build 0.1", description: "Fast Grok coding model tuned for agentic engineering and iterative edits", family: "grok-build", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-16", last_updated: "2026-04-16", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 256e3, output: 256e3 } }, "xai/grok-4.3": { id: "xai/grok-4.3", name: "Grok 4.3", description: "xAI's default Grok for chat, coding, agentic tools, and lower hallucination risk", family: "grok", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-17", last_updated: "2026-04-17", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 3e4 }, benchmarks: [{ name: "Artificial Analysis Intelligence Index", score: 53, metric: "index score", version: "4.0", source: "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing", date: "2026-04-30" }, { name: "GDPval-AA", score: 1500, metric: "Elo", source: "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing", date: "2026-04-30" }, { name: "\u03C4\xB2-Bench Telecom", score: 98, metric: "success rate", source: "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing", date: "2026-04-30" }, { name: "IFBench", score: 81, metric: "accuracy", source: "https://artificialanalysis.ai/articles/xai-launches-grok-4-3-with-improved-agentic-performance-and-lower-pricing", date: "2026-04-30" }] }, "meituan/longcat-2.0": { id: "meituan/longcat-2.0", name: "LongCat-2.0", description: "Meituan LongCat-2.0, a reasoning model with tool calling and a 1M-token context window", family: "longcat", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-30", last_updated: "2026-06-30", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 131072 }, benchmarks: [{ name: "SWE-Bench Pro", score: 59.5, metric: "resolve rate", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "SWE-Bench Multilingual", score: 77.3, metric: "resolve rate", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "Terminal-Bench", score: 70.8, metric: "success rate", version: "2.1", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "GPQA Diamond", score: 88.9, metric: "accuracy", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "BrowseComp", score: 79.9, metric: "accuracy", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "IFEval", score: 90, metric: "accuracy", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }, { name: "FORTE", score: 73.2, metric: "success rate", source: "https://github.com/meituan-longcat/longcat-2.0", date: "2026-06-30" }] }, "swiss-ai/apertus-70b": { id: "swiss-ai/apertus-70b", name: "Apertus 70B", description: "Fully open 70B multilingual LLM supporting 1800+ languages with 65K context. Trained on 15T tokens of compliant open data. Apache 2.0, EU AI Act compliant.", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-09", release_date: "2025-09-02", last_updated: "2025-09-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 65536, output: 8192 }, license: "Apache-2.0", links: [{ label: "Paper", url: "https://arxiv.org/abs/2509.14233", type: "paper" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/swiss-ai/Apertus-70B-Instruct-2509" }] }, "swiss-ai/apertus-8b": { id: "swiss-ai/apertus-8b", name: "Apertus 8B", description: "Fully open 8B multilingual LLM supporting 1800+ languages with 65K context. Trained on compliant open data. Apache 2.0, EU AI Act compliant.", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-09", release_date: "2025-09-02", last_updated: "2025-09-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 65536, output: 8192 }, license: "Apache-2.0", links: [{ label: "Paper", url: "https://arxiv.org/abs/2509.14233", type: "paper" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/swiss-ai/Apertus-8B-Instruct-2509" }] }, "stepfun/step-3.5-flash-2603": { id: "stepfun/step-3.5-flash-2603", name: "Step 3.5 Flash 2603", description: "StepFun flash model for efficient multimodal reasoning, coding, and tool use", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-01", release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, input: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/stepfun-ai/Step-3.5-Flash" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 34.6, metric: "index", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 38.5, metric: "percent correct", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 32.6, metric: "success rate", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }] }, "stepfun/step-3.5-flash": { id: "stepfun/step-3.5-flash", name: "Step 3.5 Flash", description: "StepFun flash lane for quick multimodal reasoning and coding assistance", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-01", release_date: "2026-01-29", last_updated: "2026-02-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, input: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/stepfun-ai/Step-3.5-Flash" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 31.6, metric: "index", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 40.4, metric: "percent correct", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 27.3, metric: "success rate", source: "https://openrouter.ai/stepfun/step-3.5-flash/benchmarks", date: "2026-06-02" }, { name: "SWE-Bench Verified", score: 74.4, metric: "resolved", source: "https://arxiv.org/abs/2602.10604" }] }, "stepfun/step-3.7-flash": { id: "stepfun/step-3.7-flash", name: "Step 3.7 Flash", description: "Newer StepFun flash model for faster agents, coding, and multimodal prompts", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2026-03-01", release_date: "2026-05-29", last_updated: "2026-05-29", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 256e3, input: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/stepfun-ai/Step-3.7-Flash" }], benchmarks: [{ name: "SWE-Bench Pro", score: 56.3, metric: "resolve rate", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "SWE-Bench Verified", score: 76.5, metric: "resolved", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "Terminal-Bench", score: 59.6, metric: "success rate", version: "2.1", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "Humanity's Last Exam", score: 47.2, metric: "accuracy", variant: "with tools", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "BrowseComp", score: 75.8, metric: "accuracy", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "Toolathlon", score: 49.5, metric: "success rate", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "GDPval", score: 45.8, metric: "wins or ties", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "ClawEval", score: 67.1, metric: "pass^3", version: "1.1", source: "https://static.stepfun.com/blog/step-3.7-flash/", date: "2026-05-29" }, { name: "Artificial Analysis Coding Index", score: 37.1, metric: "index", source: "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks", date: "2026-06-15" }, { name: "SciCode", score: 40, metric: "percent correct", source: "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks", date: "2026-06-15" }, { name: "Terminal-Bench Hard", score: 35.6, metric: "success rate", source: "https://openrouter.ai/stepfun/step-3.7-flash/benchmarks", date: "2026-06-15" }] }, "sarvam/sarvam-105b": { id: "sarvam/sarvam-105b", name: "Sarvam 105B", description: "Flagship Indian-language reasoning model for enterprise multilingual applications", family: "sarvam", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-09-01", last_updated: "2025-09-01", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 } }, "sarvam/sarvam-30b": { id: "sarvam/sarvam-30b", name: "Sarvam 30B", description: "Efficient Indian-language reasoning model for chat, coding, and multilingual work", family: "sarvam", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-18", last_updated: "2026-02-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 } }, "trendyol/asure-12b": { id: "trendyol/asure-12b", name: "Trendyol Asure 12B", description: "Turkish-language multimodal instruct model built on Gemma 3 12B for e-commerce text, chat, and image-text tasks", family: "gemma", attachment: true, reasoning: false, tool_call: false, temperature: true, release_date: "2026-02-19", last_updated: "2026-02-20", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072 }, license: "Gemma", weights: [{ label: "Hugging Face", url: "https://huggingface.co/Trendyol/Trendyol-LLM-Asure-12B" }] }, "deepreinforce/ornith-1.0-35b": { id: "deepreinforce/ornith-1.0-35b", name: "Ornith 1.0 35B", description: "Large coding-reasoning model for agentic software tasks and RL search", family: "ornith", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-25", last_updated: "2026-06-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144 }, license: "MIT", links: [{ label: "Model card", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B", type: "model_card" }, { label: "Announcement", url: "https://deep-reinforce.com/ornith_1_0.html", type: "announcement" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 75.6, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "SWE-Bench Pro", score: 50.4, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "SWE-Bench Multilingual", score: 69.3, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "Terminal-Bench 2.1", score: 64.2, metric: "percent", variant: "Terminus-2", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "Terminal-Bench 2.1", score: 62.8, metric: "percent", variant: "Claude Code", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "NL2Repo", score: 34.6, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }, { name: "Claw-eval", score: 69.8, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-35B" }] }, "deepreinforce/ornith-1.0-397b": { id: "deepreinforce/ornith-1.0-397b", name: "Ornith 1.0 397B", description: "Large coding-reasoning model for agentic software tasks and RL search", family: "ornith", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-25", last_updated: "2026-06-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144 }, license: "MIT", links: [{ label: "Model card", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B", type: "model_card" }, { label: "Announcement", url: "https://deep-reinforce.com/ornith_1_0.html", type: "announcement" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { label: "Hugging Face (FP8)", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B-FP8", quantization: "fp8" }], benchmarks: [{ name: "SWE-Bench Verified", score: 82.4, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "SWE-Bench Pro", score: 62.2, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "SWE-Bench Multilingual", score: 78.9, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "Terminal-Bench 2.1", score: 77.5, metric: "percent", variant: "Terminus-2", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "Terminal-Bench 2.1", score: 78.2, metric: "percent", variant: "Claude Code", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "NL2Repo", score: 48.2, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }, { name: "Claw-eval", score: 77.1, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-397B" }] }, "deepreinforce/ornith-1.0-31b": { id: "deepreinforce/ornith-1.0-31b", name: "Ornith 1.0 31B", description: "Open coding-reasoning model for repository tasks and self-improving agents", family: "ornith", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-25", last_updated: "2026-06-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144 }, license: "MIT", links: [{ label: "Announcement", url: "https://deep-reinforce.com/ornith_1_0.html", type: "announcement" }] }, "deepreinforce/ornith-1.0-9b": { id: "deepreinforce/ornith-1.0-9b", name: "Ornith 1.0 9B", description: "Open coding-reasoning model for repository tasks and self-improving agents", family: "ornith", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-25", last_updated: "2026-06-25", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144 }, license: "MIT", links: [{ label: "Model card", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B", type: "model_card" }, { label: "Announcement", url: "https://deep-reinforce.com/ornith_1_0.html", type: "announcement" }], weights: [{ label: "Hugging Face", url: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 69.4, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "SWE-Bench Pro", score: 42.9, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "SWE-Bench Multilingual", score: 52, metric: "percent resolved", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "Terminal-Bench 2.1", score: 43.1, metric: "percent", variant: "Terminus-2", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "Terminal-Bench 2.1", score: 40.6, metric: "percent", variant: "Claude Code", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "NL2Repo", score: 27.2, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }, { name: "Claw-eval", score: 63.1, metric: "percent", source: "https://huggingface.co/deepreinforce-ai/Ornith-1.0-9B" }] }, "xiaomi/mimo-v2.5-pro-ultraspeed": { id: "xiaomi/mimo-v2.5-pro-ultraspeed", name: "MiMo-V2.5-Pro-UltraSpeed", description: "MiMo pro model for strong multimodal reasoning and agent execution", family: "mimo", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-06-08", last_updated: "2026-06-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro-FP4-DFlash" }] }, "xiaomi/mimo-v2.5": { id: "xiaomi/mimo-v2.5", name: "MiMo-V2.5", description: "Open MiMo model for multimodal coding agents and long-context automation", family: "mimo", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-04-22", last_updated: "2026-04-22", modalities: { input: ["text", "image", "audio", "video"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/XiaomiMiMo/MiMo-V2.5" }] }, "xiaomi/mimo-v2-pro": { id: "xiaomi/mimo-v2-pro", name: "MiMo-V2-Pro", description: "Earlier MiMo Pro model for multimodal agents, reasoning, and code tasks", family: "mimo", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-03-18", last_updated: "2026-03-18", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 131072 } }, "xiaomi/mimo-v2-omni": { id: "xiaomi/mimo-v2-omni", name: "MiMo-V2-Omni", description: "MiMo omni model for text, image, video, audio, and agents", family: "mimo", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-03-18", last_updated: "2026-03-18", modalities: { input: ["text", "image", "audio", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 131072 } }, "xiaomi/mimo-v2-flash": { id: "xiaomi/mimo-v2-flash", name: "MiMo-V2-Flash", description: "MiMo flash model for fast multimodal assistance and agent workflows", family: "mimo", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12-01", release_date: "2025-12-16", last_updated: "2026-02-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/XiaomiMiMo/MiMo-V2-Flash" }] }, "xiaomi/mimo-v2.5-pro": { id: "xiaomi/mimo-v2.5-pro", name: "MiMo-V2.5-Pro", description: "Stronger MiMo Pro tier for multimodal reasoning and coding-agent execution", family: "mimo", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-12", release_date: "2026-04-22", last_updated: "2026-04-22", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro" }], benchmarks: [{ name: "SWE-Bench Verified", score: 78.9, metric: "resolved", source: "https://huggingface.co/XiaomiMiMo/MiMo-V2.5-Pro" }, { name: "SWE-Bench Pro", score: 57.2, metric: "resolve rate", source: "https://mimo.xiaomi.com/mimo-v2-5-pro/", date: "2026-04-22" }, { name: "GPQA Diamond", score: 86.6, metric: "accuracy", source: "https://mimo.xiaomi.com/mimo-v2-5-pro/", date: "2026-04-22" }] }, "alibaba/qwen3-vl-plus": { id: "alibaba/qwen3-vl-plus", name: "Qwen3-VL Plus", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 32768 } }, "alibaba/qwen3.8-max-preview": { id: "alibaba/qwen3.8-max-preview", name: "Qwen3.8 Max Preview", description: "Preview Qwen flagship for million-token multimodal reasoning and long-horizon agentic workflows", family: "qwen", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-07-19", last_updated: "2026-07-19", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 131072 }, benchmarks: [{ name: "Terminal-Bench", score: 86.6, metric: "accuracy", variant: "xhigh", version: "2.1", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "SWE-Bench Pro", score: 67.7, metric: "resolve rate", harness: "Claude Code", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "DeepSWE", score: 56.6, metric: "resolve rate", harness: "Claude Code", variant: "xhigh", version: "1.1", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "NL2Repo", score: 55.9, metric: "resolve rate", harness: "Claude Code", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "FrontierSWE", score: 73.5, metric: "dominance score", harness: "Claude Code", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "MLS-Bench-Lite", score: 41, metric: "score", harness: "Claude Code", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "AutomationBench", score: 27.3, metric: "pass@1", variant: "xhigh", dataset: "600-task public subset", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "Toolathlon Verified", score: 72.5, metric: "pass@1", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "WideSearch", score: 81.9, metric: "F1", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "Humanity's Last Exam", score: 56.2, metric: "accuracy", variant: "xhigh, with tools", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "GPQA Diamond", score: 92.6, metric: "accuracy", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "Humanity's Last Exam", score: 43.6, metric: "accuracy", variant: "xhigh, no tools", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "IFBench", score: 82.8, metric: "score", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "OSWorld-Verified", score: 86.1, metric: "success rate", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }, { name: "MMMU Pro", score: 82.3, metric: "accuracy", variant: "xhigh", source: "https://www.alibabacloud.com/blog/qwen3-8-max-a-new-bar-for-coding-and-cowork_603421", date: "2026-08-03" }] }, "alibaba/qwen3-coder-30b-a3b-instruct": { id: "alibaba/qwen3-coder-30b-a3b-instruct", name: "Qwen3-Coder 30B-A3B Instruct", description: "Smaller Qwen coder for efficient local agents and repo-level fixes", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-04", last_updated: "2025-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Coder-30B-A3B-Instruct" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 19.4, metric: "index", source: "https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct/benchmarks", date: "2026-06-02" }, { name: "SciCode", score: 27.8, metric: "percent correct", source: "https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct/benchmarks", date: "2026-06-02" }, { name: "Terminal-Bench Hard", score: 15.2, metric: "success rate", source: "https://openrouter.ai/qwen/qwen3-coder-30b-a3b-instruct/benchmarks", date: "2026-06-02" }] }, "alibaba/qwen3.7-max": { id: "alibaba/qwen3.7-max", name: "Qwen3.7 Max", description: "Qwen frontier model tuned for agent frameworks, coding assistants, and long tasks", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-05-21", last_updated: "2026-05-21", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 }, benchmarks: [{ name: "SWE-Bench Verified", score: 80.4, metric: "resolved", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "SWE-Bench Pro", score: 60.6, metric: "resolve rate", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "SWE-Bench Multilingual", score: 78.3, metric: "resolve rate", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "Terminal-Bench", score: 69.7, metric: "success rate", harness: "Terminus-2", version: "2.0", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "GPQA Diamond", score: 92.4, metric: "accuracy", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "Humanity's Last Exam", score: 41.4, metric: "accuracy", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "SciCode", score: 53.5, source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "MCP Atlas", score: 76.4, metric: "success rate", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }, { name: "NL2Repo", score: 47.2, harness: "Claude Code", source: "https://qwen.ai/blog?id=qwen3.7", date: "2026-05-19" }] }, "alibaba/qwen-turbo": { id: "alibaba/qwen-turbo", name: "Qwen Turbo", description: "Efficient Qwen model for fast chat, extraction, and high-volume workloads", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-11-01", last_updated: "2025-04-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 16384 } }, "alibaba/qwen-omni-turbo": { id: "alibaba/qwen-omni-turbo", name: "Qwen-Omni Turbo", description: "Qwen omni model for text, vision, audio, and multimodal agent tasks", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2025-01-19", last_updated: "2025-03-26", modalities: { input: ["text", "image", "audio", "video"], output: ["text", "audio"] }, open_weights: false, limit: { context: 32768, output: 2048 } }, "alibaba/qwen-vl-max": { id: "alibaba/qwen-vl-max", name: "Qwen-VL Max", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-04-08", last_updated: "2025-08-13", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "alibaba/qwen3-32b": { id: "alibaba/qwen3-32b", name: "Qwen3 32B", description: "Dense open Qwen model for self-hosted chat, reasoning, and coding", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-04", last_updated: "2025-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-32B" }], benchmarks: [{ name: "Aider Polyglot", score: 40, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-08" }] }, "alibaba/qwen3-235b-a22b": { id: "alibaba/qwen3-235b-a22b", name: "Qwen3 235B-A22B", description: "Large open Qwen MoE for multilingual reasoning, coding, and tool use", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-04", last_updated: "2025-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-235B-A22B" }], benchmarks: [{ name: "Aider Polyglot", score: 59.6, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-05-09" }, { name: "SWE-Bench Pro", score: 21.41, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "alibaba/qwen-max": { id: "alibaba/qwen-max", name: "Qwen Max", description: "Flagship Qwen model for complex reasoning, coding, and agentic workflows", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-04-03", last_updated: "2025-01-25", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 32768, output: 8192 }, benchmarks: [{ name: "Aider Polyglot", score: 21.8, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-28" }] }, "alibaba/qwen3.5-35b-a3b": { id: "alibaba/qwen3.5-35b-a3b", name: "Qwen3.5 35B-A3B", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-35B-A3B" }] }, "alibaba/qwen3-vl-235b-a22b-thinking": { id: "alibaba/qwen3-vl-235b-a22b-thinking", name: "Qwen3 VL 235B A22B Thinking", description: "Qwen vision-language thinking model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Thinking" }] }, "alibaba/qwen3-30b-a3b": { id: "alibaba/qwen3-30b-a3b", name: "Qwen3 30B A3B", description: "Sparse MoE Qwen model with 3B active parameters for efficient chat and reasoning", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-04-28", last_updated: "2025-04-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-30B-A3B" }] }, "alibaba/qwen3.5-397b-a17b": { id: "alibaba/qwen3.5-397b-a17b", name: "Qwen3.5 397B-A17B", description: "Large open Qwen multimodal MoE for visual agents and long technical tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-15", last_updated: "2026-02-15", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-397B-A17B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 76.4, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.5-397B-A17B" }] }, "alibaba/qwen3.5-flash": { id: "alibaba/qwen3.5-flash", name: "Qwen3.5 Flash", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen2.5-coder-0.5b": { id: "alibaba/qwen2.5-coder-0.5b", name: "Qwen2.5-Coder-0.5B", description: "Tiny open Qwen code model for lightweight completion and on-device coding", family: "qwen", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2024-11-12", last_updated: "2024-11-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 32768, output: 8192 }, license: "Apache 2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen2.5-Coder-0.5B" }] }, "alibaba/qwen-plus": { id: "alibaba/qwen-plus", name: "Qwen Plus", description: "Qwen instruction model for multilingual chat, reasoning, and tool use", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-01-25", last_updated: "2025-09-11", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 32768 } }, "alibaba/qwen3-coder-next": { id: "alibaba/qwen3-coder-next", name: "Qwen3 Coder Next", description: "Open-weight Qwen coding model for agents, repository edits, and multi-turn tool use", family: "qwen", attachment: false, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-09", release_date: "2026-02-03", last_updated: "2026-02-03", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Coder-Next" }] }, "alibaba/qwen3.5-plus": { id: "alibaba/qwen3.5-plus", name: "Qwen3.5 Plus", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-02-16", last_updated: "2026-02-16", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen2-5-vl-72b-instruct": { id: "alibaba/qwen2-5-vl-72b-instruct", name: "Qwen2.5-VL 72B Instruct", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-09", last_updated: "2024-09", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen2.5-VL-72B-Instruct" }] }, "alibaba/qwen3-coder-flash": { id: "alibaba/qwen3-coder-flash", name: "Qwen3 Coder Flash", description: "Qwen coding model for software agents, repository edits, and code reasoning", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen3.6-max-preview": { id: "alibaba/qwen3.6-max-preview", name: "Qwen3.6 Max Preview", description: "Flagship Qwen model for complex reasoning, coding, and agentic workflows", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-04-20", last_updated: "2026-04-20", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 65536 } }, "alibaba/qwen3.8-max": { id: "alibaba/qwen3.8-max", name: "Qwen3.8 Max", description: "2.4-trillion-parameter MoE flagship for coding, professional work, multimodal understanding, and long-horizon agentic workflows", family: "qwen", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-08-03", last_updated: "2026-08-03", modalities: { input: ["text", "image", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 131072 } }, "alibaba/qwen3.7-plus": { id: "alibaba/qwen3.7-plus", name: "Qwen3.7 Plus", description: "Multimodal Qwen workhorse for long-context agents, visual inputs, and coding", family: "qwen", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-06-02", last_updated: "2026-06-02", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 64e3 } }, "alibaba/qwq-32b": { id: "alibaba/qwq-32b", name: "QwQ 32B", description: "Open reasoning model from the Qwen team for math, coding, and step-by-step problem solving", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2025-03-05", last_updated: "2025-03-05", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/QwQ-32B" }] }, "alibaba/qwen3.7-flash": { id: "alibaba/qwen3.7-flash", name: "Qwen3.7 Flash", description: "Lightweight multimodal Qwen model for high-throughput text, image, and video tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-07-15", last_updated: "2026-07-15", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, input: 991e3, output: 65536 } }, "alibaba/qwen3-next-80b-a3b-thinking": { id: "alibaba/qwen3-next-80b-a3b-thinking", name: "Qwen3-Next 80B-A3B (Thinking)", description: "Efficient Qwen thinking model for local reasoning, math, and coding agents", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09", last_updated: "2025-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Thinking" }] }, "alibaba/qwen3-coder-480b-a35b-instruct": { id: "alibaba/qwen3-coder-480b-a35b-instruct", name: "Qwen3-Coder 480B-A35B Instruct", description: "Open Qwen coding heavyweight for repository reasoning and agentic engineering", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-04", last_updated: "2025-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Coder-480B-A35B-Instruct" }], benchmarks: [{ name: "SWE-Bench Pro", score: 38.7, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "alibaba/qwen2.5-coder-32b-instruct": { id: "alibaba/qwen2.5-coder-32b-instruct", name: "Qwen2.5-Coder-32B-Instruct", description: "Open coding-focused Qwen model for code generation, repair, and repository reasoning", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2024-11-12", last_updated: "2024-11-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen2.5-Coder-32B-Instruct" }] }, "alibaba/qwen-flash": { id: "alibaba/qwen-flash", name: "Qwen Flash", description: "Efficient Qwen model for fast chat, extraction, and high-volume workloads", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 32768 } }, "alibaba/qwen3-max": { id: "alibaba/qwen3-max", name: "Qwen3 Max", description: "Flagship Qwen3 model for coding agents, complex reasoning, and tool use", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 65536 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 26.4, metric: "index", source: "https://openrouter.ai/qwen/qwen3-max/benchmarks", date: "2026-05-30" }, { name: "SciCode", score: 38.3, metric: "percent correct", source: "https://openrouter.ai/qwen/qwen3-max/benchmarks", date: "2026-05-30" }, { name: "Terminal-Bench Hard", score: 20.5, metric: "success rate", source: "https://openrouter.ai/qwen/qwen3-max/benchmarks", date: "2026-05-30" }] }, "alibaba/qwen3.6-35b-a3b": { id: "alibaba/qwen3.6-35b-a3b", name: "Qwen3.6 35B-A3B", description: "Open multimodal Qwen MoE for local agents that need vision, audio, and code", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-17", last_updated: "2026-04-17", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.6-35B-A3B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 73.4, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.6-35B-A3B" }] }, "alibaba/qwen3-235b-a22b-instruct-2507": { id: "alibaba/qwen3-235b-a22b-instruct-2507", name: "Qwen3 235B-A22B Instruct 2507", description: "Updated large open Qwen3 MoE instruct model for multilingual chat, coding, and tool use", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2025-07-21", last_updated: "2025-07-21", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 16384 }, license: "Apache 2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-235B-A22B-Instruct-2507" }] }, "alibaba/qwen3.6-27b": { id: "alibaba/qwen3.6-27b", name: "Qwen3.6 27B", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-22", last_updated: "2026-04-22", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.6-27B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 77.2, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.6-27B" }] }, "alibaba/qwen3-next-80b-a3b-instruct": { id: "alibaba/qwen3-next-80b-a3b-instruct", name: "Qwen3-Next 80B-A3B Instruct", description: "Qwen instruction model for multilingual chat, reasoning, and tool use", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09", last_updated: "2025-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-Next-80B-A3B-Instruct" }] }, "alibaba/qwen3.8-27b": { id: "alibaba/qwen3.8-27b", name: "Qwen3.8 27B", description: "Dense 27B vision-language model for coding, agent tasks, and image and video understanding", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-14", last_updated: "2026-08-14", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.8-27B" }], benchmarks: [{ name: "SWE-bench Pro", score: 61.7, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.8-27B" }] }, "alibaba/qwen3.6-plus": { id: "alibaba/qwen3.6-plus", name: "Qwen3.6 Plus", description: "Earlier Qwen multimodal workhorse for million-token agent and document tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-04-02", last_updated: "2026-04-02", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen3.5-9b": { id: "alibaba/qwen3.5-9b", name: "Qwen3.5 9B", description: "Qwen instruction model for multilingual chat, reasoning, and tool use", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-9B" }] }, "alibaba/qwen3.5-27b": { id: "alibaba/qwen3.5-27b", name: "Qwen3.5 27B", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-27B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 72.4, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.5-27B" }] }, "alibaba/qwq-plus": { id: "alibaba/qwq-plus", name: "QwQ Plus", description: "Qwen reasoning model for deliberate problem solving, math, and coding", family: "qwen", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2025-03-05", last_updated: "2025-03-05", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "alibaba/qwen3.5-122b-a10b": { id: "alibaba/qwen3.5-122b-a10b", name: "Qwen3.5 122B-A10B", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-02-23", last_updated: "2026-02-23", modalities: { input: ["text", "image", "video", "audio"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 65536 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.5-122B-A10B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 72, metric: "resolved", source: "https://huggingface.co/Qwen/Qwen3.5-122B-A10B" }] }, "alibaba/qwen3-coder-plus": { id: "alibaba/qwen3-coder-plus", name: "Qwen3 Coder Plus", description: "Hosted Qwen coder for software agents, repo edits, and long-context code", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-23", last_updated: "2025-07-23", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1048576, output: 65536 } }, "alibaba/qwen3-vl-235b-a22b-instruct": { id: "alibaba/qwen3-vl-235b-a22b-instruct", name: "Qwen3 VL 235B A22B Instruct", description: "Qwen vision-language instruct model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2025-03-31", release_date: "2025-09-23", last_updated: "2025-09-23", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3-VL-235B-A22B-Instruct" }] }, "alibaba/qwen-vl-plus": { id: "alibaba/qwen-vl-plus", name: "Qwen-VL Plus", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-04", release_date: "2024-01-25", last_updated: "2025-08-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "alibaba/qwen3.6-flash": { id: "alibaba/qwen3.6-flash", name: "Qwen3.6 Flash", description: "Qwen vision-language model for visual reasoning, documents, and agent tasks", family: "qwen3.6", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-27", last_updated: "2026-04-27", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 65536 } }, "alibaba/qwen3.8-2.4t-a95b": { id: "alibaba/qwen3.8-2.4t-a95b", name: "Qwen3.8 2.4T A95B", description: "Open-weight sparse MoE (2.4T total, 95B active), the open-weight twin of Qwen3.8 Max for coding, research, complex reasoning, and agentic workflows", family: "qwen", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-12", last_updated: "2026-08-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 131072 }, license: "qwen3.8-max", weights: [{ label: "Hugging Face", url: "https://huggingface.co/Qwen/Qwen3.8-2.4T-A95B" }] }, "sakana/fugu": { id: "sakana/fugu", name: "Fugu", description: "Multi-agent model for routing expert agents across complex analytical tasks", family: "fugu", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, release_date: "2026-06-15", last_updated: "2026-06-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 1e6 }, links: [{ label: "Official model catalog", url: "https://raw.githubusercontent.com/SakanaAI/fugu/refs/heads/main/configs/files/fugu.json", type: "docs" }], benchmarks: [{ name: "SWE Bench Pro", score: 59, source: "https://console.sakana.ai/models" }, { name: "Terminal Bench 2.1", score: 80.2, source: "https://console.sakana.ai/models" }, { name: "LiveCodeBench", score: 92.9, source: "https://console.sakana.ai/models" }, { name: "LiveCodeBench Pro", score: 87.8, source: "https://console.sakana.ai/models" }, { name: "Humanity\u2019s Last Exam", score: 47.2, source: "https://console.sakana.ai/models" }, { name: "CharXiv Reasoning", score: 85.1, source: "https://console.sakana.ai/models" }, { name: "GPQA Diamond", score: 95.5, source: "https://console.sakana.ai/models" }, { name: "SciCode", score: 60.1, source: "https://console.sakana.ai/models" }, { name: "\u03C43 Banking", score: 21.7, source: "https://console.sakana.ai/models" }, { name: "Long Context Reasoning", score: 74.7, source: "https://console.sakana.ai/models" }, { name: "MRCRv2", score: 86.6, source: "https://console.sakana.ai/models" }, { name: "CTI-REALM", score: 67.5, source: "https://console.sakana.ai/models" }] }, "sakana/sakana-namazu": { id: "sakana/sakana-namazu", name: "Sakana Namazu", description: "Japanese-specialized reasoning model based on Kimi K2.6 and tuned for Japanese language, culture, and business workflows", family: "sakana-namazu", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-03", last_updated: "2026-08-03", modalities: { input: ["text", "image", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 262144, output: 65536 }, links: [{ label: "Official product page", url: "https://sakana.ai/namazu/", type: "announcement" }, { label: "Official model documentation", url: "https://console.sakana.ai/models?model=sakana-namazu", type: "docs" }], benchmarks: [{ name: "AIME26", score: 96.67, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "MMLU-Pro", score: 90.33, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "LiveCodeBench v6", score: 90.33, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "JFBench", score: 37.4, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "Translation", score: 52.2, source: "https://console.sakana.ai/models?model=sakana-namazu" }, { name: "FairPoliticsQA", score: 56.3, source: "https://console.sakana.ai/models?model=sakana-namazu" }] }, "sakana/fugu-ultra": { id: "sakana/fugu-ultra", name: "Fugu Ultra", description: "Quality-first multi-agent model for hard research, analysis, and competitions", family: "fugu", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: false, release_date: "2026-06-15", last_updated: "2026-06-15", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 1e6 }, links: [{ label: "Official model catalog", url: "https://raw.githubusercontent.com/SakanaAI/fugu/refs/heads/main/configs/files/fugu.json", type: "docs" }], benchmarks: [{ name: "SWE Bench Pro", score: 73.7, source: "https://console.sakana.ai/models" }, { name: "Terminal Bench 2.1", score: 82.1, source: "https://console.sakana.ai/models" }, { name: "LiveCodeBench", score: 93.2, source: "https://console.sakana.ai/models" }, { name: "LiveCodeBench Pro", score: 90.8, source: "https://console.sakana.ai/models" }, { name: "Humanity\u2019s Last Exam", score: 50, source: "https://console.sakana.ai/models" }, { name: "CharXiv Reasoning", score: 86.6, source: "https://console.sakana.ai/models" }, { name: "GPQA Diamond", score: 95.5, source: "https://console.sakana.ai/models" }, { name: "SciCode", score: 58.7, source: "https://console.sakana.ai/models" }, { name: "\u03C43 Banking", score: 20.6, source: "https://console.sakana.ai/models" }, { name: "Long Context Reasoning", score: 73.3, source: "https://console.sakana.ai/models" }, { name: "MRCRv2", score: 93.6, source: "https://console.sakana.ai/models" }, { name: "CTI-REALM", score: 69.4, source: "https://console.sakana.ai/models" }] }, "perplexity/sonar-deep-research": { id: "perplexity/sonar-deep-research", name: "Sonar Deep Research", description: "Sonar search model for autonomous research and citation-backed long-form reports", family: "sonar", attachment: false, reasoning: true, tool_call: false, temperature: false, knowledge: "2025-01", release_date: "2025-02-01", last_updated: "2025-09-01", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 32768 } }, "perplexity/sonar": { id: "perplexity/sonar", name: "Sonar", description: "Fast web-grounded Sonar for current answers, citations, and lightweight retrieval", family: "sonar", attachment: false, reasoning: false, tool_call: false, temperature: true, knowledge: "2025-09-01", release_date: "2024-01-01", last_updated: "2025-09-01", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 4096 }, benchmarks: [{ name: "SciCode", score: 22.9, metric: "percent correct", source: "https://openrouter.ai/perplexity/sonar/benchmarks", date: "2026-03-11" }] }, "perplexity/sonar-reasoning-pro": { id: "perplexity/sonar-reasoning-pro", name: "Sonar Reasoning Pro", description: "Web-grounded Sonar for multi-step research questions that need cited reasoning", family: "sonar-reasoning", attachment: true, reasoning: true, tool_call: false, temperature: true, knowledge: "2025-09-01", release_date: "2024-01-01", last_updated: "2025-09-01", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 4096 } }, "perplexity/sonar-pro": { id: "perplexity/sonar-pro", name: "Sonar Pro", description: "Deeper Sonar search model with broader retrieval and stronger synthesis", family: "sonar-pro", attachment: true, reasoning: false, tool_call: false, temperature: true, knowledge: "2025-09-01", release_date: "2024-01-01", last_updated: "2025-09-01", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 8192 }, benchmarks: [{ name: "SciCode", score: 22.6, metric: "percent correct", source: "https://openrouter.ai/perplexity/sonar-pro/benchmarks", date: "2026-03-11" }] }, "zhipuai/glm-4.6": { id: "zhipuai/glm-4.6", name: "GLM-4.6", description: "Late GLM-4 workhorse for coding agents, reasoning, and structured tasks", family: "glm", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-09-30", last_updated: "2025-09-30", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.6" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 29.5, metric: "index", source: "https://openrouter.ai/z-ai/glm-4.6/benchmarks", date: "2026-05-22" }, { name: "SciCode", score: 38.4, metric: "percent correct", source: "https://openrouter.ai/z-ai/glm-4.6/benchmarks", date: "2026-05-22" }, { name: "Terminal-Bench Hard", score: 25, metric: "success rate", source: "https://openrouter.ai/z-ai/glm-4.6/benchmarks", date: "2026-05-22" }, { name: "SWE-Bench Pro", score: 9.67, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "zhipuai/glm-4.5-flash": { id: "zhipuai/glm-4.5-flash", name: "GLM-4.5-Flash", description: "Efficient GLM model for fast reasoning, coding, and agent workflows", family: "glm-flash", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 98304 } }, "zhipuai/glm-5v-turbo": { id: "zhipuai/glm-5v-turbo", name: "GLM-5V-Turbo", description: "Fast GLM vision model for screenshots, documents, and multimodal agent tasks", family: "glm", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-04-01", last_updated: "2026-04-01", modalities: { input: ["text", "image", "video", "pdf"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 131072 } }, "zhipuai/glm-4.7": { id: "zhipuai/glm-4.7", name: "GLM-4.7", description: "Mature GLM model for dependable coding, reasoning, and structured agent tasks", family: "glm", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-12-22", last_updated: "2025-12-22", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.7" }], benchmarks: [{ name: "SWE-Bench Verified", score: 73.8, metric: "resolved", source: "https://huggingface.co/zai-org/GLM-4.7" }, { name: "Terminal Bench 2.0", score: 33.4, metric: "score", source: "https://huggingface.co/zai-org/GLM-4.7" }] }, "zhipuai/glm-4.5v": { id: "zhipuai/glm-4.5v", name: "GLM-4.5V", description: "GLM vision model for visual reasoning, documents, and multimodal agents", family: "glm", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-08-11", last_updated: "2025-08-11", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 64e3, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.5V" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 10.9, metric: "index", source: "https://openrouter.ai/z-ai/glm-4.5v/benchmarks", date: "2026-04-29" }, { name: "SciCode", score: 22.1, metric: "percent correct", source: "https://openrouter.ai/z-ai/glm-4.5v/benchmarks", date: "2026-04-29" }, { name: "Terminal-Bench Hard", score: 5.3, metric: "success rate", source: "https://openrouter.ai/z-ai/glm-4.5v/benchmarks", date: "2026-04-29" }] }, "zhipuai/glm-4.6v-flash": { id: "zhipuai/glm-4.6v-flash", name: "GLM-4.6V-Flash", description: "Lightweight GLM vision model for visual reasoning, documents, and multimodal agents", family: "glm", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-08", last_updated: "2025-12-08", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.6V-Flash" }] }, "zhipuai/glm-4.7-flashx": { id: "zhipuai/glm-4.7-flashx", name: "GLM-4.7-FlashX", description: "Efficient GLM model for fast reasoning, coding, and agent workflows", family: "glm-flash", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-01-19", last_updated: "2026-01-19", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 2e5, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.7-Flash" }] }, "zhipuai/glm-5": { id: "zhipuai/glm-5", name: "GLM-5", description: "General GLM flagship for coding, analysis, and tool-heavy engineering workflows", family: "glm", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-12", last_updated: "2026-02-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-5" }], benchmarks: [{ name: "SWE-Bench Verified", score: 72.8, metric: "resolved", source: "https://www.swebench.com/" }, { name: "SWE-Atlas Codebase QnA", score: 20.5, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 24.24, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 28.74, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] }, "zhipuai/glm-5-turbo": { id: "zhipuai/glm-5-turbo", name: "GLM-5-Turbo", description: "Faster GLM-5 lane for coding agents that need lower latency", family: "glm", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-03-16", last_updated: "2026-03-16", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 131072 } }, "zhipuai/glm-5.3": { id: "zhipuai/glm-5.3", name: "GLM-5.3", description: "Flagship GLM model for long-horizon coding, agents, and complex project delivery", family: "glm", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-08-14", last_updated: "2026-08-14", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 1e6, output: 131072 } }, "zhipuai/glm-5.2": { id: "zhipuai/glm-5.2", name: "GLM-5.2", description: "Open flagship GLM for long-horizon coding agents and million-token context work", family: "glm", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-06-13", last_updated: "2026-06-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 1e6, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-5.2" }], benchmarks: [{ name: "SWE-Bench Pro", score: 62.1, metric: "resolve rate", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Terminal-Bench", score: 82.7, metric: "success rate", harness: "Claude Code", version: "2.1", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "FrontierSWE", score: 74.4, metric: "dominance", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Humanity's Last Exam", score: 40.5, metric: "accuracy", dataset: "text-only subset", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Humanity's Last Exam", score: 54.7, metric: "accuracy", variant: "with tools", dataset: "text-only subset", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "CritPt", score: 20.9, metric: "accuracy", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "AIME", score: 99.2, metric: "accuracy", version: "2026", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "HMMT", score: 94.4, metric: "accuracy", version: "November 2025", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "HMMT", score: 92.5, metric: "accuracy", version: "February 2026", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "IMOAnswerBench", score: 91, metric: "accuracy", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "GPQA Diamond", score: 91.2, metric: "accuracy", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "NL2Repo", score: 48.9, metric: "resolve rate", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "DeepSWE", score: 46.2, metric: "resolve rate", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Program Bench", score: 63.7, metric: "score", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Terminal-Bench", score: 81, metric: "success rate", harness: "Terminus 2", version: "2.1", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "PostTrainBench", score: 34.3, metric: "score", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "SWE Marathon", score: 13, metric: "resolve rate", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "MCP Atlas", score: 76.8, metric: "score", dataset: "public subset", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }, { name: "Tool-Decathlon", score: 48.2, metric: "score", source: "https://z.ai/blog/glm-5.2", date: "2026-06-16" }] }, "zhipuai/glm-4.6v": { id: "zhipuai/glm-4.6v", name: "GLM-4.6V", description: "GLM vision model for visual reasoning, documents, and multimodal agents", family: "glm", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-12-08", last_updated: "2025-12-08", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 32768 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.6V" }] }, "zhipuai/glm-4.5-air": { id: "zhipuai/glm-4.5-air", name: "GLM-4.5-Air", description: "Lighter GLM-4.5 variant for fast coding assistance and cheaper agents", family: "glm-air", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 98304 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.5-Air" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 23.8, metric: "index", source: "https://openrouter.ai/z-ai/glm-4.5-air/benchmarks", date: "2026-05-30" }, { name: "SciCode", score: 30.6, metric: "percent correct", source: "https://openrouter.ai/z-ai/glm-4.5-air/benchmarks", date: "2026-05-30" }, { name: "Terminal-Bench Hard", score: 20.5, metric: "success rate", source: "https://openrouter.ai/z-ai/glm-4.5-air/benchmarks", date: "2026-05-30" }] }, "zhipuai/glm-5.1": { id: "zhipuai/glm-5.1", name: "GLM-5.1", description: "Strong GLM coding model for agentic engineering, terminals, and repository generation", family: "glm", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-07", last_updated: "2026-04-07", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 2e5, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-5.1" }], benchmarks: [{ name: "Artificial Analysis Coding Agent Index", score: 52.7, metric: "average pass@1", harness: "Claude Code", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Atlas Codebase QnA", score: 73.2, metric: "pass@1", harness: "Claude Code", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "SWE-Bench Pro", score: 19.8, metric: "pass@1", harness: "Claude Code", dataset: "hard-aa", source: "https://artificialanalysis.ai/agents/coding-agents" }, { name: "Terminal-Bench", score: 65.1, metric: "pass@1", harness: "Claude Code", version: "2.1", source: "https://artificialanalysis.ai/agents/coding-agents" }] }, "zhipuai/glm-4.7-flash": { id: "zhipuai/glm-4.7-flash", name: "GLM-4.7-Flash", description: "Budget GLM lane for fast coding help, routing, and everyday automation", family: "glm-flash", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2026-01-19", last_updated: "2026-01-19", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 2e5, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.7-Flash" }], benchmarks: [{ name: "SWE-Bench Verified", score: 59.2, metric: "resolved", source: "https://huggingface.co/zai-org/GLM-4.7-Flash" }] }, "zhipuai/glm-4.5": { id: "zhipuai/glm-4.5", name: "GLM-4.5", description: "Hybrid-reasoning GLM release that made the 4.5 line broadly useful", family: "glm", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-04", release_date: "2025-07-28", last_updated: "2025-07-28", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 98304 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/zai-org/GLM-4.5" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 26.3, metric: "index", source: "https://openrouter.ai/z-ai/glm-4.5/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 34.8, metric: "percent correct", source: "https://openrouter.ai/z-ai/glm-4.5/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 22, metric: "success rate", source: "https://openrouter.ai/z-ai/glm-4.5/benchmarks", date: "2026-03-11" }] }, "ibm/granite-4-h-micro": { id: "ibm/granite-4-h-micro", name: "Granite-4.0-H-Micro", description: "Compact open-weight hybrid Granite model for lightweight enterprise chat and tool calling", family: "granite", attachment: false, reasoning: false, tool_call: true, structured_output: true, temperature: true, release_date: "2025-10-02", last_updated: "2025-10-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/ibm-granite/granite-4.0-h-micro" }] }, "ibm/granite-4-h-small": { id: "ibm/granite-4-h-small", name: "Granite-4.0-H-Small", description: "Open-weight hybrid model for enterprise chat, coding, retrieval-augmented generation, and tool-calling workloads", family: "granite", attachment: false, reasoning: false, tool_call: true, structured_output: true, temperature: true, release_date: "2025-10-02", last_updated: "2025-10-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/ibm-granite/granite-4.0-h-small" }] }, "thinkingmachines/inkling-small": { id: "thinkingmachines/inkling-small", name: "Inkling Small", description: "Multimodal MoE reasoning model (276B total, 12B active) for text, image, and audio", family: "ling", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-07-30", last_updated: "2026-07-30", modalities: { input: ["text", "image", "audio"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 1048576 }, license: "Apache-2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/thinkingmachines/Inkling-Small" }] }, "thinkingmachines/inkling": { id: "thinkingmachines/inkling", name: "Inkling", description: "Multimodal MoE reasoning model (975B total, 41B active) for text, image, and audio", family: "ling", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-07-15", last_updated: "2026-07-15", modalities: { input: ["text", "image", "audio"], output: ["text"] }, open_weights: true, limit: { context: 1048576, output: 1048576 }, license: "Apache-2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/thinkingmachines/Inkling" }] }, "mistral/magistral-small-2506": { id: "mistral/magistral-small-2506", name: "Magistral Small", description: "Open Mistral reasoning model for transparent step-by-step problem solving", family: "magistral", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-06-10", last_updated: "2025-06-10", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "Apache 2.0", weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Magistral-Small-2506" }] }, "mistral/mistral-small-latest": { id: "mistral/mistral-small-latest", name: "Mistral Small (latest)", description: "Efficient Mistral model for fast chat, extraction, and production assistants", family: "mistral-small", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-06", release_date: "2026-03-16", last_updated: "2026-03-16", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603" }] }, "mistral/devstral-medium-2507": { id: "mistral/devstral-medium-2507", name: "Devstral Medium", description: "Mistral coding agent model for repository tasks and software engineering workflows", family: "devstral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-07-10", last_updated: "2025-07-10", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 128e3 }, benchmarks: [{ name: "SWE-Bench Verified", score: 61.6, metric: "resolved", source: "https://mistral.ai/news/devstral-2507", date: "2025-07-10" }] }, "mistral/mistral-small-3-1-24b-instruct-2503": { id: "mistral/mistral-small-3-1-24b-instruct-2503", name: "Mistral Small 3.1 24B", description: "Efficient multimodal model for instruction following, coding, reasoning, and function calling", family: "mistral-small", attachment: true, reasoning: false, tool_call: true, structured_output: true, temperature: true, knowledge: "2024-06", release_date: "2025-03-17", last_updated: "2025-03-17", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Small-3.1-24B-Instruct-2503" }] }, "mistral/mistral-large-2411": { id: "mistral/mistral-large-2411", name: "Mistral Large 2.1", description: "Flagship Mistral model for advanced reasoning, coding, and multilingual work", family: "mistral-large", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-11", release_date: "2024-11-18", last_updated: "2024-11-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Large-Instruct-2411" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 13.8, metric: "index", source: "https://openrouter.ai/mistralai/mistral-large-2407/benchmarks", date: "2026-03-11" }, { name: "SciCode", score: 29.2, metric: "percent correct", source: "https://openrouter.ai/mistralai/mistral-large-2407/benchmarks", date: "2026-03-11" }, { name: "Terminal-Bench Hard", score: 6.1, metric: "success rate", source: "https://openrouter.ai/mistralai/mistral-large-2407/benchmarks", date: "2026-03-11" }] }, "mistral/mistral-nemo": { id: "mistral/mistral-nemo", name: "Mistral Nemo", description: "Efficient Mistral-NVIDIA open model for multilingual chat and local deployment", family: "mistral-nemo", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-07", release_date: "2024-07-01", last_updated: "2024-07-01", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Nemo-Instruct-2407" }] }, "mistral/mistral-large-latest": { id: "mistral/mistral-large-latest", name: "Mistral Large (latest)", description: "Flagship Mistral model for advanced reasoning, coding, and multilingual work", family: "mistral-large", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-11", release_date: "2024-11-01", last_updated: "2025-12-02", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512" }] }, "mistral/ministral-8b-instruct-2410": { id: "mistral/ministral-8b-instruct-2410", name: "Ministral 8B Instruct", description: "Efficient open Mistral edge model for on-device chat and function calling", family: "ministral", attachment: false, reasoning: false, tool_call: true, temperature: true, release_date: "2024-10-16", last_updated: "2024-10-16", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 131072, output: 8192 }, license: "Mistral Research License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Ministral-8B-Instruct-2410" }] }, "mistral/devstral-small-2507": { id: "mistral/devstral-small-2507", name: "Devstral Small", description: "Mistral coding agent model for repository tasks and software engineering workflows", family: "devstral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-07-10", last_updated: "2025-07-10", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Devstral-Small-2507" }], benchmarks: [{ name: "SWE-Bench Verified", score: 53.6, metric: "resolved", source: "https://mistral.ai/news/devstral-2507", date: "2025-07-10" }] }, "mistral/devstral-2512": { id: "mistral/devstral-2512", name: "Devstral 2", description: "Mistral's coding-agent model for repository work, terminal tasks, and software fixes", family: "devstral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-12", release_date: "2025-12-09", last_updated: "2025-12-09", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 23.7, metric: "index", source: "https://openrouter.ai/mistralai/devstral-2512/benchmarks", date: "2026-05-31" }, { name: "SciCode", score: 33.1, metric: "percent correct", source: "https://openrouter.ai/mistralai/devstral-2512/benchmarks", date: "2026-05-31" }, { name: "Terminal-Bench Hard", score: 18.9, metric: "success rate", source: "https://openrouter.ai/mistralai/devstral-2512/benchmarks", date: "2026-05-31" }] }, "mistral/mistral-small-2603": { id: "mistral/mistral-small-2603", name: "Mistral Small 4", description: "Fast Mistral production model for chat, extraction, and cost-sensitive agents", family: "mistral-small", attachment: true, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-06", release_date: "2026-03-16", last_updated: "2026-03-16", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 256e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Small-4-119B-2603" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 24.3, metric: "index", source: "https://openrouter.ai/mistralai/mistral-small-2603/benchmarks", date: "2026-06-01" }, { name: "SciCode", score: 38, metric: "percent correct", source: "https://openrouter.ai/mistralai/mistral-small-2603/benchmarks", date: "2026-06-01" }, { name: "Terminal-Bench Hard", score: 17.4, metric: "success rate", source: "https://openrouter.ai/mistralai/mistral-small-2603/benchmarks", date: "2026-06-01" }] }, "mistral/mistral-small-2506": { id: "mistral/mistral-small-2506", name: "Mistral Small 3.2", description: "Efficient Mistral model for fast chat, extraction, and production assistants", family: "mistral-small", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-03", release_date: "2025-06-20", last_updated: "2025-06-20", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 16384 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Small-3.2-24B-Instruct-2506" }] }, "mistral/pixtral-large-latest": { id: "mistral/pixtral-large-latest", name: "Pixtral Large (latest)", description: "Mistral's larger vision model for document-heavy image understanding and chat", family: "pixtral", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-11", release_date: "2024-11-01", last_updated: "2024-11-04", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Pixtral-Large-Instruct-2411" }] }, "mistral/mistral-large-2512": { id: "mistral/mistral-large-2512", name: "Mistral Large 3", description: "Mistral's largest general model for enterprise agents, coding, and multilingual reasoning", family: "mistral-large", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-11", release_date: "2024-11-01", last_updated: "2025-12-02", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Large-3-675B-Instruct-2512" }], benchmarks: [{ name: "Artificial Analysis Coding Index", score: 22.7, metric: "index", source: "https://openrouter.ai/mistralai/mistral-large-2512/benchmarks", date: "2026-06-04" }, { name: "SciCode", score: 36.2, metric: "percent correct", source: "https://openrouter.ai/mistralai/mistral-large-2512/benchmarks", date: "2026-06-04" }, { name: "Terminal-Bench Hard", score: 15.9, metric: "success rate", source: "https://openrouter.ai/mistralai/mistral-large-2512/benchmarks", date: "2026-06-04" }] }, "mistral/mistral-medium-2604": { id: "mistral/mistral-medium-2604", name: "Mistral Medium 3.5", description: "Balanced Mistral model for enterprise assistants, multilingual work, and tools", family: "mistral-medium", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-29", last_updated: "2026-04-29", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 77.6, metric: "resolved", source: "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B" }, { name: "\u03C4\xB3-Telecom", score: 91.4, metric: "accuracy", variant: "public preview", source: "https://mistral.ai/news/vibe-remote-agents-mistral-medium-3-5/", date: "2026-05-22" }] }, "mistral/mistral-medium-2505": { id: "mistral/mistral-medium-2505", name: "Mistral Medium 3", description: "Mistral model for multilingual chat, reasoning, and tool-assisted workflows", family: "mistral-medium", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-05", release_date: "2025-05-07", last_updated: "2025-05-07", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 131072 }, benchmarks: [{ name: "Artificial Analysis Coding Index", score: 13.6, metric: "index", source: "https://openrouter.ai/mistralai/mistral-medium-3/benchmarks", date: "2026-05-30" }, { name: "SciCode", score: 33.1, metric: "percent correct", source: "https://openrouter.ai/mistralai/mistral-medium-3/benchmarks", date: "2026-05-30" }, { name: "Terminal-Bench Hard", score: 3.8, metric: "success rate", source: "https://openrouter.ai/mistralai/mistral-medium-3/benchmarks", date: "2026-05-30" }] }, "mistral/devstral-medium-latest": { id: "mistral/devstral-medium-latest", name: "Devstral 2 (latest)", description: "Mistral coding agent model for repository tasks and software engineering workflows", family: "devstral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2025-12", release_date: "2025-12-02", last_updated: "2025-12-02", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Devstral-2-123B-Instruct-2512" }] }, "mistral/pixtral-12b": { id: "mistral/pixtral-12b", name: "Pixtral 12B", description: "Mistral vision-language model for image understanding and multimodal chat", family: "pixtral", attachment: true, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-09", release_date: "2024-09-01", last_updated: "2024-09-01", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 128e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Pixtral-12B-2409" }] }, "mistral/voxtral-small-latest": { id: "mistral/voxtral-small-latest", name: "Voxtral Small (latest)", description: "Instruct model with native audio input for speech understanding and tool use", family: "voxtral", attachment: true, reasoning: false, tool_call: true, temperature: true, release_date: "2025-07-15", last_updated: "2025-07-15", modalities: { input: ["text", "audio"], output: ["text"] }, open_weights: true, limit: { context: 32e3, output: 32e3 } }, "mistral/codestral-22b-v0.1": { id: "mistral/codestral-22b-v0.1", name: "Codestral-22B-v0.1", description: "Open Mistral code model for fill-in-the-middle and 80+ programming languages", family: "codestral", attachment: false, reasoning: false, tool_call: false, temperature: true, release_date: "2024-05-29", last_updated: "2024-05-29", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 32768, output: 8192 }, license: "Mistral AI Non-Production License", weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Codestral-22B-v0.1" }] }, "mistral/codestral-latest": { id: "mistral/codestral-latest", name: "Codestral (latest)", description: "Mistral code model for completions, refactors, and developer IDE workflows", family: "codestral", attachment: false, reasoning: false, tool_call: true, temperature: true, knowledge: "2024-10", release_date: "2024-05-29", last_updated: "2025-01-04", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 256e3, output: 4096 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Codestral-22B-v0.1" }], benchmarks: [{ name: "Aider Polyglot", score: 11.1, metric: "percent correct", source: "https://aider.chat/docs/leaderboards/", date: "2025-01-13" }] }, "mistral/magistral-medium-latest": { id: "mistral/magistral-medium-latest", name: "Magistral Medium (latest)", description: "Mistral reasoning model for transparent analysis, math, and complex decisions", family: "magistral-medium", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-06", release_date: "2025-03-17", last_updated: "2025-03-20", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 128e3, output: 16384 } }, "mistral/mistral-medium-latest": { id: "mistral/mistral-medium-latest", name: "Mistral Medium (latest)", description: "Balanced Mistral model for enterprise assistants, multilingual work, and tools", family: "mistral-medium", attachment: true, reasoning: true, tool_call: true, structured_output: true, temperature: true, release_date: "2026-04-29", last_updated: "2026-04-29", modalities: { input: ["text", "image"], output: ["text"] }, open_weights: true, limit: { context: 262144, output: 262144 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B" }], benchmarks: [{ name: "SWE-Bench Verified", score: 77.6, metric: "resolved", source: "https://huggingface.co/mistralai/Mistral-Medium-3.5-128B" }] }, "upstage/solar-pro2": { id: "upstage/solar-pro2", name: "Solar Pro 2", description: "Flagship model for demanding analysis, coding, and production agent workflows", family: "solar-pro", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03", release_date: "2025-05-20", last_updated: "2025-05-20", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 65536, output: 8192 } }, "upstage/solar-pro4": { id: "upstage/solar-pro4", name: "Solar Pro 4", description: "Upstage's flagship model, specialized for agentic use", family: "solar-pro", attachment: false, reasoning: true, tool_call: true, structured_output: true, temperature: true, knowledge: "2026-02", release_date: "2026-08-06", last_updated: "2026-08-06", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 524288, output: 131072 } }, "upstage/solar-pro3": { id: "upstage/solar-pro3", name: "Solar Pro 3", description: "Flagship model for demanding analysis, coding, and production agent workflows", family: "solar-pro", attachment: false, reasoning: true, tool_call: true, temperature: true, knowledge: "2025-03", release_date: "2026-01", last_updated: "2026-01", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 131072, output: 8192 } }, "minimax/MiniMax-M2.7": { id: "minimax/MiniMax-M2.7", name: "MiniMax-M2.7", description: "Open MiniMax flagship for coding agents, office automation, and complex environments", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-03-18", last_updated: "2026-03-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.7" }], benchmarks: [{ name: "SWE-Bench Verified", score: 79.9, metric: "resolved", harness: "Claude Code", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "SWE-Bench Pro", score: 56.2, metric: "resolve rate", harness: "Claude Code", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "Terminal-Bench", score: 51.1, metric: "success rate", version: "2.1", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }] }, "minimax/MiniMax-M2.7-highspeed": { id: "minimax/MiniMax-M2.7-highspeed", name: "MiniMax-M2.7-highspeed", description: "Low-latency M2.7 variant for interactive coding plans and agent loops", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-03-18", last_updated: "2026-03-18", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.7" }] }, "minimax/MiniMax-M2.1": { id: "minimax/MiniMax-M2.1", name: "MiniMax-M2.1", description: "Earlier MiniMax agent model for practical coding and productivity tasks", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-12-23", last_updated: "2025-12-23", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.1" }], benchmarks: [{ name: "SWE-Bench Verified", score: 74, metric: "resolved", source: "https://huggingface.co/MiniMaxAI/MiniMax-M2.1" }, { name: "SWE-Bench Pro", score: 36.81, metric: "resolve rate", dataset: "public", source: "https://labs.scale.com/leaderboard/swe_bench_pro_public" }] }, "minimax/MiniMax-M2": { id: "minimax/MiniMax-M2", name: "MiniMax-M2", description: "Efficient open MiniMax model built for coding agents and tool-heavy workflows", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2025-10-27", last_updated: "2025-10-27", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 196608, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2" }], benchmarks: [{ name: "SWE-Bench Verified", score: 69.4, metric: "resolved", source: "https://huggingface.co/MiniMaxAI/MiniMax-M2" }] }, "minimax/MiniMax-M2.5-highspeed": { id: "minimax/MiniMax-M2.5-highspeed", name: "MiniMax-M2.5-highspeed", description: "High-speed MiniMax model for low-latency coding and agent workflows", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-13", last_updated: "2026-02-13", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.5" }] }, "minimax/MiniMax-M3": { id: "minimax/MiniMax-M3", name: "MiniMax-M3", description: "MiniMax multimodal model for long-context coding, perception, and agent planning", family: "minimax", attachment: true, reasoning: true, tool_call: true, temperature: true, release_date: "2026-06-01", last_updated: "2026-06-01", modalities: { input: ["text", "image", "video"], output: ["text"] }, open_weights: true, limit: { context: 512e3, output: 128e3 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M3" }], benchmarks: [{ name: "SWE-Bench Verified", score: 80.5, metric: "resolved", harness: "Claude Code", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "SWE-Bench Pro", score: 59, metric: "resolve rate", harness: "Claude Code", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "Terminal-Bench", score: 66, metric: "success rate", version: "2.1", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "BrowseComp", score: 83.52, metric: "accuracy", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "MCP Atlas", score: 74.2, metric: "success rate", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }, { name: "OSWorld-Verified", score: 70.06, metric: "success rate", source: "https://www.minimax.io/blog/minimax-m3", date: "2026-06-01" }] }, "minimax/MiniMax-M2-Her": { id: "minimax/MiniMax-M2-Her", name: "MiniMax-M2 Her", description: "MiniMax M2 variant tuned for conversational and character-driven agent interactions", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-01-23", last_updated: "2026-01-23", modalities: { input: ["text"], output: ["text"] }, open_weights: false, limit: { context: 2e5, output: 131e3 } }, "minimax/MiniMax-M2.5": { id: "minimax/MiniMax-M2.5", name: "MiniMax-M2.5", description: "Prior MiniMax coding model for agent workflows, office edits, and automation", family: "minimax", attachment: false, reasoning: true, tool_call: true, temperature: true, release_date: "2026-02-12", last_updated: "2026-02-12", modalities: { input: ["text"], output: ["text"] }, open_weights: true, limit: { context: 204800, output: 131072 }, weights: [{ label: "Hugging Face", url: "https://huggingface.co/MiniMaxAI/MiniMax-M2.5" }], benchmarks: [{ name: "SWE-Bench Verified", score: 75.8, metric: "resolved", source: "https://www.swebench.com/" }, { name: "SWE-Atlas Codebase QnA", score: 10.3, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-qna" }, { name: "SWE-Atlas Refactoring", score: 19.52, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-refactoring" }, { name: "SWE-Atlas Test Writing", score: 18.6, metric: "score", harness: "Mini-SWE-Agent", source: "https://labs.scale.com/leaderboard/sweatlas-tw" }] } } };
|
|
48228
48351
|
|
|
@@ -48283,7 +48406,7 @@ function diskCacheFresh() {
|
|
|
48283
48406
|
}
|
|
48284
48407
|
}
|
|
48285
48408
|
async function fetchFresh() {
|
|
48286
|
-
const dispatcher = proxyDispatcher(process.env.https_proxy || process.env.HTTPS_PROXY || process.env.http_proxy || process.env.HTTP_PROXY);
|
|
48409
|
+
const dispatcher = proxyDispatcher(process.env.https_proxy || process.env.HTTPS_PROXY || process.env.http_proxy || process.env.HTTP_PROXY, 15e3);
|
|
48287
48410
|
const attempts = dispatcher ? [{ opts: { dispatcher }, label: "via proxy" }, { opts: {}, label: "direct" }] : [{ opts: {}, label: "direct" }];
|
|
48288
48411
|
for (const { opts, label } of attempts) {
|
|
48289
48412
|
try {
|
|
@@ -49950,8 +50073,17 @@ var SessionStore = class {
|
|
|
49950
50073
|
async loadAll() {
|
|
49951
50074
|
const out = /* @__PURE__ */ new Map();
|
|
49952
50075
|
if (!this.enabled) return out;
|
|
50076
|
+
let clamped = 0;
|
|
49953
50077
|
for (const [id, envelope] of await this.store.loadAll()) {
|
|
49954
|
-
|
|
50078
|
+
const session = buildSession(envelope.payload);
|
|
50079
|
+
if (hasNegativePersistedTokens(envelope.payload)) {
|
|
50080
|
+
await this.store.writeNow(id, () => buildRecord(session));
|
|
50081
|
+
clamped++;
|
|
50082
|
+
}
|
|
50083
|
+
out.set(id, session);
|
|
50084
|
+
}
|
|
50085
|
+
if (clamped > 0) {
|
|
50086
|
+
log("info", `[persist] one-time migration (#408): clamped negative lastInputTokens/contextTokens in ${clamped} session(s) to 0`);
|
|
49955
50087
|
}
|
|
49956
50088
|
return out;
|
|
49957
50089
|
}
|
|
@@ -50073,7 +50205,14 @@ var SessionStore = class {
|
|
|
50073
50205
|
meta?.protocol ? this.store.loadSync(id, relPathFor(id)) : null
|
|
50074
50206
|
];
|
|
50075
50207
|
for (const envelope of envelopes) {
|
|
50076
|
-
if (envelope)
|
|
50208
|
+
if (envelope) {
|
|
50209
|
+
const session = buildSession(envelope.payload);
|
|
50210
|
+
if (hasNegativePersistedTokens(envelope.payload)) {
|
|
50211
|
+
this.scheduleSave(session);
|
|
50212
|
+
log("info", `[persist] clamped negative token stats on reload for ${id} (#408)`);
|
|
50213
|
+
}
|
|
50214
|
+
return session;
|
|
50215
|
+
}
|
|
50077
50216
|
}
|
|
50078
50217
|
return null;
|
|
50079
50218
|
}
|
|
@@ -50115,13 +50254,15 @@ var SessionStore = class {
|
|
|
50115
50254
|
}
|
|
50116
50255
|
};
|
|
50117
50256
|
function buildRecord(session) {
|
|
50257
|
+
const snapshot = boundedFoldedSnapshot(session);
|
|
50118
50258
|
return {
|
|
50119
50259
|
version: PERSIST_VERSION,
|
|
50120
50260
|
savedAt: Date.now(),
|
|
50121
50261
|
id: session.id,
|
|
50122
50262
|
meta: { ...session.meta },
|
|
50123
50263
|
stats: { ...session.stats },
|
|
50124
|
-
messages:
|
|
50264
|
+
messages: snapshot,
|
|
50265
|
+
messagesFolded: snapshot ? true : void 0,
|
|
50125
50266
|
metadata: { ...session.metadata },
|
|
50126
50267
|
state: session.state,
|
|
50127
50268
|
blockContents: Object.fromEntries(session.blockContents),
|
|
@@ -50161,10 +50302,14 @@ function buildSession(parsed) {
|
|
|
50161
50302
|
cachedTokens: stats.cachedTokens ?? parsed.cachedTokens ?? 0,
|
|
50162
50303
|
outputTokens: stats.outputTokens ?? parsed.outputTokens ?? 0,
|
|
50163
50304
|
cacheSamples: stats.cacheSamples ?? parsed.cacheSamples ?? 0,
|
|
50164
|
-
|
|
50305
|
+
// #408: clamp at restore — pre-clamp versions persisted negative
|
|
50306
|
+
// values (lastInputTokens = total − credit before the Math.max
|
|
50307
|
+
// guard existed) which would otherwise revive after upgrade and
|
|
50308
|
+
// feed the /acp panel + web stats as negative percentages.
|
|
50309
|
+
lastInputTokens: Math.max(0, stats.lastInputTokens ?? parsed.lastInputTokens ?? 0),
|
|
50165
50310
|
// In-memory only — a fresh process has no pending compress fold.
|
|
50166
50311
|
compressCreditTokens: 0,
|
|
50167
|
-
contextTokens: stats.contextTokens ?? parsed.contextTokens ?? 0
|
|
50312
|
+
contextTokens: Math.max(0, stats.contextTokens ?? parsed.contextTokens ?? 0)
|
|
50168
50313
|
},
|
|
50169
50314
|
metadata: parsed.metadata ?? {},
|
|
50170
50315
|
state: mergeState(parsed.state),
|
|
@@ -50178,6 +50323,7 @@ function buildSession(parsed) {
|
|
|
50178
50323
|
restored: true,
|
|
50179
50324
|
blockContents,
|
|
50180
50325
|
lastMessages: Array.isArray(parsed.messages) ? parsed.messages : void 0,
|
|
50326
|
+
lastMessagesFolded: parsed.messagesFolded === true,
|
|
50181
50327
|
inFlight: 0,
|
|
50182
50328
|
persisted: true
|
|
50183
50329
|
};
|
|
@@ -50187,6 +50333,12 @@ function isValidRecord(parsed) {
|
|
|
50187
50333
|
const r = parsed;
|
|
50188
50334
|
return typeof r.id === "string" && typeof r.state === "object" && r.state !== null && Array.isArray(r.state.blocks);
|
|
50189
50335
|
}
|
|
50336
|
+
function hasNegativePersistedTokens(parsed) {
|
|
50337
|
+
const stats = parsed.stats ?? {};
|
|
50338
|
+
const last = stats.lastInputTokens ?? parsed.lastInputTokens;
|
|
50339
|
+
const ctx = stats.contextTokens ?? parsed.contextTokens;
|
|
50340
|
+
return typeof last === "number" && last < 0 || typeof ctx === "number" && ctx < 0;
|
|
50341
|
+
}
|
|
50190
50342
|
function defaultDir() {
|
|
50191
50343
|
return sessionsDir();
|
|
50192
50344
|
}
|
|
@@ -50203,6 +50355,36 @@ function persistEnabled() {
|
|
|
50203
50355
|
if (env === "0" || env === "false") return false;
|
|
50204
50356
|
return true;
|
|
50205
50357
|
}
|
|
50358
|
+
function persistTailTokens() {
|
|
50359
|
+
const env = process.env.BILI_PERSIST_TAIL_TOKENS;
|
|
50360
|
+
if (env) {
|
|
50361
|
+
const n = Number.parseInt(env, 10);
|
|
50362
|
+
if (Number.isFinite(n) && n >= 0) return n;
|
|
50363
|
+
}
|
|
50364
|
+
return 16384;
|
|
50365
|
+
}
|
|
50366
|
+
function boundedFoldedSnapshot(session) {
|
|
50367
|
+
const msgs = session.lastMessages;
|
|
50368
|
+
if (!msgs || msgs.length === 0) return void 0;
|
|
50369
|
+
const budget = persistTailTokens();
|
|
50370
|
+
if (budget === 0) return void 0;
|
|
50371
|
+
let view = prune(msgs, session.state);
|
|
50372
|
+
let total = 0;
|
|
50373
|
+
for (const m2 of view) total += defaultCountTokens(m2.text ?? "");
|
|
50374
|
+
if (total > budget) {
|
|
50375
|
+
let acc = 0;
|
|
50376
|
+
let start = 0;
|
|
50377
|
+
for (let i = view.length - 1; i >= 0; i--) {
|
|
50378
|
+
acc += defaultCountTokens(view[i].text ?? "");
|
|
50379
|
+
if (acc > budget) {
|
|
50380
|
+
start = Math.min(i + 1, view.length - 1);
|
|
50381
|
+
break;
|
|
50382
|
+
}
|
|
50383
|
+
}
|
|
50384
|
+
if (start > 0) view = view.slice(start);
|
|
50385
|
+
}
|
|
50386
|
+
return view;
|
|
50387
|
+
}
|
|
50206
50388
|
function epermAlertThreshold() {
|
|
50207
50389
|
const env = process.env.BILI_PERSIST_EPERM_ALERT_THRESHOLD;
|
|
50208
50390
|
if (env) {
|
|
@@ -50322,7 +50504,10 @@ function peekSession(id) {
|
|
|
50322
50504
|
return sessions.get(id);
|
|
50323
50505
|
}
|
|
50324
50506
|
function snapshotMessages(session, messages) {
|
|
50325
|
-
if (messages.length > 0)
|
|
50507
|
+
if (messages.length > 0) {
|
|
50508
|
+
session.lastMessages = messages;
|
|
50509
|
+
session.lastMessagesFolded = false;
|
|
50510
|
+
}
|
|
50326
50511
|
}
|
|
50327
50512
|
function markDirty(session) {
|
|
50328
50513
|
getStore().scheduleSave(session);
|
|
@@ -50335,6 +50520,8 @@ function resetSessionCompression(session) {
|
|
|
50335
50520
|
session.blockContents.clear();
|
|
50336
50521
|
session.stats.lastInputTokens = 0;
|
|
50337
50522
|
session.stats.contextTokens = 0;
|
|
50523
|
+
session.hostCreditTokens = 0;
|
|
50524
|
+
session.hostContextTokens = 0;
|
|
50338
50525
|
session.metadata.nativeCompactionAt = Date.now();
|
|
50339
50526
|
markDirty(session);
|
|
50340
50527
|
}
|
|
@@ -50448,6 +50635,50 @@ function withStagedCompressGuidance(text) {
|
|
|
50448
50635
|
return text + STAGED_COMPRESS_GUIDANCE;
|
|
50449
50636
|
}
|
|
50450
50637
|
|
|
50638
|
+
// src/acp-status.ts
|
|
50639
|
+
function handleAcpStatus(args, ctx) {
|
|
50640
|
+
const scope = typeof args.scope === "string" ? args.scope : void 0;
|
|
50641
|
+
const view = typeof args.view === "string" ? args.view : void 0;
|
|
50642
|
+
const tool = typeof args.tool === "string" ? args.tool : void 0;
|
|
50643
|
+
const sort = typeof args.sort === "string" ? args.sort : void 0;
|
|
50644
|
+
const limit = typeof args.limit === "number" ? args.limit : void 0;
|
|
50645
|
+
const base = buildStatusReport(ctx.session.state, ctx.messages, defaultCountTokens, { scope, view, tool, sort, limit });
|
|
50646
|
+
if (scope) return base;
|
|
50647
|
+
const extra = [];
|
|
50648
|
+
try {
|
|
50649
|
+
const turn = ctx.core.processTurn({
|
|
50650
|
+
messages: ctx.messages,
|
|
50651
|
+
state: ctx.session.state,
|
|
50652
|
+
config: ctx.config,
|
|
50653
|
+
tokenCount: ctx.session.stats.lastInputTokens,
|
|
50654
|
+
renderTags: "none"
|
|
50655
|
+
});
|
|
50656
|
+
const nudge = turn.nudge;
|
|
50657
|
+
if (nudge) {
|
|
50658
|
+
extra.push("");
|
|
50659
|
+
extra.push(nudge.shouldInject ? `Nudge: ACTIVE \u2014 ${nudge.reason}` : `Nudge: idle \u2014 ${nudge.reason}`);
|
|
50660
|
+
const ranges = viableRanges(nudge.compressibleRanges);
|
|
50661
|
+
const protectedRanges = nudge.protectedRanges ?? [];
|
|
50662
|
+
if (ranges.length > 0 || protectedRanges.length > 0) {
|
|
50663
|
+
extra.push("");
|
|
50664
|
+
extra.push(formatRanges(ranges, protectedRanges));
|
|
50665
|
+
}
|
|
50666
|
+
}
|
|
50667
|
+
} catch {
|
|
50668
|
+
}
|
|
50669
|
+
const archive = preCompactionArchiveOf(ctx.session);
|
|
50670
|
+
const archivedIds = Object.keys(archive);
|
|
50671
|
+
if (archivedIds.length > 0) {
|
|
50672
|
+
extra.push("");
|
|
50673
|
+
extra.push(`PRE-COMPACTION ARCHIVE \u2014 ${archivedIds.length} block(s): content was replaced by the client's native compaction summary, so it is no longer in the session history and decompress is unavailable.`);
|
|
50674
|
+
for (const id of archivedIds) {
|
|
50675
|
+
extra.push(` ${id} \u2014 ${archive[id].reason}`);
|
|
50676
|
+
}
|
|
50677
|
+
}
|
|
50678
|
+
return extra.length > 0 ? `${base}
|
|
50679
|
+
${extra.join("\n")}` : base;
|
|
50680
|
+
}
|
|
50681
|
+
|
|
50451
50682
|
// src/decompress-shared.ts
|
|
50452
50683
|
import { mkdirSync as mkdirSync4, unlinkSync as unlinkSync2, writeFileSync as writeFileSync3 } from "fs";
|
|
50453
50684
|
import { dirname as dirname2, join as join2 } from "path";
|
|
@@ -50769,7 +51000,7 @@ function executeAnthropicProxyTool(toolName, args, ctx) {
|
|
|
50769
51000
|
${lines.join("\n\n")}`;
|
|
50770
51001
|
}
|
|
50771
51002
|
if (toolName === "acp_status") {
|
|
50772
|
-
return
|
|
51003
|
+
return handleAcpStatus(args, ctx);
|
|
50773
51004
|
}
|
|
50774
51005
|
return `[Unknown proxy tool: ${toolName}]`;
|
|
50775
51006
|
}
|
|
@@ -50876,6 +51107,7 @@ var CHUNK_FRACTION = 0.6;
|
|
|
50876
51107
|
var MIN_CHUNK_TOKENS = 2e3;
|
|
50877
51108
|
var MIN_SUMMARY_CHARS = 50;
|
|
50878
51109
|
var MAX_SUMMARY_OUTPUT_TOKENS = 8192;
|
|
51110
|
+
var MAX_SUMMARY_CALLS_PER_PREFLIGHT = 8;
|
|
50879
51111
|
function refMaps(messages, state) {
|
|
50880
51112
|
const refToIdx = /* @__PURE__ */ new Map();
|
|
50881
51113
|
const idxToRef = /* @__PURE__ */ new Map();
|
|
@@ -50896,6 +51128,25 @@ function estimateCoreMessages(messages) {
|
|
|
50896
51128
|
for (const m2 of messages) tokens += defaultCountTokens(m2.text ?? "");
|
|
50897
51129
|
return tokens;
|
|
50898
51130
|
}
|
|
51131
|
+
var NON_TEXT_BODY_KEYS = /* @__PURE__ */ new Set(["data", "url", "b64_json", "file_data"]);
|
|
51132
|
+
function estimateRawBodyTokens(parsed) {
|
|
51133
|
+
let tokens = 0;
|
|
51134
|
+
const walk = (value, key) => {
|
|
51135
|
+
if (typeof value === "string") {
|
|
51136
|
+
if (!key || !NON_TEXT_BODY_KEYS.has(key)) tokens += defaultCountTokens(value);
|
|
51137
|
+
return;
|
|
51138
|
+
}
|
|
51139
|
+
if (Array.isArray(value)) {
|
|
51140
|
+
for (const item of value) walk(item, key);
|
|
51141
|
+
return;
|
|
51142
|
+
}
|
|
51143
|
+
if (value && typeof value === "object") {
|
|
51144
|
+
for (const [k2, v2] of Object.entries(value)) walk(v2, k2);
|
|
51145
|
+
}
|
|
51146
|
+
};
|
|
51147
|
+
walk(parsed);
|
|
51148
|
+
return tokens;
|
|
51149
|
+
}
|
|
50899
51150
|
function rangeChars2(messages, startIdx, endIdx) {
|
|
50900
51151
|
let chars = 0;
|
|
50901
51152
|
for (let i = startIdx; i <= endIdx && i < messages.length; i++) {
|
|
@@ -51003,15 +51254,28 @@ TASK: The conversation segment below (messages ${startRef}\u2013${endRef}) must
|
|
|
51003
51254
|
}
|
|
51004
51255
|
}
|
|
51005
51256
|
var ABORTED_FAILURE = { kind: "aborted", detail: "the client disconnected during preflight compression" };
|
|
51257
|
+
function relaxedConfig(config) {
|
|
51258
|
+
return { ...config, preserveRecentMessages: 0, preserveRecentTokens: 0 };
|
|
51259
|
+
}
|
|
51260
|
+
function noEmergencyTruncate(config) {
|
|
51261
|
+
return { ...config, modelContextLimit: config.modelContextLimit * 100 };
|
|
51262
|
+
}
|
|
51006
51263
|
async function preflightCompress(deps, messages) {
|
|
51007
51264
|
const limit = deps.config.modelContextLimit;
|
|
51008
|
-
const result = { compressedRanges: 0, savedTokens: 0, payloadEstimate: estimateCoreMessages(messages) + (deps.imageFloor ?? 0) };
|
|
51265
|
+
const result = { compressedRanges: 0, savedTokens: 0, payloadEstimate: estimateCoreMessages(messages) + (deps.imageFloor ?? 0) + (deps.wireOverhead ?? 0) };
|
|
51009
51266
|
if (limit <= 0) return result;
|
|
51010
51267
|
const budget = Math.max(MIN_CHUNK_TOKENS, Math.floor(limit * CHUNK_FRACTION));
|
|
51011
51268
|
const minChars = deps.config.compress.minCompressRange;
|
|
51012
51269
|
let currentTokens = deps.session.stats.lastInputTokens;
|
|
51013
51270
|
let startTokens = -1;
|
|
51014
51271
|
let failure;
|
|
51272
|
+
let activeConfig = deps.config;
|
|
51273
|
+
let relaxed = false;
|
|
51274
|
+
const relaxedExhaustedDetail = `the payload still exceeds the window after folding everything compressible, including the soft-protected recent zone (last ${deps.config.preserveRecentMessages} messages + most recent user message), which was relaxed under overflow; hard protectedTools remain excluded. Raise the model context window or restart the session to recover.`;
|
|
51275
|
+
const skipSet = /* @__PURE__ */ new Set();
|
|
51276
|
+
let summaryCalls = 0;
|
|
51277
|
+
let budgetHit = false;
|
|
51278
|
+
let rangesTried = 0;
|
|
51015
51279
|
for (let round = 0; round < MAX_PREFLIGHT_ROUNDS; round++) {
|
|
51016
51280
|
if (deps.signal?.aborted) {
|
|
51017
51281
|
failure = ABORTED_FAILURE;
|
|
@@ -51020,91 +51284,127 @@ async function preflightCompress(deps, messages) {
|
|
|
51020
51284
|
const turn = deps.core.processTurn({
|
|
51021
51285
|
messages,
|
|
51022
51286
|
state: deps.session.state,
|
|
51023
|
-
config:
|
|
51287
|
+
config: noEmergencyTruncate(activeConfig),
|
|
51024
51288
|
tokenCount: currentTokens,
|
|
51025
51289
|
renderTags: "text-only"
|
|
51026
51290
|
});
|
|
51027
51291
|
deps.session.state = turn.state;
|
|
51028
|
-
currentTokens = Math.max(deps.session.stats.lastInputTokens, estimateCoreMessages(turn.messages) + (deps.imageFloor ?? 0));
|
|
51029
|
-
result.payloadEstimate = estimateCoreMessages(turn.messages) + (deps.imageFloor ?? 0);
|
|
51292
|
+
currentTokens = Math.max(deps.session.stats.lastInputTokens, estimateCoreMessages(turn.messages) + (deps.imageFloor ?? 0) + (deps.wireOverhead ?? 0));
|
|
51293
|
+
result.payloadEstimate = estimateCoreMessages(turn.messages) + (deps.imageFloor ?? 0) + (deps.wireOverhead ?? 0);
|
|
51030
51294
|
if (startTokens < 0) startTokens = currentTokens;
|
|
51031
51295
|
if (currentTokens < limit) break;
|
|
51032
51296
|
const ranges = viableRanges(turn.nudge?.compressibleRanges ?? []);
|
|
51033
51297
|
if (ranges.length === 0) {
|
|
51034
|
-
|
|
51035
|
-
|
|
51036
|
-
|
|
51037
|
-
|
|
51038
|
-
|
|
51039
|
-
|
|
51040
|
-
|
|
51041
|
-
|
|
51042
|
-
failure = { kind: "exhausted", detail:
|
|
51298
|
+
if (!relaxed && result.payloadEstimate >= limit) {
|
|
51299
|
+
activeConfig = relaxedConfig(deps.config);
|
|
51300
|
+
relaxed = true;
|
|
51301
|
+
summaryCalls = 0;
|
|
51302
|
+
budgetHit = false;
|
|
51303
|
+
deps.log("warn", "[preflight] no compressible ranges outside the protected recent zone; relaxing soft protection (preserveRecentMessages/Tokens -> 0) and retrying");
|
|
51304
|
+
continue;
|
|
51305
|
+
}
|
|
51306
|
+
failure = { kind: "exhausted", detail: relaxed ? relaxedExhaustedDetail : "no compressible ranges remain in the conversation" };
|
|
51043
51307
|
break;
|
|
51044
51308
|
}
|
|
51309
|
+
const ordered = [...ranges].sort((a, b2) => refNum3(a.startRef) - refNum3(b2.startRef));
|
|
51045
51310
|
let appliedThisRound = 0;
|
|
51046
|
-
for (const
|
|
51311
|
+
for (const range of ordered) {
|
|
51047
51312
|
if (currentTokens < limit) break;
|
|
51048
51313
|
if (deps.signal?.aborted) {
|
|
51049
51314
|
failure = ABORTED_FAILURE;
|
|
51050
51315
|
break;
|
|
51051
51316
|
}
|
|
51052
|
-
|
|
51053
|
-
const
|
|
51054
|
-
|
|
51055
|
-
|
|
51056
|
-
|
|
51057
|
-
const
|
|
51058
|
-
if (
|
|
51059
|
-
|
|
51060
|
-
|
|
51061
|
-
|
|
51062
|
-
|
|
51063
|
-
|
|
51064
|
-
|
|
51065
|
-
|
|
51066
|
-
status: err2.status,
|
|
51067
|
-
detail: err2.status === 429 ? `the summarization call was rate-limited by the upstream (HTTP 429)` : `the summarization call was rejected by the upstream (HTTP ${err2.status})`
|
|
51068
|
-
};
|
|
51069
|
-
deps.log("warn", `[preflight] summarization failed: HTTP ${err2.status} ${err2.body.slice(0, 200)}`);
|
|
51070
|
-
} else if (deps.signal?.aborted) {
|
|
51317
|
+
if (budgetHit) break;
|
|
51318
|
+
const skipKey = `${range.startRef}:${range.endRef}`;
|
|
51319
|
+
if (skipSet.has(skipKey)) continue;
|
|
51320
|
+
const { refToIdx } = refMaps(messages, deps.session.state);
|
|
51321
|
+
const startIdx = refToIdx.get(range.startRef);
|
|
51322
|
+
const endIdx = refToIdx.get(range.endRef);
|
|
51323
|
+
if (startIdx === void 0 || endIdx === void 0 || startIdx > endIdx) {
|
|
51324
|
+
skipSet.add(skipKey);
|
|
51325
|
+
continue;
|
|
51326
|
+
}
|
|
51327
|
+
rangesTried += 1;
|
|
51328
|
+
for (const [cs2, ce2] of splitChunks(messages, startIdx, endIdx, budget)) {
|
|
51329
|
+
if (currentTokens < limit) break;
|
|
51330
|
+
if (deps.signal?.aborted) {
|
|
51071
51331
|
failure = ABORTED_FAILURE;
|
|
51072
|
-
|
|
51073
|
-
} else {
|
|
51074
|
-
failure = { kind: "upstream", detail: `the summarization call failed: ${String(err2)}` };
|
|
51075
|
-
deps.log("warn", `[preflight] summarization failed: ${String(err2)}`);
|
|
51332
|
+
break;
|
|
51076
51333
|
}
|
|
51334
|
+
if (budgetHit) break;
|
|
51335
|
+
const maps = refMaps(messages, deps.session.state);
|
|
51336
|
+
const startRef = maps.idxToRef.get(cs2);
|
|
51337
|
+
const endRef = maps.idxToRef.get(ce2);
|
|
51338
|
+
if (!startRef || !endRef) continue;
|
|
51339
|
+
if (rangeChars2(messages, cs2, ce2) < minChars) continue;
|
|
51340
|
+
const content = renderRange(messages, cs2, ce2);
|
|
51341
|
+
if (content.length === 0) continue;
|
|
51342
|
+
if (summaryCalls >= MAX_SUMMARY_CALLS_PER_PREFLIGHT) {
|
|
51343
|
+
budgetHit = true;
|
|
51344
|
+
break;
|
|
51345
|
+
}
|
|
51346
|
+
summaryCalls += 1;
|
|
51347
|
+
let summary;
|
|
51348
|
+
try {
|
|
51349
|
+
summary = await summarizeRange(deps, content, startRef, endRef);
|
|
51350
|
+
} catch (err2) {
|
|
51351
|
+
if (err2 instanceof UpstreamHttpError) {
|
|
51352
|
+
failure = {
|
|
51353
|
+
kind: "upstream",
|
|
51354
|
+
status: err2.status,
|
|
51355
|
+
detail: err2.status === 429 ? `the summarization call was rate-limited by the upstream (HTTP 429)` : `the summarization call was rejected by the upstream (HTTP ${err2.status})`
|
|
51356
|
+
};
|
|
51357
|
+
deps.log("warn", `[preflight] summarization failed: HTTP ${err2.status} ${err2.body.slice(0, 200)}`);
|
|
51358
|
+
} else if (deps.signal?.aborted) {
|
|
51359
|
+
failure = ABORTED_FAILURE;
|
|
51360
|
+
deps.log("warn", `[preflight] summarization aborted: client disconnected`);
|
|
51361
|
+
} else {
|
|
51362
|
+
failure = { kind: "upstream", detail: `the summarization call failed: ${String(err2)}` };
|
|
51363
|
+
deps.log("warn", `[preflight] summarization failed: ${String(err2)}`);
|
|
51364
|
+
}
|
|
51365
|
+
break;
|
|
51366
|
+
}
|
|
51367
|
+
if (!summary) {
|
|
51368
|
+
deps.log("warn", `[preflight] range ${skipKey} produced no usable summary; skipping it`);
|
|
51369
|
+
skipSet.add(skipKey);
|
|
51370
|
+
break;
|
|
51371
|
+
}
|
|
51372
|
+
const ctx = {
|
|
51373
|
+
core: deps.core,
|
|
51374
|
+
config: activeConfig,
|
|
51375
|
+
messages,
|
|
51376
|
+
session: deps.session,
|
|
51377
|
+
log: (msg) => deps.log("info", msg)
|
|
51378
|
+
};
|
|
51379
|
+
const creditBefore = deps.session.stats.compressCreditTokens;
|
|
51380
|
+
const applied = applyRanges(parseCompressInput({ content: [{ startId: startRef, endId: endRef, summary, topic: "preflight overflow compress" }] }), ctx);
|
|
51381
|
+
if (applied.startsWith("[Compression FAILED")) {
|
|
51382
|
+
deps.log("warn", `[preflight] ${applied}`);
|
|
51383
|
+
skipSet.add(skipKey);
|
|
51384
|
+
break;
|
|
51385
|
+
}
|
|
51386
|
+
const compressed = deps.session.stats.compressCreditTokens - creditBefore;
|
|
51387
|
+
currentTokens = Math.max(0, currentTokens - compressed + defaultCountTokens(summary));
|
|
51388
|
+
deps.session.stats.lastInputTokens += defaultCountTokens(summary);
|
|
51389
|
+
appliedThisRound += 1;
|
|
51390
|
+
result.compressedRanges += 1;
|
|
51077
51391
|
break;
|
|
51078
51392
|
}
|
|
51079
|
-
if (
|
|
51080
|
-
|
|
51081
|
-
core: deps.core,
|
|
51082
|
-
config: deps.config,
|
|
51083
|
-
messages,
|
|
51084
|
-
session: deps.session,
|
|
51085
|
-
log: (msg) => deps.log("info", msg)
|
|
51086
|
-
};
|
|
51087
|
-
const creditBefore = deps.session.stats.compressCreditTokens;
|
|
51088
|
-
const applied = applyRanges(parseCompressInput({ content: [{ startId: startRef, endId: endRef, summary, topic: "preflight overflow compress" }] }), ctx);
|
|
51089
|
-
if (applied.startsWith("[Compression FAILED")) {
|
|
51090
|
-
deps.log("warn", `[preflight] ${applied}`);
|
|
51091
|
-
continue;
|
|
51092
|
-
}
|
|
51093
|
-
const compressed = deps.session.stats.compressCreditTokens - creditBefore;
|
|
51094
|
-
currentTokens = Math.max(0, currentTokens - compressed + defaultCountTokens(summary));
|
|
51095
|
-
deps.session.stats.lastInputTokens += defaultCountTokens(summary);
|
|
51096
|
-
appliedThisRound += 1;
|
|
51097
|
-
result.compressedRanges += 1;
|
|
51098
|
-
}
|
|
51099
|
-
if (appliedThisRound === 0) {
|
|
51100
|
-
if (!failure) {
|
|
51101
|
-
failure = { kind: "exhausted", detail: "no range could be compressed (chunks below minCompressRange or the summarization responses were unusable)" };
|
|
51102
|
-
}
|
|
51103
|
-
break;
|
|
51393
|
+
if (appliedThisRound > 0) break;
|
|
51394
|
+
if (failure || budgetHit) break;
|
|
51104
51395
|
}
|
|
51396
|
+
if (appliedThisRound === 0) break;
|
|
51105
51397
|
}
|
|
51106
|
-
if (
|
|
51107
|
-
|
|
51398
|
+
if (currentTokens >= limit && !failure) {
|
|
51399
|
+
if (budgetHit) {
|
|
51400
|
+
failure = { kind: "exhausted", detail: `the preflight summarization budget (${MAX_SUMMARY_CALLS_PER_PREFLIGHT} calls per protection regime) was exhausted before the payload fit the window` };
|
|
51401
|
+
} else if (relaxed && result.compressedRanges > 0) {
|
|
51402
|
+
failure = { kind: "exhausted", detail: relaxedExhaustedDetail };
|
|
51403
|
+
} else if (result.compressedRanges === 0) {
|
|
51404
|
+
failure = { kind: "exhausted", detail: `no range could be compressed across ${rangesTried} viable range${rangesTried === 1 ? "" : "s"} (each was below minCompressRange, had an unusable summary, or failed to apply)` };
|
|
51405
|
+
} else {
|
|
51406
|
+
failure = { kind: "exhausted", detail: `the compress budget was exhausted after ${MAX_PREFLIGHT_ROUNDS} rounds` };
|
|
51407
|
+
}
|
|
51108
51408
|
}
|
|
51109
51409
|
if (result.compressedRanges > 0) deps.session.stats.lastInputTokens = currentTokens;
|
|
51110
51410
|
result.savedTokens = Math.max(0, startTokens - currentTokens);
|
|
@@ -51634,7 +51934,7 @@ function reapOrphanBlocks(session, visible, deactivate) {
|
|
|
51634
51934
|
}
|
|
51635
51935
|
|
|
51636
51936
|
// src/instance.ts
|
|
51637
|
-
import { randomUUID as randomUUID3 } from "crypto";
|
|
51937
|
+
import { createHash as createHash4, randomUUID as randomUUID3 } from "crypto";
|
|
51638
51938
|
import fs3 from "fs";
|
|
51639
51939
|
import path7 from "path";
|
|
51640
51940
|
function isProxyInstanceFile(v2) {
|
|
@@ -51680,14 +51980,13 @@ function readProxyInstanceFile(file) {
|
|
|
51680
51980
|
function instanceFilePath() {
|
|
51681
51981
|
return path7.join(stateDir(), "proxy-origin");
|
|
51682
51982
|
}
|
|
51683
|
-
function
|
|
51684
|
-
const filePath = file ?? instanceFilePath();
|
|
51983
|
+
function atomicWriteJson(obj, filePath) {
|
|
51685
51984
|
fs3.mkdirSync(path7.dirname(filePath), { recursive: true });
|
|
51686
51985
|
const tempPath = `${filePath}.${process.pid}.${randomUUID3()}.tmp`;
|
|
51687
51986
|
let descriptor;
|
|
51688
51987
|
try {
|
|
51689
51988
|
descriptor = fs3.openSync(tempPath, "wx", 420);
|
|
51690
|
-
fs3.writeSync(descriptor, JSON.stringify(
|
|
51989
|
+
fs3.writeSync(descriptor, JSON.stringify(obj) + "\n", null, "utf8");
|
|
51691
51990
|
fs3.fsyncSync(descriptor);
|
|
51692
51991
|
fs3.closeSync(descriptor);
|
|
51693
51992
|
descriptor = void 0;
|
|
@@ -51706,6 +52005,9 @@ function atomicWriteInstanceFile(info, file) {
|
|
|
51706
52005
|
throw error;
|
|
51707
52006
|
}
|
|
51708
52007
|
}
|
|
52008
|
+
function atomicWriteInstanceFile(info, file) {
|
|
52009
|
+
atomicWriteJson(info, file ?? instanceFilePath());
|
|
52010
|
+
}
|
|
51709
52011
|
function clearProxyInstanceFile(instanceId, file) {
|
|
51710
52012
|
const filePath = file ?? instanceFilePath();
|
|
51711
52013
|
const current = readProxyInstanceFile(filePath);
|
|
@@ -51725,44 +52027,99 @@ function isPidAlive(pid) {
|
|
|
51725
52027
|
return err2.code === "EPERM";
|
|
51726
52028
|
}
|
|
51727
52029
|
}
|
|
51728
|
-
function
|
|
52030
|
+
function registryDirPath() {
|
|
52031
|
+
return path7.join(stateDir(), "instances");
|
|
52032
|
+
}
|
|
52033
|
+
function legacyRegistryFilePath() {
|
|
51729
52034
|
return path7.join(stateDir(), "instances.json");
|
|
51730
52035
|
}
|
|
51731
|
-
|
|
51732
|
-
|
|
51733
|
-
|
|
52036
|
+
var SAFE_NAME_RE = /^[A-Za-z0-9][A-Za-z0-9._-]*$/;
|
|
52037
|
+
function safeRegistryName(instanceId) {
|
|
52038
|
+
if (instanceId !== "." && instanceId !== ".." && instanceId.length <= 128 && SAFE_NAME_RE.test(instanceId)) {
|
|
52039
|
+
return instanceId;
|
|
52040
|
+
}
|
|
52041
|
+
return createHash4("sha256").update(instanceId).digest("hex");
|
|
52042
|
+
}
|
|
52043
|
+
function registryEntryFile(instanceId) {
|
|
52044
|
+
return path7.join(registryDirPath(), `${safeRegistryName(instanceId)}.json`);
|
|
52045
|
+
}
|
|
52046
|
+
function safeReadJson2(file) {
|
|
51734
52047
|
try {
|
|
51735
|
-
|
|
51736
|
-
|
|
51737
|
-
|
|
51738
|
-
|
|
51739
|
-
|
|
51740
|
-
|
|
52048
|
+
return JSON.parse(fs3.readFileSync(file, "utf8"));
|
|
52049
|
+
} catch {
|
|
52050
|
+
return void 0;
|
|
52051
|
+
}
|
|
52052
|
+
}
|
|
52053
|
+
function coerceEntry(value) {
|
|
52054
|
+
if (!value || typeof value !== "object") return void 0;
|
|
52055
|
+
const o = value;
|
|
52056
|
+
if (typeof o.instanceId !== "string" || o.instanceId === "") return void 0;
|
|
52057
|
+
return {
|
|
52058
|
+
instanceId: o.instanceId,
|
|
52059
|
+
pid: typeof o.pid === "number" ? o.pid : 0,
|
|
52060
|
+
port: typeof o.port === "number" ? o.port : 0,
|
|
52061
|
+
origin: typeof o.origin === "string" ? o.origin : "",
|
|
52062
|
+
startedAt: typeof o.startedAt === "number" ? o.startedAt : 0
|
|
52063
|
+
};
|
|
52064
|
+
}
|
|
52065
|
+
function readMarkerNames() {
|
|
52066
|
+
try {
|
|
52067
|
+
return fs3.readdirSync(registryDirPath());
|
|
52068
|
+
} catch {
|
|
52069
|
+
return [];
|
|
52070
|
+
}
|
|
52071
|
+
}
|
|
52072
|
+
function readAllRegistryEntries() {
|
|
52073
|
+
const seen = /* @__PURE__ */ new Set();
|
|
52074
|
+
const out = [];
|
|
52075
|
+
for (const name of readMarkerNames()) {
|
|
52076
|
+
if (!name.endsWith(".json")) continue;
|
|
52077
|
+
const entry = coerceEntry(safeReadJson2(path7.join(registryDirPath(), name)));
|
|
52078
|
+
if (entry && !seen.has(entry.instanceId)) {
|
|
52079
|
+
seen.add(entry.instanceId);
|
|
52080
|
+
out.push(entry);
|
|
52081
|
+
}
|
|
52082
|
+
}
|
|
52083
|
+
const legacy = safeReadJson2(legacyRegistryFilePath());
|
|
52084
|
+
if (legacy && Array.isArray(legacy.instances)) {
|
|
52085
|
+
for (const raw of legacy.instances) {
|
|
52086
|
+
const entry = coerceEntry(raw);
|
|
52087
|
+
if (entry && !seen.has(entry.instanceId)) {
|
|
52088
|
+
seen.add(entry.instanceId);
|
|
52089
|
+
out.push(entry);
|
|
51741
52090
|
}
|
|
51742
52091
|
}
|
|
51743
|
-
} catch {
|
|
51744
52092
|
}
|
|
51745
|
-
|
|
52093
|
+
return out;
|
|
52094
|
+
}
|
|
52095
|
+
function reapDeadMarkers(ours) {
|
|
52096
|
+
for (const name of readMarkerNames()) {
|
|
52097
|
+
if (!name.endsWith(".json")) continue;
|
|
52098
|
+
const file = path7.join(registryDirPath(), name);
|
|
52099
|
+
const entry = coerceEntry(safeReadJson2(file));
|
|
52100
|
+
if (!entry || entry.instanceId === ours || isPidAlive(entry.pid)) continue;
|
|
52101
|
+
try {
|
|
52102
|
+
fs3.unlinkSync(file);
|
|
52103
|
+
} catch {
|
|
52104
|
+
}
|
|
52105
|
+
}
|
|
52106
|
+
}
|
|
52107
|
+
function registerInstanceAndWarn(entry, warn) {
|
|
52108
|
+
const others = readAllRegistryEntries().filter((e) => e.instanceId !== entry.instanceId && isPidAlive(e.pid));
|
|
52109
|
+
for (const other of others) {
|
|
51746
52110
|
warn(
|
|
51747
52111
|
`another bili instance is running (pid ${other.pid}, ${other.origin}) \u2014 both processes will write the same sessions directory; stop one to avoid state pollution (#394)`
|
|
51748
52112
|
);
|
|
51749
52113
|
}
|
|
51750
|
-
|
|
52114
|
+
reapDeadMarkers(entry.instanceId);
|
|
51751
52115
|
try {
|
|
51752
|
-
|
|
51753
|
-
fs3.writeFileSync(file, JSON.stringify({ instances: entries }) + "\n");
|
|
52116
|
+
atomicWriteJson(entry, registryEntryFile(entry.instanceId));
|
|
51754
52117
|
} catch {
|
|
51755
52118
|
}
|
|
51756
52119
|
}
|
|
51757
52120
|
function unregisterInstance(instanceId) {
|
|
51758
|
-
const file = registryFilePath();
|
|
51759
52121
|
try {
|
|
51760
|
-
|
|
51761
|
-
if (!Array.isArray(parsed.instances)) return;
|
|
51762
|
-
const kept = parsed.instances.filter(
|
|
51763
|
-
(e) => !(e && typeof e.instanceId === "string" && e.instanceId === instanceId) && isPidAlive(e.pid)
|
|
51764
|
-
);
|
|
51765
|
-
fs3.writeFileSync(file, JSON.stringify({ instances: kept }) + "\n");
|
|
52122
|
+
fs3.unlinkSync(registryEntryFile(instanceId));
|
|
51766
52123
|
} catch {
|
|
51767
52124
|
}
|
|
51768
52125
|
}
|
|
@@ -51879,6 +52236,126 @@ function warnCacheCollapse(session, input, cached) {
|
|
|
51879
52236
|
);
|
|
51880
52237
|
}
|
|
51881
52238
|
|
|
52239
|
+
// src/util.ts
|
|
52240
|
+
import { createHash as createHash5 } from "crypto";
|
|
52241
|
+
function hashId2(s3) {
|
|
52242
|
+
return createHash5("sha256").update(s3, "utf8").digest("hex").slice(0, 16);
|
|
52243
|
+
}
|
|
52244
|
+
function isLoopbackAddress(addr) {
|
|
52245
|
+
return !!addr && (addr.startsWith("127.") || addr === "::1" || addr.startsWith("::ffff:127."));
|
|
52246
|
+
}
|
|
52247
|
+
function usageTotals(protocol, usage) {
|
|
52248
|
+
const num3 = (v2) => typeof v2 === "number" && Number.isFinite(v2) ? v2 : void 0;
|
|
52249
|
+
if (protocol === "anthropic") {
|
|
52250
|
+
const fresh = num3(usage["input_tokens"]);
|
|
52251
|
+
const read = num3(usage["cache_read_input_tokens"]);
|
|
52252
|
+
const creation = num3(usage["cache_creation_input_tokens"]);
|
|
52253
|
+
const any = fresh !== void 0 || read !== void 0 || creation !== void 0;
|
|
52254
|
+
return {
|
|
52255
|
+
total: any ? (fresh ?? 0) + (read ?? 0) + (creation ?? 0) : void 0,
|
|
52256
|
+
cached: read
|
|
52257
|
+
};
|
|
52258
|
+
}
|
|
52259
|
+
if (protocol === "openai") {
|
|
52260
|
+
const prompt = num3(usage["prompt_tokens"]);
|
|
52261
|
+
const cached = num3(usage["prompt_tokens_details"]?.["cached_tokens"]);
|
|
52262
|
+
return {
|
|
52263
|
+
total: prompt !== void 0 ? promptInputTotal("openai", prompt, cached) : void 0,
|
|
52264
|
+
cached
|
|
52265
|
+
};
|
|
52266
|
+
}
|
|
52267
|
+
return {
|
|
52268
|
+
total: num3(usage["input_tokens"]),
|
|
52269
|
+
cached: num3(usage["input_tokens_details"]?.["cached_tokens"])
|
|
52270
|
+
};
|
|
52271
|
+
}
|
|
52272
|
+
function promptInputTotal(protocol, input, cached) {
|
|
52273
|
+
if (input === void 0) return 0;
|
|
52274
|
+
const includesCached = protocol === "openai" || protocol === "responses";
|
|
52275
|
+
const splitSemantics = !includesCached || typeof cached === "number" && input < cached;
|
|
52276
|
+
return input + (splitSemantics && typeof cached === "number" ? cached : 0);
|
|
52277
|
+
}
|
|
52278
|
+
function backfillHostUsage(protocol, usage, credit) {
|
|
52279
|
+
if (!Number.isFinite(credit) || credit <= 0) return false;
|
|
52280
|
+
let patched = false;
|
|
52281
|
+
const add = (key) => {
|
|
52282
|
+
if (typeof usage[key] === "number" && Number.isFinite(usage[key])) {
|
|
52283
|
+
usage[key] = usage[key] + credit;
|
|
52284
|
+
patched = true;
|
|
52285
|
+
}
|
|
52286
|
+
};
|
|
52287
|
+
if (protocol === "openai") {
|
|
52288
|
+
add("prompt_tokens");
|
|
52289
|
+
add("total_tokens");
|
|
52290
|
+
} else {
|
|
52291
|
+
add("input_tokens");
|
|
52292
|
+
}
|
|
52293
|
+
return patched;
|
|
52294
|
+
}
|
|
52295
|
+
var CONTEXT_OVERFLOW_PATTERNS = [
|
|
52296
|
+
/context_length_exceeded/i,
|
|
52297
|
+
/context_window_exceeded/i,
|
|
52298
|
+
/context length exceeded/i,
|
|
52299
|
+
/maximum context length/i,
|
|
52300
|
+
/max context length/i,
|
|
52301
|
+
/maximum context size/i,
|
|
52302
|
+
/longer than the model'?s context length/i,
|
|
52303
|
+
/exceeds the context window/i,
|
|
52304
|
+
/out of room in the model/i,
|
|
52305
|
+
/exceeded model token limit/i,
|
|
52306
|
+
/prompt is too long/i,
|
|
52307
|
+
/prompt_too_long/i,
|
|
52308
|
+
/prompt_is_too_long/i,
|
|
52309
|
+
/request_too_large/i,
|
|
52310
|
+
/token limit exceeded/i,
|
|
52311
|
+
// #554: llama.cpp-family "exceed_context_size_error (A / B > W)" — carried by
|
|
52312
|
+
// side requests that bypass preflight; without it the learned channel learns nothing.
|
|
52313
|
+
/exceed[_\s]?context[_\s]?size/i
|
|
52314
|
+
];
|
|
52315
|
+
function toTokenNumber(s3) {
|
|
52316
|
+
const n = parseInt(s3.replace(/,/g, ""), 10);
|
|
52317
|
+
return Number.isFinite(n) && n >= 1e3 ? n : void 0;
|
|
52318
|
+
}
|
|
52319
|
+
function parseOverflowWindow(text) {
|
|
52320
|
+
let m2 = text.match(/>\s*(\d[\d,]*)\s*maximum/i);
|
|
52321
|
+
if (m2) return toTokenNumber(m2[1]);
|
|
52322
|
+
m2 = text.match(/\(\s*\d[\d,]*\s*\/\s*\d[\d,]*\s*>\s*(\d[\d,]+)\s*\)/);
|
|
52323
|
+
if (m2) return toTokenNumber(m2[1]);
|
|
52324
|
+
m2 = text.match(/maximum context length is (\d[\d,]*)/i) ?? text.match(/maximum context length of (\d[\d,]*)/i) ?? text.match(/maximum context size (?:is|of) (\d[\d,]*)/i) ?? text.match(/(?:maximum|max)\s+(?:context\s+)?length\s+(?:is\s+)?(\d[\d,]*)/i) ?? text.match(/context length\s*\((\d[\d,]*)\s*token/i) ?? text.match(/limit of (\d[\d,]*)\s*token/i) ?? text.match(/(\d[\d,]*)\s*maximum\b/i);
|
|
52325
|
+
if (m2) return toTokenNumber(m2[1]);
|
|
52326
|
+
return void 0;
|
|
52327
|
+
}
|
|
52328
|
+
function inspectContextOverflow(status, bodyText) {
|
|
52329
|
+
const message = (bodyText ?? "").slice(0, 300);
|
|
52330
|
+
if (status !== 400 && status !== 413) return { isOverflow: false, message };
|
|
52331
|
+
if (!bodyText) return { isOverflow: false, message };
|
|
52332
|
+
const isOverflow = CONTEXT_OVERFLOW_PATTERNS.some((p2) => p2.test(bodyText));
|
|
52333
|
+
if (!isOverflow) return { isOverflow: false, message };
|
|
52334
|
+
return { isOverflow: true, window: parseOverflowWindow(bodyText), message };
|
|
52335
|
+
}
|
|
52336
|
+
function reserveOutputHeadroom(window2, maxOutput) {
|
|
52337
|
+
if (Number.isFinite(window2) && window2 > 0 && Number.isFinite(maxOutput) && maxOutput > 0 && maxOutput < window2) {
|
|
52338
|
+
return window2 - maxOutput;
|
|
52339
|
+
}
|
|
52340
|
+
return window2;
|
|
52341
|
+
}
|
|
52342
|
+
function systemToUser(messages) {
|
|
52343
|
+
let hasSys = false;
|
|
52344
|
+
for (const m2 of messages) {
|
|
52345
|
+
if (m2.role === "system" || m2.role === "developer") {
|
|
52346
|
+
hasSys = true;
|
|
52347
|
+
break;
|
|
52348
|
+
}
|
|
52349
|
+
}
|
|
52350
|
+
if (!hasSys) return messages;
|
|
52351
|
+
return messages.map(
|
|
52352
|
+
(m2) => m2.role === "system" || m2.role === "developer" ? { ...m2, role: "user" } : m2
|
|
52353
|
+
);
|
|
52354
|
+
}
|
|
52355
|
+
function shouldReserveOutputHeadroom(protocol) {
|
|
52356
|
+
return protocol !== "anthropic";
|
|
52357
|
+
}
|
|
52358
|
+
|
|
51882
52359
|
// src/loop/core.ts
|
|
51883
52360
|
var MAX_LOOP_ROUNDS = 10;
|
|
51884
52361
|
function isLoopThinking(m2) {
|
|
@@ -51932,46 +52409,14 @@ ${lines.join("\n\n")}`;
|
|
|
51932
52409
|
}
|
|
51933
52410
|
return `[Unknown proxy tool: ${toolName}]`;
|
|
51934
52411
|
}
|
|
51935
|
-
function handleAcpStatus(args, ctx) {
|
|
51936
|
-
const scope = typeof args.scope === "string" ? args.scope : void 0;
|
|
51937
|
-
const view = typeof args.view === "string" ? args.view : void 0;
|
|
51938
|
-
const tool = typeof args.tool === "string" ? args.tool : void 0;
|
|
51939
|
-
const sort = typeof args.sort === "string" ? args.sort : void 0;
|
|
51940
|
-
const limit = typeof args.limit === "number" ? args.limit : void 0;
|
|
51941
|
-
const base = buildStatusReport(ctx.session.state, ctx.messages, defaultCountTokens, { scope, view, tool, sort, limit });
|
|
51942
|
-
if (scope) return base;
|
|
51943
|
-
const nudge = ctx.nudge;
|
|
51944
|
-
const ranges = nudge?.compressibleRanges ?? [];
|
|
51945
|
-
const protectedRanges = nudge?.protectedRanges ?? [];
|
|
51946
|
-
const archive = preCompactionArchiveOf(ctx.session);
|
|
51947
|
-
const archivedIds = Object.keys(archive);
|
|
51948
|
-
const extra = [];
|
|
51949
|
-
if (nudge) {
|
|
51950
|
-
extra.push("");
|
|
51951
|
-
extra.push(nudge.shouldInject ? `Nudge: ACTIVE \u2014 ${nudge.reason}` : `Nudge: idle \u2014 ${nudge.reason}`);
|
|
51952
|
-
}
|
|
51953
|
-
if (ranges.length > 0 || protectedRanges.length > 0) {
|
|
51954
|
-
extra.push("");
|
|
51955
|
-
extra.push(formatRanges(ranges, protectedRanges));
|
|
51956
|
-
}
|
|
51957
|
-
if (archivedIds.length > 0) {
|
|
51958
|
-
extra.push("");
|
|
51959
|
-
extra.push(`PRE-COMPACTION ARCHIVE \u2014 ${archivedIds.length} block(s): content was replaced by the client's native compaction summary, so it is no longer in the session history and decompress is unavailable.`);
|
|
51960
|
-
for (const id of archivedIds) {
|
|
51961
|
-
extra.push(` ${id} \u2014 ${archive[id].reason}`);
|
|
51962
|
-
}
|
|
51963
|
-
}
|
|
51964
|
-
return extra.length > 0 ? `${base}
|
|
51965
|
-
${extra.join("\n")}` : base;
|
|
51966
|
-
}
|
|
51967
52412
|
function recordUsage(ctx, usage, round) {
|
|
51968
52413
|
const prompt = usage.inputTokens;
|
|
51969
52414
|
const cached = usage.cachedTokens;
|
|
51970
52415
|
const out = usage.outputTokens;
|
|
51971
|
-
const
|
|
51972
|
-
const total = (typeof prompt === "number" ? prompt : 0) + (!includesCached && typeof cached === "number" ? cached : 0);
|
|
52416
|
+
const total = promptInputTotal(ctx.protocol, prompt, cached);
|
|
51973
52417
|
if (total > 0) ctx.session.stats.inputTokens += total;
|
|
51974
52418
|
ctx.session.stats.lastInputTokens = Math.max(0, total - (ctx.session.stats.compressCreditTokens ?? 0));
|
|
52419
|
+
ctx.session.hostContextTokens = total + (ctx.session.hostCreditTokens ?? 0);
|
|
51975
52420
|
if (typeof cached === "number") {
|
|
51976
52421
|
ctx.session.stats.cachedTokens += cached;
|
|
51977
52422
|
ctx.session.stats.cacheSamples += 1;
|
|
@@ -51995,7 +52440,7 @@ async function* runCompressLoop(upstream, ctx, requestBody, requestOptions, adap
|
|
|
51995
52440
|
{
|
|
51996
52441
|
method: "POST",
|
|
51997
52442
|
headers: requestOptions.headers,
|
|
51998
|
-
body: JSON.stringify(body),
|
|
52443
|
+
body: JSON.stringify(requestOptions.wireTransform ? requestOptions.wireTransform(body) : body),
|
|
51999
52444
|
...ctx.proxyUrl ? { dispatcher: proxyDispatcher(ctx.proxyUrl) } : {}
|
|
52000
52445
|
},
|
|
52001
52446
|
void 0,
|
|
@@ -52110,6 +52555,10 @@ async function* runCompressLoop(upstream, ctx, requestBody, requestOptions, adap
|
|
|
52110
52555
|
if (usage.inputTokens !== void 0 || usage.outputTokens !== void 0 || usage.cachedTokens !== void 0) {
|
|
52111
52556
|
recordUsage(ctx, usage, round);
|
|
52112
52557
|
}
|
|
52558
|
+
const hostCredit = ctx.session.hostCreditTokens ?? 0;
|
|
52559
|
+
if (hostCredit > 0 && typeof usage.inputTokens === "number") {
|
|
52560
|
+
usage.inputTokens += hostCredit;
|
|
52561
|
+
}
|
|
52113
52562
|
let resolvedText = assistantText;
|
|
52114
52563
|
let allCalls = calls;
|
|
52115
52564
|
if (ctx.textProtocol && assistantText.length > 0 && adapter.extractTextTriggers) {
|
|
@@ -52337,96 +52786,6 @@ async function* runCompressLoop(upstream, ctx, requestBody, requestOptions, adap
|
|
|
52337
52786
|
}
|
|
52338
52787
|
}
|
|
52339
52788
|
|
|
52340
|
-
// src/util.ts
|
|
52341
|
-
import { createHash as createHash4 } from "crypto";
|
|
52342
|
-
function hashId2(s3) {
|
|
52343
|
-
return createHash4("sha256").update(s3, "utf8").digest("hex").slice(0, 16);
|
|
52344
|
-
}
|
|
52345
|
-
function isLoopbackAddress(addr) {
|
|
52346
|
-
return !!addr && (addr.startsWith("127.") || addr === "::1" || addr.startsWith("::ffff:127."));
|
|
52347
|
-
}
|
|
52348
|
-
function usageTotals(protocol, usage) {
|
|
52349
|
-
const num3 = (v2) => typeof v2 === "number" && Number.isFinite(v2) ? v2 : void 0;
|
|
52350
|
-
if (protocol === "anthropic") {
|
|
52351
|
-
const fresh = num3(usage["input_tokens"]);
|
|
52352
|
-
const read = num3(usage["cache_read_input_tokens"]);
|
|
52353
|
-
const creation = num3(usage["cache_creation_input_tokens"]);
|
|
52354
|
-
const any = fresh !== void 0 || read !== void 0 || creation !== void 0;
|
|
52355
|
-
return {
|
|
52356
|
-
total: any ? (fresh ?? 0) + (read ?? 0) + (creation ?? 0) : void 0,
|
|
52357
|
-
cached: read
|
|
52358
|
-
};
|
|
52359
|
-
}
|
|
52360
|
-
if (protocol === "openai") {
|
|
52361
|
-
return {
|
|
52362
|
-
total: num3(usage["prompt_tokens"]),
|
|
52363
|
-
cached: num3(usage["prompt_tokens_details"]?.["cached_tokens"])
|
|
52364
|
-
};
|
|
52365
|
-
}
|
|
52366
|
-
return {
|
|
52367
|
-
total: num3(usage["input_tokens"]),
|
|
52368
|
-
cached: num3(usage["input_tokens_details"]?.["cached_tokens"])
|
|
52369
|
-
};
|
|
52370
|
-
}
|
|
52371
|
-
var CONTEXT_OVERFLOW_PATTERNS = [
|
|
52372
|
-
/context_length_exceeded/i,
|
|
52373
|
-
/context_window_exceeded/i,
|
|
52374
|
-
/context length exceeded/i,
|
|
52375
|
-
/maximum context length/i,
|
|
52376
|
-
/max context length/i,
|
|
52377
|
-
/maximum context size/i,
|
|
52378
|
-
/longer than the model'?s context length/i,
|
|
52379
|
-
/exceeds the context window/i,
|
|
52380
|
-
/out of room in the model/i,
|
|
52381
|
-
/exceeded model token limit/i,
|
|
52382
|
-
/prompt is too long/i,
|
|
52383
|
-
/prompt_too_long/i,
|
|
52384
|
-
/prompt_is_too_long/i,
|
|
52385
|
-
/request_too_large/i,
|
|
52386
|
-
/token limit exceeded/i
|
|
52387
|
-
];
|
|
52388
|
-
function toTokenNumber(s3) {
|
|
52389
|
-
const n = parseInt(s3.replace(/,/g, ""), 10);
|
|
52390
|
-
return Number.isFinite(n) && n >= 1e3 ? n : void 0;
|
|
52391
|
-
}
|
|
52392
|
-
function parseOverflowWindow(text) {
|
|
52393
|
-
let m2 = text.match(/>\s*(\d[\d,]*)\s*maximum/i);
|
|
52394
|
-
if (m2) return toTokenNumber(m2[1]);
|
|
52395
|
-
m2 = text.match(/maximum context length is (\d[\d,]*)/i) ?? text.match(/maximum context length of (\d[\d,]*)/i) ?? text.match(/maximum context size (?:is|of) (\d[\d,]*)/i) ?? text.match(/(?:maximum|max)\s+(?:context\s+)?length\s+(?:is\s+)?(\d[\d,]*)/i) ?? text.match(/context length\s*\((\d[\d,]*)\s*token/i) ?? text.match(/limit of (\d[\d,]*)\s*token/i) ?? text.match(/(\d[\d,]*)\s*maximum\b/i);
|
|
52396
|
-
if (m2) return toTokenNumber(m2[1]);
|
|
52397
|
-
return void 0;
|
|
52398
|
-
}
|
|
52399
|
-
function inspectContextOverflow(status, bodyText) {
|
|
52400
|
-
const message = (bodyText ?? "").slice(0, 300);
|
|
52401
|
-
if (status !== 400 && status !== 413) return { isOverflow: false, message };
|
|
52402
|
-
if (!bodyText) return { isOverflow: false, message };
|
|
52403
|
-
const isOverflow = CONTEXT_OVERFLOW_PATTERNS.some((p2) => p2.test(bodyText));
|
|
52404
|
-
if (!isOverflow) return { isOverflow: false, message };
|
|
52405
|
-
return { isOverflow: true, window: parseOverflowWindow(bodyText), message };
|
|
52406
|
-
}
|
|
52407
|
-
function reserveOutputHeadroom(window2, maxOutput) {
|
|
52408
|
-
if (Number.isFinite(window2) && window2 > 0 && Number.isFinite(maxOutput) && maxOutput > 0 && maxOutput < window2) {
|
|
52409
|
-
return window2 - maxOutput;
|
|
52410
|
-
}
|
|
52411
|
-
return window2;
|
|
52412
|
-
}
|
|
52413
|
-
function systemToUser(messages) {
|
|
52414
|
-
let hasSys = false;
|
|
52415
|
-
for (const m2 of messages) {
|
|
52416
|
-
if (m2.role === "system" || m2.role === "developer") {
|
|
52417
|
-
hasSys = true;
|
|
52418
|
-
break;
|
|
52419
|
-
}
|
|
52420
|
-
}
|
|
52421
|
-
if (!hasSys) return messages;
|
|
52422
|
-
return messages.map(
|
|
52423
|
-
(m2) => m2.role === "system" || m2.role === "developer" ? { ...m2, role: "user" } : m2
|
|
52424
|
-
);
|
|
52425
|
-
}
|
|
52426
|
-
function shouldReserveOutputHeadroom(protocol) {
|
|
52427
|
-
return protocol !== "anthropic";
|
|
52428
|
-
}
|
|
52429
|
-
|
|
52430
52789
|
// src/loop/adapter-responses.ts
|
|
52431
52790
|
var RESPONSES_ITEM_ID_MAX = 64;
|
|
52432
52791
|
function normalizeResponsesMessageItems(input) {
|
|
@@ -53058,7 +53417,7 @@ function stripFinishReasonChunk(buf) {
|
|
|
53058
53417
|
return buf;
|
|
53059
53418
|
}
|
|
53060
53419
|
}
|
|
53061
|
-
function createOpenaiAdapter(requestBody, clientSystem) {
|
|
53420
|
+
function createOpenaiAdapter(requestBody, clientSystem, hostCredit = 0) {
|
|
53062
53421
|
const model = requestBody.model ?? "unknown";
|
|
53063
53422
|
let responseId = `chatcmpl-proxy-${Date.now()}`;
|
|
53064
53423
|
let toolIndex = 0;
|
|
@@ -53252,7 +53611,24 @@ function createOpenaiAdapter(requestBody, clientSystem) {
|
|
|
53252
53611
|
cachedTokens: typeof pd?.cached_tokens === "number" ? pd.cached_tokens : void 0
|
|
53253
53612
|
};
|
|
53254
53613
|
if (sawRealToolCall) {
|
|
53255
|
-
|
|
53614
|
+
let chunk = rawBuf;
|
|
53615
|
+
if (hostCredit > 0 && u2) {
|
|
53616
|
+
const pu = typeof u2.prompt_tokens === "number" ? u2.prompt_tokens : void 0;
|
|
53617
|
+
const tu = typeof u2.total_tokens === "number" ? u2.total_tokens : void 0;
|
|
53618
|
+
if (pu !== void 0 || tu !== void 0) {
|
|
53619
|
+
const patched = {
|
|
53620
|
+
...parsed,
|
|
53621
|
+
usage: {
|
|
53622
|
+
...u2,
|
|
53623
|
+
...pu !== void 0 ? { prompt_tokens: pu + hostCredit } : {},
|
|
53624
|
+
...tu !== void 0 ? { total_tokens: tu + hostCredit } : {}
|
|
53625
|
+
}
|
|
53626
|
+
};
|
|
53627
|
+
const out = eventStr.split("\n").map((l) => l.startsWith("data:") ? `data: ${JSON.stringify(patched)}` : l).join("\n");
|
|
53628
|
+
chunk = Buffer.from(out + "\n\n", "utf8");
|
|
53629
|
+
}
|
|
53630
|
+
}
|
|
53631
|
+
yield { kind: "meta", chunk };
|
|
53256
53632
|
yield { kind: "done", finishReason, suppressCompletion: true };
|
|
53257
53633
|
continue;
|
|
53258
53634
|
} else {
|
|
@@ -53439,7 +53815,7 @@ data: ${JSON.stringify({ type: "content_block_delta", index, delta: { type: "tex
|
|
|
53439
53815
|
"utf8"
|
|
53440
53816
|
);
|
|
53441
53817
|
}
|
|
53442
|
-
function createAnthropicAdapter(requestBody, originalSystem) {
|
|
53818
|
+
function createAnthropicAdapter(requestBody, originalSystem, hostCredit = 0) {
|
|
53443
53819
|
const model = requestBody.model ?? void 0;
|
|
53444
53820
|
let messageId;
|
|
53445
53821
|
let clientIndex = 0;
|
|
@@ -53553,8 +53929,19 @@ ${systemPrompt}` : systemPrompt;
|
|
|
53553
53929
|
if (typeof u2.input_tokens === "number") roundInput = u2.input_tokens;
|
|
53554
53930
|
if (typeof u2.cache_read_input_tokens === "number") roundCached = u2.cache_read_input_tokens;
|
|
53555
53931
|
if (round === 1) {
|
|
53932
|
+
let chunk = rawBuf;
|
|
53933
|
+
if (hostCredit > 0 && typeof u2.input_tokens === "number") {
|
|
53934
|
+
const patched = structuredClone(data);
|
|
53935
|
+
const pmsg = patched["message"];
|
|
53936
|
+
const pu = pmsg?.["usage"] ?? {};
|
|
53937
|
+
if (typeof pu.input_tokens === "number") {
|
|
53938
|
+
pu.input_tokens += hostCredit;
|
|
53939
|
+
const out = eventStr.split("\n").map((l) => l.startsWith("data:") ? `data: ${JSON.stringify(patched)}` : l).join("\n");
|
|
53940
|
+
chunk = Buffer.from(out + "\n\n", "utf8");
|
|
53941
|
+
}
|
|
53942
|
+
}
|
|
53556
53943
|
messageStartForwarded = true;
|
|
53557
|
-
yield { kind: "meta", chunk
|
|
53944
|
+
yield { kind: "meta", chunk, firstRoundOnly: true };
|
|
53558
53945
|
}
|
|
53559
53946
|
} else if (type === "ping") {
|
|
53560
53947
|
yield { kind: "meta", chunk: rawBuf };
|
|
@@ -53737,10 +54124,10 @@ data: ${JSON.stringify({ type: "content_block_stop", index })}
|
|
|
53737
54124
|
}
|
|
53738
54125
|
|
|
53739
54126
|
// src/loop/index.ts
|
|
53740
|
-
function pickAdapter(protocol, requestBody, textProtocol, responsesProjection, anthropicSystem, openaiSystem) {
|
|
54127
|
+
function pickAdapter(protocol, requestBody, textProtocol, responsesProjection, anthropicSystem, openaiSystem, hostCredit = 0) {
|
|
53741
54128
|
if (protocol === "responses") return createResponsesAdapter(textProtocol, responsesProjection);
|
|
53742
|
-
if (protocol === "openai") return createOpenaiAdapter(requestBody, openaiSystem);
|
|
53743
|
-
if (protocol === "anthropic") return createAnthropicAdapter(requestBody, anthropicSystem);
|
|
54129
|
+
if (protocol === "openai") return createOpenaiAdapter(requestBody, openaiSystem, hostCredit);
|
|
54130
|
+
if (protocol === "anthropic") return createAnthropicAdapter(requestBody, anthropicSystem, hostCredit);
|
|
53744
54131
|
throw new Error(`[acp-loop] unknown protocol: ${protocol}`);
|
|
53745
54132
|
}
|
|
53746
54133
|
|
|
@@ -53801,7 +54188,7 @@ function executeProxyTool2(toolName, args, ctx) {
|
|
|
53801
54188
|
${lines.join("\n\n")}`;
|
|
53802
54189
|
}
|
|
53803
54190
|
if (toolName === "acp_status") {
|
|
53804
|
-
return
|
|
54191
|
+
return handleAcpStatus(args, ctx);
|
|
53805
54192
|
}
|
|
53806
54193
|
return `[Unknown proxy tool: ${toolName}]`;
|
|
53807
54194
|
}
|
|
@@ -53898,7 +54285,7 @@ async function compressLoopResponsesJson(initialResponse, ctx, requestBody, requ
|
|
|
53898
54285
|
const result = await fetchWithRetry(requestOptions.url, {
|
|
53899
54286
|
method: "POST",
|
|
53900
54287
|
headers: requestOptions.headers,
|
|
53901
|
-
body: JSON.stringify(requestBody),
|
|
54288
|
+
body: JSON.stringify(requestOptions.wireTransform ? requestOptions.wireTransform(requestBody) : requestBody),
|
|
53902
54289
|
...ctx.proxyUrl ? { dispatcher: proxyDispatcher(ctx.proxyUrl) } : {}
|
|
53903
54290
|
}, void 0, void 0, (info) => {
|
|
53904
54291
|
const lc = lastCompressSuffix(ctx.session.lastCompress);
|
|
@@ -54316,6 +54703,35 @@ data: ${JSON.stringify({ type: "message_delta", delta: { stop_reason: "end_turn"
|
|
|
54316
54703
|
safeWrite(res, `event: message_stop
|
|
54317
54704
|
data: ${JSON.stringify({ type: "message_stop" })}
|
|
54318
54705
|
|
|
54706
|
+
`);
|
|
54707
|
+
}
|
|
54708
|
+
} catch {
|
|
54709
|
+
} finally {
|
|
54710
|
+
try {
|
|
54711
|
+
res.end();
|
|
54712
|
+
} catch {
|
|
54713
|
+
}
|
|
54714
|
+
}
|
|
54715
|
+
}
|
|
54716
|
+
function emitPreflightError(res, protocol, error, log2) {
|
|
54717
|
+
const err2 = { type: "server_error", code: "preflight_compress_failed", message: error.message, retryable: error.retryable };
|
|
54718
|
+
log2?.(`[acp-proxy: preflight failed after early response commit \u2014 delivering in-band: ${error.message}]`);
|
|
54719
|
+
try {
|
|
54720
|
+
if (protocol === "openai") {
|
|
54721
|
+
safeWrite(res, `data: ${JSON.stringify({ error: err2 })}
|
|
54722
|
+
|
|
54723
|
+
data: [DONE]
|
|
54724
|
+
|
|
54725
|
+
`);
|
|
54726
|
+
} else if (protocol === "responses") {
|
|
54727
|
+
safeWrite(res, `event: error
|
|
54728
|
+
data: ${JSON.stringify({ type: "error", code: err2.code, message: err2.message })}
|
|
54729
|
+
|
|
54730
|
+
`);
|
|
54731
|
+
} else {
|
|
54732
|
+
safeWrite(res, `event: error
|
|
54733
|
+
data: ${JSON.stringify({ type: "error", error: { type: "server_error", code: err2.code, message: err2.message } })}
|
|
54734
|
+
|
|
54319
54735
|
`);
|
|
54320
54736
|
}
|
|
54321
54737
|
} catch {
|
|
@@ -54371,7 +54787,7 @@ function codexTurnIdentity(headers) {
|
|
|
54371
54787
|
}
|
|
54372
54788
|
|
|
54373
54789
|
// src/prefix-affinity.ts
|
|
54374
|
-
import { createHash as
|
|
54790
|
+
import { createHash as createHash6 } from "crypto";
|
|
54375
54791
|
var MIN_CANONICAL_BYTES = 24;
|
|
54376
54792
|
var MAX_TRACKED_SESSIONS3 = 256;
|
|
54377
54793
|
var TTL_MS2 = 7 * 24 * 60 * 60 * 1e3;
|
|
@@ -54393,7 +54809,7 @@ function sortKeys2(value) {
|
|
|
54393
54809
|
return value;
|
|
54394
54810
|
}
|
|
54395
54811
|
function sha256(text) {
|
|
54396
|
-
return
|
|
54812
|
+
return createHash6("sha256").update(text, "utf8").digest("hex");
|
|
54397
54813
|
}
|
|
54398
54814
|
function hasUserMessage(messages) {
|
|
54399
54815
|
return messages.some((m2) => !!m2 && typeof m2 === "object" && m2.role === "user");
|
|
@@ -55006,7 +55422,7 @@ function handlePluginManifest(res) {
|
|
|
55006
55422
|
statusEndpoint: "/__bili/plugin/status"
|
|
55007
55423
|
}));
|
|
55008
55424
|
}
|
|
55009
|
-
function handlePluginStatus(conversationId2, res, fallbackLatest = false) {
|
|
55425
|
+
function handlePluginStatus(conversationId2, res, deps, fallbackLatest = false) {
|
|
55010
55426
|
let entry = conversations.get(conversationId2);
|
|
55011
55427
|
let session = entry ? peekSession(entry.sessionId) : void 0;
|
|
55012
55428
|
let viaFallback = false;
|
|
@@ -55030,16 +55446,33 @@ function handlePluginStatus(conversationId2, res, fallbackLatest = false) {
|
|
|
55030
55446
|
const limit = session.metadata.effectiveContextLimit;
|
|
55031
55447
|
const mem = remembered.get(session.id);
|
|
55032
55448
|
const modelContextLimit = typeof limit === "number" && limit > 0 ? limit : 0;
|
|
55449
|
+
let nudge;
|
|
55450
|
+
try {
|
|
55451
|
+
const messages = mem ? mem.processed.length > 0 ? mem.processed : mem.original : [];
|
|
55452
|
+
if (messages.length > 0) {
|
|
55453
|
+
nudge = deps.core.processTurn({
|
|
55454
|
+
messages,
|
|
55455
|
+
state: session.state,
|
|
55456
|
+
config: deps.config,
|
|
55457
|
+
tokenCount: session.stats.lastInputTokens,
|
|
55458
|
+
renderTags: "none"
|
|
55459
|
+
}).nudge;
|
|
55460
|
+
}
|
|
55461
|
+
} catch {
|
|
55462
|
+
nudge = void 0;
|
|
55463
|
+
}
|
|
55033
55464
|
let panel;
|
|
55034
55465
|
try {
|
|
55466
|
+
const sysTokRaw = session.metadata.systemPromptTokens;
|
|
55467
|
+
const systemPromptTokens = typeof sysTokRaw === "number" && Number.isFinite(sysTokRaw) && sysTokRaw > 0 ? sysTokRaw : 0;
|
|
55035
55468
|
panel = buildStatusPanel({
|
|
55036
55469
|
version: `billion-context@${PROXY_VERSION}`,
|
|
55037
|
-
tokenCount: session.stats.lastInputTokens,
|
|
55038
|
-
systemPromptTokens
|
|
55470
|
+
tokenCount: session.hostContextTokens ?? session.stats.lastInputTokens,
|
|
55471
|
+
systemPromptTokens,
|
|
55039
55472
|
state: session.state,
|
|
55040
|
-
nudge
|
|
55473
|
+
nudge,
|
|
55041
55474
|
modelContextLimit,
|
|
55042
|
-
unprunedTokens: mem && mem.original.length > 0 ? mem.original.reduce((sum, m2) => sum + defaultCountTokens(m2.text ?? ""), 0) : void 0
|
|
55475
|
+
unprunedTokens: mem && mem.original.length > 0 ? mem.original.reduce((sum, m2) => sum + defaultCountTokens(m2.text ?? ""), 0) + systemPromptTokens : void 0
|
|
55043
55476
|
});
|
|
55044
55477
|
} catch {
|
|
55045
55478
|
panel = void 0;
|
|
@@ -55052,7 +55485,8 @@ function handlePluginStatus(conversationId2, res, fallbackLatest = false) {
|
|
|
55052
55485
|
label: session.meta.label ?? null,
|
|
55053
55486
|
pluginAgent: session.metadata.pluginAgent ?? null,
|
|
55054
55487
|
contextLimit: typeof limit === "number" ? limit : null,
|
|
55055
|
-
contextTokens: session.stats.lastInputTokens,
|
|
55488
|
+
contextTokens: session.hostContextTokens ?? session.stats.lastInputTokens,
|
|
55489
|
+
hostCredit: session.hostCreditTokens ?? 0,
|
|
55056
55490
|
inputTokens: session.stats.inputTokens,
|
|
55057
55491
|
outputTokens: session.stats.outputTokens,
|
|
55058
55492
|
cachedTokens: session.stats.cachedTokens,
|
|
@@ -55104,8 +55538,7 @@ async function handlePluginTool(payload, res, deps) {
|
|
|
55104
55538
|
config: deps.config,
|
|
55105
55539
|
messages,
|
|
55106
55540
|
session,
|
|
55107
|
-
log: (m2) => deps.log("info", `[${session.id}] [plugin] ${m2}`)
|
|
55108
|
-
nudge: mem?.nudge
|
|
55541
|
+
log: (m2) => deps.log("info", `[${session.id}] [plugin] ${m2}`)
|
|
55109
55542
|
}, callId);
|
|
55110
55543
|
});
|
|
55111
55544
|
} catch (err2) {
|
|
@@ -55159,16 +55592,16 @@ function usageFromSseEvent(obj) {
|
|
|
55159
55592
|
return void 0;
|
|
55160
55593
|
}
|
|
55161
55594
|
function applyUsageSample(session, sample, protocol) {
|
|
55162
|
-
const includesCached = protocol === "openai" || protocol === "responses";
|
|
55163
55595
|
if (sample.cachedTokens !== void 0) {
|
|
55164
55596
|
session.stats.cachedTokens += sample.cachedTokens;
|
|
55165
55597
|
session.stats.cacheSamples += 1;
|
|
55166
55598
|
}
|
|
55167
55599
|
if (sample.inputTokens !== void 0) {
|
|
55168
|
-
const total =
|
|
55600
|
+
const total = promptInputTotal(protocol, sample.inputTokens, sample.cachedTokens);
|
|
55169
55601
|
session.stats.inputTokens += total;
|
|
55170
55602
|
session.stats.lastInputTokens = Math.max(0, total - (session.stats.compressCreditTokens ?? 0));
|
|
55171
55603
|
warnCacheCollapse(session, total, sample.cachedTokens ?? 0);
|
|
55604
|
+
session.hostContextTokens = total + (session.hostCreditTokens ?? 0);
|
|
55172
55605
|
}
|
|
55173
55606
|
if (sample.outputTokens !== void 0) session.stats.outputTokens += sample.outputTokens;
|
|
55174
55607
|
}
|
|
@@ -55182,6 +55615,7 @@ async function pipePluginChatWithStrip(stream2, res, protocol, session, log2) {
|
|
|
55182
55615
|
const decoder = new TextDecoder("utf-8");
|
|
55183
55616
|
let buf = "";
|
|
55184
55617
|
const acc = {};
|
|
55618
|
+
const credit = session?.hostCreditTokens ?? 0;
|
|
55185
55619
|
const onDrop = (snippet) => {
|
|
55186
55620
|
log("warn", `[tag-echo] stripped model-emitted render tag (plugin passthrough): ${snippet.slice(0, 80).replace(/\n/g, " ")}`);
|
|
55187
55621
|
log2?.(`[tag-echo] stripped model-emitted render tag from plugin passthrough text`);
|
|
@@ -55339,12 +55773,23 @@ data: ${JSON.stringify({ type: "content_block_delta", index, delta: { type: delt
|
|
|
55339
55773
|
if (ev["type"] === "message_stop") sawTerminal = true;
|
|
55340
55774
|
const sample = usageFromSseEvent(ev);
|
|
55341
55775
|
if (sample) mergeUsageSample(acc, sample);
|
|
55776
|
+
let backfilled = false;
|
|
55777
|
+
if (credit > 0 && protocol) {
|
|
55778
|
+
const usage = ev["type"] === "message_start" ? ev["message"]?.["usage"] : ev["usage"];
|
|
55779
|
+
const deltaEcho = ev["type"] === "message_delta" && (num2(usage?.["input_tokens"]) ?? 0) <= 0;
|
|
55780
|
+
if (usage && !deltaEcho && backfillHostUsage(protocol, usage, credit)) backfilled = true;
|
|
55781
|
+
}
|
|
55342
55782
|
const out = protocol === "anthropic" ? processAnthropic(ev, rawEvent) : processOpenai(ev, rawEvent);
|
|
55343
|
-
if (out
|
|
55783
|
+
if (backfilled && out === rawEvent + "\n\n") {
|
|
55784
|
+
await write(rebuildEvent(rawEvent, ev));
|
|
55785
|
+
} else if (out.length > 0) {
|
|
55786
|
+
await write(out);
|
|
55787
|
+
}
|
|
55344
55788
|
}
|
|
55345
55789
|
}
|
|
55346
55790
|
if (res.destroyed || res.writableEnded) break;
|
|
55347
55791
|
}
|
|
55792
|
+
if (buf.length > 0 && !res.destroyed && !res.writableEnded) await write(buf);
|
|
55348
55793
|
const rest = flushTails();
|
|
55349
55794
|
if (rest.length > 0 && !res.destroyed && !res.writableEnded) await write(rest);
|
|
55350
55795
|
settleUsage();
|
|
@@ -55389,6 +55834,7 @@ async function pipePluginResponsesWithStrip(stream2, res, session, log2) {
|
|
|
55389
55834
|
if (!res.write(Buffer.from(s3, "utf8"))) {
|
|
55390
55835
|
return new Promise((r) => res.once("drain", () => r()));
|
|
55391
55836
|
}
|
|
55837
|
+
return Promise.resolve();
|
|
55392
55838
|
};
|
|
55393
55839
|
const settleUsage = () => {
|
|
55394
55840
|
if (session && (acc.inputTokens !== void 0 || acc.outputTokens !== void 0 || acc.cachedTokens !== void 0)) {
|
|
@@ -55445,7 +55891,17 @@ async function pipePluginResponsesWithStrip(stream2, res, session, log2) {
|
|
|
55445
55891
|
const type = ev["type"];
|
|
55446
55892
|
if (type === "response.output_text.done" || type === "response.content_part.done" || type === "response.output_item.done" || type === "response.completed" || type === "response.failed" || type === "response.incomplete") {
|
|
55447
55893
|
if (type === "response.completed" || type === "response.failed" || type === "response.incomplete") sawTerminal = true;
|
|
55448
|
-
|
|
55894
|
+
let evOut = ev;
|
|
55895
|
+
let rebuild = containsRenderTagText(jsonStr);
|
|
55896
|
+
if (rebuild) evOut = stripResponsesText(ev);
|
|
55897
|
+
if (type === "response.completed") {
|
|
55898
|
+
const credit = session?.hostCreditTokens ?? 0;
|
|
55899
|
+
const usage = evOut["response"]?.["usage"];
|
|
55900
|
+
if (credit > 0 && usage && backfillHostUsage("responses", usage, credit)) {
|
|
55901
|
+
rebuild = true;
|
|
55902
|
+
}
|
|
55903
|
+
}
|
|
55904
|
+
const out = rebuild ? rebuildEvent(rawEvent, evOut) : rawEvent + "\n\n";
|
|
55449
55905
|
await write(flushTail(out));
|
|
55450
55906
|
continue;
|
|
55451
55907
|
}
|
|
@@ -55526,8 +55982,10 @@ async function pipePluginJson(stream2, res, session, protocol) {
|
|
|
55526
55982
|
}
|
|
55527
55983
|
reader.releaseLock();
|
|
55528
55984
|
const text = Buffer.concat(chunks).toString("utf8");
|
|
55985
|
+
let json;
|
|
55986
|
+
let mutated = false;
|
|
55529
55987
|
try {
|
|
55530
|
-
|
|
55988
|
+
json = JSON.parse(text);
|
|
55531
55989
|
const usage = json["usage"];
|
|
55532
55990
|
if (session && usage) {
|
|
55533
55991
|
const input = num2(usage["prompt_tokens"]) ?? num2(usage["input_tokens"]);
|
|
@@ -55538,18 +55996,21 @@ async function pipePluginJson(stream2, res, session, protocol) {
|
|
|
55538
55996
|
cachedTokens: num2(usage["prompt_tokens_details"]?.["cached_tokens"]) ?? num2(usage["input_tokens_details"]?.["cached_tokens"]) ?? num2(usage["cache_read_input_tokens"])
|
|
55539
55997
|
}, protocol);
|
|
55540
55998
|
markDirty(session);
|
|
55999
|
+
const credit = session.hostCreditTokens ?? 0;
|
|
56000
|
+
if (credit > 0 && protocol && backfillHostUsage(protocol, usage, credit)) {
|
|
56001
|
+
mutated = true;
|
|
56002
|
+
}
|
|
55541
56003
|
}
|
|
55542
56004
|
}
|
|
55543
56005
|
} catch {
|
|
55544
56006
|
}
|
|
55545
|
-
if (containsRenderTagText(text)) {
|
|
55546
|
-
|
|
55547
|
-
|
|
55548
|
-
|
|
55549
|
-
|
|
55550
|
-
|
|
55551
|
-
|
|
55552
|
-
}
|
|
56007
|
+
if (json && containsRenderTagText(text)) {
|
|
56008
|
+
json = protocol === "responses" ? stripResponsesText(json) : protocol === "anthropic" ? stripAnthropicText(json) : stripOpenaiChatText(json);
|
|
56009
|
+
mutated = true;
|
|
56010
|
+
}
|
|
56011
|
+
if (mutated && json) {
|
|
56012
|
+
res.end(Buffer.from(JSON.stringify(json), "utf8"));
|
|
56013
|
+
return;
|
|
55553
56014
|
}
|
|
55554
56015
|
res.end(text);
|
|
55555
56016
|
}
|
|
@@ -56223,13 +56684,24 @@ function parseIpLiteral(s3) {
|
|
|
56223
56684
|
if (t.includes(":")) return t;
|
|
56224
56685
|
return null;
|
|
56225
56686
|
}
|
|
56687
|
+
function normalizeIpLiteral(ip) {
|
|
56688
|
+
const t = ip.trim().toLowerCase();
|
|
56689
|
+
if (!t.startsWith("::ffff:")) return t;
|
|
56690
|
+
const dotted = t.match(/^::ffff:(\d{1,3}(?:\.\d{1,3}){3})$/);
|
|
56691
|
+
if (dotted) return dotted[1];
|
|
56692
|
+
const hex = t.match(/^::ffff:0*([0-9a-f]{1,4}):0*([0-9a-f]{1,4})$/);
|
|
56693
|
+
if (hex) {
|
|
56694
|
+
const hi2 = parseInt(hex[1], 16);
|
|
56695
|
+
const lo2 = parseInt(hex[2], 16);
|
|
56696
|
+
return `${hi2 >> 8 & 255}.${hi2 & 255}.${lo2 >> 8 & 255}.${lo2 & 255}`;
|
|
56697
|
+
}
|
|
56698
|
+
return t;
|
|
56699
|
+
}
|
|
56226
56700
|
function classifyIp(ip) {
|
|
56227
|
-
const lit = parseIpLiteral(ip);
|
|
56701
|
+
const lit = normalizeIpLiteral(parseIpLiteral(ip) ?? "");
|
|
56228
56702
|
if (!lit) return "public";
|
|
56229
56703
|
if (lit.includes(":")) {
|
|
56230
56704
|
if (lit === "::1" || lit === "::") return "loopback";
|
|
56231
|
-
const mapped = lit.match(/^::ffff:(\d+\.\d+\.\d+\.\d+)$/);
|
|
56232
|
-
if (mapped) return classifyIp(mapped[1]);
|
|
56233
56705
|
if (lit.startsWith("fe8") || lit.startsWith("fe9") || lit.startsWith("fea") || lit.startsWith("feb")) return "linkLocal";
|
|
56234
56706
|
if (lit.startsWith("fc") || lit.startsWith("fd")) return "private";
|
|
56235
56707
|
return "public";
|
|
@@ -56280,7 +56752,7 @@ async function checkTunnelDestination(origin, ctx) {
|
|
|
56280
56752
|
let ips;
|
|
56281
56753
|
const literal = parseIpLiteral(host);
|
|
56282
56754
|
if (literal) {
|
|
56283
|
-
ips = [literal];
|
|
56755
|
+
ips = [normalizeIpLiteral(literal)];
|
|
56284
56756
|
} else {
|
|
56285
56757
|
try {
|
|
56286
56758
|
ips = await (ctx.resolveHost ?? dnsResolveHost)(host);
|
|
@@ -56291,7 +56763,7 @@ async function checkTunnelDestination(origin, ctx) {
|
|
|
56291
56763
|
}
|
|
56292
56764
|
if (ctx.selfPort !== void 0 && port === ctx.selfPort) {
|
|
56293
56765
|
const mine = (ctx.localIps ?? localMachineIps)();
|
|
56294
|
-
if (ips.some((ip) => mine.has(ip.toLowerCase()) || mine.has(`::ffff:${ip.toLowerCase()}`))) {
|
|
56766
|
+
if (ips.some((ip) => ip === "0.0.0.0" || ip === "::" || mine.has(ip.toLowerCase()) || mine.has(`::ffff:${ip.toLowerCase()}`))) {
|
|
56295
56767
|
return { ok: false, code: "self", message: "the bili tunnel may not target the proxy itself" };
|
|
56296
56768
|
}
|
|
56297
56769
|
}
|
|
@@ -57168,10 +57640,7 @@ function resolveUpstream(_opts, reqUrl, req) {
|
|
|
57168
57640
|
if (reqUrl.startsWith("http://") || reqUrl.startsWith("https://")) {
|
|
57169
57641
|
try {
|
|
57170
57642
|
const u2 = new URL(reqUrl);
|
|
57171
|
-
|
|
57172
|
-
if (u2.host.toLowerCase() !== ownHost) {
|
|
57173
|
-
return { upstream: `${u2.protocol}//${u2.host}`, rewrittenUrl: reqUrl, tunnel: true };
|
|
57174
|
-
}
|
|
57643
|
+
return { upstream: `${u2.protocol}//${u2.host}`, rewrittenUrl: reqUrl, tunnel: true };
|
|
57175
57644
|
} catch {
|
|
57176
57645
|
}
|
|
57177
57646
|
}
|
|
@@ -57387,6 +57856,16 @@ function restoreOutputBudget(parsed, session, log2) {
|
|
|
57387
57856
|
log2("info", `[${session.id}] output budget restored ${value} -> ${highWater} (#546: client shrank it from its raw-history estimate)`);
|
|
57388
57857
|
}
|
|
57389
57858
|
}
|
|
57859
|
+
var SIDE_REQUEST_GUARD_TOLERANCE = 1.15;
|
|
57860
|
+
function sideRequestGuard(parsed, protocol, modelContextLimit, learnedLimit) {
|
|
57861
|
+
let limit = modelContextLimit;
|
|
57862
|
+
if (typeof learnedLimit === "number" && learnedLimit > 0 && learnedLimit < limit) limit = learnedLimit;
|
|
57863
|
+
const field = outputBudgetField(parsed);
|
|
57864
|
+
const maxOut = field ? parsed[field] : 0;
|
|
57865
|
+
if (limit > 0 && shouldReserveOutputHeadroom(protocol)) limit = reserveOutputHeadroom(limit, maxOut);
|
|
57866
|
+
const estimate = estimateRawBodyTokens(parsed) + imageTokensInParsedBody(protocol, parsed);
|
|
57867
|
+
return { blocked: limit > 0 && estimate >= limit * SIDE_REQUEST_GUARD_TOLERANCE, estimate, limit };
|
|
57868
|
+
}
|
|
57390
57869
|
function isTrustedAdminOrigin(origin, host, trustedHosts) {
|
|
57391
57870
|
if (!host || !trustedHosts.has(host.toLowerCase())) return false;
|
|
57392
57871
|
if (!origin) return true;
|
|
@@ -57468,6 +57947,7 @@ async function handle(req, res, opts, core, config, log2, instanceId, instanceSt
|
|
|
57468
57947
|
opts.proxySource = fresh.proxySource;
|
|
57469
57948
|
opts.proxyFallback = fresh.proxyFallback;
|
|
57470
57949
|
opts.compress = fresh.compress;
|
|
57950
|
+
opts.compat = fresh.compat;
|
|
57471
57951
|
resetProxyCache();
|
|
57472
57952
|
for (const k2 of Object.keys(opts.routes)) delete opts.routes[k2];
|
|
57473
57953
|
Object.assign(opts.routes, loadRoutes());
|
|
@@ -57508,7 +57988,7 @@ async function handle(req, res, opts, core, config, log2, instanceId, instanceSt
|
|
|
57508
57988
|
try {
|
|
57509
57989
|
const result = await fetchWithTimeout(targetUrl2, {
|
|
57510
57990
|
method: "HEAD",
|
|
57511
|
-
...proxyUrl ? { dispatcher: proxyDispatcher(proxyUrl) } : {}
|
|
57991
|
+
...proxyUrl ? { dispatcher: proxyDispatcher(proxyUrl, 15e3) } : {}
|
|
57512
57992
|
}, 15e3);
|
|
57513
57993
|
result.clearTimer();
|
|
57514
57994
|
recordUpstreamConnection(targetUrl2, proxyUrl);
|
|
@@ -57531,7 +58011,7 @@ async function handle(req, res, opts, core, config, log2, instanceId, instanceSt
|
|
|
57531
58011
|
res.end(JSON.stringify({ ok: false, error: "conversationId query parameter is required" }));
|
|
57532
58012
|
return;
|
|
57533
58013
|
}
|
|
57534
|
-
return handlePluginStatus(conversationId2, res, params.get("fallback") === "latest");
|
|
58014
|
+
return handlePluginStatus(conversationId2, res, { core, config, log: log2 }, params.get("fallback") === "latest");
|
|
57535
58015
|
}
|
|
57536
58016
|
if (req.method === "POST" && req.url === "/__bili/plugin/tool") {
|
|
57537
58017
|
try {
|
|
@@ -57787,6 +58267,25 @@ ${bodyBuffer.toString("utf8")}`);
|
|
|
57787
58267
|
const pluginMode = pluginAgent !== void 0;
|
|
57788
58268
|
restoreOutputBudget(parsed, session, log2);
|
|
57789
58269
|
if (!countTokens && !responsesCompact && protocol !== null && isSideRequest(parsed)) {
|
|
58270
|
+
const reqModel2 = parsed.model;
|
|
58271
|
+
const learnedMap2 = session.metadata.learnedContextLimits;
|
|
58272
|
+
const learnedLimit2 = (reqModel2 && learnedMap2 ? learnedMap2[reqModel2] : void 0) ?? session.metadata.learnedContextLimit;
|
|
58273
|
+
const guard = sideRequestGuard(parsed, protocol, reqConfig.modelContextLimit, learnedLimit2);
|
|
58274
|
+
if (guard.blocked) {
|
|
58275
|
+
log2("warn", `[${session.id}] side request (~${guard.estimate} tokens) \u2265 effective window ${guard.limit} (model=${reqModel2 ?? "?"}) \u2014 NOT forwarded: guaranteed upstream 400 (side requests bypass preflight by design, #388)`);
|
|
58276
|
+
if (!res.headersSent && !res.writableEnded && !res.destroyed) {
|
|
58277
|
+
res.writeHead(413, { "content-type": "application/json" });
|
|
58278
|
+
res.end(JSON.stringify({
|
|
58279
|
+
error: {
|
|
58280
|
+
type: "server_error",
|
|
58281
|
+
code: "side_request_payload_too_large",
|
|
58282
|
+
message: `side request payload ~${guard.estimate} tokens reaches the effective context window ${guard.limit} (model=${reqModel2 ?? "unknown"}); NOT forwarded \u2014 side requests (max_tokens<=${SIDE_REQUEST_MAX_TOKENS}) bypass compression by design (#388). Shrink the conversation or raise the model's context window.`,
|
|
58283
|
+
retryable: false
|
|
58284
|
+
}
|
|
58285
|
+
}));
|
|
58286
|
+
}
|
|
58287
|
+
return;
|
|
58288
|
+
}
|
|
57790
58289
|
log2("info", `[${session.id}] side request (max_tokens<=${SIDE_REQUEST_MAX_TOKENS}) \u2192 passthrough + tag strip only, kernel state untouched`);
|
|
57791
58290
|
const sidePrepared = {
|
|
57792
58291
|
body: bodyBuffer,
|
|
@@ -57876,19 +58375,37 @@ ${bodyBuffer.toString("utf8")}`);
|
|
|
57876
58375
|
instanceId
|
|
57877
58376
|
);
|
|
57878
58377
|
if (isPreflightFailFast(outcome)) {
|
|
57879
|
-
if (outcome.respond && !res.
|
|
57880
|
-
res.
|
|
57881
|
-
|
|
57882
|
-
|
|
57883
|
-
|
|
57884
|
-
|
|
57885
|
-
|
|
57886
|
-
|
|
57887
|
-
|
|
57888
|
-
|
|
57889
|
-
|
|
58378
|
+
if (outcome.respond && !res.destroyed) {
|
|
58379
|
+
if (res.headersSent) {
|
|
58380
|
+
if (prepared.stream) {
|
|
58381
|
+
emitPreflightError(res, prepared.protocol, { message: outcome.message, retryable: outcome.retryable }, (m2) => log2("warn", m2));
|
|
58382
|
+
} else {
|
|
58383
|
+
try {
|
|
58384
|
+
res.end(JSON.stringify({
|
|
58385
|
+
error: {
|
|
58386
|
+
type: "server_error",
|
|
58387
|
+
code: "preflight_compress_failed",
|
|
58388
|
+
message: outcome.message,
|
|
58389
|
+
retryable: outcome.retryable
|
|
58390
|
+
}
|
|
58391
|
+
}));
|
|
58392
|
+
} catch {
|
|
58393
|
+
}
|
|
57890
58394
|
}
|
|
57891
|
-
}
|
|
58395
|
+
} else {
|
|
58396
|
+
res.writeHead(outcome.status, {
|
|
58397
|
+
"content-type": "application/json",
|
|
58398
|
+
...outcome.status === 503 ? { "retry-after": "30" } : {}
|
|
58399
|
+
});
|
|
58400
|
+
res.end(JSON.stringify({
|
|
58401
|
+
error: {
|
|
58402
|
+
type: "server_error",
|
|
58403
|
+
code: "preflight_compress_failed",
|
|
58404
|
+
message: outcome.message,
|
|
58405
|
+
retryable: outcome.retryable
|
|
58406
|
+
}
|
|
58407
|
+
}));
|
|
58408
|
+
}
|
|
57892
58409
|
}
|
|
57893
58410
|
return;
|
|
57894
58411
|
}
|
|
@@ -57921,6 +58438,59 @@ function stripKernelSummaries(messages, state) {
|
|
|
57921
58438
|
}
|
|
57922
58439
|
return messages.filter((m2) => !(m2.id ?? "").startsWith("acp_summary_") || !carried.has(m2.id));
|
|
57923
58440
|
}
|
|
58441
|
+
var RESPONSES_TURN_SEPARATOR = "[The exchange between these two assistant turns was compressed.]";
|
|
58442
|
+
function repairResponsesAssistantOrdering(folded, original) {
|
|
58443
|
+
const runOf = /* @__PURE__ */ new Map();
|
|
58444
|
+
const runHasBody = /* @__PURE__ */ new Map();
|
|
58445
|
+
let run = 0;
|
|
58446
|
+
let inRun = false;
|
|
58447
|
+
for (const m2 of original) {
|
|
58448
|
+
if (m2.role === "assistant") {
|
|
58449
|
+
if (!inRun) {
|
|
58450
|
+
run++;
|
|
58451
|
+
inRun = true;
|
|
58452
|
+
}
|
|
58453
|
+
runOf.set(m2.id, run);
|
|
58454
|
+
if (m2.contentType !== "reasoning") runHasBody.set(run, true);
|
|
58455
|
+
} else {
|
|
58456
|
+
inRun = false;
|
|
58457
|
+
}
|
|
58458
|
+
}
|
|
58459
|
+
const survivorCount = /* @__PURE__ */ new Map();
|
|
58460
|
+
for (const m2 of folded) {
|
|
58461
|
+
const r = m2.role === "assistant" ? runOf.get(m2.id) : void 0;
|
|
58462
|
+
if (r !== void 0) survivorCount.set(r, (survivorCount.get(r) ?? 0) + 1);
|
|
58463
|
+
}
|
|
58464
|
+
const out = [];
|
|
58465
|
+
let phase = -1;
|
|
58466
|
+
let seenReasoning = false;
|
|
58467
|
+
let sepSeq = 0;
|
|
58468
|
+
const pushSeparator = () => {
|
|
58469
|
+
sepSeq++;
|
|
58470
|
+
out.push({ id: `acp_turn_sep_${sepSeq}`, role: "user", contentType: "text", text: RESPONSES_TURN_SEPARATOR });
|
|
58471
|
+
phase = -1;
|
|
58472
|
+
seenReasoning = false;
|
|
58473
|
+
};
|
|
58474
|
+
for (const m2 of folded) {
|
|
58475
|
+
if (m2.role !== "assistant") {
|
|
58476
|
+
out.push(m2);
|
|
58477
|
+
phase = -1;
|
|
58478
|
+
seenReasoning = false;
|
|
58479
|
+
continue;
|
|
58480
|
+
}
|
|
58481
|
+
const kind = m2.contentType === "reasoning" ? "reasoning" : m2.contentType === "tool-call" ? "tool-call" : "message";
|
|
58482
|
+
const r = runOf.get(m2.id);
|
|
58483
|
+
if (kind === "reasoning" && r !== void 0 && runHasBody.get(r) && survivorCount.get(r) === 1) continue;
|
|
58484
|
+
if (kind === "reasoning" && (phase > 0 || seenReasoning) || kind === "message" && phase === 2) pushSeparator();
|
|
58485
|
+
out.push(m2);
|
|
58486
|
+
if (kind === "reasoning") {
|
|
58487
|
+
phase = Math.max(phase, 0);
|
|
58488
|
+
seenReasoning = true;
|
|
58489
|
+
} else if (kind === "message") phase = Math.max(phase, 1);
|
|
58490
|
+
else phase = Math.max(phase, 2);
|
|
58491
|
+
}
|
|
58492
|
+
return out;
|
|
58493
|
+
}
|
|
57924
58494
|
function diagTagSummary(messages, sessionId, strategy) {
|
|
57925
58495
|
let textTagged = 0;
|
|
57926
58496
|
let toolTagged = 0;
|
|
@@ -57946,10 +58516,19 @@ function diagNudge(turn, sessionId, tokenCount, limit, model, willInject) {
|
|
|
57946
58516
|
const modelTag = model ? ` model=${model}` : "";
|
|
57947
58517
|
return `[${sessionId}] nudge ${inject}: usage=${pct2} (${tokenCount}/${limit}), growth=${growth}/${floor} (ref=${ref}, interval=${interval}), pendingT1=${pendingT1}/${interval}${modelTag}, reason="${n.reason.slice(0, 120)}"`;
|
|
57948
58518
|
}
|
|
58519
|
+
function armHostUsageCredit(session, originalMessages, processedMessages, log2) {
|
|
58520
|
+
session.hostCreditTokens = 0;
|
|
58521
|
+
if (session.metadata.pluginAgent === "pi") return;
|
|
58522
|
+
session.hostCreditTokens = processedMessages.length > 0 ? Math.max(0, estimateCoreMessages(originalMessages) - estimateCoreMessages(processedMessages)) : 0;
|
|
58523
|
+
if (session.hostCreditTokens > 0) {
|
|
58524
|
+
log2("info", `[${session.id}] host usage backfill armed: +${session.hostCreditTokens} tok (forwarded view is folded); host usage will report the uncompressed baseline`);
|
|
58525
|
+
}
|
|
58526
|
+
}
|
|
57949
58527
|
function prepareAnthropic(parsed, req, opts, core, config, prompts, log2, session, pluginMode) {
|
|
57950
58528
|
const sessionId = session.id;
|
|
57951
58529
|
const stream2 = parsed.stream === true;
|
|
57952
58530
|
++session.stats.requests;
|
|
58531
|
+
session.hostCreditTokens = 0;
|
|
57953
58532
|
const injectTools = opts.compress.injectTool && !pluginMode;
|
|
57954
58533
|
if (isAutoModeClassifier(parsed)) {
|
|
57955
58534
|
log2("info", `[${sessionId}] auto-mode classifier passthrough (skipping compress injection)`);
|
|
@@ -58005,20 +58584,46 @@ function prepareAnthropic(parsed, req, opts, core, config, prompts, log2, sessio
|
|
|
58005
58584
|
log2("warn", `[${sessionId}] kernel transform failed, forwarding unchanged: ${String(err2)}`);
|
|
58006
58585
|
processedMessages = [];
|
|
58007
58586
|
}
|
|
58587
|
+
session.metadata.systemPromptTokens = countSystemAndToolsTokens(extractSystem(systemOut), toolsOut);
|
|
58008
58588
|
snapshotMessages(session, originalMessages);
|
|
58009
58589
|
markDirty(session);
|
|
58010
58590
|
const rebuilt = { ...parsed, messages: rebuiltMessages, system: systemOut, tools: toolsOut };
|
|
58011
58591
|
delete rebuilt.prompt_cache_key;
|
|
58592
|
+
armHostUsageCredit(session, originalMessages, processedMessages, log2);
|
|
58012
58593
|
return { body: JSON.stringify(rebuilt), session, processedMessages, originalMessages, anthropicSystem: parsed.system, protocol: "anthropic", stream: stream2, compressInjected: injectTools, pluginMode, nudge, prompts, renderTags: "text-only" };
|
|
58013
58594
|
}
|
|
58014
58595
|
var OUTPUT_CLAMP_MARGIN_PCT = 0.05;
|
|
58015
58596
|
var OUTPUT_CLAMP_MIN_MARGIN = 2048;
|
|
58016
58597
|
var OUTPUT_CLAMP_FLOOR = 1024;
|
|
58017
58598
|
var EMERGENCY_NUDGE_ESCALATION_PCT = 0.7;
|
|
58599
|
+
function countSystemAndToolsTokens(systemText, tools) {
|
|
58600
|
+
return defaultCountTokens(systemText ?? "") + defaultCountTokens(JSON.stringify(tools ?? []));
|
|
58601
|
+
}
|
|
58018
58602
|
function estimateInputTokens(processedMessages, systemText, tools, lastInputTokens) {
|
|
58019
|
-
const est = estimateCoreMessages(processedMessages) +
|
|
58603
|
+
const est = estimateCoreMessages(processedMessages) + countSystemAndToolsTokens(systemText, tools);
|
|
58020
58604
|
return Math.max(lastInputTokens > 0 ? lastInputTokens : 0, est);
|
|
58021
58605
|
}
|
|
58606
|
+
function estimateWireOverhead(protocol, body) {
|
|
58607
|
+
let parsed;
|
|
58608
|
+
try {
|
|
58609
|
+
parsed = JSON.parse(typeof body === "string" ? body : body.toString("utf8"));
|
|
58610
|
+
} catch {
|
|
58611
|
+
return 0;
|
|
58612
|
+
}
|
|
58613
|
+
const sysRaw = protocol === "responses" ? parsed.instructions : parsed.system;
|
|
58614
|
+
let sysText = "";
|
|
58615
|
+
if (typeof sysRaw === "string") {
|
|
58616
|
+
sysText = sysRaw;
|
|
58617
|
+
} else if (Array.isArray(sysRaw)) {
|
|
58618
|
+
sysText = sysRaw.map((part) => typeof part?.text === "string" ? part.text : "").join("\n");
|
|
58619
|
+
}
|
|
58620
|
+
if (protocol === "openai" && Array.isArray(parsed.messages)) {
|
|
58621
|
+
const hoisted = parsed.messages.filter((m2) => m2.role === "system" || m2.role === "developer").map((m2) => typeof m2.content === "string" ? m2.content : "").join("\n");
|
|
58622
|
+
sysText = sysText ? `${sysText}
|
|
58623
|
+
${hoisted}` : hoisted;
|
|
58624
|
+
}
|
|
58625
|
+
return defaultCountTokens(sysText) + defaultCountTokens(JSON.stringify(parsed.tools ?? []));
|
|
58626
|
+
}
|
|
58022
58627
|
function clampOutputBudget(requested, inputEstimate, nativeWindow) {
|
|
58023
58628
|
const margin = Math.max(OUTPUT_CLAMP_MIN_MARGIN, Math.ceil(inputEstimate * OUTPUT_CLAMP_MARGIN_PCT));
|
|
58024
58629
|
const cap = nativeWindow - inputEstimate - margin;
|
|
@@ -58044,7 +58649,9 @@ function prepareOpenai(parsed, req, opts, core, config, prompts, log2, session,
|
|
|
58044
58649
|
const sessionId = session.id;
|
|
58045
58650
|
const stream2 = parsed.stream === true;
|
|
58046
58651
|
++session.stats.requests;
|
|
58652
|
+
session.hostCreditTokens = 0;
|
|
58047
58653
|
let openaiSystemText = "";
|
|
58654
|
+
let openaiOutboundSystem;
|
|
58048
58655
|
let processedMessages = [];
|
|
58049
58656
|
let originalMessages = [];
|
|
58050
58657
|
let nudge;
|
|
@@ -58085,6 +58692,7 @@ function prepareOpenai(parsed, req, opts, core, config, prompts, log2, session,
|
|
|
58085
58692
|
if (systemText) sysParts.push(systemText);
|
|
58086
58693
|
if (shouldInject) sysParts.push(buildCompressSystemPrompt(prompts));
|
|
58087
58694
|
rebuiltMessages = injectOpenaiSystem(rebuiltMessages, sysParts);
|
|
58695
|
+
openaiOutboundSystem = sysParts.join("\n\n");
|
|
58088
58696
|
if (injectTools) {
|
|
58089
58697
|
toolsOut = injectOpenaiTool(parsed.tools);
|
|
58090
58698
|
}
|
|
@@ -58107,6 +58715,10 @@ function prepareOpenai(parsed, req, opts, core, config, prompts, log2, session,
|
|
|
58107
58715
|
if (stream2 && rebuilt.stream_options === void 0) {
|
|
58108
58716
|
rebuilt.stream_options = { include_usage: true };
|
|
58109
58717
|
}
|
|
58718
|
+
armHostUsageCredit(session, originalMessages, processedMessages, log2);
|
|
58719
|
+
if (!isTitleGen && openaiOutboundSystem !== void 0) {
|
|
58720
|
+
session.metadata.systemPromptTokens = countSystemAndToolsTokens(openaiOutboundSystem, toolsOut);
|
|
58721
|
+
}
|
|
58110
58722
|
snapshotMessages(session, originalMessages);
|
|
58111
58723
|
markDirty(session);
|
|
58112
58724
|
return { body: JSON.stringify(rebuilt), session, processedMessages, originalMessages, protocol: "openai", stream: stream2, compressInjected: injectTools, pluginMode, nudge, prompts, openaiSystemText, renderTags: "text-only" };
|
|
@@ -58115,6 +58727,7 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
|
|
|
58115
58727
|
const sessionId = session.id;
|
|
58116
58728
|
const stream2 = parsed.stream === true;
|
|
58117
58729
|
++session.stats.requests;
|
|
58730
|
+
session.hostCreditTokens = 0;
|
|
58118
58731
|
if (reconcileNativeCompactionBoundary(session)) {
|
|
58119
58732
|
log2("info", `[${sessionId}] reconciled ACP state after native Responses compact boundary`);
|
|
58120
58733
|
}
|
|
@@ -58134,6 +58747,7 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
|
|
|
58134
58747
|
let rebuiltInput = parsed.input;
|
|
58135
58748
|
let toolsOut = parsed.tools;
|
|
58136
58749
|
let transformOk = false;
|
|
58750
|
+
let responsesDevContent;
|
|
58137
58751
|
const typedItems = normalizeResponsesMessageItems(parsed.input);
|
|
58138
58752
|
if (typedItems > 0) {
|
|
58139
58753
|
log2("info", `[${sessionId}] stamped type:"message" on ${typedItems} type-less input item(s) before projection (omp wire form)`);
|
|
@@ -58174,19 +58788,21 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
|
|
|
58174
58788
|
log2("info", diagTagSummary(turn.messages, sessionId, "text-only"));
|
|
58175
58789
|
const willInjectNudge = opts.compress.injectNudge && !!turn.nudge && shouldInject && !isCompactionTrigger && (turn.nudge.shouldInject || emergencyNudge(turn.nudge));
|
|
58176
58790
|
log2("info", diagNudge(turn, sessionId, tokenCount, config.modelContextLimit, parsed.model, willInjectNudge));
|
|
58177
|
-
processedMessages = stripKernelSummaries(turn.messages, turn.state);
|
|
58791
|
+
processedMessages = repairResponsesAssistantOrdering(stripKernelSummaries(turn.messages, turn.state), originalMessages);
|
|
58178
58792
|
reapOrphanBlocks(session, msgs, deactivateBlock);
|
|
58179
58793
|
rebuiltInput = patchResponsesInput(projection, processedMessages);
|
|
58180
58794
|
const forgedSummaries = echoReplaced ? [] : session.metadata.codexForgedSummaries ?? [];
|
|
58181
58795
|
if (shouldInject && !isCompactionTrigger && !process.env.ACP_NO_COMPRESS_PROMPT) {
|
|
58182
58796
|
const prompt = responsesTextProtocol ? buildCompressHybridSystemPrompt(prompts) : buildCompressSystemPrompt(prompts);
|
|
58183
58797
|
const devContent = [...projection.systemParts, ...forgedSummaries, prompt].join("\n\n---\n\n");
|
|
58798
|
+
responsesDevContent = devContent;
|
|
58184
58799
|
rebuiltInput = injectResponsesDeveloperMessage(rebuiltInput, devContent);
|
|
58185
58800
|
if (!process.env.ACP_NO_INJECT_TOOL && injectTools) {
|
|
58186
58801
|
toolsOut = responsesTextProtocol ? injectResponsesTool(parsed.tools, ACP_READONLY_TOOLS_RESPONSES) : injectResponsesTool(parsed.tools);
|
|
58187
58802
|
}
|
|
58188
58803
|
} else if (projection.systemParts.length > 0 || forgedSummaries.length > 0) {
|
|
58189
58804
|
const devContent = [...projection.systemParts, ...forgedSummaries].join("\n\n---\n\n");
|
|
58805
|
+
responsesDevContent = devContent;
|
|
58190
58806
|
rebuiltInput = injectResponsesDeveloperMessage(rebuiltInput, devContent);
|
|
58191
58807
|
}
|
|
58192
58808
|
if (willInjectNudge && turn.nudge) {
|
|
@@ -58244,6 +58860,10 @@ function prepareResponses(parsed, req, opts, core, config, prompts, log2, sessio
|
|
|
58244
58860
|
});
|
|
58245
58861
|
log2("info", `[${sessionId}] responses forward tools=[${fwdTools.join(",")}] injectTool=${injectTools}${pluginMode ? " (plugin mode: wire injection suppressed)" : ""} NO_INJECT_TOOL=${!!process.env.ACP_NO_INJECT_TOOL} NO_COMPRESS_PROMPT=${!!process.env.ACP_NO_COMPRESS_PROMPT}`);
|
|
58246
58862
|
}
|
|
58863
|
+
armHostUsageCredit(session, originalMessages, processedMessages, log2);
|
|
58864
|
+
if (transformOk) {
|
|
58865
|
+
session.metadata.systemPromptTokens = countSystemAndToolsTokens(responsesDevContent ?? "", toolsOut);
|
|
58866
|
+
}
|
|
58247
58867
|
snapshotMessages(session, originalMessages);
|
|
58248
58868
|
markDirty(session);
|
|
58249
58869
|
return {
|
|
@@ -58330,7 +58950,7 @@ function prepareResponsesCompact(body, parsed, session, req, core, config, log2)
|
|
|
58330
58950
|
session.state = prevState;
|
|
58331
58951
|
return base;
|
|
58332
58952
|
}
|
|
58333
|
-
const processed = stripKernelSummaries(turn.messages, turn.state);
|
|
58953
|
+
const processed = repairResponsesAssistantOrdering(stripKernelSummaries(turn.messages, turn.state), projection.msgs);
|
|
58334
58954
|
const output = patchResponsesInput(projection, processed);
|
|
58335
58955
|
if (typeof output === "string") {
|
|
58336
58956
|
session.state = prevState;
|
|
@@ -58415,6 +59035,12 @@ function logUpstreamProxyDecision(opts, upstreamUrl, decision) {
|
|
|
58415
59035
|
const via = decision.proxy ? `via ${maskUrlForLog(decision.proxy)}` : "direct";
|
|
58416
59036
|
logMsg(opts, "info", `[upstream-proxy] ${maskHostPortForLog(host)} ${via} (source=${decision.source})`);
|
|
58417
59037
|
}
|
|
59038
|
+
function inferWireProtocol(path18) {
|
|
59039
|
+
const p2 = path18.split("?", 2)[0];
|
|
59040
|
+
if (p2.endsWith("/chat/completions")) return "openai";
|
|
59041
|
+
if (p2.endsWith("/responses") || p2.endsWith("/responses/compact")) return "responses";
|
|
59042
|
+
return null;
|
|
59043
|
+
}
|
|
58418
59044
|
function buildForwardTarget(req, opts, route, affinity, hopMarker) {
|
|
58419
59045
|
const reqUrl = req.url ?? "";
|
|
58420
59046
|
const isAbsoluteUrl = /^https?:\/\//i.test(reqUrl);
|
|
@@ -58446,12 +59072,53 @@ function buildForwardTarget(req, opts, route, affinity, hopMarker) {
|
|
|
58446
59072
|
function isPreflightFailFast(outcome) {
|
|
58447
59073
|
return "failFast" in outcome;
|
|
58448
59074
|
}
|
|
59075
|
+
var PREFLIGHT_HOLD_GRACE_DEFAULT_MS = 3e4;
|
|
59076
|
+
var PREFLIGHT_KEEPALIVE_MS = 15e3;
|
|
59077
|
+
function preflightHoldGraceMs() {
|
|
59078
|
+
const raw = process.env.BILI_PREFLIGHT_HOLD_MS;
|
|
59079
|
+
if (!raw) return PREFLIGHT_HOLD_GRACE_DEFAULT_MS;
|
|
59080
|
+
const v2 = Number(raw);
|
|
59081
|
+
return Number.isFinite(v2) && v2 >= 0 ? Math.floor(v2) : PREFLIGHT_HOLD_GRACE_DEFAULT_MS;
|
|
59082
|
+
}
|
|
59083
|
+
function beginPreflightHold(res, prepared, log2) {
|
|
59084
|
+
if (res.headersSent || res.destroyed || res.writableEnded) return void 0;
|
|
59085
|
+
const sid = prepared.session.id;
|
|
59086
|
+
const keepAlive = prepared.stream ? ": bili-preflight\n\n" : " ";
|
|
59087
|
+
try {
|
|
59088
|
+
if (prepared.stream) {
|
|
59089
|
+
res.writeHead(200, {
|
|
59090
|
+
"content-type": "text/event-stream",
|
|
59091
|
+
"cache-control": "no-cache",
|
|
59092
|
+
"x-accel-buffering": "no",
|
|
59093
|
+
"x-bili-preflight": "compressing"
|
|
59094
|
+
});
|
|
59095
|
+
} else {
|
|
59096
|
+
res.writeHead(200, { "content-type": "application/json", "x-bili-preflight": "compressing" });
|
|
59097
|
+
}
|
|
59098
|
+
} catch {
|
|
59099
|
+
return void 0;
|
|
59100
|
+
}
|
|
59101
|
+
log2("info", `[${sid}] preflight still running after ${preflightHoldGraceMs()}ms grace \u2014 committed early ${prepared.stream ? "SSE" : "JSON"} headers + keep-alive to hold the client (#568)`);
|
|
59102
|
+
try {
|
|
59103
|
+
res.write(keepAlive);
|
|
59104
|
+
} catch {
|
|
59105
|
+
}
|
|
59106
|
+
const iv = setInterval(() => {
|
|
59107
|
+
try {
|
|
59108
|
+
res.write(keepAlive);
|
|
59109
|
+
} catch {
|
|
59110
|
+
clearInterval(iv);
|
|
59111
|
+
}
|
|
59112
|
+
}, PREFLIGHT_KEEPALIVE_MS);
|
|
59113
|
+
return () => clearInterval(iv);
|
|
59114
|
+
}
|
|
58449
59115
|
async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, core, config, model, route, affinity, log2, instanceId) {
|
|
58450
59116
|
const session = prepared.session;
|
|
58451
59117
|
const limit = config.modelContextLimit;
|
|
58452
59118
|
const imageTokens = imageTokensInRawBody(prepared.protocol, prepared.body);
|
|
58453
59119
|
const textEstimate = estimateCoreMessages(prepared.processedMessages);
|
|
58454
|
-
const
|
|
59120
|
+
const overheadEstimate = estimateWireOverhead(prepared.protocol, prepared.body);
|
|
59121
|
+
const payloadEstimate = textEstimate + overheadEstimate + imageTokens;
|
|
58455
59122
|
const tokenCount = Math.max(session.stats.lastInputTokens, payloadEstimate);
|
|
58456
59123
|
if (limit <= 0 || !model || tokenCount < limit) return prepared;
|
|
58457
59124
|
const learnedMap = session.metadata.learnedContextLimits;
|
|
@@ -58467,12 +59134,9 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
|
|
|
58467
59134
|
log2("error", `[${session.id}] preflight fail-fast ${status2} (retryable=${retryable}): ${message}`);
|
|
58468
59135
|
return { failFast: true, status: status2, message, retryable, respond: !res.writableEnded };
|
|
58469
59136
|
};
|
|
58470
|
-
if ((prepared.nudge?.compressibleRanges ?? []).length === 0) {
|
|
58471
|
-
|
|
58472
|
-
|
|
58473
|
-
return prepared;
|
|
58474
|
-
}
|
|
58475
|
-
return failFast(502, "no part of the conversation is compressible (nothing left to fold)", false);
|
|
59137
|
+
if ((prepared.nudge?.compressibleRanges ?? []).length === 0 && payloadEstimate < limit) {
|
|
59138
|
+
log2("warn", `[${session.id}] preflight trigger fired on a stale baseline (~${tokenCount}) but the payload fits (~${payloadEstimate}/${limit}); forwarding as-is`);
|
|
59139
|
+
return prepared;
|
|
58476
59140
|
}
|
|
58477
59141
|
log2("warn", `[${session.id}] context ${tokenCount} tokens exceeds model window ${limit} (model=${model}); preflight compressing before forward`);
|
|
58478
59142
|
const { upstreamUrl, headers, proxyUrl } = buildForwardTarget(req, opts, route, affinity, instanceId);
|
|
@@ -58481,31 +59145,42 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
|
|
|
58481
59145
|
if (!res.writableEnded) clientAbort.abort();
|
|
58482
59146
|
});
|
|
58483
59147
|
const started = Date.now();
|
|
58484
|
-
|
|
58485
|
-
|
|
58486
|
-
|
|
58487
|
-
|
|
58488
|
-
|
|
58489
|
-
|
|
58490
|
-
|
|
58491
|
-
|
|
58492
|
-
|
|
58493
|
-
|
|
58494
|
-
|
|
58495
|
-
|
|
58496
|
-
|
|
58497
|
-
|
|
58498
|
-
|
|
58499
|
-
|
|
58500
|
-
|
|
58501
|
-
|
|
58502
|
-
|
|
58503
|
-
|
|
58504
|
-
|
|
58505
|
-
|
|
58506
|
-
|
|
58507
|
-
|
|
58508
|
-
|
|
59148
|
+
let stopHold;
|
|
59149
|
+
const holdTimer = setTimeout(() => {
|
|
59150
|
+
stopHold = beginPreflightHold(res, prepared, log2);
|
|
59151
|
+
}, preflightHoldGraceMs());
|
|
59152
|
+
holdTimer.unref();
|
|
59153
|
+
let result;
|
|
59154
|
+
try {
|
|
59155
|
+
result = await preflightCompress(
|
|
59156
|
+
{
|
|
59157
|
+
core,
|
|
59158
|
+
session,
|
|
59159
|
+
config,
|
|
59160
|
+
prompts: prepared.prompts ?? defaultPrompts,
|
|
59161
|
+
protocol: prepared.protocol,
|
|
59162
|
+
url: upstreamUrl,
|
|
59163
|
+
headers,
|
|
59164
|
+
model,
|
|
59165
|
+
proxyUrl,
|
|
59166
|
+
signal: clientAbort.signal,
|
|
59167
|
+
log: log2,
|
|
59168
|
+
imageFloor: imageTokens,
|
|
59169
|
+
wireOverhead: overheadEstimate
|
|
59170
|
+
},
|
|
59171
|
+
prepared.originalMessages
|
|
59172
|
+
);
|
|
59173
|
+
} finally {
|
|
59174
|
+
clearTimeout(holdTimer);
|
|
59175
|
+
stopHold?.();
|
|
59176
|
+
}
|
|
59177
|
+
if (result.compressedRanges > 0) {
|
|
59178
|
+
log2("info", `[${session.id}] preflight compressed ${result.compressedRanges} range(s), ~${result.savedTokens} tokens saved (${tokenCount} \u2192 ${session.stats.lastInputTokens}) in ${Date.now() - started}ms; rebuilding payload`);
|
|
59179
|
+
const rebuilt = runPrepare();
|
|
59180
|
+
session.stats.requests -= 1;
|
|
59181
|
+
if (estimateCoreMessages(rebuilt.processedMessages) + overheadEstimate + imageTokens < limit) return rebuilt;
|
|
59182
|
+
} else if (estimateCoreMessages(prepared.processedMessages) + overheadEstimate + imageTokens < limit) {
|
|
59183
|
+
log2("warn", `[${session.id}] preflight made no progress but the payload fits; forwarding as-is`);
|
|
58509
59184
|
return prepared;
|
|
58510
59185
|
}
|
|
58511
59186
|
const f2 = result.failure;
|
|
@@ -58514,16 +59189,40 @@ async function preflightCompressIfNeeded(prepared, runPrepare, req, res, opts, c
|
|
|
58514
59189
|
return { failFast: true, status: 0, message: f2.detail, retryable: false, respond: false };
|
|
58515
59190
|
}
|
|
58516
59191
|
const status = f2?.kind === "upstream" && f2.status === 429 ? 503 : 502;
|
|
58517
|
-
return failFast(status, f2?.detail ?? "
|
|
59192
|
+
return failFast(status, f2?.detail ?? "the payload still exceeds the window after preflight compression", status === 503);
|
|
58518
59193
|
}
|
|
58519
59194
|
async function forward(req, res, opts, body, prepared, core, config, log2, route, instanceId, affinity) {
|
|
58520
59195
|
if (prepared?.codexForge) {
|
|
58521
59196
|
log2("info", `[${prepared.session.id}] codex compact served locally (${prepared.codexForge.kind}); upstream not contacted`);
|
|
58522
|
-
res.writeHead(200, { "content-type": prepared.codexForge.contentType });
|
|
59197
|
+
if (!res.headersSent) res.writeHead(200, { "content-type": prepared.codexForge.contentType });
|
|
58523
59198
|
res.end(prepared.codexForge.body);
|
|
58524
59199
|
return;
|
|
58525
59200
|
}
|
|
59201
|
+
let wireBody = body;
|
|
59202
|
+
let compatRoles = null;
|
|
59203
|
+
let compatProtocol = null;
|
|
58526
59204
|
const { upstreamUrl, headers, proxyUrl } = buildForwardTarget(req, opts, route, affinity, prepared !== null ? instanceId : void 0);
|
|
59205
|
+
if (typeof body === "string") {
|
|
59206
|
+
const configured = resolveCompatRoles(opts.routes, upstreamUrl, opts.compat?.roles);
|
|
59207
|
+
const learned = prepared?.session.metadata.learnedCompatRoles ?? {};
|
|
59208
|
+
const roles = { ...configured, ...learned };
|
|
59209
|
+
const protocol = prepared?.protocol ?? route?.explicitProtocol ?? inferWireProtocol(req.url ?? "");
|
|
59210
|
+
if (protocol === "openai" || protocol === "responses") {
|
|
59211
|
+
compatProtocol = protocol;
|
|
59212
|
+
if (Object.keys(roles).length > 0) {
|
|
59213
|
+
compatRoles = roles;
|
|
59214
|
+
const applied = applyCompatRoles(body, protocol, roles);
|
|
59215
|
+
if (applied.rewritten > 0) {
|
|
59216
|
+
wireBody = applied.body;
|
|
59217
|
+
log2("info", `[${prepared?.session.id ?? "passthrough"}] [compat] rewrote ${applied.rewritten} message role(s) per compat.roles (${Object.entries(roles).map(([f2, t]) => `${f2}\u2192${t}`).join(",")})`);
|
|
59218
|
+
}
|
|
59219
|
+
}
|
|
59220
|
+
}
|
|
59221
|
+
}
|
|
59222
|
+
const wireTransform = compatProtocol ? (b2) => {
|
|
59223
|
+
if (compatRoles) applyCompatRolesJson(b2, compatProtocol, compatRoles);
|
|
59224
|
+
return b2;
|
|
59225
|
+
} : void 0;
|
|
58527
59226
|
log2("info", `forward ${req.method} \u2192 ${maskUrlForLog(upstreamUrl)}`);
|
|
58528
59227
|
if (process.env.ACP_DEBUG && prepared) {
|
|
58529
59228
|
const sid = prepared.session.id;
|
|
@@ -58538,9 +59237,9 @@ async function forward(req, res, opts, body, prepared, core, config, log2, route
|
|
|
58538
59237
|
}
|
|
58539
59238
|
}
|
|
58540
59239
|
}
|
|
58541
|
-
if (typeof
|
|
59240
|
+
if (typeof wireBody === "string" && (opts.debug || bodyDumpEnabled())) {
|
|
58542
59241
|
try {
|
|
58543
|
-
const parsed = JSON.parse(
|
|
59242
|
+
const parsed = JSON.parse(wireBody);
|
|
58544
59243
|
if (opts.debug) {
|
|
58545
59244
|
const toolNames = (parsed.tools ?? []).map((t) => {
|
|
58546
59245
|
const fn = t.function;
|
|
@@ -58557,10 +59256,10 @@ async function forward(req, res, opts, body, prepared, core, config, log2, route
|
|
|
58557
59256
|
const sid = prepared?.session.id ?? "unknown";
|
|
58558
59257
|
const out = `${dumpDir}/req-${Date.now()}-${safeSessionId(sid)}.json`;
|
|
58559
59258
|
try {
|
|
58560
|
-
const pretty = JSON.stringify(JSON.parse(
|
|
59259
|
+
const pretty = JSON.stringify(JSON.parse(wireBody), null, 2);
|
|
58561
59260
|
fs8.writeFileSync(out, pretty);
|
|
58562
59261
|
} catch {
|
|
58563
|
-
fs8.writeFileSync(out,
|
|
59262
|
+
fs8.writeFileSync(out, wireBody);
|
|
58564
59263
|
}
|
|
58565
59264
|
log2("info", `[debug] forwarded body written to ${out}`);
|
|
58566
59265
|
}
|
|
@@ -58589,7 +59288,7 @@ async function forward(req, res, opts, body, prepared, core, config, log2, route
|
|
|
58589
59288
|
if (rawBase) {
|
|
58590
59289
|
try {
|
|
58591
59290
|
const hdrText = Object.entries(maskHeadersForLog(headers)).map(([k2, v2]) => `${k2}: ${v2}`).join("\n");
|
|
58592
|
-
const bodyText = req.method === "GET" || req.method === "HEAD" ? "" : typeof
|
|
59291
|
+
const bodyText = req.method === "GET" || req.method === "HEAD" ? "" : typeof wireBody === "string" ? wireBody : Buffer.from(wireBody).toString("utf8");
|
|
58593
59292
|
const reqPath = `${rawBase}-REQ.txt`;
|
|
58594
59293
|
fs8.writeFileSync(reqPath, `${req.method ?? "POST"} ${maskUrlForLog(upstreamUrl)}
|
|
58595
59294
|
${hdrText}
|
|
@@ -58604,7 +59303,7 @@ ${bodyText}`);
|
|
|
58604
59303
|
const init = {
|
|
58605
59304
|
method: req.method ?? "GET",
|
|
58606
59305
|
headers,
|
|
58607
|
-
body: req.method === "GET" || req.method === "HEAD" ? void 0 :
|
|
59306
|
+
body: req.method === "GET" || req.method === "HEAD" ? void 0 : wireBody
|
|
58608
59307
|
};
|
|
58609
59308
|
if (dispatcher) init.dispatcher = dispatcher;
|
|
58610
59309
|
const clientAbort = new AbortController();
|
|
@@ -58619,6 +59318,49 @@ ${bodyText}`);
|
|
|
58619
59318
|
recordUpstreamConnection(upstreamUrl, proxyUrl, error);
|
|
58620
59319
|
throw new Error(`upstream request failed: ${formatUpstreamError(error, upstreamUrl, proxyUrl)}`, { cause: error });
|
|
58621
59320
|
}
|
|
59321
|
+
if (compatProtocol && typeof wireBody === "string" && upstreamResult.response.status === 400 && upstreamResult.response.body) {
|
|
59322
|
+
let roleErrText = null;
|
|
59323
|
+
try {
|
|
59324
|
+
roleErrText = (await readStreamToBuffer(upstreamResult.response.body)).toString("utf8");
|
|
59325
|
+
} catch {
|
|
59326
|
+
roleErrText = null;
|
|
59327
|
+
}
|
|
59328
|
+
if (roleErrText !== null) {
|
|
59329
|
+
upstreamResult = {
|
|
59330
|
+
response: new Response(roleErrText, {
|
|
59331
|
+
status: upstreamResult.response.status,
|
|
59332
|
+
statusText: upstreamResult.response.statusText,
|
|
59333
|
+
headers: new Headers(upstreamResult.response.headers)
|
|
59334
|
+
}),
|
|
59335
|
+
clearTimer: upstreamResult.clearTimer
|
|
59336
|
+
};
|
|
59337
|
+
const rejection = detectRoleRejection(upstreamResult.response.status, roleErrText);
|
|
59338
|
+
if (rejection && rejection.role !== "system") {
|
|
59339
|
+
const fixed = applyCompatRoles(wireBody, compatProtocol, { [rejection.role]: "system" });
|
|
59340
|
+
if (fixed.rewritten > 0) {
|
|
59341
|
+
try {
|
|
59342
|
+
const retry = await fetchWithTimeout(upstreamUrl, { ...init, body: fixed.body }, void 0, clientAbort.signal);
|
|
59343
|
+
if (retry.response.ok) {
|
|
59344
|
+
upstreamResult.clearTimer();
|
|
59345
|
+
const s3 = prepared?.session;
|
|
59346
|
+
if (s3) {
|
|
59347
|
+
const prev = s3.metadata.learnedCompatRoles ?? {};
|
|
59348
|
+
s3.metadata.learnedCompatRoles = { ...prev, [rejection.role]: "system" };
|
|
59349
|
+
markDirty(s3);
|
|
59350
|
+
}
|
|
59351
|
+
compatRoles = { ...compatRoles, [rejection.role]: "system" };
|
|
59352
|
+
const providerKey = new URL(upstreamUrl).origin;
|
|
59353
|
+
log2("info", `[${prepared?.session.id ?? "passthrough"}] [compat] upstream rejected role "${rejection.role}" \u2014 auto-rewrote ${fixed.rewritten} message role(s) to "system", retry OK (remembered for this session only). To make permanent, add: {"providers":{"${providerKey}":{"compat":{"roles":{"${rejection.role}":"system"}}}}`);
|
|
59354
|
+
upstreamResult = retry;
|
|
59355
|
+
} else {
|
|
59356
|
+
retry.clearTimer();
|
|
59357
|
+
}
|
|
59358
|
+
} catch {
|
|
59359
|
+
}
|
|
59360
|
+
}
|
|
59361
|
+
}
|
|
59362
|
+
}
|
|
59363
|
+
}
|
|
58622
59364
|
const { response: upstream, clearTimer: clearUpstreamTimer } = upstreamResult;
|
|
58623
59365
|
const respHeaders = {};
|
|
58624
59366
|
const respConnNamed = connectionNamedHeaders(upstream.headers.get("connection") ?? void 0);
|
|
@@ -58664,9 +59406,10 @@ ${hdrText}
|
|
|
58664
59406
|
if (info.isOverflow) {
|
|
58665
59407
|
let reqModel;
|
|
58666
59408
|
let rejectedImageTokens = 0;
|
|
59409
|
+
let parsedBody;
|
|
58667
59410
|
try {
|
|
58668
59411
|
const rawBody = typeof prepared.body === "string" ? prepared.body : prepared.body.toString("utf8");
|
|
58669
|
-
|
|
59412
|
+
parsedBody = JSON.parse(rawBody);
|
|
58670
59413
|
reqModel = typeof parsedBody.model === "string" ? parsedBody.model : void 0;
|
|
58671
59414
|
rejectedImageTokens = imageTokensInParsedBody(prepared.protocol, parsedBody);
|
|
58672
59415
|
} catch {
|
|
@@ -58680,7 +59423,7 @@ ${hdrText}
|
|
|
58680
59423
|
s3.metadata.learnedContextLimits = learnedMap;
|
|
58681
59424
|
log2("warn", `[${s3.id}] upstream context overflow \u2014 learned real window ${info.window} for ${reqModel ?? "(unknown model)"} (was ${prev ?? "unset"}); arming emergency shrink`);
|
|
58682
59425
|
} else {
|
|
58683
|
-
const payloadEstimate = estimateCoreMessages(prepared.processedMessages) + rejectedImageTokens;
|
|
59426
|
+
const payloadEstimate = (prepared.processedMessages.length > 0 ? estimateCoreMessages(prepared.processedMessages) : estimateRawBodyTokens(parsedBody)) + rejectedImageTokens;
|
|
58684
59427
|
const prev = (reqModel ? learnedMap[reqModel] : void 0) ?? s3.metadata.learnedContextLimit;
|
|
58685
59428
|
if (payloadEstimate >= 1e3 && (prev === void 0 || payloadEstimate < prev)) {
|
|
58686
59429
|
if (reqModel) learnedMap[reqModel] = payloadEstimate;
|
|
@@ -58704,6 +59447,18 @@ ${hdrText}
|
|
|
58704
59447
|
if (bodyText.length > 600) snippet += " \u2026";
|
|
58705
59448
|
if (!snippet) snippet = "(no body)";
|
|
58706
59449
|
log("warn", `[${errSid}] \u2190 upstream ${upstream.status}${reqIdText}: ${snippet}`);
|
|
59450
|
+
if (res.headersSent) {
|
|
59451
|
+
if (prepared?.stream) {
|
|
59452
|
+
emitStreamError(res, prepared.protocol, `upstream HTTP ${upstream.status}: ${snippet}`, (m2) => log("info", m2));
|
|
59453
|
+
} else {
|
|
59454
|
+
try {
|
|
59455
|
+
res.end(errBody ?? void 0);
|
|
59456
|
+
} catch {
|
|
59457
|
+
}
|
|
59458
|
+
}
|
|
59459
|
+
clearUpstreamTimer();
|
|
59460
|
+
return;
|
|
59461
|
+
}
|
|
58707
59462
|
const errHeaders = { ...respHeaders };
|
|
58708
59463
|
delete errHeaders["content-length"];
|
|
58709
59464
|
delete errHeaders["transfer-encoding"];
|
|
@@ -58712,7 +59467,7 @@ ${hdrText}
|
|
|
58712
59467
|
clearUpstreamTimer();
|
|
58713
59468
|
return;
|
|
58714
59469
|
}
|
|
58715
|
-
res.writeHead(upstream.status, respHeaders);
|
|
59470
|
+
if (!res.headersSent) res.writeHead(upstream.status, respHeaders);
|
|
58716
59471
|
if (!upstream.body) {
|
|
58717
59472
|
res.end();
|
|
58718
59473
|
clearUpstreamTimer();
|
|
@@ -58820,7 +59575,7 @@ ${hdrText}
|
|
|
58820
59575
|
const reqHeaders = buildForwardHeaders(headers);
|
|
58821
59576
|
const textProtocol = prepared.protocol === "responses" && !!prepared.responsesTextProtocol;
|
|
58822
59577
|
const systemPrompt = textProtocol ? buildCompressHybridSystemPrompt(prepared.prompts ?? defaultPrompts) : buildCompressSystemPrompt(prepared.prompts ?? defaultPrompts);
|
|
58823
|
-
const adapter = pickAdapter(prepared.protocol, parsedReq, textProtocol, prepared.responsesProjection, prepared.anthropicSystem, prepared.openaiSystemText);
|
|
59578
|
+
const adapter = pickAdapter(prepared.protocol, parsedReq, textProtocol, prepared.responsesProjection, prepared.anthropicSystem, prepared.openaiSystemText, prepared.session.hostCreditTokens ?? 0);
|
|
58824
59579
|
const refreshFolded = (current) => {
|
|
58825
59580
|
const turn = core.processTurn({
|
|
58826
59581
|
messages: prepared.originalMessages,
|
|
@@ -58831,13 +59586,13 @@ ${hdrText}
|
|
|
58831
59586
|
});
|
|
58832
59587
|
prepared.session.state = turn.state;
|
|
58833
59588
|
const records = current.filter((m2) => typeof m2.id === "string" && m2.id.startsWith("acp_loop_"));
|
|
58834
|
-
return stripKernelSummaries([...turn.messages, ...records], turn.state);
|
|
59589
|
+
return repairResponsesAssistantOrdering(stripKernelSummaries([...turn.messages, ...records], turn.state), prepared.originalMessages);
|
|
58835
59590
|
};
|
|
58836
59591
|
const loop = runCompressLoop(
|
|
58837
59592
|
streamToRead,
|
|
58838
|
-
{ core, config, messages: prepared.processedMessages.length > 0 ? prepared.processedMessages : prepared.originalMessages, compressMessages: prepared.originalMessages, session: prepared.session, log: ctx.log, proxyUrl, protocol: prepared.protocol, textProtocol, debug: opts.debug,
|
|
59593
|
+
{ core, config, messages: prepared.processedMessages.length > 0 ? prepared.processedMessages : prepared.originalMessages, compressMessages: prepared.originalMessages, session: prepared.session, log: ctx.log, proxyUrl, protocol: prepared.protocol, textProtocol, debug: opts.debug, refreshFolded },
|
|
58839
59594
|
parsedReq,
|
|
58840
|
-
{ url: upstreamUrl, headers: reqHeaders },
|
|
59595
|
+
{ url: upstreamUrl, headers: reqHeaders, wireTransform },
|
|
58841
59596
|
adapter,
|
|
58842
59597
|
systemPrompt,
|
|
58843
59598
|
clientAbort.signal
|
|
@@ -58884,7 +59639,7 @@ ${hdrText}
|
|
|
58884
59639
|
json,
|
|
58885
59640
|
{ core, config, messages: prepared.originalMessages, session: prepared.session, log: ctx.log, proxyUrl, textProtocol: true },
|
|
58886
59641
|
requestBody,
|
|
58887
|
-
{ url: upstreamUrl, headers: requestHeaders }
|
|
59642
|
+
{ url: upstreamUrl, headers: requestHeaders, wireTransform }
|
|
58888
59643
|
);
|
|
58889
59644
|
}
|
|
58890
59645
|
const u2 = json.usage ?? {};
|
|
@@ -58902,6 +59657,10 @@ ${hdrText}
|
|
|
58902
59657
|
const out = u2.completion_tokens ?? u2.output_tokens;
|
|
58903
59658
|
if (typeof out === "number") prepared.session.stats.outputTokens += out;
|
|
58904
59659
|
}
|
|
59660
|
+
const credit = prepared.session.hostCreditTokens ?? 0;
|
|
59661
|
+
if (credit > 0 && backfillHostUsage(prepared.protocol, u2, credit)) {
|
|
59662
|
+
prepared.session.hostContextTokens = (typeof total === "number" ? total : 0) + credit;
|
|
59663
|
+
}
|
|
58905
59664
|
if (prepared.protocol === "openai") {
|
|
58906
59665
|
rewriteOpenaiJsonResponse(json, ctx);
|
|
58907
59666
|
} else if (prepared.protocol === "responses") {
|
|
@@ -59045,6 +59804,7 @@ function handleConfigReload(opts, res, log2) {
|
|
|
59045
59804
|
for (const k2 of Object.keys(opts.routes)) delete opts.routes[k2];
|
|
59046
59805
|
Object.assign(opts.routes, fresh);
|
|
59047
59806
|
opts.compress = loadOptions().compress;
|
|
59807
|
+
opts.compat = loadOptions().compat;
|
|
59048
59808
|
resetProxyCache();
|
|
59049
59809
|
const names = Object.keys(fresh);
|
|
59050
59810
|
log2("info", `[acp-web] routes hot-reloaded (${names.length} providers): ${names.join(", ") || "(none)"}`);
|
|
@@ -62148,6 +62908,14 @@ async function findInstallDir(packageName) {
|
|
|
62148
62908
|
dir = parent;
|
|
62149
62909
|
}
|
|
62150
62910
|
}
|
|
62911
|
+
async function isGitWorkingTree(dir) {
|
|
62912
|
+
try {
|
|
62913
|
+
await access(path12.join(dir, ".git"));
|
|
62914
|
+
return true;
|
|
62915
|
+
} catch {
|
|
62916
|
+
return false;
|
|
62917
|
+
}
|
|
62918
|
+
}
|
|
62151
62919
|
async function readDiskVersion(installDir) {
|
|
62152
62920
|
try {
|
|
62153
62921
|
const pkg = JSON.parse(await readFile2(path12.join(installDir, "package.json"), "utf-8"));
|
|
@@ -62304,6 +63072,11 @@ async function checkForUpdate(opts, force = false) {
|
|
|
62304
63072
|
}
|
|
62305
63073
|
await writeLastCheck(now);
|
|
62306
63074
|
firstCheckDone = true;
|
|
63075
|
+
const installDir = await findInstallDir(opts.packageName);
|
|
63076
|
+
if (installDir && await isGitWorkingTree(installDir)) {
|
|
63077
|
+
log("info", `[update] running from a source checkout (${installDir}) \u2014 skipping auto-update (use npm install -g ${opts.packageName})`);
|
|
63078
|
+
return;
|
|
63079
|
+
}
|
|
62307
63080
|
log("info", `[update] checking npm registry for ${opts.packageName}${sinceLastSec < 0 ? " (startup check)" : sinceLastSec === 0 ? "" : ` (last check ${sinceLastSec}s ago)`}\u2026`);
|
|
62308
63081
|
const url = `${REGISTRY_BASE}/${opts.packageName}/latest`;
|
|
62309
63082
|
const res = await fetch(url, {
|
|
@@ -62320,7 +63093,6 @@ async function checkForUpdate(opts, force = false) {
|
|
|
62320
63093
|
log("warn", `[update] registry response had no version, skipping`);
|
|
62321
63094
|
return;
|
|
62322
63095
|
}
|
|
62323
|
-
const installDir = await findInstallDir(opts.packageName);
|
|
62324
63096
|
const diskVersion = installDir ? await readDiskVersion(installDir) : void 0;
|
|
62325
63097
|
const currentVersion = diskVersion ?? opts.currentVersion;
|
|
62326
63098
|
if (!isNewer(latest, currentVersion)) {
|
|
@@ -62391,6 +63163,9 @@ async function installViaTarball(version2, tarballUrl, installDir, integrity, sh
|
|
|
62391
63163
|
} catch {
|
|
62392
63164
|
return { ok: false, error: `install dir not writable: ${installDir}` };
|
|
62393
63165
|
}
|
|
63166
|
+
if (await isGitWorkingTree(installDir)) {
|
|
63167
|
+
return { ok: false, error: `install dir is a git working tree (${installDir}) \u2014 refusing to overwrite a source checkout (use npm install -g)` };
|
|
63168
|
+
}
|
|
62394
63169
|
const MAX_TARBALL_BYTES = 100 * 1024 * 1024;
|
|
62395
63170
|
let tgzBuffer;
|
|
62396
63171
|
try {
|
|
@@ -63104,6 +63879,11 @@ function unwrapUpstream2(url) {
|
|
|
63104
63879
|
const idx = url.indexOf("/bili/");
|
|
63105
63880
|
return idx >= 0 ? url.slice(idx + "/bili/".length) : url;
|
|
63106
63881
|
}
|
|
63882
|
+
function isLoopbackHost(host) {
|
|
63883
|
+
const h = host.toLowerCase();
|
|
63884
|
+
if (h === "localhost" || h === "::1" || h === "[::1]") return true;
|
|
63885
|
+
return /^127\.\d+\.\d+\.\d+$/.test(h);
|
|
63886
|
+
}
|
|
63107
63887
|
function resolveCaCertPath(env) {
|
|
63108
63888
|
const base = env.XDG_DATA_HOME || path15.join(os5.homedir(), ".local/share");
|
|
63109
63889
|
return path15.join(base, "billion-context", "ca", "root-ca.pem");
|
|
@@ -63116,6 +63896,7 @@ function discoverRoutes(client, config) {
|
|
|
63116
63896
|
const httpsDomains = [];
|
|
63117
63897
|
const httpRewrites = [];
|
|
63118
63898
|
const httpsRewrites = [];
|
|
63899
|
+
const httpEnvRoutes = [];
|
|
63119
63900
|
const httpsSeen = /* @__PURE__ */ new Set();
|
|
63120
63901
|
const rewriteKeys = /* @__PURE__ */ new Set();
|
|
63121
63902
|
const httpsRewriteKeys = /* @__PURE__ */ new Set();
|
|
@@ -63201,8 +63982,18 @@ function discoverRoutes(client, config) {
|
|
|
63201
63982
|
if (dshSeen.has(real)) continue;
|
|
63202
63983
|
dshSeen.add(real);
|
|
63203
63984
|
anon += 1;
|
|
63204
|
-
|
|
63205
|
-
|
|
63985
|
+
if (isLoopbackHost(url.hostname)) {
|
|
63986
|
+
rewriteKeys.add(`dsh-${anon}`);
|
|
63987
|
+
httpRewrites.push({ key: `dsh-${anon}`, realUpstream: real });
|
|
63988
|
+
} else if (url.protocol === "https:") {
|
|
63989
|
+
const host = url.hostname.toLowerCase();
|
|
63990
|
+
if (host && !httpsSeen.has(host)) {
|
|
63991
|
+
httpsSeen.add(host);
|
|
63992
|
+
httpsDomains.push(host);
|
|
63993
|
+
}
|
|
63994
|
+
} else if (!httpEnvRoutes.includes(real)) {
|
|
63995
|
+
httpEnvRoutes.push(real);
|
|
63996
|
+
}
|
|
63206
63997
|
} catch {
|
|
63207
63998
|
}
|
|
63208
63999
|
}
|
|
@@ -63212,7 +64003,7 @@ function discoverRoutes(client, config) {
|
|
|
63212
64003
|
}
|
|
63213
64004
|
classify(config.codex?.openaiBaseUrl, "openai_base_url");
|
|
63214
64005
|
}
|
|
63215
|
-
return { httpsDomains, httpRewrites, httpsRewrites };
|
|
64006
|
+
return { httpsDomains, httpRewrites, httpsRewrites, httpEnvRoutes };
|
|
63216
64007
|
}
|
|
63217
64008
|
function discoverDomains(client, config) {
|
|
63218
64009
|
return discoverRoutes(client, config).httpsDomains;
|
|
@@ -64116,7 +64907,7 @@ async function runLaunch(params, deps = {}) {
|
|
|
64116
64907
|
const domains = dedupeInOrder([...routes.httpsDomains, ...params.mitmDomains ?? []]);
|
|
64117
64908
|
const handle2 = await ensureProxyRunning({ host, port, passthrough, debug, mitmDomains: domains, modelWindows: collectModelWindows(config, base) }, deps);
|
|
64118
64909
|
console.error(
|
|
64119
|
-
`bili: started proxy at ${handle2.origin} (MITM domains: ${domains.length ? domains.join(", ") : "defaults"})` + (routes.httpRewrites.length > 0 ? ` (HTTP /bili/ rewrites: ${routes.httpRewrites.length})` : "") + (routes.httpsRewrites.length > 0 ? ` (HTTPS cert rewrites: ${routes.httpsRewrites.length})` : "") + (params.client === "pi-test" ? " (no extensions)" : "")
|
|
64910
|
+
`bili: started proxy at ${handle2.origin} (MITM domains: ${domains.length ? domains.join(", ") : "defaults"})` + (routes.httpRewrites.length > 0 ? ` (HTTP /bili/ rewrites: ${routes.httpRewrites.length})` : "") + (routes.httpsRewrites.length > 0 ? ` (HTTPS cert rewrites: ${routes.httpsRewrites.length})` : "") + (routes.httpEnvRoutes.length > 0 ? ` (HTTP proxy-env routes: ${routes.httpEnvRoutes.length})` : "") + (params.client === "pi-test" ? " (no extensions)" : "")
|
|
64120
64911
|
);
|
|
64121
64912
|
if (handle2.logPath) {
|
|
64122
64913
|
console.error(`bili: proxy log: ${handle2.logPath}`);
|
|
@@ -64181,18 +64972,25 @@ async function runLaunch(params, deps = {}) {
|
|
|
64181
64972
|
);
|
|
64182
64973
|
}
|
|
64183
64974
|
} else if (base === "dsh") {
|
|
64184
|
-
|
|
64975
|
+
const usesProxyEnv = routes.httpsDomains.length > 0 || routes.httpEnvRoutes.length > 0;
|
|
64976
|
+
env = usesProxyEnv ? stripInheritedProxy(process.env) : { ...process.env };
|
|
64977
|
+
env.BILLION_CONTEXT_PROXY = origin;
|
|
64185
64978
|
env.DEEPSEEK_BASE_URL = wrapUpstream(origin, "https://api.deepseek.com");
|
|
64979
|
+
if (usesProxyEnv) {
|
|
64980
|
+
env.HTTPS_PROXY = origin;
|
|
64981
|
+
env.SSL_CERT_FILE = resolveCombinedCaPath(process.env);
|
|
64982
|
+
}
|
|
64983
|
+
if (routes.httpEnvRoutes.length > 0) env.HTTP_PROXY = origin;
|
|
64186
64984
|
env.PI_CACHE_RETENTION = "long";
|
|
64187
64985
|
const dshHomeDir = resolveDshHome(process.env);
|
|
64188
|
-
dshOverlayHome = prepareDshHome(dshHomeDir, origin, routes.httpRewrites);
|
|
64986
|
+
dshOverlayHome = routes.httpRewrites.length > 0 ? prepareDshHome(dshHomeDir, origin, routes.httpRewrites) : void 0;
|
|
64189
64987
|
if (dshOverlayHome) {
|
|
64190
64988
|
env.DSH_HOME = dshOverlayHome;
|
|
64191
64989
|
} else if (routes.httpRewrites.length > 0) {
|
|
64192
64990
|
console.error(
|
|
64193
|
-
"bili: dsh settings.yaml could not be rewritten (unreadable or no matching endpoints) \u2014 custom providers will NOT go through the proxy;
|
|
64991
|
+
"bili: dsh settings.yaml could not be rewritten (unreadable or no matching endpoints) \u2014 loopback custom providers will NOT go through the proxy; other routes still do."
|
|
64194
64992
|
);
|
|
64195
|
-
} else {
|
|
64993
|
+
} else if (!usesProxyEnv) {
|
|
64196
64994
|
console.error(
|
|
64197
64995
|
"bili: no custom providers found in ~/.dsh/settings.yaml \u2014 proxying the built-in deepseek route via DEEPSEEK_BASE_URL only."
|
|
64198
64996
|
);
|
|
@@ -64355,10 +65153,13 @@ function renderHandoff(s3, full) {
|
|
|
64355
65153
|
lines.push("");
|
|
64356
65154
|
const messages = s3.lastMessages;
|
|
64357
65155
|
if (messages && messages.length > 0) {
|
|
64358
|
-
|
|
65156
|
+
const folded = s3.lastMessagesFolded === true;
|
|
65157
|
+
const view = full || folded ? messages : prune(messages, s3.state);
|
|
65158
|
+
lines.push(
|
|
65159
|
+
full && !folded ? `## Full conversation (${messages.length} messages)` : folded ? `## Conversation (persisted folded snapshot, ${messages.length} messages)` : `## Conversation (folded view as the model saw it, ${messages.length} client messages)`
|
|
65160
|
+
);
|
|
64359
65161
|
lines.push("");
|
|
64360
65162
|
let lastRole = "";
|
|
64361
|
-
const view = full ? messages : prune(messages, s3.state);
|
|
64362
65163
|
for (const m2 of view) {
|
|
64363
65164
|
if (m2.role !== lastRole) {
|
|
64364
65165
|
lines.push(`### ${m2.role}`);
|
|
@@ -64368,6 +65169,18 @@ function renderHandoff(s3, full) {
|
|
|
64368
65169
|
lines.push(renderMessage2(m2));
|
|
64369
65170
|
}
|
|
64370
65171
|
lines.push("");
|
|
65172
|
+
if (full && folded) {
|
|
65173
|
+
for (const b2 of s3.state.blocks.filter((x) => x.active)) {
|
|
65174
|
+
const content = s3.blockContents.get(b2.blockId);
|
|
65175
|
+
if (!content) continue;
|
|
65176
|
+
lines.push(`## Block ${b2.blockId}${b2.topic ? ` \u2014 ${b2.topic}` : ""}`);
|
|
65177
|
+
lines.push("");
|
|
65178
|
+
lines.push(`### Original messages (${content.full.count})`);
|
|
65179
|
+
lines.push("");
|
|
65180
|
+
lines.push(content.full.text.trim());
|
|
65181
|
+
lines.push("");
|
|
65182
|
+
}
|
|
65183
|
+
}
|
|
64371
65184
|
return lines.join("\n");
|
|
64372
65185
|
}
|
|
64373
65186
|
const active = s3.state.blocks.filter((b2) => b2.active);
|
|
@@ -64429,7 +65242,7 @@ async function exportSession(selector, opts = {}) {
|
|
|
64429
65242
|
const store = new SessionStore({ dir: opts.dir, enabled: true });
|
|
64430
65243
|
const all = [...(await store.loadAll()).values()];
|
|
64431
65244
|
if (all.length === 0) {
|
|
64432
|
-
return "No persisted sessions found. Sessions are written under the sessions directory once the proxy has served a request (compression state
|
|
65245
|
+
return "No persisted sessions found. Sessions are written under the sessions directory once the proxy has served a request (compression state, compressed originals, and a bounded folded-view snapshot of the recent conversation).";
|
|
64433
65246
|
}
|
|
64434
65247
|
if (!selector) {
|
|
64435
65248
|
const list = await listSessions2(opts);
|